diff --git a/docker/npu_patch/miles.patch b/docker/npu_patch/miles.patch index 9b9a02d61d0..0708d4d6717 100644 --- a/docker/npu_patch/miles.patch +++ b/docker/npu_patch/miles.patch @@ -340,7 +340,7 @@ diff --git a/miles/utils/external_utils/command_utils.py b/miles/utils/external_ index d016e01ac..08b4d6eff 100644 --- a/miles/utils/external_utils/command_utils.py +++ b/miles/utils/external_utils/command_utils.py -@@ -193,6 +193,112 @@ def execute_train( +@@ -193,6 +193,107 @@ def execute_train( ) @@ -434,18 +434,13 @@ index d016e01ac..08b4d6eff 100644 + ) + + if get_bool_env_var("SLIME_SCRIPT_ENABLE_RAY_SUBMIT", "1"): -+ cmd_megatron_model_source = ( -+ f'source "{repo_base_dir}/scripts/models/{megatron_model_type}.sh" && ' -+ if megatron_model_type is not None -+ else "" -+ ) ++ model_args = load_model_args(megatron_model_type) if megatron_model_type is not None else "" + exec_command_cpu( + f"export no_proxy=127.0.0.1 && export PYTHONUNBUFFERED=1 && " -+ f"{cmd_megatron_model_source}" + f'ray job submit --address="http://127.0.0.1:8265" ' + f"--runtime-env-json='{runtime_env_json}' " + f"-- python3 {train_script} " -+ f"{'${MODEL_ARGS[@]}' if megatron_model_type is not None else ''} " ++ f"{model_args} " + f"{train_args}" + ) + diff --git a/docs/advanced/on-policy-distillation.md b/docs/advanced/on-policy-distillation.md index 6723abb78ac..9235c7316eb 100644 --- a/docs/advanced/on-policy-distillation.md +++ b/docs/advanced/on-policy-distillation.md @@ -144,7 +144,8 @@ hf download --repo-type dataset zhuzilin/dapo-math-17k --local-dir /root/dapo-ma # 2. Convert student model cd /root/miles -source scripts/models/qwen3-8B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3-8B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ --hf-checkpoint /root/Qwen3-8B \ diff --git a/docs/examples/openhermes-sft.md b/docs/examples/openhermes-sft.md index c3893afaab7..b5b7ca5bd57 100644 --- a/docs/examples/openhermes-sft.md +++ b/docs/examples/openhermes-sft.md @@ -27,7 +27,8 @@ If you don't already have it: hf download Qwen/Qwen3-4B-Base --local-dir /root/Qwen3-4B-Base cd /root/miles -source scripts/models/qwen3-4B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3-4B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ --hf-checkpoint /root/Qwen3-4B-Base \ diff --git a/docs/examples/reproducibility.md b/docs/examples/reproducibility.md index b87bd45baf6..ad1ff17b350 100644 --- a/docs/examples/reproducibility.md +++ b/docs/examples/reproducibility.md @@ -70,7 +70,8 @@ hf download --repo-type dataset openai/gsm8k --local-dir /root/gsm8k hf download Qwen/Qwen2.5-0.5B-Instruct --local-dir /root/Qwen2.5-0.5B-Instruct cd /root/miles -source scripts/models/qwen2.5-0.5B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen2.5-0.5B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ --hf-checkpoint /root/Qwen2.5-0.5B-Instruct \ diff --git a/docs/examples/search-r1.md b/docs/examples/search-r1.md index e77a9689e57..c66ea664727 100644 --- a/docs/examples/search-r1.md +++ b/docs/examples/search-r1.md @@ -56,7 +56,8 @@ python $WORK_DIR/scripts/data_process/qa_search_train_merge.py \ ```bash hf download Qwen/Qwen2.5-3B --local-dir /root/Qwen2.5-3B cd /root/miles -source scripts/models/qwen2.5-3B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen2.5-3B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ --hf-checkpoint /root/Qwen2.5-3B \ diff --git a/docs/getting-started/quick-start.md b/docs/getting-started/quick-start.md index 21228399a7b..4207d4e98cd 100644 --- a/docs/getting-started/quick-start.md +++ b/docs/getting-started/quick-start.md @@ -65,7 +65,8 @@ map the HuggingFace weights into a sharded `torch_dist` checkpoint. ```bash cd /root/miles -source scripts/models/qwen3-4B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3-4B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ diff --git a/docs/models/deepseek/deepseek-v4-flash.md b/docs/models/deepseek/deepseek-v4-flash.md index e927666dc20..c1160c94d92 100644 --- a/docs/models/deepseek/deepseek-v4-flash.md +++ b/docs/models/deepseek/deepseek-v4-flash.md @@ -88,7 +88,8 @@ python tools/fp8_cast_bf16.py \ --input-fp8-hf-path /root/models/DeepSeek-V4-Flash-FP8 \ --output-bf16-hf-path /root/models/DeepSeek-V4-Flash-FP8-bf16/ -source scripts/models/deepseek-v4-flash.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py deepseek-v4-flash)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM torchrun \ --nproc-per-node 4 --nnodes 8 \ --master-addr ${MASTER_ADDR} --master-port 12345 \ diff --git a/docs/models/deepseek/deepseek.md b/docs/models/deepseek/deepseek.md index ffc3d4cba9c..da922a7b15d 100644 --- a/docs/models/deepseek/deepseek.md +++ b/docs/models/deepseek/deepseek.md @@ -52,7 +52,8 @@ Then convert BF16 HF → Megatron `torch_dist`. Run on **4 separate nodes** (`NO ```bash cd miles/ -source scripts/models/deepseek-v3.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py deepseek-v3)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM/ torchrun \ --nproc-per-node 8 \ --master-addr ${MASTER_ADDR} --master-port 12345 \ diff --git a/docs/models/glm/glm4-5.md b/docs/models/glm/glm4-5.md index 7aeb2492515..35a4e8d0458 100644 --- a/docs/models/glm/glm4-5.md +++ b/docs/models/glm/glm4-5.md @@ -49,7 +49,8 @@ The bash launcher does **not** convert for you — produce `$BASE_DIR/GLM-4.5-35 ```bash cd /root/miles -source scripts/models/glm4.5-355B-A32B.sh +MODEL_ARGS_LINE="$(python3 scripts/model_args.py glm4.5-355B-A32B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM torchrun --nproc-per-node 8 \ tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ diff --git a/docs/models/glm/glm4-7-flash.md b/docs/models/glm/glm4-7-flash.md index 07162c31fa3..31b368b3f15 100644 --- a/docs/models/glm/glm4-7-flash.md +++ b/docs/models/glm/glm4-7-flash.md @@ -35,7 +35,8 @@ The bash launcher hardcodes `BASE_DIR=/root/shared`. The Python launcher downloa ```bash cd /root/miles -source scripts/models/glm4.7-flash.sh +MODEL_ARGS_LINE="$(python3 scripts/model_args.py glm4.7-flash)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM torchrun --nproc-per-node 8 \ tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ diff --git a/docs/models/glm/glm4.md b/docs/models/glm/glm4.md index 688ba67ce82..04dcd74629d 100644 --- a/docs/models/glm/glm4.md +++ b/docs/models/glm/glm4.md @@ -32,7 +32,8 @@ hf download --repo-type dataset zhuzilin/aime-2024 --local-dir /root/aime-20 ```bash cd /root/miles -source scripts/models/glm4-9B.sh +MODEL_ARGS_LINE="$(python3 scripts/model_args.py glm4-9B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ --hf-checkpoint /root/GLM-Z1-9B-0414 \ diff --git a/docs/models/kimi/kimi-k2.5.md b/docs/models/kimi/kimi-k2.5.md index 5b766462c22..f4d924ff428 100644 --- a/docs/models/kimi/kimi-k2.5.md +++ b/docs/models/kimi/kimi-k2.5.md @@ -2,7 +2,7 @@ title: Kimi K2.5 / K2.6 description: Launch recipe for Kimi-K2.5, running full-parameter GRPO on 32 × 8 H200 with an INT4 actor and a BF16 reference. --- -The reference launcher is [`scripts/run-kimi-k25.sh`](https://github.com/radixark/miles/blob/main/scripts/run-kimi-k25.sh), which sources the shared model definition in `scripts/models/kimi-k2-thinking.sh`. +The reference launcher is [`scripts/run-kimi-k25.sh`](https://github.com/radixark/miles/blob/main/scripts/run-kimi-k25.sh), which loads the shared model definition from `scripts/models/kimi-k2-thinking.py`. ## 1. Model Introduction @@ -79,7 +79,7 @@ ray start --address=${MASTER_ADDR}:6379 --num-gpus 8 --node-ip-address ${WORKER_ ## 4. Script breakdown -The launcher groups its flags into the arrays that are passed to `train.py`. The model shape comes from `MODEL_ARGS`, which is sourced from `scripts/models/kimi-k2-thinking.sh`. That definition sets the MLA latent ranks (`q_lora_rank=1536`, `kv_lora_rank=512`, `qk_head_dim=128`, `qk_pos_emb_head_dim=64`, `v_head_dim=128`), the MoE routing (384 experts, top-8, sigmoid pre-softmax scoring, FP32 router, `--moe-router-topk-scaling-factor 2.827`), and RoPE (`--rotary-base 50000`, `--rotary-scaling-factor 64.0`). The K2.5 recipe then layers the following on top: +The launcher groups its flags into the arrays that are passed to `train.py`. The model shape comes from `MODEL_ARGS`, which is loaded from `scripts/models/kimi-k2-thinking.py`. That definition sets the MLA latent ranks (`q_lora_rank=1536`, `kv_lora_rank=512`, `qk_head_dim=128`, `qk_pos_emb_head_dim=64`, `v_head_dim=128`), the MoE routing (384 experts, top-8, sigmoid pre-softmax scoring, FP32 router, `--moe-router-topk-scaling-factor 2.827`), and RoPE (`--rotary-base 50000`, `--rotary-scaling-factor 64.0`). The K2.5 recipe then layers the following on top: - **`CKPT_ARGS`** wires up the dual checkpoint (INT4 actor via `--hf-checkpoint`, BF16 reference via `--ref-load`) together with `--megatron-to-hf-mode bridge` and `--model-name kimi_k25`. - **`ROLLOUT_ARGS`** and **`EVAL_ARGS`** configure GRPO sampling and periodic AIME evaluation (covered in §5.2). diff --git a/docs/models/kimi/kimi-k2.md b/docs/models/kimi/kimi-k2.md index 7bb410b834b..3315577492c 100644 --- a/docs/models/kimi/kimi-k2.md +++ b/docs/models/kimi/kimi-k2.md @@ -47,7 +47,8 @@ Convert across 4 nodes (mirror the DeepSeek-V3 procedure): ```bash cd /root/miles -source scripts/models/kimi-k2.sh # or kimi-k2-thinking.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py kimi-k2)" || exit 1 # or kimi-k2-thinking +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM/ torchrun \ --nproc-per-node 8 \ --master-addr ${MASTER_ADDR} --master-port 12345 \ diff --git a/docs/models/kimi/moonlight.md b/docs/models/kimi/moonlight.md index 5f1e781ba5c..b4244127b07 100644 --- a/docs/models/kimi/moonlight.md +++ b/docs/models/kimi/moonlight.md @@ -33,7 +33,8 @@ hf download --repo-type dataset zhuzilin/aime-2024 --local-dir /root/aime-20 ```bash cd /root/miles -source scripts/models/moonlight.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py moonlight)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM torchrun --nproc-per-node 8 \ tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ diff --git a/docs/models/mimo/mimo.md b/docs/models/mimo/mimo.md index bec36877023..0e35bb6f458 100644 --- a/docs/models/mimo/mimo.md +++ b/docs/models/mimo/mimo.md @@ -33,7 +33,8 @@ hf download --repo-type dataset zhuzilin/aime-2024 --local-dir /root/aime-20 ```bash cd /root/miles -source scripts/models/mimo-7B-rl.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py mimo-7B-rl)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ --hf-checkpoint /root/MiMo-7B-RL \ diff --git a/docs/models/nemotron/nemotron-3-nano-moe.md b/docs/models/nemotron/nemotron-3-nano-moe.md index ff4efdbf4af..77baa943c58 100644 --- a/docs/models/nemotron/nemotron-3-nano-moe.md +++ b/docs/models/nemotron/nemotron-3-nano-moe.md @@ -125,7 +125,7 @@ memory pressure rises. ### 5.5 Notable quirks -From `scripts/models/nemotron-3-nano-30b-a3b.sh` and `scripts/run-nemotron-3-nano-30b-a3b.sh`: +From `scripts/models/nemotron-3-nano-30b-a3b.py` and `scripts/run-nemotron-3-nano-30b-a3b.sh`: - **No `--spec`**: AutoBridge + the NemotronH shim synthesize the Megatron MoE spec from HF config. - 128 experts, `--moe-router-topk 6`, shared expert (3712-dim). diff --git a/docs/models/nemotron/nemotron-3-nano.md b/docs/models/nemotron/nemotron-3-nano.md index f45af012534..75dbe1f3147 100644 --- a/docs/models/nemotron/nemotron-3-nano.md +++ b/docs/models/nemotron/nemotron-3-nano.md @@ -112,7 +112,7 @@ memory pressure rises. ### 5.5 Notable quirks -From `scripts/models/nemotron-3-nano-4b.sh` and `scripts/run-nemotron-3-nano-4b.sh`: +From `scripts/models/nemotron-3-nano-4b.py` and `scripts/run-nemotron-3-nano-4b.sh`: - **No `--spec`**: the AutoBridge synthesizes the Megatron spec from HF config. - `--position-embedding-type none` (no RoPE). diff --git a/docs/models/qwen/qwen3-5-moe.md b/docs/models/qwen/qwen3-5-moe.md index 0c0fd7092b5..f5f24d66c78 100644 --- a/docs/models/qwen/qwen3-5-moe.md +++ b/docs/models/qwen/qwen3-5-moe.md @@ -32,7 +32,8 @@ hf download --repo-type dataset zhuzilin/aime-2024 --local-dir /root/aime-20 ```bash cd /root/miles -source scripts/models/qwen3.5-35B-A3B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3.5-35B-A3B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM torchrun --nproc-per-node 8 \ tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ @@ -98,7 +99,7 @@ CPU Adam is enabled (`--optimizer-cpu-offload --overlap-cpu-optimizer-d2h-h2d -- ### 5.5 Notable quirks - The Megatron side uses `--moe-token-dispatcher-type flex`; DeepEP isn't enabled here, unlike Qwen3-Next. -- The model config (`scripts/models/qwen3.5-35B-A3B.sh`) reuses the Qwen3.5 spec: `--attention-output-gate`, `--rotary-base 10000000`, `--rotary-percent 0.25`, `A_log` kept in FP32 via the bridge. See [Backends Beyond Megatron](/advanced/architecture-support). +- The model config (`scripts/models/qwen3.5-35B-A3B.py`) reuses the Qwen3.5 spec: `--attention-output-gate`, `--rotary-base 10000000`, `--rotary-percent 0.25`, `A_log` kept in FP32 via the bridge. See [Backends Beyond Megatron](/advanced/architecture-support). ## 6. Pairs Well With diff --git a/docs/models/qwen/qwen3-5.md b/docs/models/qwen/qwen3-5.md index 8169df811d9..9cfb1419c68 100644 --- a/docs/models/qwen/qwen3-5.md +++ b/docs/models/qwen/qwen3-5.md @@ -34,7 +34,8 @@ hf download --repo-type dataset zhuzilin/aime-2024 --local-dir /root/aime-20 ```bash cd /root/miles -source scripts/models/qwen3.5-4B.sh # or qwen3.5-9B.sh / qwen3.5-27B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3.5-4B)" || exit 1 # or qwen3.5-9B / qwen3.5-27B +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ --hf-checkpoint /root/Qwen3.5-4B \ @@ -92,7 +93,7 @@ Only the 27 B script enables CPU Adam (`--optimizer-cpu-offload --overlap-cpu-op ### 5.5 Notable quirks -From `scripts/models/qwen3.5-4B.sh` (and analogous configs for 9 B / 27 B): +From `scripts/models/qwen3.5-4B.py` (and analogous configs for 9 B / 27 B): - `--spec miles_plugins.models.qwen3_5 get_qwen3_5_spec` — attention-output gate, `A_log` parameter handling. - `--rotary-base 10000000`, `--rotary-percent 0.25`. diff --git a/docs/models/qwen/qwen3-6-moe.md b/docs/models/qwen/qwen3-6-moe.md index 1e0b21c87d9..d8b988d49cd 100644 --- a/docs/models/qwen/qwen3-6-moe.md +++ b/docs/models/qwen/qwen3-6-moe.md @@ -47,7 +47,8 @@ hf download Qwen/Qwen3.6-35B-A3B --local-dir /root/models/Qwen3.6-35B-A3B ```bash cd /root/miles -source scripts/models/qwen3.6-35B-A3B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3.6-35B-A3B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM torchrun --nproc-per-node 8 \ tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ @@ -136,7 +137,7 @@ CPU Adam is enabled (`--optimizer-cpu-offload --overlap-cpu-optimizer-d2h-h2d -- ### 5.5 Notable quirks -From `scripts/models/qwen3.6-35B-A3B.sh` and `scripts/run_qwen3_6_35b_a3b_mtp.py`: +From `scripts/models/qwen3.6-35B-A3B.py` and `scripts/run_qwen3_6_35b_a3b_mtp.py`: - `--spec miles_plugins.models.qwen3_5 get_qwen3_5_spec` — Qwen3.6 reuses the Qwen3.5 spec. - 256 experts, `--moe-router-topk 8`, `--moe-router-score-function softmax`. diff --git a/docs/models/qwen/qwen3-6.md b/docs/models/qwen/qwen3-6.md index adac7cbba70..9829ed746a7 100644 --- a/docs/models/qwen/qwen3-6.md +++ b/docs/models/qwen/qwen3-6.md @@ -45,7 +45,8 @@ hf download --repo-type dataset zhuzilin/aime-2024 --local-dir /root/aime-20 ```bash cd /root/miles -source scripts/models/qwen3.6-27B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3.6-27B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ --hf-checkpoint /root/Qwen3.6-27B \ @@ -109,7 +110,7 @@ CPU Adam is enabled (`--optimizer-cpu-offload --overlap-cpu-optimizer-d2h-h2d -- ### 5.5 Notable quirks -From `scripts/models/qwen3.6-27B.sh`: +From `scripts/models/qwen3.6-27B.py`: - `--spec miles_plugins.models.qwen3_5 get_qwen3_5_spec` — Qwen3.6 reuses the Qwen3.5 spec (gated attention, FP32 `A_log`). - `--rotary-base 10000000`, `--rotary-percent 0.25`. diff --git a/docs/models/qwen/qwen3-moe.md b/docs/models/qwen/qwen3-moe.md index 4c8f92c2b42..f043fc6ecca 100644 --- a/docs/models/qwen/qwen3-moe.md +++ b/docs/models/qwen/qwen3-moe.md @@ -47,7 +47,8 @@ hf download Qwen/Qwen3-235B-A22B-FP8 --local-dir $BASE_FOLDER/Qwen3-235B-A22B-FP ### 3.3 HF → Megatron `torch_dist` conversion ```bash -source scripts/models/qwen3-30B-A3B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3-30B-A3B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM torchrun --nproc-per-node 8 \ tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ diff --git a/docs/models/qwen/qwen3-next.md b/docs/models/qwen/qwen3-next.md index 3aff187dbe9..bc050a9073f 100644 --- a/docs/models/qwen/qwen3-next.md +++ b/docs/models/qwen/qwen3-next.md @@ -42,7 +42,8 @@ hf download --repo-type dataset zhuzilin/aime-2024 --local-dir $BASE_FOLDER/ ```bash cd /root/miles -source scripts/models/qwen3-next-80B-A3B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3-next-80B-A3B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM torchrun --nproc-per-node 8 \ tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ diff --git a/docs/models/qwen/qwen3.md b/docs/models/qwen/qwen3.md index b2f07f0e839..03f410a3488 100644 --- a/docs/models/qwen/qwen3.md +++ b/docs/models/qwen/qwen3.md @@ -39,7 +39,8 @@ hf download --repo-type dataset zhuzilin/aime-2024 --local-dir /root/aime-20 ```bash cd /root/miles -source scripts/models/qwen3-4B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3-4B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ --hf-checkpoint /root/Qwen3-4B \ @@ -57,9 +58,9 @@ cd /root/miles bash scripts/run-qwen3-4B.sh ``` -Other variants follow the same pattern — replace the script name (`run-qwen3-32B.sh`, etc.) and the `qwen3-XB.sh` model config. +Other variants follow the same pattern — replace the script name (`run-qwen3-32B.sh`, etc.) and the `qwen3-XB.py` model config. -The Qwen3-4B-Instruct-2507 config (`scripts/models/qwen3-4B-Instruct-2507.sh`) just sets `MODEL_ARGS_ROTARY_BASE=5000000` and re-sources `qwen3-4B.sh` — source it when converting / launching the Instruct-2507 checkpoint. +The Qwen3-4B-Instruct-2507 config (`scripts/models/qwen3-4B-Instruct-2507.py`) just calls `qwen3-4B` with `rotary_base=5000000` (`MODEL_ARGS_ROTARY_BASE` still works as an environment override) — load it when converting / launching the Instruct-2507 checkpoint. ## 5. Recipe Configuration diff --git a/docs/models/thinkingmachines/inkling-small.md b/docs/models/thinkingmachines/inkling-small.md index 5c0f5bc1aea..398ac67fa01 100644 --- a/docs/models/thinkingmachines/inkling-small.md +++ b/docs/models/thinkingmachines/inkling-small.md @@ -39,7 +39,7 @@ python scripts/run_inkling.py train \ --sglang-context-length 4096 --rollout-max-response-len 2048 ``` -The model definition lives in `scripts/models/inkling-small.sh` (`MODEL_ARGS_NUM_LAYERS` overrides the layer count for sliced smoke/parity checkpoints). HF → `torch_dist` conversion uses the same tool as Inkling with this recipe file — a single 8-GPU node (TP8 EP8) converts it in one pass. +The model definition lives in `scripts/models/inkling-small.py` (`MODEL_ARGS_NUM_LAYERS` overrides the layer count for sliced smoke/parity checkpoints). HF → `torch_dist` conversion uses the same tool as Inkling with this recipe file — a single 8-GPU node (TP8 EP8) converts it in one pass. ## 4. Validated parallelism diff --git a/docs/models/thinkingmachines/inkling.md b/docs/models/thinkingmachines/inkling.md index 31416e01328..801f9653f00 100644 --- a/docs/models/thinkingmachines/inkling.md +++ b/docs/models/thinkingmachines/inkling.md @@ -72,11 +72,12 @@ Pass `--hf-checkpoint ` to the launcher when the weights are already on a ### 4.2 HF → Megatron `torch_dist` conversion -Inkling ships in BF16, so conversion is a single distributed `torch_dist` shard (no precision cast). The model definition comes from `scripts/models/inkling.sh`: +Inkling ships in BF16, so conversion is a single distributed `torch_dist` shard (no precision cast). The model definition comes from `scripts/models/inkling.py`: ```bash cd /root/miles -source scripts/models/inkling.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py inkling)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CONVERT_KEEP_PP1=1 PYTHONPATH=/root/Megatron-LM torchrun \ --nproc-per-node 4 --nnodes 4 \ --master-addr ${MASTER_ADDR} --master-port 12345 \ diff --git a/docs/platforms/amd.md b/docs/platforms/amd.md index f1cd7bd3806..07f9baa2c8c 100644 --- a/docs/platforms/amd.md +++ b/docs/platforms/amd.md @@ -68,7 +68,8 @@ ROCm converter is in development. ```bash cd /root/miles -source scripts/models/qwen3-4B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3-4B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" MEGATRON_LM_PATH=$(pip list | grep megatron-core | awk '{print $NF}') PYTHONPATH=${MEGATRON_LM_PATH} python tools/convert_hf_to_torch_dist.py \ diff --git a/docs/user-guide/argument-groups.md b/docs/user-guide/argument-groups.md index 11e52e7e6b8..bb20a53b766 100644 --- a/docs/user-guide/argument-groups.md +++ b/docs/user-guide/argument-groups.md @@ -11,7 +11,7 @@ when you need the full default and type for an individual flag. | Group | Owns | Typical source | |---|---|---| -| [`MODEL_ARGS`](#model-args) | Architecture constants and plugin specs | `scripts/models/.sh` | +| [`MODEL_ARGS`](#model-args) | Architecture constants and plugin specs | `scripts/models/.py` | | [`CKPT_ARGS`](#ckpt-args) | Actor, reference, HF tokenizer/config, save paths | Launch script | | [`ROLLOUT_ARGS`](#rollout-args) | Prompt data, sampling, reward, train/eval batch flow | Launch script | | [`EVAL_ARGS`](#eval-args) | Evaluation datasets and eval-only sampling overrides | Launch script | @@ -24,7 +24,7 @@ when you need the full default and type for an individual flag. ## MODEL_ARGS - architecture constants `MODEL_ARGS` tells Megatron what model it is instantiating. Megatron cannot infer all -architecture details from a HuggingFace checkpoint, so each recipe sources a matching +architecture details from a HuggingFace checkpoint, so each recipe loads a matching file from `scripts/models/`. Common entries: diff --git a/docs/user-guide/training-script-walkthrough.md b/docs/user-guide/training-script-walkthrough.md index 341aef2dcc8..6499e630ae0 100644 --- a/docs/user-guide/training-script-walkthrough.md +++ b/docs/user-guide/training-script-walkthrough.md @@ -29,15 +29,16 @@ off to `train.py`: ## MODEL_ARGS — architecture constants Megatron needs the model architecture hardcoded at launch because it cannot introspect -a HuggingFace checkpoint. Miles therefore sources a matching bash file from -`scripts/models/.sh`: +a HuggingFace checkpoint. Miles therefore loads a matching python file from +`scripts/models/.py`: ```bash SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/glm4-9B.sh" +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" glm4-9B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" ``` -The sourced file sets `MODEL_ARGS=(--num-layers ... --hidden-size ... --rotary-base ...)`. +The loaded file prints `--num-layers ... --hidden-size ... --rotary-base ...` on one line. @@ -46,7 +47,8 @@ padding, or normalization epsilon. Diff the `config.json` against the file in `scripts/models/` before you run, and override anything that drifts: ```bash -source "${SCRIPT_DIR}/models/glm4-9B.sh" +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" glm4-9B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" MODEL_ARGS+=(--rotary-base 10000) ``` diff --git a/docs/user-guide/usage.md b/docs/user-guide/usage.md index 7dccf4d29f6..359dee8655a 100644 --- a/docs/user-guide/usage.md +++ b/docs/user-guide/usage.md @@ -106,7 +106,8 @@ the next run. Requires `--save` to be set. ### HuggingFace → torch_dist ```bash -source scripts/models/.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py )" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ --hf-checkpoint /root/ \ diff --git a/examples/experimental/eval/nemo_skills/README.md b/examples/experimental/eval/nemo_skills/README.md index e2a3aaad703..c9d5bf79f48 100644 --- a/examples/experimental/eval/nemo_skills/README.md +++ b/examples/experimental/eval/nemo_skills/README.md @@ -138,7 +138,8 @@ You need to convert the HF model to the format required by Megatron-LM. Ensure y ```bash # Source model arguments -source scripts/models/qwen3-4B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3-4B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" # Convert model PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ diff --git a/examples/experimental/eval/scripts/run-qwen3-32B.sh b/examples/experimental/eval/scripts/run-qwen3-32B.sh index 525bfe357d5..cd09054754a 100644 --- a/examples/experimental/eval/scripts/run-qwen3-32B.sh +++ b/examples/experimental/eval/scripts/run-qwen3-32B.sh @@ -30,7 +30,8 @@ echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" REPO_ROOT="$(cd -- "${SCRIPT_DIR}/../../../.." &>/dev/null && pwd)" -source "${REPO_ROOT}/scripts/models/qwen3-32B.sh" +MODEL_ARGS_LINE="$(python3 "${REPO_ROOT}/miles/utils/external_utils/model_args_utils.py" "qwen3-32B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" # Store eval/delegate settings in a YAML config similar to examples/experimental/eval_multi_task. EVAL_CONFIG_PATH=${SKILLS_EVAL_CONFIG_PATH:-"${REPO_ROOT}/examples/experimental/eval/scripts/multi_tasks.yaml"} diff --git a/examples/experimental/eval/scripts/run-qwen3-4B.sh b/examples/experimental/eval/scripts/run-qwen3-4B.sh index e2647973c80..250e6c996a8 100644 --- a/examples/experimental/eval/scripts/run-qwen3-4B.sh +++ b/examples/experimental/eval/scripts/run-qwen3-4B.sh @@ -31,7 +31,8 @@ echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" REPO_ROOT="$(cd -- "${SCRIPT_DIR}/../../../.." &>/dev/null && pwd)" -source "${REPO_ROOT}/scripts/models/qwen3-4B.sh" +MODEL_ARGS_LINE="$(python3 "${REPO_ROOT}/miles/utils/external_utils/model_args_utils.py" "qwen3-4B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" # Store eval/delegate settings in a YAML config similar to examples/experimental/eval_multi_task. EVAL_CONFIG_PATH=${SKILLS_EVAL_CONFIG_PATH:-"${REPO_ROOT}/examples/experimental/eval/scripts/multi_tasks.yaml"} diff --git a/examples/experimental/eval_multi_task/multi_task.sh b/examples/experimental/eval_multi_task/multi_task.sh index 090c461a005..6a5aeca0bb8 100644 --- a/examples/experimental/eval_multi_task/multi_task.sh +++ b/examples/experimental/eval_multi_task/multi_task.sh @@ -25,7 +25,8 @@ echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" REPO_ROOT="$(cd -- "${SCRIPT_DIR}/../../.." &>/dev/null && pwd)" -source "${REPO_ROOT}/scripts/models/qwen3-4B.sh" +MODEL_ARGS_LINE="$(python3 "${REPO_ROOT}/miles/utils/external_utils/model_args_utils.py" "qwen3-4B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" EVAL_CONFIG_PATH="${REPO_ROOT}/examples/experimental/eval_multi_task/multi_task.yaml" CKPT_ARGS=( diff --git a/examples/experimental/formal_math/single_round/run_minimal.py b/examples/experimental/formal_math/single_round/run_minimal.py index fa19ccf4524..7e7d7745387 100644 --- a/examples/experimental/formal_math/single_round/run_minimal.py +++ b/examples/experimental/formal_math/single_round/run_minimal.py @@ -8,6 +8,8 @@ import subprocess from pathlib import Path +from miles.utils.external_utils.model_args_utils import load_model_args + repo_base_dir = Path(os.path.abspath(__file__)).resolve().parents[4] MODEL_NAME, MODEL_TYPE = "Qwen3-8B", "qwen3-8B" @@ -131,11 +133,10 @@ cmd = ( f"export PYTHONUNBUFFERED=1 && " - f'source "{repo_base_dir}/scripts/models/{MODEL_TYPE}.sh" && ' f'ray job submit --address="http://127.0.0.1:8265" ' f"--runtime-env-json='{runtime_env_json}' " f"-- python3 train.py " - "${MODEL_ARGS[@]} " + f"{load_model_args(MODEL_TYPE)} " f"{train_args}" ) diff --git a/examples/experimental/multi_agent/run-qwen3-30B-A3B-multi-agent.sh b/examples/experimental/multi_agent/run-qwen3-30B-A3B-multi-agent.sh index fa484f0468d..3ca29aaefce 100644 --- a/examples/experimental/multi_agent/run-qwen3-30B-A3B-multi-agent.sh +++ b/examples/experimental/multi_agent/run-qwen3-30B-A3B-multi-agent.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../../scripts/models/qwen3-30B-A3B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../../miles/utils/external_utils/model_args_utils.py" "qwen3-30B-A3B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-30B-A3B #--hf-checkpoint /root/Qwen3-30B-A3B-FP8 diff --git a/examples/experimental/reproducibility/README.md b/examples/experimental/reproducibility/README.md index 08759d7aa16..cedfc7d366e 100644 --- a/examples/experimental/reproducibility/README.md +++ b/examples/experimental/reproducibility/README.md @@ -34,7 +34,8 @@ hf download Qwen/Qwen2.5-0.5B-Instruct --local-dir /root/Qwen2.5-0.5B-Instruct # convert ckpt cd miles/ -source scripts/models/qwen2.5-0.5B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen2.5-0.5B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM/ python \ tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ diff --git a/examples/experimental/reproducibility/run-qwen2.5-0.5B-gsm8k.sh b/examples/experimental/reproducibility/run-qwen2.5-0.5B-gsm8k.sh index f8d498ee2ee..b84b66c243b 100644 --- a/examples/experimental/reproducibility/run-qwen2.5-0.5B-gsm8k.sh +++ b/examples/experimental/reproducibility/run-qwen2.5-0.5B-gsm8k.sh @@ -16,7 +16,8 @@ set -ex export PYTHONUNBUFFERED=1 SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../../scripts/models/qwen2.5-0.5B.sh" +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../../miles/utils/external_utils/model_args_utils.py" "qwen2.5-0.5B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen2.5-0.5B-Instruct/ diff --git a/examples/experimental/search-r1/README.md b/examples/experimental/search-r1/README.md index 4cab49e363d..0089e021e5c 100644 --- a/examples/experimental/search-r1/README.md +++ b/examples/experimental/search-r1/README.md @@ -51,7 +51,8 @@ hf download Qwen/Qwen2.5-3B --local-dir /root/Qwen2.5-3B # mcore checkpoint cd /root/miles -source scripts/models/qwen2.5-3B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen2.5-3B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ --hf-checkpoint /root/Qwen2.5-3B \ @@ -70,17 +71,14 @@ SEARCH_R1_CONFIGS = { "max_turns": 2, "topk": 3, "search_concurrency": 256, - # ============== Search Backend Selection ============== "search_backend": "local", # Options: "local" or "google" - # ============== Local Search Configuration ============== # (Only used when search_backend="local") "local": { "search_url": "http://127.0.0.1:8000/retrieve", # URL of your local retrieval server "proxy": None, }, - # ============== Google Search Configuration ============== # (Only used when search_backend="google") "google": { @@ -88,10 +86,8 @@ SEARCH_R1_CONFIGS = { "snippet_only": True, "proxy": None, }, - # ============== Log Probability Collection ============== "return_logprob": True, # Set to True to collect log probabilities (required for TIS) - # ============== Reward Model Configuration ============== "format_score": 0.2, } diff --git a/examples/experimental/search-r1/run_qwen2.5_3B.sh b/examples/experimental/search-r1/run_qwen2.5_3B.sh index 798a29d75ba..ab0dd75cb3c 100644 --- a/examples/experimental/search-r1/run_qwen2.5_3B.sh +++ b/examples/experimental/search-r1/run_qwen2.5_3B.sh @@ -16,7 +16,8 @@ set -ex export PYTHONUNBUFFERED=1 SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../../scripts/models/qwen2.5-3B.sh" +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../../miles/utils/external_utils/model_args_utils.py" "qwen2.5-3B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen2.5-3B/ diff --git a/examples/experimental/strands_sglang/README.md b/examples/experimental/strands_sglang/README.md index 6fe4cce4d47..158b113d011 100644 --- a/examples/experimental/strands_sglang/README.md +++ b/examples/experimental/strands_sglang/README.md @@ -36,7 +36,8 @@ hf download Qwen/Qwen3-8B --local-dir /root/models/Qwen/Qwen3-8B # mcore checkpoint cd /root/miles -source scripts/models/qwen3-8B.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3-8B)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ --hf-checkpoint /root/models/Qwen/Qwen3-8B \ @@ -49,6 +50,7 @@ Following [Retool](https://arxiv.org/abs/2504.11536), we use `dapo-math-17k` as ```python from datasets import load_dataset + ds = load_dataset("zhuzilin/dapo-math-17k", split="train") ds.to_json("/root/data/dapo-math-17k.jsonl", orient="records", lines=True) ``` @@ -57,6 +59,7 @@ and `aime-2024` as eval data: ```python from datasets import load_dataset + ds = load_dataset("zhuzilin/aime-2024", split="train") ds.to_json("/root/data/aime-2024.jsonl", orient="records", lines=True) ``` diff --git a/examples/experimental/strands_sglang/strands_qwen3_8b.sh b/examples/experimental/strands_sglang/strands_qwen3_8b.sh index 9e4aa1f45e4..79a49a6467a 100644 --- a/examples/experimental/strands_sglang/strands_qwen3_8b.sh +++ b/examples/experimental/strands_sglang/strands_qwen3_8b.sh @@ -28,8 +28,8 @@ echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" MILES_ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../../scripts/models/qwen3-8B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../../miles/utils/external_utils/model_args_utils.py" "qwen3-8B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" # Generate timestamp suffix for save path TIMESTAMP_SUFFIX=$(date +%Y%m%d_%H%M%S) diff --git a/examples/experimental/tau-bench/README.md b/examples/experimental/tau-bench/README.md index 2959ef1dd7f..b0179ad7017 100644 --- a/examples/experimental/tau-bench/README.md +++ b/examples/experimental/tau-bench/README.md @@ -33,7 +33,8 @@ hf download Qwen/Qwen3-4B-Instruct-2507 --local-dir /root/Qwen3-4B-Instruct-2507 # mcore checkpoint cd /root/miles -source scripts/models/qwen3-4B-Instruct-2507.sh +MODEL_ARGS_LINE="$(python3 miles/utils/external_utils/model_args_utils.py qwen3-4B-Instruct-2507)" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" PYTHONPATH=/root/Megatron-LM python tools/convert_hf_to_torch_dist.py \ ${MODEL_ARGS[@]} \ --hf-checkpoint /root/Qwen3-4B-Instruct-2507 \ diff --git a/examples/experimental/tau-bench/run_qwen3_4B.sh b/examples/experimental/tau-bench/run_qwen3_4B.sh index 172834e79f1..184c0fa5e9d 100644 --- a/examples/experimental/tau-bench/run_qwen3_4B.sh +++ b/examples/experimental/tau-bench/run_qwen3_4B.sh @@ -24,7 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../../scripts/models/qwen3-4B-Instruct-2507.sh" +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../../miles/utils/external_utils/model_args_utils.py" "qwen3-4B-Instruct-2507")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-4B-Instruct-2507/ diff --git a/examples/fully_async/run-qwen3-4b-fully_async.sh b/examples/fully_async/run-qwen3-4b-fully_async.sh index 44445b86472..07dd2b6ee5f 100644 --- a/examples/fully_async/run-qwen3-4b-fully_async.sh +++ b/examples/fully_async/run-qwen3-4b-fully_async.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../scripts/models/qwen3-4B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../miles/utils/external_utils/model_args_utils.py" "qwen3-4B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-4B #--hf-checkpoint /root/Qwen3-4B-FP8 diff --git a/examples/geo3k_vlm/run_geo3k_vlm.sh b/examples/geo3k_vlm/run_geo3k_vlm.sh index e95c55ebb4b..3da32be58c6 100644 --- a/examples/geo3k_vlm/run_geo3k_vlm.sh +++ b/examples/geo3k_vlm/run_geo3k_vlm.sh @@ -188,7 +188,8 @@ else MILES_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." &>/dev/null && pwd)" MODEL_ARGS_FILE=$(echo "$MODEL_NAME" | sed 's/-Instruct//g; s/-Thinking//g; s/Qwen3-VL-/qwen3-/g; s/-2B/-1.7B/g') # VL models require rotary-base 5000000 - MODEL_ARGS_ROTARY_BASE=5000000 source "${MILES_DIR}/scripts/models/${MODEL_ARGS_FILE}.sh" + MODEL_ARGS_LINE="$(MODEL_ARGS_ROTARY_BASE=5000000 python3 "${MILES_DIR}/miles/utils/external_utils/model_args_utils.py" "${MODEL_ARGS_FILE}")" || exit 1 + read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" fi diff --git a/examples/geo3k_vlm/run_geo3k_vlm_sft.sh b/examples/geo3k_vlm/run_geo3k_vlm_sft.sh index 7975c4c07f7..023f816a695 100644 --- a/examples/geo3k_vlm/run_geo3k_vlm_sft.sh +++ b/examples/geo3k_vlm/run_geo3k_vlm_sft.sh @@ -152,7 +152,8 @@ else MILES_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." &>/dev/null && pwd)" MODEL_ARGS_FILE=$(echo "$MODEL_NAME" | sed 's/-Instruct//g; s/-Thinking//g; s/Qwen3-VL-/qwen3-/g; s/-2B/-1.7B/g') # VL models require rotary-base 5000000 - MODEL_ARGS_ROTARY_BASE=5000000 source "${MILES_DIR}/scripts/models/${MODEL_ARGS_FILE}.sh" + MODEL_ARGS_LINE="$(MODEL_ARGS_ROTARY_BASE=5000000 python3 "${MILES_DIR}/miles/utils/external_utils/model_args_utils.py" "${MODEL_ARGS_FILE}")" || exit 1 + read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" fi # Start Ray if not using external Ray diff --git a/examples/infra_features/low_precision/run-kimi-k2-Thinking-int4.sh b/examples/infra_features/low_precision/run-kimi-k2-Thinking-int4.sh index f7fa4840e2f..0572681da73 100644 --- a/examples/infra_features/low_precision/run-kimi-k2-Thinking-int4.sh +++ b/examples/infra_features/low_precision/run-kimi-k2-Thinking-int4.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../../scripts/models/kimi-k2-thinking.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../../miles/utils/external_utils/model_args_utils.py" "kimi-k2-thinking")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Kimi-K2-Thinking/ --ref-load /root/Kimi-K2_thinking_torch_dist/ diff --git a/examples/infra_features/low_precision/run-moonlight-16B-A3B-int4.sh b/examples/infra_features/low_precision/run-moonlight-16B-A3B-int4.sh index 17b81f2ee83..f15a4245710 100644 --- a/examples/infra_features/low_precision/run-moonlight-16B-A3B-int4.sh +++ b/examples/infra_features/low_precision/run-moonlight-16B-A3B-int4.sh @@ -25,8 +25,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../../scripts/models/moonlight.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../../miles/utils/external_utils/model_args_utils.py" "moonlight")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Moonlight-16B-A3B-Instruct-INT4 --ref-load /root/Moonlight-16B-A3B-Instruct-INT4_torch_dist diff --git a/examples/infra_features/low_precision/run-qwen3-235B-A22B-int4.sh b/examples/infra_features/low_precision/run-qwen3-235B-A22B-int4.sh index 490f8e41418..54a47fba575 100644 --- a/examples/infra_features/low_precision/run-qwen3-235B-A22B-int4.sh +++ b/examples/infra_features/low_precision/run-qwen3-235B-A22B-int4.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../../scripts/models/qwen3-235B-A22B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../../miles/utils/external_utils/model_args_utils.py" "qwen3-235B-A22B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-235B-A22B-INT4/ --ref-load /root/Qwen3-235B-A22B_torch_dist/ diff --git a/examples/infra_features/low_precision/run-qwen3-30B-A3B-int4.sh b/examples/infra_features/low_precision/run-qwen3-30B-A3B-int4.sh index 0ff20072ec8..9b633d938fc 100644 --- a/examples/infra_features/low_precision/run-qwen3-30B-A3B-int4.sh +++ b/examples/infra_features/low_precision/run-qwen3-30B-A3B-int4.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../../scripts/models/qwen3-30B-A3B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../../miles/utils/external_utils/model_args_utils.py" "qwen3-30B-A3B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-30B-A3B-INT4/ --ref-load /root/Qwen3-30B-A3B_torch_dist/ diff --git a/examples/infra_features/low_precision/run-qwen3-30b-a3b-fp8-two-nodes.sh b/examples/infra_features/low_precision/run-qwen3-30b-a3b-fp8-two-nodes.sh index 0f6fbf6b5be..19cca96e043 100644 --- a/examples/infra_features/low_precision/run-qwen3-30b-a3b-fp8-two-nodes.sh +++ b/examples/infra_features/low_precision/run-qwen3-30b-a3b-fp8-two-nodes.sh @@ -25,7 +25,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../../scripts/models/qwen3-30B-A3B.sh" +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../../miles/utils/external_utils/model_args_utils.py" "qwen3-30B-A3B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" # Base directory for checkpoints and related files (adjust if necessary) BASE_DIR="/root" diff --git a/examples/infra_features/low_precision/run-qwen3-4b-fp8.sh b/examples/infra_features/low_precision/run-qwen3-4b-fp8.sh index bf8f6407aeb..205cda89a72 100644 --- a/examples/infra_features/low_precision/run-qwen3-4b-fp8.sh +++ b/examples/infra_features/low_precision/run-qwen3-4b-fp8.sh @@ -24,7 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../../scripts/models/qwen3-4B.sh" +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../../miles/utils/external_utils/model_args_utils.py" "qwen3-4B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-4B-FP8 diff --git a/examples/infra_features/p2p_weight_transfer/run-glm4.5-air-8node-profile.sh b/examples/infra_features/p2p_weight_transfer/run-glm4.5-air-8node-profile.sh index b6981ced2f3..652b310a393 100755 --- a/examples/infra_features/p2p_weight_transfer/run-glm4.5-air-8node-profile.sh +++ b/examples/infra_features/p2p_weight_transfer/run-glm4.5-air-8node-profile.sh @@ -72,8 +72,8 @@ MODEL_NAME="GLM-4.5-Air" MODEL_TYPE="glm4.5-106B-A12B" MILES_ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../../.." &>/dev/null && pwd)" -source "${MILES_ROOT}/scripts/models/${MODEL_TYPE}.sh" - +MODEL_ARGS_LINE="$(python3 "${MILES_ROOT}/miles/utils/external_utils/model_args_utils.py" "${MODEL_TYPE}")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" # Rotary base override export MODEL_ARGS_ROTARY_BASE=1000000 diff --git a/examples/infra_features/p2p_weight_transfer/run-glm4.7-flash-2node-profile.sh b/examples/infra_features/p2p_weight_transfer/run-glm4.7-flash-2node-profile.sh index b2604b5c24f..9062fa0cfd3 100644 --- a/examples/infra_features/p2p_weight_transfer/run-glm4.7-flash-2node-profile.sh +++ b/examples/infra_features/p2p_weight_transfer/run-glm4.7-flash-2node-profile.sh @@ -80,8 +80,8 @@ MODEL_TYPE="glm4.7-flash" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" MILES_ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../../.." &>/dev/null && pwd)" -source "${MILES_ROOT}/scripts/models/${MODEL_TYPE}.sh" - +MODEL_ARGS_LINE="$(python3 "${MILES_ROOT}/miles/utils/external_utils/model_args_utils.py" "${MODEL_TYPE}")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" # --------------------------------------------------------------------------- # Determine modes to run # --------------------------------------------------------------------------- diff --git a/examples/infra_features/p2p_weight_transfer/run-glm5-disagg-profile.sh b/examples/infra_features/p2p_weight_transfer/run-glm5-disagg-profile.sh index 0a279c28dc0..7ce1df610f8 100755 --- a/examples/infra_features/p2p_weight_transfer/run-glm5-disagg-profile.sh +++ b/examples/infra_features/p2p_weight_transfer/run-glm5-disagg-profile.sh @@ -116,8 +116,8 @@ esac NUM_TRAIN_NODES=$((NUM_TRAIN_GPUS / GPUS_PER_NODE)) MILES_ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../../.." &>/dev/null && pwd)" -source "${MILES_ROOT}/scripts/models/${MODEL_TYPE}.sh" - +MODEL_ARGS_LINE="$(python3 "${MILES_ROOT}/miles/utils/external_utils/model_args_utils.py" "${MODEL_TYPE}")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" echo "" echo "============================================================" echo " Model : ${MODEL_NAME} (${MODEL_TYPE})" diff --git a/examples/infra_features/p2p_weight_transfer/run-kimi-k2-64node-profile.sh b/examples/infra_features/p2p_weight_transfer/run-kimi-k2-64node-profile.sh index 8dc4e008533..ddcdd9c4473 100644 --- a/examples/infra_features/p2p_weight_transfer/run-kimi-k2-64node-profile.sh +++ b/examples/infra_features/p2p_weight_transfer/run-kimi-k2-64node-profile.sh @@ -85,9 +85,8 @@ MODEL_TYPE="kimi-k2" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" MILES_ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../../.." &>/dev/null && pwd)" -source "${MILES_ROOT}/scripts/models/${MODEL_TYPE}.sh" - - +MODEL_ARGS_LINE="$(python3 "${MILES_ROOT}/miles/utils/external_utils/model_args_utils.py" "${MODEL_TYPE}")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" # --------------------------------------------------------------------------- # Determine modes to run # --------------------------------------------------------------------------- diff --git a/examples/infra_features/p2p_weight_transfer/run-qwen3-235B-A22B-16node-profile.sh b/examples/infra_features/p2p_weight_transfer/run-qwen3-235B-A22B-16node-profile.sh index 129d0abb990..57a28d1ca7c 100644 --- a/examples/infra_features/p2p_weight_transfer/run-qwen3-235B-A22B-16node-profile.sh +++ b/examples/infra_features/p2p_weight_transfer/run-qwen3-235B-A22B-16node-profile.sh @@ -82,9 +82,8 @@ MODEL_TYPE="qwen3-235B-A22B" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" MILES_ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../../.." &>/dev/null && pwd)" export MODEL_ARGS_ROTARY_BASE=5000000 -source "${MILES_ROOT}/scripts/models/${MODEL_TYPE}.sh" - - +MODEL_ARGS_LINE="$(python3 "${MILES_ROOT}/miles/utils/external_utils/model_args_utils.py" "${MODEL_TYPE}")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" # --------------------------------------------------------------------------- # Determine modes to run # --------------------------------------------------------------------------- diff --git a/examples/infra_features/p2p_weight_transfer/run-qwen3-30B-A3B-4node-profile.sh b/examples/infra_features/p2p_weight_transfer/run-qwen3-30B-A3B-4node-profile.sh index e17a48445fd..43b610f3f6e 100644 --- a/examples/infra_features/p2p_weight_transfer/run-qwen3-30B-A3B-4node-profile.sh +++ b/examples/infra_features/p2p_weight_transfer/run-qwen3-30B-A3B-4node-profile.sh @@ -84,8 +84,8 @@ MODEL_TYPE="qwen3-30B-A3B" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" MILES_ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../../.." &>/dev/null && pwd)" -source "${MILES_ROOT}/scripts/models/${MODEL_TYPE}.sh" - +MODEL_ARGS_LINE="$(python3 "${MILES_ROOT}/miles/utils/external_utils/model_args_utils.py" "${MODEL_TYPE}")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" # Rotary base override export MODEL_ARGS_ROTARY_BASE=1000000 diff --git a/examples/infra_features/p2p_weight_transfer/run.py b/examples/infra_features/p2p_weight_transfer/run.py index dd14e9b145f..1014b13a97b 100644 --- a/examples/infra_features/p2p_weight_transfer/run.py +++ b/examples/infra_features/p2p_weight_transfer/run.py @@ -35,7 +35,7 @@ class PrepareConfig: """Configuration for the `prepare` subcommand.""" hf_repo: str - model_type: str # megatron model type (maps to scripts/models/.sh) + model_type: str # megatron model type (maps to scripts/models/.py) datasets: list[str] = field(default_factory=lambda: ["zhuzilin/dapo-math-17k"]) convert_gpus_per_node: int = 8 convert_multinode: bool = False @@ -1101,7 +1101,11 @@ def cmd_run( def build_model_args_command(cfg: RunConfig) -> str: """A shell snippet leaving MODEL_ARGS set; the knobs must reach it, not only ray's runtime env.""" prefix = "".join(f"{name}={shlex.quote(value)} " for name, value in build_model_args_env(cfg).items()) - return f'{prefix}source "{MILES_ROOT}/scripts/models/{cfg.model_type}.sh"' + return ( + f'MODEL_ARGS_LINE="$({prefix}python3 "{MILES_ROOT}/miles/utils/external_utils/model_args_utils.py"' + f' {cfg.model_type})" || exit 1; ' + 'read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}"' + ) def build_model_args_env(cfg: RunConfig) -> dict[str, str]: diff --git a/examples/infra_features/train_infer_mismatch_helper/run-qwen3-4b-mis.sh b/examples/infra_features/train_infer_mismatch_helper/run-qwen3-4b-mis.sh index 4a3f0aeda77..1390d3ac836 100644 --- a/examples/infra_features/train_infer_mismatch_helper/run-qwen3-4b-mis.sh +++ b/examples/infra_features/train_infer_mismatch_helper/run-qwen3-4b-mis.sh @@ -24,7 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../../scripts/models/qwen3-4B.sh" +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../../miles/utils/external_utils/model_args_utils.py" "qwen3-4B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-4B diff --git a/examples/lora/dev.sh b/examples/lora/dev.sh index 80648cfb810..0b2145efb94 100644 --- a/examples/lora/dev.sh +++ b/examples/lora/dev.sh @@ -19,8 +19,8 @@ set -ex SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../scripts/models/qwen2.5-3B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../miles/utils/external_utils/model_args_utils.py" "qwen2.5-3B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen2.5-3B-Instruct/ --megatron-to-hf-mode bridge diff --git a/examples/lora/run-gpt-oss-20B-megatron-moe-lora.sh b/examples/lora/run-gpt-oss-20B-megatron-moe-lora.sh index e26d663fa23..098a0e76865 100644 --- a/examples/lora/run-gpt-oss-20B-megatron-moe-lora.sh +++ b/examples/lora/run-gpt-oss-20B-megatron-moe-lora.sh @@ -16,8 +16,8 @@ GPUS_PER_NODE=$(echo "$CUDA_VISIBLE_DEVICES" | tr ',' '\n' | wc -l) # Load model architecture config SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../scripts/models/gpt-oss-20b.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../miles/utils/external_utils/model_args_utils.py" "gpt-oss-20b")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/models/gpt-oss-20b --megatron-to-hf-mode bridge diff --git a/examples/lora/run-kimi-k25-megatron-lora.sh b/examples/lora/run-kimi-k25-megatron-lora.sh index f43f141dabd..ec25afdf91c 100755 --- a/examples/lora/run-kimi-k25-megatron-lora.sh +++ b/examples/lora/run-kimi-k25-megatron-lora.sh @@ -28,8 +28,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../scripts/models/kimi-k2-thinking.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../miles/utils/external_utils/model_args_utils.py" "kimi-k2-thinking")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint $BASE_DIR/Kimi-K2.5-int4 --ref-load $BASE_DIR/Kimi-K2.5-bf16 diff --git a/examples/lora/run-qwen2.5-0.5B-megatron-lora.sh b/examples/lora/run-qwen2.5-0.5B-megatron-lora.sh index bc287acd695..be29ca03790 100644 --- a/examples/lora/run-qwen2.5-0.5B-megatron-lora.sh +++ b/examples/lora/run-qwen2.5-0.5B-megatron-lora.sh @@ -18,8 +18,8 @@ set -ex SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../scripts/models/qwen2.5-0.5B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../miles/utils/external_utils/model_args_utils.py" "qwen2.5-0.5B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen2.5-0.5B-Instruct/ --megatron-to-hf-mode bridge diff --git a/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated-multi-node.sh b/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated-multi-node.sh index 592c77148a2..ab8556f719d 100644 --- a/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated-multi-node.sh +++ b/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated-multi-node.sh @@ -134,8 +134,8 @@ pkill -9 python || true set -ex SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../scripts/models/qwen2.5-3B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../miles/utils/external_utils/model_args_utils.py" "qwen2.5-3B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen2.5-3B-Instruct/ --megatron-to-hf-mode bridge diff --git a/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated.sh b/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated.sh index f04fabf65bf..7687a83d3c2 100644 --- a/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated.sh +++ b/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated.sh @@ -18,8 +18,8 @@ set -ex SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../scripts/models/qwen2.5-3B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../miles/utils/external_utils/model_args_utils.py" "qwen2.5-3B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen2.5-3B-Instruct/ --megatron-to-hf-mode bridge diff --git a/examples/lora/run-qwen3-4B-megatron-lora.sh b/examples/lora/run-qwen3-4B-megatron-lora.sh index 48ed58344e6..e47a50fc1f5 100644 --- a/examples/lora/run-qwen3-4B-megatron-lora.sh +++ b/examples/lora/run-qwen3-4B-megatron-lora.sh @@ -30,7 +30,8 @@ echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" REPO_ROOT="$(cd -- "${SCRIPT_DIR}/../.." &>/dev/null && pwd)" -source "${REPO_ROOT}/scripts/models/qwen3-4B.sh" +MODEL_ARGS_LINE="$(python3 "${REPO_ROOT}/miles/utils/external_utils/model_args_utils.py" "qwen3-4B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" # Store eval/delegate settings in a YAML config similar to examples/experimental/eval_multi_task. # EVAL_CONFIG_PATH=${SKILLS_EVAL_CONFIG_PATH:-"${REPO_ROOT}/examples/experimental/eval/scripts/multi_tasks.yaml"} diff --git a/examples/lora/run-qwen3-4b-megatron-lora-result.sh b/examples/lora/run-qwen3-4b-megatron-lora-result.sh index 0d76d807cb3..624a76892d5 100644 --- a/examples/lora/run-qwen3-4b-megatron-lora-result.sh +++ b/examples/lora/run-qwen3-4b-megatron-lora-result.sh @@ -33,8 +33,8 @@ echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" LR=2e-5 SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../scripts/models/qwen3-4B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../miles/utils/external_utils/model_args_utils.py" "qwen3-4B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-4B --save /root/Qwen3-4B-lora-ckpt diff --git a/examples/on_policy_distillation/run-qwen3-8B-opd-megatron.sh b/examples/on_policy_distillation/run-qwen3-8B-opd-megatron.sh index 0e971aae269..df65334a7d2 100644 --- a/examples/on_policy_distillation/run-qwen3-8B-opd-megatron.sh +++ b/examples/on_policy_distillation/run-qwen3-8B-opd-megatron.sh @@ -22,9 +22,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../scripts/models/qwen3-8B.sh" - - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../miles/utils/external_utils/model_args_utils.py" "qwen3-8B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-8B --ref-load /root/Qwen3-8B_torch_dist diff --git a/examples/on_policy_distillation/run-qwen3-8B-opd-multi-teacher.sh b/examples/on_policy_distillation/run-qwen3-8B-opd-multi-teacher.sh index fad2c0aadee..1dae8db8a02 100644 --- a/examples/on_policy_distillation/run-qwen3-8B-opd-multi-teacher.sh +++ b/examples/on_policy_distillation/run-qwen3-8B-opd-multi-teacher.sh @@ -97,7 +97,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../scripts/models/qwen3-8B.sh" +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../miles/utils/external_utils/model_args_utils.py" "qwen3-8B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( diff --git a/examples/on_policy_distillation/run-qwen3-8B-opd.sh b/examples/on_policy_distillation/run-qwen3-8B-opd.sh index 1389138133f..953c380c00d 100644 --- a/examples/on_policy_distillation/run-qwen3-8B-opd.sh +++ b/examples/on_policy_distillation/run-qwen3-8B-opd.sh @@ -45,9 +45,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../../scripts/models/qwen3-8B.sh" - - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../../miles/utils/external_utils/model_args_utils.py" "qwen3-8B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-8B --ref-load /root/Qwen3-8B_torch_dist diff --git a/miles/utils/debug_utils/run_megatron/cli/commands/args.py b/miles/utils/debug_utils/run_megatron/cli/commands/args.py index a851b0e400c..818a709eaaf 100644 --- a/miles/utils/debug_utils/run_megatron/cli/commands/args.py +++ b/miles/utils/debug_utils/run_megatron/cli/commands/args.py @@ -15,7 +15,7 @@ def _field( @dataclasses.dataclass class CommonRunArgs: - model_type: str = _field(help="Model type matching scripts/models/{model_type}.sh") + model_type: str = _field(help="Model type matching scripts/models/{model_type}.py") hf_checkpoint: Path = _field(help="HuggingFace checkpoint path") ref_load: Path | None = _field(default=None, help="Megatron checkpoint path") sp: bool = _field(default=False, help="Enable sequence parallelism") diff --git a/miles/utils/debug_utils/run_megatron/cli/commands/run.py b/miles/utils/debug_utils/run_megatron/cli/commands/run.py index 4f1a53659f8..e83e98e1499 100644 --- a/miles/utils/debug_utils/run_megatron/cli/commands/run.py +++ b/miles/utils/debug_utils/run_megatron/cli/commands/run.py @@ -7,7 +7,7 @@ from miles.utils.debug_utils.run_megatron.cli.commands.args import RunArgs from miles.utils.debug_utils.run_megatron.cli.parallel_utils import ParallelConfig -from miles.utils.debug_utils.run_megatron.cli.path_utils import resolve_megatron_path, resolve_model_script +from miles.utils.debug_utils.run_megatron.cli.path_utils import resolve_megatron_path from miles.utils.debug_utils.run_megatron.cli.prompt_utils import ( PromptConfig, generate_token_ids, @@ -19,7 +19,8 @@ build_worker_args, ) from miles.utils.debug_utils.run_megatron.worker.script_args import WorkerScriptArgs -from miles.utils.external_utils.exec_command import exec_command_cpu, exec_command_gpu +from miles.utils.external_utils.exec_command import exec_command_gpu +from miles.utils.external_utils.model_args_utils import load_model_args from miles.utils.typer_utils import dataclass_cli @@ -94,12 +95,7 @@ def run(args: RunArgs) -> None: def show_model_args( - model_type: Annotated[str, typer.Option(help="Model type matching scripts/models/{model_type}.sh")], + model_type: Annotated[str, typer.Option(help="Model type matching scripts/models/{model_type}.py")], ) -> None: """Show the MODEL_ARGS for a given model type (debug helper).""" - output: str | None = exec_command_cpu( - f'source "{resolve_model_script(model_type)}" && echo "${{MODEL_ARGS[@]}}"', - capture_output=True, - ) - if output: - print(output.strip()) + print(load_model_args(model_type)) diff --git a/miles/utils/debug_utils/run_megatron/cli/path_utils.py b/miles/utils/debug_utils/run_megatron/cli/path_utils.py index b34f91d1459..810f0539632 100644 --- a/miles/utils/debug_utils/run_megatron/cli/path_utils.py +++ b/miles/utils/debug_utils/run_megatron/cli/path_utils.py @@ -19,7 +19,7 @@ def resolve_megatron_path(megatron_path: Path | None) -> Path: def resolve_model_script(model_type: str) -> Path: repo_base: Path = _resolve_repo_base() - script: Path = repo_base / "scripts" / "models" / f"{model_type}.sh" + script: Path = repo_base / "scripts" / "models" / f"{model_type}.py" if not script.exists(): raise typer.BadParameter(f"Model script not found: {script}") return script diff --git a/miles/utils/debug_utils/run_megatron/cli/worker_executor.py b/miles/utils/debug_utils/run_megatron/cli/worker_executor.py index 211ae089067..5bb6644bbf9 100644 --- a/miles/utils/debug_utils/run_megatron/cli/worker_executor.py +++ b/miles/utils/debug_utils/run_megatron/cli/worker_executor.py @@ -3,8 +3,8 @@ from pathlib import Path from miles.utils.debug_utils.run_megatron.cli.parallel_utils import ParallelConfig -from miles.utils.debug_utils.run_megatron.cli.path_utils import resolve_model_script from miles.utils.debug_utils.run_megatron.worker.script_args import WORKER_SCRIPT_ARGS_BRIDGE, WorkerScriptArgs +from miles.utils.external_utils.model_args_utils import load_model_args def build_torchrun_cmd( @@ -15,16 +15,14 @@ def build_torchrun_cmd( worker_args: str, ) -> str: """Build the full shell command to launch the worker via torchrun.""" - model_script: Path = resolve_model_script(model_type) worker_module: str = "miles.utils.debug_utils.run_megatron.worker.main" cmd: str = ( - f'source "{model_script}" && ' f"PYTHONPATH={megatron_path}:$PYTHONPATH " f"CUDA_DEVICE_MAX_CONNECTIONS=1 " f"torchrun --nproc-per-node {nproc} " f"-m {worker_module} " - f"${{MODEL_ARGS[@]}} " + f"{load_model_args(model_type)} " f"--hidden-dropout 0 --attention-dropout 0 " f"{worker_args}" ) diff --git a/miles/utils/external_utils/command_utils.py b/miles/utils/external_utils/command_utils.py index d296734ba69..ae0e96ee2a6 100644 --- a/miles/utils/external_utils/command_utils.py +++ b/miles/utils/external_utils/command_utils.py @@ -9,12 +9,12 @@ import random import shlex import socket -import subprocess from dataclasses import dataclass, field from functools import partial from pathlib import Path from miles.utils.external_utils.exec_command import exec_command_cpu, exec_command_gpu, exec_command_multi_node +from miles.utils.external_utils.model_args_utils import load_model_args from miles.utils.file_arg_utils import PSEUDO_FILE_PREFIX from miles.utils.http_utils import wait_for_server_ready from miles.utils.typer_utils import dataclass_cli @@ -32,18 +32,6 @@ def _pythonpath_with_sources(megatron_path: str, *additional_pythonpaths: str | return os.pathsep.join(dict.fromkeys(entries)) -def load_model_args(megatron_model_type: str) -> list[str]: - """Expand the MODEL_ARGS array that scripts/models/.sh declares.""" - script = f"{repo_base_dir}/scripts/models/{megatron_model_type}.sh" - assert os.path.exists(script), f"no model args script at {script}" - expansion = f'source {shlex.quote(script)} && printf "%s\\0" "${{MODEL_ARGS[@]}}"' - result = subprocess.run(["bash", "-c", expansion], capture_output=True, text=True, check=True) - tokens = result.stdout.split("\0")[:-1] - for token in tokens: - assert token.split() == [token], f"model args token must be one whitespace-free word: {token!r}" - return tokens - - def convert_checkpoint( model_name, megatron_model_type, @@ -81,7 +69,7 @@ def convert_checkpoint( f"--nproc-per-node {num_gpus_per_node} " f"{multinode_args}" f"{repo_base_dir}/tools/convert_hf_to_torch_dist.py " - f"{' '.join(load_model_args(megatron_model_type))} " + f"{load_model_args(megatron_model_type)} " f"--hf-checkpoint {hf_checkpoint} " f"--save {path_dst} " f"{extra_args}" @@ -207,7 +195,7 @@ def execute_train( runtime_env_json = json.dumps({"env_vars": runtime_env_vars}) if get_bool_env_var("MILES_SCRIPT_ENABLE_RAY_SUBMIT", "1"): - model_args = " ".join(load_model_args(megatron_model_type)) if megatron_model_type is not None else "" + model_args = load_model_args(megatron_model_type) if megatron_model_type is not None else "" exec_command_cpu( f"export no_proxy=127.0.0.1 && export PYTHONUNBUFFERED=1 && " f"""ray job submit {'' if 'RAY_ADDRESS' in os.environ else '--address="http://127.0.0.1:8265" '}""" diff --git a/miles/utils/external_utils/model_args_utils.py b/miles/utils/external_utils/model_args_utils.py new file mode 100644 index 00000000000..698f38aa2ad --- /dev/null +++ b/miles/utils/external_utils/model_args_utils.py @@ -0,0 +1,59 @@ +import importlib.util +import sys +from pathlib import Path +from types import ModuleType + +REPO_ROOT = Path(__file__).resolve().parents[3] +MODEL_SCRIPT_DIR = REPO_ROOT / "scripts" / "models" + + +# ==================== loading a model script ==================== + + +def load_model_args(model_type: str, model_script_dir: Path | None = None, **kwargs: object) -> str: + """Collapse scripts/models/.py to one line; a newline would truncate the shell's read -ra.""" + path = (model_script_dir or MODEL_SCRIPT_DIR) / f"{model_type}.py" + assert path.exists(), f"no model args script at {path}" + sys.modules.setdefault("model_args_utils", sys.modules[__name__]) + module = import_module_from_path(path, f"miles_model_args_{path.stem.replace('.', '_').replace('-', '_')}") + args = " ".join(module.model_args(**kwargs).split()) + assert args, f"{path} declared no model args" + return args + + +def load_sibling_model_args(model_script: str, model_type: str, **kwargs: object) -> str: + """Load the model a variant is derived from, out of the same checkout as the variant itself.""" + return load_model_args(model_type, model_script_dir=Path(model_script).resolve().parent, **kwargs) + + +# ==================== what a model script may call ==================== + + +def moe_layer_freq(*, nlayers: int, first_k_dense_replace: int) -> str: + """Render megatron's --moe-layer-freq pattern: the first K layers dense, the rest MoE.""" + dense = min(first_k_dense_replace, nlayers) + return "[" + ",".join(["0"] * dense + ["1"] * (nlayers - dense)) + "]" + + +# ==================== importing a file by path ==================== + + +def import_module_from_path(path: Path, module_name: str) -> ModuleType: + """Import a python file that is not reachable as a dotted module path.""" + spec = importlib.util.spec_from_file_location(module_name, path) + assert spec is not None and spec.loader is not None, f"cannot load {path}" + module = importlib.util.module_from_spec(spec) + sys.modules[module_name] = module + try: + spec.loader.exec_module(module) + finally: + del sys.modules[module_name] + return module + + +# ==================== command line ==================== + + +if __name__ == "__main__": + (_MODEL_TYPE,) = sys.argv[1:] + print(load_model_args(_MODEL_TYPE)) diff --git a/scripts/amd/run-qwen3-4B-amd.sh b/scripts/amd/run-qwen3-4B-amd.sh index d3251fcf50a..9bb31865373 100644 --- a/scripts/amd/run-qwen3-4B-amd.sh +++ b/scripts/amd/run-qwen3-4B-amd.sh @@ -30,8 +30,8 @@ if [[ -n "${CUDA_VISIBLE_DEVICES:-}" ]]; then fi SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3-4B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3-4B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-4B --ref-load /root/Qwen3-4B_torch_dist diff --git a/scripts/models/deepseek-v3-20layer.py b/scripts/models/deepseek-v3-20layer.py new file mode 100644 index 00000000000..19bc7a48c94 --- /dev/null +++ b/scripts/models/deepseek-v3-20layer.py @@ -0,0 +1,5 @@ +from model_args_utils import load_sibling_model_args + + +def model_args() -> str: + return load_sibling_model_args(__file__, "deepseek-v3", nlayers=20) diff --git a/scripts/models/deepseek-v3-20layer.sh b/scripts/models/deepseek-v3-20layer.sh deleted file mode 100644 index 6fdde1820c8..00000000000 --- a/scripts/models/deepseek-v3-20layer.sh +++ /dev/null @@ -1 +0,0 @@ -MODEL_ARGS_NUM_LAYERS=20 source "$(dirname -- "${BASH_SOURCE[0]}")/deepseek-v3.sh" diff --git a/scripts/models/deepseek-v3-5layer.py b/scripts/models/deepseek-v3-5layer.py new file mode 100644 index 00000000000..910b1b951f8 --- /dev/null +++ b/scripts/models/deepseek-v3-5layer.py @@ -0,0 +1,5 @@ +from model_args_utils import load_sibling_model_args + + +def model_args() -> str: + return load_sibling_model_args(__file__, "deepseek-v3", nlayers=5) diff --git a/scripts/models/deepseek-v3-5layer.sh b/scripts/models/deepseek-v3-5layer.sh deleted file mode 100644 index a5e5d2522ce..00000000000 --- a/scripts/models/deepseek-v3-5layer.sh +++ /dev/null @@ -1 +0,0 @@ -MODEL_ARGS_NUM_LAYERS=5 source "$(dirname -- "${BASH_SOURCE[0]}")/deepseek-v3.sh" diff --git a/scripts/models/deepseek-v3.py b/scripts/models/deepseek-v3.py new file mode 100644 index 00000000000..ce0bc96f2ef --- /dev/null +++ b/scripts/models/deepseek-v3.py @@ -0,0 +1,56 @@ +import os + +from model_args_utils import moe_layer_freq + + +FIRST_K_DENSE_REPLACE = 3 + + +def model_args(nlayers: int | None = None) -> str: + nlayers = nlayers if nlayers is not None else int(os.environ.get("MODEL_ARGS_NUM_LAYERS") or 61) + return ( + "--disable-bias-linear " + f"--num-layers {nlayers} " + "--hidden-size 7168 " + "--ffn-hidden-size 18432 " + "--num-attention-heads 128 " + "--kv-channels 128 " + "--normalization RMSNorm " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 129280 " + "--multi-latent-attention " + "--q-lora-rank 1536 " + "--kv-lora-rank 512 " + "--qk-head-dim 128 " + "--qk-pos-emb-head-dim 64 " + "--v-head-dim 128 " + "--qk-layernorm " + "--rotary-scaling-factor 40 " + "--rotary-base 10000 " + "--mscale 1.0 " + "--mscale-all-dim 1.0 " + "--attention-softmax-in-fp32 " + "--no-rope-fusion " + # moe + "--num-experts 256 " + f"--moe-layer-freq {moe_layer_freq(nlayers=nlayers, first_k_dense_replace=FIRST_K_DENSE_REPLACE)} " + "--moe-ffn-hidden-size 2048 " + "--moe-router-topk 8 " + "--moe-shared-expert-intermediate-size 2048 " + "--moe-router-pre-softmax " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-token-dispatcher-type alltoall " + "--moe-aux-loss-coeff 0 " + "--moe-router-bias-update-rate 0 " + "--moe-router-group-topk 4 " + "--moe-router-num-groups 8 " + "--moe-grouped-gemm " + "--moe-router-topk-scaling-factor 2.5 " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + ) diff --git a/scripts/models/deepseek-v3.sh b/scripts/models/deepseek-v3.sh deleted file mode 100644 index 8c50d2c9405..00000000000 --- a/scripts/models/deepseek-v3.sh +++ /dev/null @@ -1,63 +0,0 @@ -NLAYERS="${MODEL_ARGS_NUM_LAYERS:-61}" -FIRST_K_DENSE_REPLACE=3 - -arr=() -for ((i=0; i str: + return load_sibling_model_args(__file__, "deepseek-v32", nlayers=5) diff --git a/scripts/models/deepseek-v32-5layer.sh b/scripts/models/deepseek-v32-5layer.sh deleted file mode 100644 index 2466640afd5..00000000000 --- a/scripts/models/deepseek-v32-5layer.sh +++ /dev/null @@ -1 +0,0 @@ -MODEL_ARGS_NUM_LAYERS=5 source "$(dirname -- "${BASH_SOURCE[0]}")/deepseek-v32.sh" diff --git a/scripts/models/deepseek-v32.py b/scripts/models/deepseek-v32.py new file mode 100644 index 00000000000..d2e02f006dc --- /dev/null +++ b/scripts/models/deepseek-v32.py @@ -0,0 +1,58 @@ +import os + +from model_args_utils import moe_layer_freq + + +def model_args(nlayers: int | None = None, first_k_dense_replace: int | None = None) -> str: + nlayers = nlayers if nlayers is not None else int(os.environ.get("MODEL_ARGS_NUM_LAYERS") or 61) + first_k_dense_replace = ( + first_k_dense_replace + if first_k_dense_replace is not None + else int(os.environ.get("MODEL_ARGS_FIRST_K_DENSE_REPLACE") or 3) + ) + return ( + "--spec miles_plugins.models.glm5.glm5 get_glm5_spec " + "--disable-bias-linear " + f"--num-layers {nlayers} " + "--hidden-size 7168 " + "--ffn-hidden-size 18432 " + "--num-attention-heads 128 " + "--kv-channels 128 " + "--normalization RMSNorm " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 129280 " + "--multi-latent-attention " + "--q-lora-rank 1536 " + "--kv-lora-rank 512 " + "--qk-head-dim 128 " + "--qk-pos-emb-head-dim 64 " + "--v-head-dim 128 " + "--qk-layernorm " + "--rotary-scaling-factor 40 " + "--rotary-base 10000 " + "--mscale 1.0 " + "--mscale-all-dim 1.0 " + "--attention-softmax-in-fp32 " + "--no-rope-fusion " + "--num-experts 256 " + f"--moe-layer-freq {moe_layer_freq(nlayers=nlayers, first_k_dense_replace=first_k_dense_replace)} " + "--moe-ffn-hidden-size 2048 " + "--moe-router-topk 8 " + "--moe-shared-expert-intermediate-size 2048 " + "--moe-router-pre-softmax " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-token-dispatcher-type alltoall " + "--moe-aux-loss-coeff 0 " + "--moe-router-bias-update-rate 0 " + "--moe-router-group-topk 4 " + "--moe-router-num-groups 8 " + "--moe-grouped-gemm " + "--moe-router-topk-scaling-factor 2.5 " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + ) diff --git a/scripts/models/deepseek-v32.sh b/scripts/models/deepseek-v32.sh deleted file mode 100644 index 649c3f6baca..00000000000 --- a/scripts/models/deepseek-v32.sh +++ /dev/null @@ -1,62 +0,0 @@ -NLAYERS="${MODEL_ARGS_NUM_LAYERS:-61}" -FIRST_K_DENSE_REPLACE="${MODEL_ARGS_FIRST_K_DENSE_REPLACE:-3}" - -arr=() -for ((i=0; i str: + return load_sibling_model_args(__file__, "deepseek-v4-flash", nlayers=4, compress_ratios="0 0 4 128") diff --git a/scripts/models/deepseek-v4-flash-4layer.sh b/scripts/models/deepseek-v4-flash-4layer.sh deleted file mode 100644 index 2a852460c1d..00000000000 --- a/scripts/models/deepseek-v4-flash-4layer.sh +++ /dev/null @@ -1,3 +0,0 @@ -MODEL_ARGS_NUM_LAYERS=4 -COMPRESS_RATIOS=(0 0 4 128) -source "$(dirname -- "${BASH_SOURCE[0]}")/deepseek-v4-flash.sh" diff --git a/scripts/models/deepseek-v4-flash.py b/scripts/models/deepseek-v4-flash.py new file mode 100644 index 00000000000..f9b87961e3b --- /dev/null +++ b/scripts/models/deepseek-v4-flash.py @@ -0,0 +1,77 @@ +import os + +from model_args_utils import moe_layer_freq + + +COMPRESS_RATIOS = "0 0 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 128 4 0" +SWIGLU_LIMIT_ARGS = "--activation-func-clamp-value 10 --no-bias-swiglu-fusion --no-activation-func-clamp-shared-expert" + + +def model_args( + nlayers: int | None = None, rotary_scaling_factor: str | None = None, compress_ratios: str = COMPRESS_RATIOS +) -> str: + nlayers = nlayers if nlayers is not None else int(os.environ.get("MODEL_ARGS_NUM_LAYERS") or 43) + rotary_scaling_factor = ( + rotary_scaling_factor if rotary_scaling_factor is not None else os.environ.get("ROTARY_SCALING_FACTOR") or "16" + ) + return ( + "--disable-bias-linear " + f"--num-layers {nlayers} " + "--hidden-size 4096 " + "--ffn-hidden-size 2048 " + "--num-attention-heads 64 " + "--normalization RMSNorm " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 129280 " + "--hidden-dropout 0.0 " + "--attention-dropout 0.0 " + # MLA params (reused by V4) + "--multi-latent-attention " + "--q-lora-rank 1024 " + "--kv-lora-rank 512 " + "--qk-head-dim 512 " + "--qk-pos-emb-head-dim 64 " + "--v-head-dim 512 " + "--qk-layernorm " + f"--rotary-scaling-factor {rotary_scaling_factor} " + "--rotary-base 10000 " + "--original-max-position-embeddings 65536 " + "--beta-fast 32 " + "--beta-slow 1 " + "--attention-softmax-in-fp32 " + "--no-rope-fusion " + # MoE + "--num-experts 256 " + f"--moe-layer-freq {moe_layer_freq(nlayers=nlayers, first_k_dense_replace=0)} " + "--moe-ffn-hidden-size 2048 " + "--moe-router-topk 6 " + "--moe-shared-expert-intermediate-size 2048 " + "--moe-router-pre-softmax " + "--moe-router-score-function sqrtsoftplus " + "--moe-router-enable-expert-bias " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-token-dispatcher-type alltoall " + "--moe-aux-loss-coeff 0 " + "--moe-grouped-gemm " + "--moe-router-topk-scaling-factor 1.5 " + # DSV4 specific + "--experimental-attention-variant dsv4 " + "--dsv4-hc-mult 4 " + "--dsv4-hc-sinkhorn-iters 20 " + f"--dsv4-compress-ratios {compress_ratios} " + "--dsv4-compress-rope-theta 160000 " + "--dsv4-o-groups 8 " + "--dsv4-o-lora-rank 1024 " + "--dsv4-n-hash-layers 3 " + "--dsv4-window-size 128 " + # DSA Indexer + "--dsa-indexer-n-heads 64 " + "--dsa-indexer-head-dim 128 " + "--dsa-indexer-topk 512 " + # V4 model spec (plugin) + "--spec miles_plugins.models.deepseek_v4.deepseek_v4 get_dsv4_spec " + f"{SWIGLU_LIMIT_ARGS} " + ) diff --git a/scripts/models/deepseek-v4-flash.sh b/scripts/models/deepseek-v4-flash.sh deleted file mode 100644 index 3997290de44..00000000000 --- a/scripts/models/deepseek-v4-flash.sh +++ /dev/null @@ -1,85 +0,0 @@ -NLAYERS="${MODEL_ARGS_NUM_LAYERS:-43}" - -# V4: all layers are MoE -arr=() -for ((i=0; i str: + nlayers = nlayers if nlayers is not None else int(os.environ.get("MODEL_ARGS_NUM_LAYERS") or 61) + rotary_scaling_factor = ( + rotary_scaling_factor if rotary_scaling_factor is not None else os.environ.get("ROTARY_SCALING_FACTOR") or "16" + ) + return ( + "--disable-bias-linear " + f"--num-layers {nlayers} " + "--hidden-size 7168 " + "--ffn-hidden-size 3072 " + "--num-attention-heads 128 " + "--normalization RMSNorm " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 129280 " + "--hidden-dropout 0.0 " + "--attention-dropout 0.0 " + # MLA params (reused by V4) + "--multi-latent-attention " + "--q-lora-rank 1536 " + "--kv-lora-rank 512 " + "--qk-head-dim 512 " + "--qk-pos-emb-head-dim 64 " + "--v-head-dim 512 " + "--qk-layernorm " + f"--rotary-scaling-factor {rotary_scaling_factor} " + "--rotary-base 10000 " + "--original-max-position-embeddings 65536 " + "--beta-fast 32 " + "--beta-slow 1 " + "--attention-softmax-in-fp32 " + "--no-rope-fusion " + # MoE + "--num-experts 384 " + f"--moe-layer-freq {moe_layer_freq(nlayers=nlayers, first_k_dense_replace=0)} " + "--moe-ffn-hidden-size 3072 " + "--moe-router-topk 6 " + "--moe-shared-expert-intermediate-size 3072 " + "--moe-router-pre-softmax " + "--moe-router-score-function sqrtsoftplus " + "--moe-router-enable-expert-bias " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-token-dispatcher-type alltoall " + "--moe-aux-loss-coeff 0 " + "--moe-grouped-gemm " + "--moe-router-topk-scaling-factor 2.5 " + # DSV4 specific + "--experimental-attention-variant dsv4 " + "--dsv4-hc-mult 4 " + "--dsv4-hc-sinkhorn-iters 20 " + f"--dsv4-compress-ratios {compress_ratios} " + "--dsv4-compress-rope-theta 160000 " + "--dsv4-o-groups 16 " + "--dsv4-o-lora-rank 1024 " + "--dsv4-n-hash-layers 3 " + "--dsv4-window-size 128 " + # DSA Indexer + "--dsa-indexer-n-heads 64 " + "--dsa-indexer-head-dim 128 " + "--dsa-indexer-topk 1024 " + # V4 model spec (plugin) + "--spec miles_plugins.models.deepseek_v4.deepseek_v4 get_dsv4_spec " + f"{SWIGLU_LIMIT_ARGS} " + ) diff --git a/scripts/models/deepseek-v4-pro.sh b/scripts/models/deepseek-v4-pro.sh deleted file mode 100644 index 067d887d967..00000000000 --- a/scripts/models/deepseek-v4-pro.sh +++ /dev/null @@ -1,85 +0,0 @@ -NLAYERS="${MODEL_ARGS_NUM_LAYERS:-61}" - -# V4: all layers are MoE -arr=() -for ((i=0; i str: + return ( + "--disable-bias-linear " + "--group-query-attention " + "--num-attention-heads 16 " + "--num-query-groups 8 " + "--kv-channels 256 " + "--num-layers 30 " + "--hidden-size 2816 " + "--ffn-hidden-size 2112 " + "--normalization RMSNorm " + "--norm-epsilon 1e-06 " + "--position-embedding-type rope " + "--rotary-base 1000000 " + "--vocab-size 262144 " + "--make-vocab-size-divisible-by 128 " + "--max-position-embeddings 262144 " + # tied embeddings: do not pass --untie-embeddings-and-output-weights + "--num-experts 128 " + "--moe-router-topk 8 " + "--moe-ffn-hidden-size 704 " + "--moe-router-score-function softmax " + "--moe-grouped-gemm " + "--moe-router-dtype fp32 " + "--moe-router-num-groups 1 " + "--moe-router-group-topk 1 " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-router-bias-update-rate 0 " + "--moe-aux-loss-coeff 0 " + ) diff --git a/scripts/models/gemma-4-26b-a4b-it.sh b/scripts/models/gemma-4-26b-a4b-it.sh deleted file mode 100644 index 63fcee46545..00000000000 --- a/scripts/models/gemma-4-26b-a4b-it.sh +++ /dev/null @@ -1,32 +0,0 @@ -# Google Gemma-4 26B-A4B-it (BF16, MoE: 128 experts, top-k 8). - -MODEL_ARGS=( - --disable-bias-linear - --group-query-attention - --num-attention-heads 16 - --num-query-groups 8 - --kv-channels 256 - --num-layers 30 - --hidden-size 2816 - --ffn-hidden-size 2112 - --normalization RMSNorm - --norm-epsilon 1e-06 - --position-embedding-type rope - --rotary-base 1000000 - --vocab-size 262144 - --make-vocab-size-divisible-by 128 - --max-position-embeddings 262144 - # tied embeddings: do not pass --untie-embeddings-and-output-weights - - --num-experts 128 - --moe-router-topk 8 - --moe-ffn-hidden-size 704 - --moe-router-score-function softmax - --moe-grouped-gemm - --moe-router-dtype fp32 - --moe-router-num-groups 1 - --moe-router-group-topk 1 - --moe-router-load-balancing-type seq_aux_loss - --moe-router-bias-update-rate 0 - --moe-aux-loss-coeff 0 -) diff --git a/scripts/models/gemma-4-31b-it.py b/scripts/models/gemma-4-31b-it.py new file mode 100644 index 00000000000..bc1ce66293d --- /dev/null +++ b/scripts/models/gemma-4-31b-it.py @@ -0,0 +1,19 @@ +def model_args() -> str: + return ( + "--disable-bias-linear " + "--group-query-attention " + "--num-attention-heads 32 " + "--num-query-groups 16 " + "--kv-channels 256 " + "--num-layers 60 " + "--hidden-size 5376 " + "--ffn-hidden-size 21504 " + "--normalization RMSNorm " + "--norm-epsilon 1e-06 " + "--position-embedding-type rope " + "--rotary-base 1000000 " + "--vocab-size 262144 " + "--make-vocab-size-divisible-by 128 " + "--max-position-embeddings 262144 " + # tied embeddings: do not pass --untie-embeddings-and-output-weights + ) diff --git a/scripts/models/gemma-4-31b-it.sh b/scripts/models/gemma-4-31b-it.sh deleted file mode 100644 index 3a65ddb1f5f..00000000000 --- a/scripts/models/gemma-4-31b-it.sh +++ /dev/null @@ -1,20 +0,0 @@ -# Google Gemma-4 31B-it (BF16, DENSE — no experts). - -MODEL_ARGS=( - --disable-bias-linear - --group-query-attention - --num-attention-heads 32 - --num-query-groups 16 - --kv-channels 256 - --num-layers 60 - --hidden-size 5376 - --ffn-hidden-size 21504 - --normalization RMSNorm - --norm-epsilon 1e-06 - --position-embedding-type rope - --rotary-base 1000000 - --vocab-size 262144 - --make-vocab-size-divisible-by 128 - --max-position-embeddings 262144 - # tied embeddings: do not pass --untie-embeddings-and-output-weights -) diff --git a/scripts/models/glm4-32B.py b/scripts/models/glm4-32B.py new file mode 100644 index 00000000000..53613b2065d --- /dev/null +++ b/scripts/models/glm4-32B.py @@ -0,0 +1,25 @@ +def model_args() -> str: + return ( + "--spec miles_plugins.models.glm4 get_glm_spec " + "--swiglu " + "--num-layers 64 " + "--hidden-size 6144 " + "--ffn-hidden-size 23040 " + "--num-attention-heads 48 " + "--max-position-embeddings 32768 " + "--seq-length 32768 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--normalization RMSNorm " + "--norm-epsilon 1e-5 " + "--rotary-base 10000 " + "--group-query-attention " + "--num-query-groups 8 " + "--vocab-size 151552 " + "--post-self-attn-layernorm " + "--post-mlp-layernorm " + "--rotary-interleaved " + "--rotary-percent 0.5 " + "--no-rope-fusion " + "--untie-embeddings-and-output-weights " + ) diff --git a/scripts/models/glm4-32B.sh b/scripts/models/glm4-32B.sh deleted file mode 100644 index 15cf273fc9a..00000000000 --- a/scripts/models/glm4-32B.sh +++ /dev/null @@ -1,24 +0,0 @@ -MODEL_ARGS=( - --spec "miles_plugins.models.glm4" "get_glm_spec" - --swiglu - --num-layers 64 - --hidden-size 6144 - --ffn-hidden-size 23040 - --num-attention-heads 48 - --max-position-embeddings 32768 - --seq-length 32768 - --use-rotary-position-embeddings - --disable-bias-linear - --normalization "RMSNorm" - --norm-epsilon 1e-5 - --rotary-base 10000 - --group-query-attention - --num-query-groups 8 - --vocab-size 151552 - --post-self-attn-layernorm - --post-mlp-layernorm - --rotary-interleaved - --rotary-percent 0.5 - --no-rope-fusion - --untie-embeddings-and-output-weights -) \ No newline at end of file diff --git a/scripts/models/glm4-9B.py b/scripts/models/glm4-9B.py new file mode 100644 index 00000000000..0c320b91f6d --- /dev/null +++ b/scripts/models/glm4-9B.py @@ -0,0 +1,24 @@ +def model_args() -> str: + return ( + "--spec miles_plugins.models.glm4 get_glm_spec " + "--swiglu " + "--num-layers 40 " + "--hidden-size 4096 " + "--ffn-hidden-size 13696 " + "--num-attention-heads 32 " + "--group-query-attention " + "--num-query-groups 2 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--add-qkv-bias " + "--normalization RMSNorm " + "--norm-epsilon 1e-5 " + "--rotary-base 10000 " + "--vocab-size 151552 " + "--post-self-attn-layernorm " + "--post-mlp-layernorm " + "--rotary-interleaved " + "--rotary-percent 0.5 " + "--no-rope-fusion " + "--untie-embeddings-and-output-weights " + ) diff --git a/scripts/models/glm4-9B.sh b/scripts/models/glm4-9B.sh deleted file mode 100644 index 652b579cf20..00000000000 --- a/scripts/models/glm4-9B.sh +++ /dev/null @@ -1,23 +0,0 @@ -MODEL_ARGS=( - --spec "miles_plugins.models.glm4" "get_glm_spec" - --swiglu - --num-layers 40 - --hidden-size 4096 - --ffn-hidden-size 13696 - --num-attention-heads 32 - --group-query-attention - --num-query-groups 2 - --use-rotary-position-embeddings - --disable-bias-linear - --add-qkv-bias - --normalization "RMSNorm" - --norm-epsilon 1e-5 - --rotary-base 10000 - --vocab-size 151552 - --post-self-attn-layernorm - --post-mlp-layernorm - --rotary-interleaved - --rotary-percent 0.5 - --no-rope-fusion - --untie-embeddings-and-output-weights -) \ No newline at end of file diff --git a/scripts/models/glm4.5-106B-A12B.py b/scripts/models/glm4.5-106B-A12B.py new file mode 100644 index 00000000000..ba7617c98c8 --- /dev/null +++ b/scripts/models/glm4.5-106B-A12B.py @@ -0,0 +1,39 @@ +N_DENSE_LAYERS = 1 +N_MOE_LAYERS = 45 + + +def model_args() -> str: + return ( + "--disable-bias-linear " + "--group-query-attention " + "--num-attention-heads 96 " + "--num-query-groups 8 " + "--kv-channels 128 " + f"--num-layers {N_DENSE_LAYERS + N_MOE_LAYERS} " + "--hidden-size 4096 " + "--ffn-hidden-size 10944 " + "--add-qkv-bias " + "--normalization RMSNorm " + "--position-embedding-type rope " + "--rotary-percent 0.5 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 151552 " + "--rotary-base 1000000 " + # moe + "--moe-ffn-hidden-size 1408 " + "--moe-shared-expert-intermediate-size 1408 " + "--moe-router-pre-softmax " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-router-bias-update-rate 0 " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-token-dispatcher-type alltoall " + "--moe-router-topk 8 " + f"--moe-layer-freq [0]*{N_DENSE_LAYERS}+[1]*{N_MOE_LAYERS} " + "--num-experts 128 " + "--moe-grouped-gemm " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + "--moe-aux-loss-coeff 0 " + ) diff --git a/scripts/models/glm4.5-106B-A12B.sh b/scripts/models/glm4.5-106B-A12B.sh deleted file mode 100644 index 8c244164336..00000000000 --- a/scripts/models/glm4.5-106B-A12B.sh +++ /dev/null @@ -1,40 +0,0 @@ -N_DENSE_LAYERS=1 -N_MOE_LAYERS=45 - -# glm4.5-106B-A12B -MODEL_ARGS=( - --disable-bias-linear - --group-query-attention - --num-attention-heads 96 - --num-query-groups 8 - --kv-channels 128 - --num-layers $((N_DENSE_LAYERS + N_MOE_LAYERS)) - --hidden-size 4096 - --ffn-hidden-size 10944 - - --add-qkv-bias - --normalization RMSNorm - --position-embedding-type rope - --rotary-percent 0.5 - --swiglu - --untie-embeddings-and-output-weights - --vocab-size 151552 - --rotary-base 1000000 - - # moe - --moe-ffn-hidden-size 1408 - --moe-shared-expert-intermediate-size 1408 - --moe-router-pre-softmax - --moe-router-score-function sigmoid - --moe-router-enable-expert-bias - --moe-router-bias-update-rate 0 - --moe-router-load-balancing-type seq_aux_loss - --moe-token-dispatcher-type alltoall - --moe-router-topk 8 - --moe-layer-freq "[0]*$N_DENSE_LAYERS+[1]*$N_MOE_LAYERS" - --num-experts 128 - --moe-grouped-gemm - --moe-router-dtype fp32 - --moe-permute-fusion - --moe-aux-loss-coeff 0 -) \ No newline at end of file diff --git a/scripts/models/glm4.5-355B-A32B.py b/scripts/models/glm4.5-355B-A32B.py new file mode 100644 index 00000000000..7c25722c55b --- /dev/null +++ b/scripts/models/glm4.5-355B-A32B.py @@ -0,0 +1,41 @@ +N_DENSE_LAYERS = 3 +N_MOE_LAYERS = 89 + + +def model_args() -> str: + return ( + "--disable-bias-linear " + "--qk-layernorm " + "--group-query-attention " + "--num-attention-heads 96 " + "--num-query-groups 8 " + "--kv-channels 128 " + f"--num-layers {N_DENSE_LAYERS + N_MOE_LAYERS} " + "--hidden-size 5120 " + "--ffn-hidden-size 12288 " + "--add-qkv-bias " + "--normalization RMSNorm " + "--position-embedding-type rope " + "--rotary-percent 0.5 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 151552 " + "--rotary-base 1000000 " + # moe + "--moe-ffn-hidden-size 1536 " + "--moe-shared-expert-intermediate-size 1536 " + "--moe-router-pre-softmax " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-router-bias-update-rate 0 " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-token-dispatcher-type alltoall " + "--moe-router-topk 8 " + "--moe-router-topk-scaling-factor 2.5 " + f"--moe-layer-freq [0]*{N_DENSE_LAYERS}+[1]*{N_MOE_LAYERS} " + "--num-experts 160 " + "--moe-grouped-gemm " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + "--moe-aux-loss-coeff 0 " + ) diff --git a/scripts/models/glm4.5-355B-A32B.sh b/scripts/models/glm4.5-355B-A32B.sh deleted file mode 100644 index 68b590ed4a7..00000000000 --- a/scripts/models/glm4.5-355B-A32B.sh +++ /dev/null @@ -1,43 +0,0 @@ -N_DENSE_LAYERS=3 -N_MOE_LAYERS=89 - -# glm4.5-355B-A32B -MODEL_ARGS=( - --disable-bias-linear - --qk-layernorm - --group-query-attention - --num-attention-heads 96 - --num-query-groups 8 - --kv-channels 128 - --num-layers $((N_DENSE_LAYERS + N_MOE_LAYERS)) - --hidden-size 5120 - --ffn-hidden-size 12288 - - --add-qkv-bias - --normalization RMSNorm - --position-embedding-type rope - --rotary-percent 0.5 - --swiglu - --untie-embeddings-and-output-weights - --vocab-size 151552 - - --rotary-base 1000000 - - # moe - --moe-ffn-hidden-size 1536 - --moe-shared-expert-intermediate-size 1536 - --moe-router-pre-softmax - --moe-router-score-function sigmoid - --moe-router-enable-expert-bias - --moe-router-bias-update-rate 0 - --moe-router-load-balancing-type seq_aux_loss - --moe-token-dispatcher-type alltoall - --moe-router-topk 8 - --moe-router-topk-scaling-factor 2.5 - --moe-layer-freq "[0]*$N_DENSE_LAYERS+[1]*$N_MOE_LAYERS" - --num-experts 160 - --moe-grouped-gemm - --moe-router-dtype fp32 - --moe-permute-fusion - --moe-aux-loss-coeff 0 -) diff --git a/scripts/models/glm4.7-flash.py b/scripts/models/glm4.7-flash.py new file mode 100644 index 00000000000..24363d671b5 --- /dev/null +++ b/scripts/models/glm4.7-flash.py @@ -0,0 +1,55 @@ +MOE_ROUTED_EXPERTS = 64 +MOE_ACTIVE_ROUTED_EXPERTS = 4 +MOE_SHARED_EXPERTS = 1 +NHIDDEN = 2048 +MOE_FFN_HIDDEN = 1536 +MOE_SHARED_EXPERT_INTERMEDIATE_SIZE = MOE_FFN_HIDDEN * MOE_SHARED_EXPERTS +FFN_HIDDEN = 10240 +N_DENSE_LAYERS = 1 +N_MOE_LAYERS = 46 +NHEADS = 20 + + +def model_args() -> str: + return ( + f"--moe-layer-freq [0]*{N_DENSE_LAYERS}+[1]*{N_MOE_LAYERS} " + f"--num-experts {MOE_ROUTED_EXPERTS} " + f"--moe-shared-expert-intermediate-size {MOE_SHARED_EXPERT_INTERMEDIATE_SIZE} " + f"--moe-router-topk {MOE_ACTIVE_ROUTED_EXPERTS} " + "--moe-grouped-gemm " + "--moe-permute-fusion " + f"--moe-ffn-hidden-size {MOE_FFN_HIDDEN} " + "--moe-router-score-function sigmoid " + "--moe-router-pre-softmax " + "--moe-router-enable-expert-bias " + "--moe-router-bias-update-rate 0 " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-router-topk-scaling-factor 1.8 " + "--moe-aux-loss-coeff 0 " + "--moe-router-dtype fp32 " + "--make-vocab-size-divisible-by 64 " + f"--num-layers {N_DENSE_LAYERS + N_MOE_LAYERS} " + f"--hidden-size {NHIDDEN} " + f"--ffn-hidden-size {FFN_HIDDEN} " + f"--num-attention-heads {NHEADS} " + "--disable-bias-linear " + "--add-qkv-bias " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--position-embedding-type rope " + "--no-position-embedding " + "--normalization RMSNorm " + "--norm-epsilon 1e-5 " + "--qk-layernorm " + "--multi-latent-attention " + "--q-lora-rank 768 " + "--kv-lora-rank 512 " + "--qk-head-dim 192 " + "--v-head-dim 256 " + "--kv-channels 192 " + "--qk-pos-emb-head-dim 64 " + "--vocab-size 154880 " + "--rotary-base 1000000 " + "--no-rope-fusion " + "--mtp-num-layers 1 " + ) diff --git a/scripts/models/glm4.7-flash.sh b/scripts/models/glm4.7-flash.sh deleted file mode 100644 index 763aadc387a..00000000000 --- a/scripts/models/glm4.7-flash.sh +++ /dev/null @@ -1,54 +0,0 @@ -MOE_ROUTED_EXPERTS=64 -MOE_ACTIVE_ROUTED_EXPERTS=4 -MOE_SHARED_EXPERTS=1 - -NHIDDEN=2048 -MOE_FFN_HIDDEN=1536 -MOE_SHARED_EXPERT_INTERMEDIATE_SIZE=$((MOE_FFN_HIDDEN * MOE_SHARED_EXPERTS)) -FFN_HIDDEN=10240 -N_DENSE_LAYERS=1 -N_MOE_LAYERS=46 -NHEADS=20 - -MODEL_ARGS=( - --moe-layer-freq "[0]*${N_DENSE_LAYERS}+[1]*${N_MOE_LAYERS}" - --num-experts $MOE_ROUTED_EXPERTS - --moe-shared-expert-intermediate-size $MOE_SHARED_EXPERT_INTERMEDIATE_SIZE - --moe-router-topk $MOE_ACTIVE_ROUTED_EXPERTS - --moe-grouped-gemm - --moe-permute-fusion - --moe-ffn-hidden-size $MOE_FFN_HIDDEN - --moe-router-score-function sigmoid - --moe-router-pre-softmax - --moe-router-enable-expert-bias - --moe-router-bias-update-rate 0 - --moe-router-load-balancing-type seq_aux_loss - --moe-router-topk-scaling-factor 1.8 - --moe-aux-loss-coeff 0 - --moe-router-dtype fp32 - --make-vocab-size-divisible-by 64 - --num-layers $((N_DENSE_LAYERS + N_MOE_LAYERS)) - --hidden-size $NHIDDEN - --ffn-hidden-size $FFN_HIDDEN - --num-attention-heads $NHEADS - --disable-bias-linear - --add-qkv-bias - --swiglu - --untie-embeddings-and-output-weights - --position-embedding-type rope - --no-position-embedding - --normalization RMSNorm - --norm-epsilon 1e-5 - --qk-layernorm - --multi-latent-attention - --q-lora-rank 768 - --kv-lora-rank 512 - --qk-head-dim 192 - --v-head-dim 256 - --kv-channels 192 - --qk-pos-emb-head-dim 64 - --vocab-size 154880 - --rotary-base 1000000 - --no-rope-fusion - --mtp-num-layers 1 -) \ No newline at end of file diff --git a/scripts/models/glm5-744B-A40B.py b/scripts/models/glm5-744B-A40B.py new file mode 100644 index 00000000000..9869f411e8e --- /dev/null +++ b/scripts/models/glm5-744B-A40B.py @@ -0,0 +1,52 @@ +MOE_ROUTED_EXPERTS = 256 +MOE_ACTIVE_ROUTED_EXPERTS = 8 +MOE_SHARED_EXPERTS = 1 +NHIDDEN = 6144 +MOE_FFN_HIDDEN = 2048 +MOE_SHARED_EXPERT_INTERMEDIATE_SIZE = MOE_FFN_HIDDEN * MOE_SHARED_EXPERTS +FFN_HIDDEN = 12288 +N_DENSE_LAYERS = 3 +NHEADS = 64 + + +def model_args(n_moe_layers: int = 75) -> str: + return ( + "--spec miles_plugins.models.glm5.glm5 get_glm5_spec " + f"--moe-layer-freq [0]*{N_DENSE_LAYERS}+[1]*{n_moe_layers} " + f"--num-experts {MOE_ROUTED_EXPERTS} " + f"--moe-shared-expert-intermediate-size {MOE_SHARED_EXPERT_INTERMEDIATE_SIZE} " + f"--moe-router-topk {MOE_ACTIVE_ROUTED_EXPERTS} " + "--moe-grouped-gemm " + "--moe-permute-fusion " + f"--moe-ffn-hidden-size {MOE_FFN_HIDDEN} " + "--moe-router-score-function sigmoid " + "--moe-router-pre-softmax " + "--moe-router-enable-expert-bias " + "--moe-router-bias-update-rate 0 " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-router-topk-scaling-factor 2.5 " + "--moe-aux-loss-coeff 0 " + "--moe-router-dtype fp32 " + "--make-vocab-size-divisible-by 16 " + f"--num-layers {N_DENSE_LAYERS + n_moe_layers} " + f"--hidden-size {NHIDDEN} " + f"--ffn-hidden-size {FFN_HIDDEN} " + f"--num-attention-heads {NHEADS} " + "--disable-bias-linear " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--position-embedding-type rope " + "--no-position-embedding " + "--normalization RMSNorm " + "--qk-layernorm " + "--multi-latent-attention " + "--q-lora-rank 2048 " + "--kv-lora-rank 512 " + "--qk-head-dim 192 " + "--v-head-dim 256 " + "--kv-channels 192 " + "--qk-pos-emb-head-dim 64 " + "--vocab-size 154880 " + "--rotary-base 1000000 " + "--enable-experimental " + ) diff --git a/scripts/models/glm5-744B-A40B.sh b/scripts/models/glm5-744B-A40B.sh deleted file mode 100644 index b9241e1a09c..00000000000 --- a/scripts/models/glm5-744B-A40B.sh +++ /dev/null @@ -1,52 +0,0 @@ -MOE_ROUTED_EXPERTS=256 -MOE_ACTIVE_ROUTED_EXPERTS=8 -MOE_SHARED_EXPERTS=1 - -NHIDDEN=6144 -MOE_FFN_HIDDEN=2048 -MOE_SHARED_EXPERT_INTERMEDIATE_SIZE=$(($MOE_FFN_HIDDEN * $MOE_SHARED_EXPERTS)) -FFN_HIDDEN=12288 -N_DENSE_LAYERS=3 -N_MOE_LAYERS=75 -NHEADS=64 - -MODEL_ARGS=( - --spec "miles_plugins.models.glm5.glm5" "get_glm5_spec" - --moe-layer-freq "[0]*${N_DENSE_LAYERS}+[1]*${N_MOE_LAYERS}" - --num-experts $MOE_ROUTED_EXPERTS - --moe-shared-expert-intermediate-size $MOE_SHARED_EXPERT_INTERMEDIATE_SIZE - --moe-router-topk $MOE_ACTIVE_ROUTED_EXPERTS - --moe-grouped-gemm - --moe-permute-fusion - --moe-ffn-hidden-size $MOE_FFN_HIDDEN - --moe-router-score-function sigmoid - --moe-router-pre-softmax - --moe-router-enable-expert-bias - --moe-router-bias-update-rate 0 - --moe-router-load-balancing-type seq_aux_loss - --moe-router-topk-scaling-factor 2.5 - --moe-aux-loss-coeff 0 - --moe-router-dtype fp32 - --make-vocab-size-divisible-by 16 - --num-layers $((N_DENSE_LAYERS + N_MOE_LAYERS)) - --hidden-size $NHIDDEN - --ffn-hidden-size $FFN_HIDDEN - --num-attention-heads $NHEADS - --disable-bias-linear - --swiglu - --untie-embeddings-and-output-weights - --position-embedding-type rope - --no-position-embedding - --normalization RMSNorm - --qk-layernorm - --multi-latent-attention - --q-lora-rank 2048 - --kv-lora-rank 512 - --qk-head-dim 192 - --v-head-dim 256 - --kv-channels 192 - --qk-pos-emb-head-dim 64 - --vocab-size 154880 - --rotary-base 1000000 - --enable-experimental -) \ No newline at end of file diff --git a/scripts/models/glm5-744B-A40B_20layer.py b/scripts/models/glm5-744B-A40B_20layer.py new file mode 100644 index 00000000000..6ec017c9810 --- /dev/null +++ b/scripts/models/glm5-744B-A40B_20layer.py @@ -0,0 +1,6 @@ +from model_args_utils import load_sibling_model_args + + +def model_args() -> str: + # Override for 20-layer pruned model (first 20 layers: 3 dense + 17 MoE) + return load_sibling_model_args(__file__, "glm5-744B-A40B", n_moe_layers=17) diff --git a/scripts/models/glm5-744B-A40B_20layer.sh b/scripts/models/glm5-744B-A40B_20layer.sh deleted file mode 100644 index 1eb85223d71..00000000000 --- a/scripts/models/glm5-744B-A40B_20layer.sh +++ /dev/null @@ -1,12 +0,0 @@ -SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]:-$0}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/glm5-744B-A40B.sh" - -# Override for 20-layer pruned model (first 20 layers: 3 dense + 17 MoE) -N_MOE_LAYERS=17 - -for ((i=0; i<${#MODEL_ARGS[@]}; i++)); do - case "${MODEL_ARGS[$i]}" in - --num-layers) MODEL_ARGS[$((i+1))]=$((N_DENSE_LAYERS + N_MOE_LAYERS)) ;; - --moe-layer-freq) MODEL_ARGS[$((i+1))]="[0]*${N_DENSE_LAYERS}+[1]*${N_MOE_LAYERS}" ;; - esac -done diff --git a/scripts/models/glm5-744B-A40B_4layer.py b/scripts/models/glm5-744B-A40B_4layer.py new file mode 100644 index 00000000000..0a5c3ce5625 --- /dev/null +++ b/scripts/models/glm5-744B-A40B_4layer.py @@ -0,0 +1,6 @@ +from model_args_utils import load_sibling_model_args + + +def model_args() -> str: + # Override for 4-layer pruned model (first 4 layers: 3 dense + 1 MoE) + return load_sibling_model_args(__file__, "glm5-744B-A40B", n_moe_layers=1) diff --git a/scripts/models/glm5-744B-A40B_4layer.sh b/scripts/models/glm5-744B-A40B_4layer.sh deleted file mode 100644 index 50f10056327..00000000000 --- a/scripts/models/glm5-744B-A40B_4layer.sh +++ /dev/null @@ -1,12 +0,0 @@ -SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]:-$0}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/glm5-744B-A40B.sh" - -# Override for 4-layer pruned model (first 4 layers: 3 dense + 1 MoE) -N_MOE_LAYERS=1 - -for ((i=0; i<${#MODEL_ARGS[@]}; i++)); do - case "${MODEL_ARGS[$i]}" in - --num-layers) MODEL_ARGS[$((i+1))]=$((N_DENSE_LAYERS + N_MOE_LAYERS)) ;; - --moe-layer-freq) MODEL_ARGS[$((i+1))]="[0]*${N_DENSE_LAYERS}+[1]*${N_MOE_LAYERS}" ;; - esac -done diff --git a/scripts/models/glm5.1-744B-A40B_6layer.py b/scripts/models/glm5.1-744B-A40B_6layer.py new file mode 100644 index 00000000000..7144d28349a --- /dev/null +++ b/scripts/models/glm5.1-744B-A40B_6layer.py @@ -0,0 +1,7 @@ +from model_args_utils import load_sibling_model_args + + +def model_args() -> str: + # Override for the 6-layer pruned GLM-5.1 toy (jybsuper/GLM-5.1-6layer): + # first 6 layers = 3 dense + 3 MoE. + return load_sibling_model_args(__file__, "glm5-744B-A40B", n_moe_layers=3) diff --git a/scripts/models/glm5.1-744B-A40B_6layer.sh b/scripts/models/glm5.1-744B-A40B_6layer.sh deleted file mode 100644 index a7044d3a439..00000000000 --- a/scripts/models/glm5.1-744B-A40B_6layer.sh +++ /dev/null @@ -1,14 +0,0 @@ -SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]:-$0}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/glm5-744B-A40B.sh" - -# Override for the 6-layer pruned GLM-5.1 toy (jybsuper/GLM-5.1-6layer): -# first 6 layers = 3 dense + 3 MoE. -N_MOE_LAYERS=3 - -for ((i=0; i<${#MODEL_ARGS[@]}; i++)); do - case "${MODEL_ARGS[$i]}" in - --num-layers) MODEL_ARGS[$((i+1))]=$((N_DENSE_LAYERS + N_MOE_LAYERS)) ;; - --moe-layer-freq) MODEL_ARGS[$((i+1))]="[0]*${N_DENSE_LAYERS}+[1]*${N_MOE_LAYERS}" ;; - esac -done - diff --git a/scripts/models/glm5.1-744B-A40B_6layer_lora.py b/scripts/models/glm5.1-744B-A40B_6layer_lora.py new file mode 100644 index 00000000000..efc537a2ed8 --- /dev/null +++ b/scripts/models/glm5.1-744B-A40B_6layer_lora.py @@ -0,0 +1,7 @@ +from model_args_utils import load_sibling_model_args + + +def model_args() -> str: + # Override for the 6-layer pruned GLM-5.1 toy (jybsuper/GLM-5.1-6layer): + # first 6 layers = 3 dense + 3 MoE. + return load_sibling_model_args(__file__, "glm5.1-744B-A40B_lora", n_moe_layers=3) diff --git a/scripts/models/glm5.1-744B-A40B_6layer_lora.sh b/scripts/models/glm5.1-744B-A40B_6layer_lora.sh deleted file mode 100644 index 2cac8891cac..00000000000 --- a/scripts/models/glm5.1-744B-A40B_6layer_lora.sh +++ /dev/null @@ -1,13 +0,0 @@ -SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]:-$0}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/glm5.1-744B-A40B_lora.sh" - -# Override for the 6-layer pruned GLM-5.1 toy (jybsuper/GLM-5.1-6layer): -# first 6 layers = 3 dense + 3 MoE. -N_MOE_LAYERS=3 - -for ((i=0; i<${#MODEL_ARGS[@]}; i++)); do - case "${MODEL_ARGS[$i]}" in - --num-layers) MODEL_ARGS[$((i+1))]=$((N_DENSE_LAYERS + N_MOE_LAYERS)) ;; - --moe-layer-freq) MODEL_ARGS[$((i+1))]="[0]*${N_DENSE_LAYERS}+[1]*${N_MOE_LAYERS}" ;; - esac -done diff --git a/scripts/models/glm5.1-744B-A40B_lora.py b/scripts/models/glm5.1-744B-A40B_lora.py new file mode 100644 index 00000000000..9869f411e8e --- /dev/null +++ b/scripts/models/glm5.1-744B-A40B_lora.py @@ -0,0 +1,52 @@ +MOE_ROUTED_EXPERTS = 256 +MOE_ACTIVE_ROUTED_EXPERTS = 8 +MOE_SHARED_EXPERTS = 1 +NHIDDEN = 6144 +MOE_FFN_HIDDEN = 2048 +MOE_SHARED_EXPERT_INTERMEDIATE_SIZE = MOE_FFN_HIDDEN * MOE_SHARED_EXPERTS +FFN_HIDDEN = 12288 +N_DENSE_LAYERS = 3 +NHEADS = 64 + + +def model_args(n_moe_layers: int = 75) -> str: + return ( + "--spec miles_plugins.models.glm5.glm5 get_glm5_spec " + f"--moe-layer-freq [0]*{N_DENSE_LAYERS}+[1]*{n_moe_layers} " + f"--num-experts {MOE_ROUTED_EXPERTS} " + f"--moe-shared-expert-intermediate-size {MOE_SHARED_EXPERT_INTERMEDIATE_SIZE} " + f"--moe-router-topk {MOE_ACTIVE_ROUTED_EXPERTS} " + "--moe-grouped-gemm " + "--moe-permute-fusion " + f"--moe-ffn-hidden-size {MOE_FFN_HIDDEN} " + "--moe-router-score-function sigmoid " + "--moe-router-pre-softmax " + "--moe-router-enable-expert-bias " + "--moe-router-bias-update-rate 0 " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-router-topk-scaling-factor 2.5 " + "--moe-aux-loss-coeff 0 " + "--moe-router-dtype fp32 " + "--make-vocab-size-divisible-by 16 " + f"--num-layers {N_DENSE_LAYERS + n_moe_layers} " + f"--hidden-size {NHIDDEN} " + f"--ffn-hidden-size {FFN_HIDDEN} " + f"--num-attention-heads {NHEADS} " + "--disable-bias-linear " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--position-embedding-type rope " + "--no-position-embedding " + "--normalization RMSNorm " + "--qk-layernorm " + "--multi-latent-attention " + "--q-lora-rank 2048 " + "--kv-lora-rank 512 " + "--qk-head-dim 192 " + "--v-head-dim 256 " + "--kv-channels 192 " + "--qk-pos-emb-head-dim 64 " + "--vocab-size 154880 " + "--rotary-base 1000000 " + "--enable-experimental " + ) diff --git a/scripts/models/glm5.1-744B-A40B_lora.sh b/scripts/models/glm5.1-744B-A40B_lora.sh deleted file mode 100644 index 2b77e7c71d7..00000000000 --- a/scripts/models/glm5.1-744B-A40B_lora.sh +++ /dev/null @@ -1,58 +0,0 @@ -MOE_ROUTED_EXPERTS=256 -MOE_ACTIVE_ROUTED_EXPERTS=8 -MOE_SHARED_EXPERTS=1 - -NHIDDEN=6144 -MOE_FFN_HIDDEN=2048 -MOE_SHARED_EXPERT_INTERMEDIATE_SIZE=$(($MOE_FFN_HIDDEN * $MOE_SHARED_EXPERTS)) -FFN_HIDDEN=12288 -N_DENSE_LAYERS=3 -N_MOE_LAYERS=75 -NHEADS=64 - -# GLM-5.1 744B-A40B (zai-org/GLM-5.1, glm_moe_dsa). Identical to glm5-744B-A40B.sh; the only -# architecture difference vs the glm5.2-744B-A40B* registries is --rotary-base (5.2 uses 8e6). -MODEL_ARGS=( - --spec "miles_plugins.models.glm5.glm5" "get_glm5_spec" - --moe-layer-freq "[0]*${N_DENSE_LAYERS}+[1]*${N_MOE_LAYERS}" - --num-experts $MOE_ROUTED_EXPERTS - --moe-shared-expert-intermediate-size $MOE_SHARED_EXPERT_INTERMEDIATE_SIZE - --moe-router-topk $MOE_ACTIVE_ROUTED_EXPERTS - --moe-grouped-gemm - --moe-permute-fusion - --moe-ffn-hidden-size $MOE_FFN_HIDDEN - --moe-router-score-function sigmoid - --moe-router-pre-softmax - --moe-router-enable-expert-bias - --moe-router-bias-update-rate 0 - --moe-router-load-balancing-type seq_aux_loss - --moe-router-topk-scaling-factor 2.5 - --moe-aux-loss-coeff 0 - --moe-router-dtype fp32 - --make-vocab-size-divisible-by 16 - --num-layers $((N_DENSE_LAYERS + N_MOE_LAYERS)) - --hidden-size $NHIDDEN - --ffn-hidden-size $FFN_HIDDEN - --num-attention-heads $NHEADS - --disable-bias-linear - --swiglu - --untie-embeddings-and-output-weights - --position-embedding-type rope - --no-position-embedding - --normalization RMSNorm - --qk-layernorm - --multi-latent-attention - --q-lora-rank 2048 - --kv-lora-rank 512 - --qk-head-dim 192 - --v-head-dim 256 - --kv-channels 192 - --qk-pos-emb-head-dim 64 - --vocab-size 154880 - --rotary-base 1000000 - --enable-experimental -) - -# LoRA registry for scripts/run_glm5_1_744b_a40b_lora.py: MODEL_ARGS carries the architecture only -# (identical to glm5-744B-A40B.sh); every LoRA / run-mode flag lives in the runner, which -# always wins (argparse last-occurrence). --spec above is inert under bridge LoRA. diff --git a/scripts/models/glm5.2-744B-A40B.py b/scripts/models/glm5.2-744B-A40B.py new file mode 100644 index 00000000000..6f582e55594 --- /dev/null +++ b/scripts/models/glm5.2-744B-A40B.py @@ -0,0 +1,52 @@ +MOE_ROUTED_EXPERTS = 256 +MOE_ACTIVE_ROUTED_EXPERTS = 8 +MOE_SHARED_EXPERTS = 1 +NHIDDEN = 6144 +MOE_FFN_HIDDEN = 2048 +MOE_SHARED_EXPERT_INTERMEDIATE_SIZE = MOE_FFN_HIDDEN * MOE_SHARED_EXPERTS +FFN_HIDDEN = 12288 +N_DENSE_LAYERS = 3 +NHEADS = 64 + + +def model_args(n_moe_layers: int = 75) -> str: + return ( + "--spec miles_plugins.models.glm5.glm5 get_glm5_spec " + f"--moe-layer-freq [0]*{N_DENSE_LAYERS}+[1]*{n_moe_layers} " + f"--num-experts {MOE_ROUTED_EXPERTS} " + f"--moe-shared-expert-intermediate-size {MOE_SHARED_EXPERT_INTERMEDIATE_SIZE} " + f"--moe-router-topk {MOE_ACTIVE_ROUTED_EXPERTS} " + "--moe-grouped-gemm " + "--moe-permute-fusion " + f"--moe-ffn-hidden-size {MOE_FFN_HIDDEN} " + "--moe-router-score-function sigmoid " + "--moe-router-pre-softmax " + "--moe-router-enable-expert-bias " + "--moe-router-bias-update-rate 0 " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-router-topk-scaling-factor 2.5 " + "--moe-aux-loss-coeff 0 " + "--moe-router-dtype fp32 " + "--make-vocab-size-divisible-by 16 " + f"--num-layers {N_DENSE_LAYERS + n_moe_layers} " + f"--hidden-size {NHIDDEN} " + f"--ffn-hidden-size {FFN_HIDDEN} " + f"--num-attention-heads {NHEADS} " + "--disable-bias-linear " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--position-embedding-type rope " + "--no-position-embedding " + "--normalization RMSNorm " + "--qk-layernorm " + "--multi-latent-attention " + "--q-lora-rank 2048 " + "--kv-lora-rank 512 " + "--qk-head-dim 192 " + "--v-head-dim 256 " + "--kv-channels 192 " + "--qk-pos-emb-head-dim 64 " + "--vocab-size 154880 " + "--rotary-base 8000000 " + "--enable-experimental " + ) diff --git a/scripts/models/glm5.2-744B-A40B.sh b/scripts/models/glm5.2-744B-A40B.sh deleted file mode 100644 index 2b831c6f185..00000000000 --- a/scripts/models/glm5.2-744B-A40B.sh +++ /dev/null @@ -1,59 +0,0 @@ -MOE_ROUTED_EXPERTS=256 -MOE_ACTIVE_ROUTED_EXPERTS=8 -MOE_SHARED_EXPERTS=1 - -NHIDDEN=6144 -MOE_FFN_HIDDEN=2048 -MOE_SHARED_EXPERT_INTERMEDIATE_SIZE=$(($MOE_FFN_HIDDEN * $MOE_SHARED_EXPERTS)) -FFN_HIDDEN=12288 -N_DENSE_LAYERS=3 -N_MOE_LAYERS=75 -NHEADS=64 - -# GLM-5.2 744B-A40B with DSA cross-layer index sharing. Only the computing layers -# (1,2,3,7,11,...,75 in Megatron 1-indexing) carry indexer weights and compute the -# sparse top-k; the remaining layers reuse the most recent computing layer's indices. -# The schedule (index_topk_freq=4, index_skip_topk_offset=3) is read from the HF config -# by the shared glm5 provider; cross-layer sharing activates when index_topk_freq > 1. -# allgather-CP is enabled at train time in the run script (not here) so that checkpoint -# conversion does not need to parse it. Differs from glm5-744B-A40B.sh only in rotary-base. -MODEL_ARGS=( - --spec "miles_plugins.models.glm5.glm5" "get_glm5_spec" - --moe-layer-freq "[0]*${N_DENSE_LAYERS}+[1]*${N_MOE_LAYERS}" - --num-experts $MOE_ROUTED_EXPERTS - --moe-shared-expert-intermediate-size $MOE_SHARED_EXPERT_INTERMEDIATE_SIZE - --moe-router-topk $MOE_ACTIVE_ROUTED_EXPERTS - --moe-grouped-gemm - --moe-permute-fusion - --moe-ffn-hidden-size $MOE_FFN_HIDDEN - --moe-router-score-function sigmoid - --moe-router-pre-softmax - --moe-router-enable-expert-bias - --moe-router-bias-update-rate 0 - --moe-router-load-balancing-type seq_aux_loss - --moe-router-topk-scaling-factor 2.5 - --moe-aux-loss-coeff 0 - --moe-router-dtype fp32 - --make-vocab-size-divisible-by 16 - --num-layers $((N_DENSE_LAYERS + N_MOE_LAYERS)) - --hidden-size $NHIDDEN - --ffn-hidden-size $FFN_HIDDEN - --num-attention-heads $NHEADS - --disable-bias-linear - --swiglu - --untie-embeddings-and-output-weights - --position-embedding-type rope - --no-position-embedding - --normalization RMSNorm - --qk-layernorm - --multi-latent-attention - --q-lora-rank 2048 - --kv-lora-rank 512 - --qk-head-dim 192 - --v-head-dim 256 - --kv-channels 192 - --qk-pos-emb-head-dim 64 - --vocab-size 154880 - --rotary-base 8000000 - --enable-experimental -) diff --git a/scripts/models/glm5.2-744B-A40B_5layer.py b/scripts/models/glm5.2-744B-A40B_5layer.py new file mode 100644 index 00000000000..35f03c3b9b6 --- /dev/null +++ b/scripts/models/glm5.2-744B-A40B_5layer.py @@ -0,0 +1,8 @@ +from model_args_utils import load_sibling_model_args + + +def model_args() -> str: + # Override for 5-layer pruned model (first 5 layers: 3 dense + 2 MoE). + # Keeps at least one computing + one skip layer so the DSA cross-layer index + # sharing path is exercised (computing layers 0,1,2; skip layers 3,4). + return load_sibling_model_args(__file__, "glm5.2-744B-A40B", n_moe_layers=2) diff --git a/scripts/models/glm5.2-744B-A40B_5layer.sh b/scripts/models/glm5.2-744B-A40B_5layer.sh deleted file mode 100644 index f98c2c3ed9f..00000000000 --- a/scripts/models/glm5.2-744B-A40B_5layer.sh +++ /dev/null @@ -1,14 +0,0 @@ -SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]:-$0}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/glm5.2-744B-A40B.sh" - -# Override for 5-layer pruned model (first 5 layers: 3 dense + 2 MoE). -# Keeps at least one computing + one skip layer so the DSA cross-layer index -# sharing path is exercised (computing layers 0,1,2; skip layers 3,4). -N_MOE_LAYERS=2 - -for ((i=0; i<${#MODEL_ARGS[@]}; i++)); do - case "${MODEL_ARGS[$i]}" in - --num-layers) MODEL_ARGS[$((i+1))]=$((N_DENSE_LAYERS + N_MOE_LAYERS)) ;; - --moe-layer-freq) MODEL_ARGS[$((i+1))]="[0]*${N_DENSE_LAYERS}+[1]*${N_MOE_LAYERS}" ;; - esac -done diff --git a/scripts/models/glm5.2-744B-A40B_5layer_lora.py b/scripts/models/glm5.2-744B-A40B_5layer_lora.py new file mode 100644 index 00000000000..0df3cf9e296 --- /dev/null +++ b/scripts/models/glm5.2-744B-A40B_5layer_lora.py @@ -0,0 +1,7 @@ +from model_args_utils import load_sibling_model_args + + +def model_args() -> str: + # Override for the 5-layer pruned model (first 5 layers: 3 dense + 2 MoE). Keeps at least + # one computing + one skip layer so the DSA cross-layer index sharing path is exercised. + return load_sibling_model_args(__file__, "glm5.2-744B-A40B_lora", n_moe_layers=2) diff --git a/scripts/models/glm5.2-744B-A40B_5layer_lora.sh b/scripts/models/glm5.2-744B-A40B_5layer_lora.sh deleted file mode 100644 index 8da4e10f6ed..00000000000 --- a/scripts/models/glm5.2-744B-A40B_5layer_lora.sh +++ /dev/null @@ -1,13 +0,0 @@ -SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]:-$0}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/glm5.2-744B-A40B_lora.sh" - -# Override for the 5-layer pruned model (first 5 layers: 3 dense + 2 MoE). Keeps at least -# one computing + one skip layer so the DSA cross-layer index sharing path is exercised. -N_MOE_LAYERS=2 - -for ((i=0; i<${#MODEL_ARGS[@]}; i++)); do - case "${MODEL_ARGS[$i]}" in - --num-layers) MODEL_ARGS[$((i+1))]=$((N_DENSE_LAYERS + N_MOE_LAYERS)) ;; - --moe-layer-freq) MODEL_ARGS[$((i+1))]="[0]*${N_DENSE_LAYERS}+[1]*${N_MOE_LAYERS}" ;; - esac -done diff --git a/scripts/models/glm5.2-744B-A40B_lora.py b/scripts/models/glm5.2-744B-A40B_lora.py new file mode 100644 index 00000000000..6f582e55594 --- /dev/null +++ b/scripts/models/glm5.2-744B-A40B_lora.py @@ -0,0 +1,52 @@ +MOE_ROUTED_EXPERTS = 256 +MOE_ACTIVE_ROUTED_EXPERTS = 8 +MOE_SHARED_EXPERTS = 1 +NHIDDEN = 6144 +MOE_FFN_HIDDEN = 2048 +MOE_SHARED_EXPERT_INTERMEDIATE_SIZE = MOE_FFN_HIDDEN * MOE_SHARED_EXPERTS +FFN_HIDDEN = 12288 +N_DENSE_LAYERS = 3 +NHEADS = 64 + + +def model_args(n_moe_layers: int = 75) -> str: + return ( + "--spec miles_plugins.models.glm5.glm5 get_glm5_spec " + f"--moe-layer-freq [0]*{N_DENSE_LAYERS}+[1]*{n_moe_layers} " + f"--num-experts {MOE_ROUTED_EXPERTS} " + f"--moe-shared-expert-intermediate-size {MOE_SHARED_EXPERT_INTERMEDIATE_SIZE} " + f"--moe-router-topk {MOE_ACTIVE_ROUTED_EXPERTS} " + "--moe-grouped-gemm " + "--moe-permute-fusion " + f"--moe-ffn-hidden-size {MOE_FFN_HIDDEN} " + "--moe-router-score-function sigmoid " + "--moe-router-pre-softmax " + "--moe-router-enable-expert-bias " + "--moe-router-bias-update-rate 0 " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-router-topk-scaling-factor 2.5 " + "--moe-aux-loss-coeff 0 " + "--moe-router-dtype fp32 " + "--make-vocab-size-divisible-by 16 " + f"--num-layers {N_DENSE_LAYERS + n_moe_layers} " + f"--hidden-size {NHIDDEN} " + f"--ffn-hidden-size {FFN_HIDDEN} " + f"--num-attention-heads {NHEADS} " + "--disable-bias-linear " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--position-embedding-type rope " + "--no-position-embedding " + "--normalization RMSNorm " + "--qk-layernorm " + "--multi-latent-attention " + "--q-lora-rank 2048 " + "--kv-lora-rank 512 " + "--qk-head-dim 192 " + "--v-head-dim 256 " + "--kv-channels 192 " + "--qk-pos-emb-head-dim 64 " + "--vocab-size 154880 " + "--rotary-base 8000000 " + "--enable-experimental " + ) diff --git a/scripts/models/glm5.2-744B-A40B_lora.sh b/scripts/models/glm5.2-744B-A40B_lora.sh deleted file mode 100644 index c9d2d4971f3..00000000000 --- a/scripts/models/glm5.2-744B-A40B_lora.sh +++ /dev/null @@ -1,58 +0,0 @@ -MOE_ROUTED_EXPERTS=256 -MOE_ACTIVE_ROUTED_EXPERTS=8 -MOE_SHARED_EXPERTS=1 - -NHIDDEN=6144 -MOE_FFN_HIDDEN=2048 -MOE_SHARED_EXPERT_INTERMEDIATE_SIZE=$(($MOE_FFN_HIDDEN * $MOE_SHARED_EXPERTS)) -FFN_HIDDEN=12288 -N_DENSE_LAYERS=3 -N_MOE_LAYERS=75 -NHEADS=64 - -# GLM-5.2 744B-A40B (DSA cross-layer index sharing; the schedule is read from the HF config -# by the shared glm5 provider). Differs from glm5-744B-A40B.sh only in rotary-base. -MODEL_ARGS=( - --spec "miles_plugins.models.glm5.glm5" "get_glm5_spec" - --moe-layer-freq "[0]*${N_DENSE_LAYERS}+[1]*${N_MOE_LAYERS}" - --num-experts $MOE_ROUTED_EXPERTS - --moe-shared-expert-intermediate-size $MOE_SHARED_EXPERT_INTERMEDIATE_SIZE - --moe-router-topk $MOE_ACTIVE_ROUTED_EXPERTS - --moe-grouped-gemm - --moe-permute-fusion - --moe-ffn-hidden-size $MOE_FFN_HIDDEN - --moe-router-score-function sigmoid - --moe-router-pre-softmax - --moe-router-enable-expert-bias - --moe-router-bias-update-rate 0 - --moe-router-load-balancing-type seq_aux_loss - --moe-router-topk-scaling-factor 2.5 - --moe-aux-loss-coeff 0 - --moe-router-dtype fp32 - --make-vocab-size-divisible-by 16 - --num-layers $((N_DENSE_LAYERS + N_MOE_LAYERS)) - --hidden-size $NHIDDEN - --ffn-hidden-size $FFN_HIDDEN - --num-attention-heads $NHEADS - --disable-bias-linear - --swiglu - --untie-embeddings-and-output-weights - --position-embedding-type rope - --no-position-embedding - --normalization RMSNorm - --qk-layernorm - --multi-latent-attention - --q-lora-rank 2048 - --kv-lora-rank 512 - --qk-head-dim 192 - --v-head-dim 256 - --kv-channels 192 - --qk-pos-emb-head-dim 64 - --vocab-size 154880 - --rotary-base 8000000 - --enable-experimental -) - -# LoRA registry for scripts/run_glm5_2_744b_a40b_lora.py: MODEL_ARGS carries the architecture only -# (identical to glm5.2-744B-A40B.sh); every LoRA / run-mode flag lives in the runner, which -# always wins (argparse last-occurrence). --spec above is inert under bridge LoRA. diff --git a/scripts/models/gpt-oss-20b.py b/scripts/models/gpt-oss-20b.py new file mode 100644 index 00000000000..18009a85e96 --- /dev/null +++ b/scripts/models/gpt-oss-20b.py @@ -0,0 +1,40 @@ +def model_args() -> str: + return ( + # Base architecture + "--num-layers 24 " + "--hidden-size 2880 " + "--ffn-hidden-size 2880 " + "--num-attention-heads 64 " + "--group-query-attention " + "--num-query-groups 8 " + "--kv-channels 64 " + # Positional embeddings + "--use-rotary-position-embeddings " + "--rotary-percent 1.0 " + "--rotary-base 150000 " + # Train with a 4k context, but keep max positions aligned with the HF checkpoint (YaRN scaling). + "--max-position-embeddings 131072 " + # Normalization + "--normalization RMSNorm " + "--norm-epsilon 1e-5 " + # Activation & embeddings + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 201088 " + # Note: attention_bias is true in HF config, so we may need bias + # --disable-bias-linear # commented out since attention_bias=true + # Sliding window attention + learnable softmax offset (alternating SWA/full attention). + "--softmax-type learnable " + "--window-size 128,0 " + "--window-attn-skip-freq 2 " + # Fusions can be incompatible with this attention pattern on some stacks. + "--no-masked-softmax-fusion " + "--no-rope-fusion " + # MoE parameters + "--num-experts 32 " + "--moe-router-topk 4 " + "--moe-aux-loss-coeff 0.0 " + "--moe-token-dispatcher-type alltoall " + "--moe-router-dtype fp32 " + "--moe-grouped-gemm " + ) diff --git a/scripts/models/gpt-oss-20b.sh b/scripts/models/gpt-oss-20b.sh deleted file mode 100644 index c1184f515ab..00000000000 --- a/scripts/models/gpt-oss-20b.sh +++ /dev/null @@ -1,48 +0,0 @@ -# gpt-oss-20b model architecture -# Expected to match HF config for gpt-oss-20b-BF16 (MoE + sliding window attention). - -MODEL_ARGS=( - # Base architecture - --num-layers 24 - --hidden-size 2880 - --ffn-hidden-size 2880 - --num-attention-heads 64 - --group-query-attention - --num-query-groups 8 - --kv-channels 64 - - # Positional embeddings - --use-rotary-position-embeddings - --rotary-percent 1.0 - --rotary-base 150000 - # Train with a 4k context, but keep max positions aligned with the HF checkpoint (YaRN scaling). - --max-position-embeddings 131072 - - # Normalization - --normalization "RMSNorm" - --norm-epsilon 1e-5 - - # Activation & embeddings - --swiglu - --untie-embeddings-and-output-weights - --vocab-size 201088 - - # Note: attention_bias is true in HF config, so we may need bias - # --disable-bias-linear # commented out since attention_bias=true - - # Sliding window attention + learnable softmax offset (alternating SWA/full attention). - --softmax-type learnable - --window-size 128,0 - --window-attn-skip-freq 2 - # Fusions can be incompatible with this attention pattern on some stacks. - --no-masked-softmax-fusion - --no-rope-fusion - - # MoE parameters - --num-experts 32 - --moe-router-topk 4 - --moe-aux-loss-coeff 0.0 - --moe-token-dispatcher-type alltoall - --moe-router-dtype fp32 - --moe-grouped-gemm -) diff --git a/scripts/models/inkling-small.py b/scripts/models/inkling-small.py new file mode 100644 index 00000000000..e1d0226515b --- /dev/null +++ b/scripts/models/inkling-small.py @@ -0,0 +1,44 @@ +import os + + +def model_args(nlayers: int | None = None) -> str: + # Inkling-Small 276B config (42 layers; derived from the HF config the same way + # inkling.py maps the Inkling one). + nlayers = nlayers if nlayers is not None else int(os.environ.get("MODEL_ARGS_NUM_LAYERS") or 42) + return ( + "--disable-bias-linear " + f"--num-layers {nlayers} " + "--hidden-size 4096 " + "--ffn-hidden-size 2048 " + "--num-attention-heads 32 " + "--group-query-attention " + "--num-query-groups 8 " + "--kv-channels 128 " + "--normalization RMSNorm " + "--norm-epsilon 1e-6 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 201024 " + "--hidden-dropout 0.0 " + "--attention-dropout 0.0 " + "--attention-softmax-in-fp32 " + "--position-embedding-type none " + "--no-rope-fusion " + "--no-masked-softmax-fusion " + "--max-position-embeddings 1048576 " + # MoE + "--num-experts 256 " + "--moe-ffn-hidden-size 2048 " + "--moe-router-topk 6 " + "--moe-shared-expert-intermediate-size 2048 " + "--moe-router-pre-softmax " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-token-dispatcher-type alltoall " + "--moe-aux-loss-coeff 0 " + "--moe-grouped-gemm " + "--qk-layernorm " + # Inkling model provider + "--custom-model-provider-path miles_plugins.models.inkling.model.inkling_model_provider " + ) diff --git a/scripts/models/inkling-small.sh b/scripts/models/inkling-small.sh deleted file mode 100644 index 10131973de1..00000000000 --- a/scripts/models/inkling-small.sh +++ /dev/null @@ -1,43 +0,0 @@ -NLAYERS="${MODEL_ARGS_NUM_LAYERS:-42}" - -# Inkling-Small 276B config (42 layers; derived from the HF config the same way -# inkling.sh maps the Inkling one). -MODEL_ARGS=( - --disable-bias-linear - --num-layers $NLAYERS - --hidden-size 4096 - --ffn-hidden-size 2048 - --num-attention-heads 32 - --group-query-attention - --num-query-groups 8 - --kv-channels 128 - --normalization RMSNorm - --norm-epsilon 1e-6 - --swiglu - --untie-embeddings-and-output-weights - --vocab-size 201024 - --hidden-dropout 0.0 - --attention-dropout 0.0 - --attention-softmax-in-fp32 - --position-embedding-type none - --no-rope-fusion - --no-masked-softmax-fusion - --max-position-embeddings 1048576 - - # MoE - --num-experts 256 - --moe-ffn-hidden-size 2048 - --moe-router-topk 6 - --moe-shared-expert-intermediate-size 2048 - --moe-router-pre-softmax - --moe-router-score-function sigmoid - --moe-router-enable-expert-bias - --moe-router-load-balancing-type seq_aux_loss - --moe-token-dispatcher-type alltoall - --moe-aux-loss-coeff 0 - --moe-grouped-gemm - --qk-layernorm - - # Inkling model provider - --custom-model-provider-path miles_plugins.models.inkling.model.inkling_model_provider -) diff --git a/scripts/models/inkling.py b/scripts/models/inkling.py new file mode 100644 index 00000000000..d5b2e726287 --- /dev/null +++ b/scripts/models/inkling.py @@ -0,0 +1,43 @@ +import os + + +def model_args(nlayers: int | None = None) -> str: + # Inkling config (66L full model; set MODEL_ARGS_NUM_LAYERS=4 for the 4-layer slice) + nlayers = nlayers if nlayers is not None else int(os.environ.get("MODEL_ARGS_NUM_LAYERS") or 66) + return ( + "--disable-bias-linear " + f"--num-layers {nlayers} " + "--hidden-size 6144 " + "--ffn-hidden-size 3072 " + "--num-attention-heads 64 " + "--group-query-attention " + "--num-query-groups 8 " + "--kv-channels 128 " + "--normalization RMSNorm " + "--norm-epsilon 1e-6 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 201024 " + "--hidden-dropout 0.0 " + "--attention-dropout 0.0 " + "--attention-softmax-in-fp32 " + "--position-embedding-type none " + "--no-rope-fusion " + "--no-masked-softmax-fusion " + "--max-position-embeddings 1048576 " + # MoE + "--num-experts 256 " + "--moe-ffn-hidden-size 3072 " + "--moe-router-topk 6 " + "--moe-shared-expert-intermediate-size 3072 " + "--moe-router-pre-softmax " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-token-dispatcher-type alltoall " + "--moe-aux-loss-coeff 0 " + "--moe-grouped-gemm " + "--qk-layernorm " + # Inkling model provider + "--custom-model-provider-path miles_plugins.models.inkling.model.inkling_model_provider " + ) diff --git a/scripts/models/inkling.sh b/scripts/models/inkling.sh deleted file mode 100644 index 313f4a3db88..00000000000 --- a/scripts/models/inkling.sh +++ /dev/null @@ -1,42 +0,0 @@ -NLAYERS="${MODEL_ARGS_NUM_LAYERS:-66}" - -# Inkling config (66L full model; set MODEL_ARGS_NUM_LAYERS=4 for the 4-layer slice) -MODEL_ARGS=( - --disable-bias-linear - --num-layers $NLAYERS - --hidden-size 6144 - --ffn-hidden-size 3072 - --num-attention-heads 64 - --group-query-attention - --num-query-groups 8 - --kv-channels 128 - --normalization RMSNorm - --norm-epsilon 1e-6 - --swiglu - --untie-embeddings-and-output-weights - --vocab-size 201024 - --hidden-dropout 0.0 - --attention-dropout 0.0 - --attention-softmax-in-fp32 - --position-embedding-type none - --no-rope-fusion - --no-masked-softmax-fusion - --max-position-embeddings 1048576 - - # MoE - --num-experts 256 - --moe-ffn-hidden-size 3072 - --moe-router-topk 6 - --moe-shared-expert-intermediate-size 3072 - --moe-router-pre-softmax - --moe-router-score-function sigmoid - --moe-router-enable-expert-bias - --moe-router-load-balancing-type seq_aux_loss - --moe-token-dispatcher-type alltoall - --moe-aux-loss-coeff 0 - --moe-grouped-gemm - --qk-layernorm - - # Inkling model provider - --custom-model-provider-path miles_plugins.models.inkling.model.inkling_model_provider -) diff --git a/scripts/models/joyai-llm-flash.py b/scripts/models/joyai-llm-flash.py new file mode 100644 index 00000000000..14296e4502c --- /dev/null +++ b/scripts/models/joyai-llm-flash.py @@ -0,0 +1,55 @@ +import os + +from model_args_utils import moe_layer_freq + + +FIRST_K_DENSE_REPLACE = 1 + + +def model_args(nlayers: int | None = None, rotary_base: str | None = None) -> str: + nlayers = nlayers if nlayers is not None else int(os.environ.get("MODEL_ARGS_NUM_LAYERS") or 40) + rotary_base = rotary_base if rotary_base is not None else os.environ.get("MODEL_ARGS_ROTARY_BASE") or "32000000" + return ( + "--disable-bias-linear " + f"--num-layers {nlayers} " + "--hidden-size 2048 " + "--ffn-hidden-size 7168 " + "--num-attention-heads 32 " + "--kv-channels 128 " + "--normalization RMSNorm " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 129280 " + "--multi-latent-attention " + "--q-lora-rank 1536 " + "--kv-lora-rank 512 " + "--qk-head-dim 128 " + "--qk-pos-emb-head-dim 64 " + "--v-head-dim 128 " + "--qk-layernorm " + f"--rotary-base {rotary_base} " + "--mscale 1.0 " + "--mscale-all-dim 1.0 " + "--attention-softmax-in-fp32 " + "--no-rope-fusion " + "--num-experts 256 " + f"--moe-layer-freq {moe_layer_freq(nlayers=nlayers, first_k_dense_replace=FIRST_K_DENSE_REPLACE)} " + "--moe-ffn-hidden-size 768 " + "--moe-router-topk 8 " + "--moe-shared-expert-intermediate-size 768 " + "--moe-router-pre-softmax " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-token-dispatcher-type alltoall " + "--moe-aux-loss-coeff 0 " + "--moe-router-bias-update-rate 0 " + "--moe-router-group-topk 1 " + "--moe-router-num-groups 1 " + "--moe-grouped-gemm " + "--moe-router-topk-scaling-factor 2.5 " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + ) diff --git a/scripts/models/joyai-llm-flash.sh b/scripts/models/joyai-llm-flash.sh deleted file mode 100644 index 8a2b28c537a..00000000000 --- a/scripts/models/joyai-llm-flash.sh +++ /dev/null @@ -1,60 +0,0 @@ -NLAYERS="${MODEL_ARGS_NUM_LAYERS:-40}" -FIRST_K_DENSE_REPLACE=1 - -arr=() -for ((i=0; i str: + return ( + "--disable-bias-linear " + f"--num-layers {nlayers} " + "--hidden-size 7168 " + "--ffn-hidden-size 18432 " + "--num-attention-heads 64 " + "--kv-channels 64 " + "--normalization RMSNorm " + "--position-embedding-type rope " + "--norm-epsilon 1e-5 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 163840 " + "--multi-latent-attention " + "--q-lora-rank 1536 " + "--kv-lora-rank 512 " + "--qk-head-dim 128 " + "--qk-pos-emb-head-dim 64 " + "--v-head-dim 128 " + "--qk-layernorm " + "--rotary-scaling-factor 64.0 " + "--rotary-base 50000 " + "--mscale 1.0 " + "--mscale-all-dim 1.0 " + "--attention-softmax-in-fp32 " + "--no-rope-fusion " + # moe + "--num-experts 384 " + f"--moe-layer-freq {moe_layer_freq(nlayers=nlayers, first_k_dense_replace=first_k_dense_replace)} " + "--moe-ffn-hidden-size 2048 " + "--moe-router-topk 8 " + "--moe-shared-expert-intermediate-size 2048 " + "--moe-router-pre-softmax " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-token-dispatcher-type alltoall " + "--moe-aux-loss-coeff 0 " + "--moe-router-bias-update-rate 0 " + "--moe-router-group-topk 1 " + "--moe-router-num-groups 1 " + "--moe-grouped-gemm " + "--moe-router-topk-scaling-factor 2.827 " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + ) diff --git a/scripts/models/kimi-k2-thinking.sh b/scripts/models/kimi-k2-thinking.sh deleted file mode 100644 index b7fceda5913..00000000000 --- a/scripts/models/kimi-k2-thinking.sh +++ /dev/null @@ -1,63 +0,0 @@ -NLAYERS=61 -FIRST_K_DENSE_REPLACE=1 - -arr=() -for ((i=0; i str: + return ( + "--disable-bias-linear " + "--num-layers 61 " + "--hidden-size 7168 " + "--ffn-hidden-size 18432 " + "--num-attention-heads 64 " + "--kv-channels 64 " + "--normalization RMSNorm " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 163840 " + "--multi-latent-attention " + "--q-lora-rank 1536 " + "--kv-lora-rank 512 " + "--qk-head-dim 128 " + "--qk-pos-emb-head-dim 64 " + "--v-head-dim 128 " + "--qk-layernorm " + "--rotary-scaling-factor 32.0 " + "--rotary-base 50000 " + "--mscale 1.0 " + "--mscale-all-dim 1.0 " + "--attention-softmax-in-fp32 " + "--no-rope-fusion " + # moe + "--num-experts 384 " + f"--moe-layer-freq {moe_layer_freq(nlayers=NLAYERS, first_k_dense_replace=FIRST_K_DENSE_REPLACE)} " + "--moe-ffn-hidden-size 2048 " + "--moe-router-topk 8 " + "--moe-shared-expert-intermediate-size 2048 " + "--moe-router-pre-softmax " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-token-dispatcher-type alltoall " + "--moe-aux-loss-coeff 0 " + "--moe-router-bias-update-rate 0 " + "--moe-router-group-topk 1 " + "--moe-router-num-groups 1 " + "--moe-grouped-gemm " + "--moe-router-topk-scaling-factor 2.827 " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + ) diff --git a/scripts/models/kimi-k2.sh b/scripts/models/kimi-k2.sh deleted file mode 100644 index eafb7acadd1..00000000000 --- a/scripts/models/kimi-k2.sh +++ /dev/null @@ -1,63 +0,0 @@ -NLAYERS=61 -FIRST_K_DENSE_REPLACE=1 - -arr=() -for ((i=0; i str: + # Override for the 2-layer pruned debugging model (first_k_dense_replace=1): + # 1 dense layer + 1 MoE layer. Architecture is otherwise identical to the full + # Kimi-K2.5 / K2-Thinking, so we reuse those MODEL_ARGS and only patch the + # layer count and the MoE-layer-frequency mask. + return load_sibling_model_args(__file__, "kimi-k2-thinking", nlayers=2, first_k_dense_replace=1) diff --git a/scripts/models/kimi-k25_2layer.sh b/scripts/models/kimi-k25_2layer.sh deleted file mode 100644 index f57c207b3a8..00000000000 --- a/scripts/models/kimi-k25_2layer.sh +++ /dev/null @@ -1,26 +0,0 @@ -SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]:-$0}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/kimi-k2-thinking.sh" - -# Override for the 2-layer pruned debugging model (first_k_dense_replace=1): -# 1 dense layer + 1 MoE layer. Architecture is otherwise identical to the full -# Kimi-K2.5 / K2-Thinking, so we reuse those MODEL_ARGS and only patch the -# layer count and the MoE-layer-frequency mask. -NLAYERS=2 -FIRST_K_DENSE_REPLACE=1 - -arr=() -for ((i = 0; i < NLAYERS; i++)); do - if ((i < FIRST_K_DENSE_REPLACE)); then - arr+=(0) - else - arr+=(1) - fi -done -printf -v MOE_LAYER_FREQ "[%s]" "$(IFS=', '; echo "${arr[*]}")" - -for ((i = 0; i < ${#MODEL_ARGS[@]}; i++)); do - case "${MODEL_ARGS[$i]}" in - --num-layers) MODEL_ARGS[$((i + 1))]=$NLAYERS ;; - --moe-layer-freq) MODEL_ARGS[$((i + 1))]="$MOE_LAYER_FREQ" ;; - esac -done diff --git a/scripts/models/llama3.1-8B-Instruct.py b/scripts/models/llama3.1-8B-Instruct.py new file mode 100644 index 00000000000..1375255d845 --- /dev/null +++ b/scripts/models/llama3.1-8B-Instruct.py @@ -0,0 +1,21 @@ +def model_args() -> str: + return ( + "--swiglu " + "--num-layers 32 " + "--hidden-size 4096 " + "--ffn-hidden-size 14336 " + "--num-attention-heads 32 " + "--group-query-attention " + "--num-query-groups 8 " + "--max-position-embeddings 131072 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--normalization RMSNorm " + "--norm-epsilon 1e-5 " + "--rotary-base 500000 " + "--vocab-size 128256 " + "--kv-channels 128 " + "--use-rope-scaling " + "--rotary-scaling-factor 8.0 " + "--untie-embeddings-and-output-weights " + ) diff --git a/scripts/models/llama3.1-8B-Instruct.sh b/scripts/models/llama3.1-8B-Instruct.sh deleted file mode 100644 index 0815b3e0a34..00000000000 --- a/scripts/models/llama3.1-8B-Instruct.sh +++ /dev/null @@ -1,20 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 32 - --hidden-size 4096 - --ffn-hidden-size 14336 - --num-attention-heads 32 - --group-query-attention - --num-query-groups 8 - --max-position-embeddings 131072 - --use-rotary-position-embeddings - --disable-bias-linear - --normalization "RMSNorm" - --norm-epsilon 1e-5 - --rotary-base 500000 - --vocab-size 128256 - --kv-channels 128 - --use-rope-scaling - --rotary-scaling-factor 8.0 - --untie-embeddings-and-output-weights -) \ No newline at end of file diff --git a/scripts/models/llama3.2-3B-Instruct-amd.py b/scripts/models/llama3.2-3B-Instruct-amd.py new file mode 100644 index 00000000000..ba185446bee --- /dev/null +++ b/scripts/models/llama3.2-3B-Instruct-amd.py @@ -0,0 +1,20 @@ +def model_args() -> str: + return ( + "--swiglu " + "--num-layers 28 " + "--hidden-size 3072 " + "--ffn-hidden-size 8192 " + "--num-attention-heads 24 " + "--group-query-attention " + "--num-query-groups 8 " + "--max-position-embeddings 131072 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--normalization RMSNorm " + "--norm-epsilon 1e-5 " + "--rotary-base 500000 " + "--vocab-size 128256 " + "--kv-channels 128 " + "--use-rope-scaling " + "--rotary-scaling-factor 32.0 " + ) diff --git a/scripts/models/llama3.2-3B-Instruct-amd.sh b/scripts/models/llama3.2-3B-Instruct-amd.sh deleted file mode 100644 index 654de5a3865..00000000000 --- a/scripts/models/llama3.2-3B-Instruct-amd.sh +++ /dev/null @@ -1,19 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 28 - --hidden-size 3072 - --ffn-hidden-size 8192 - --num-attention-heads 24 - --group-query-attention - --num-query-groups 8 - --max-position-embeddings 131072 - --use-rotary-position-embeddings - --disable-bias-linear - --normalization "RMSNorm" - --norm-epsilon 1e-5 - --rotary-base 500000 - --vocab-size 128256 - --kv-channels 128 - --use-rope-scaling - --rotary-scaling-factor 32.0 -) diff --git a/scripts/models/llama3.2-3B-Instruct.py b/scripts/models/llama3.2-3B-Instruct.py new file mode 100644 index 00000000000..ba185446bee --- /dev/null +++ b/scripts/models/llama3.2-3B-Instruct.py @@ -0,0 +1,20 @@ +def model_args() -> str: + return ( + "--swiglu " + "--num-layers 28 " + "--hidden-size 3072 " + "--ffn-hidden-size 8192 " + "--num-attention-heads 24 " + "--group-query-attention " + "--num-query-groups 8 " + "--max-position-embeddings 131072 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--normalization RMSNorm " + "--norm-epsilon 1e-5 " + "--rotary-base 500000 " + "--vocab-size 128256 " + "--kv-channels 128 " + "--use-rope-scaling " + "--rotary-scaling-factor 32.0 " + ) diff --git a/scripts/models/llama3.2-3B-Instruct.sh b/scripts/models/llama3.2-3B-Instruct.sh deleted file mode 100644 index ff50130c8a7..00000000000 --- a/scripts/models/llama3.2-3B-Instruct.sh +++ /dev/null @@ -1,19 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 28 - --hidden-size 3072 - --ffn-hidden-size 8192 - --num-attention-heads 24 - --group-query-attention - --num-query-groups 8 - --max-position-embeddings 131072 - --use-rotary-position-embeddings - --disable-bias-linear - --normalization "RMSNorm" - --norm-epsilon 1e-5 - --rotary-base 500000 - --vocab-size 128256 - --kv-channels 128 - --use-rope-scaling - --rotary-scaling-factor 32.0 -) \ No newline at end of file diff --git a/scripts/models/mimo-7B-rl.py b/scripts/models/mimo-7B-rl.py new file mode 100644 index 00000000000..d790dbddbc7 --- /dev/null +++ b/scripts/models/mimo-7B-rl.py @@ -0,0 +1,20 @@ +def model_args() -> str: + return ( + "--swiglu " + "--num-layers 36 " + "--hidden-size 4096 " + "--ffn-hidden-size 11008 " + "--num-attention-heads 32 " + "--group-query-attention " + "--num-query-groups 8 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--add-qkv-bias " + "--normalization RMSNorm " + "--norm-epsilon 1e-05 " + "--rotary-base 640000 " + "--vocab-size 151680 " + "--untie-embeddings-and-output-weights " + "--max-position-embeddings 32768 " + "--mtp-num-layers 1 " + ) diff --git a/scripts/models/mimo-7B-rl.sh b/scripts/models/mimo-7B-rl.sh deleted file mode 100644 index 22366935f99..00000000000 --- a/scripts/models/mimo-7B-rl.sh +++ /dev/null @@ -1,19 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 36 - --hidden-size 4096 - --ffn-hidden-size 11008 - --num-attention-heads 32 - --group-query-attention - --num-query-groups 8 - --use-rotary-position-embeddings - --disable-bias-linear - --add-qkv-bias - --normalization "RMSNorm" - --norm-epsilon 1e-05 - --rotary-base 640000 - --vocab-size 151680 - --untie-embeddings-and-output-weights - --max-position-embeddings 32768 - --mtp-num-layers 1 -) diff --git a/scripts/models/moonlight.py b/scripts/models/moonlight.py new file mode 100644 index 00000000000..432ef0b6a16 --- /dev/null +++ b/scripts/models/moonlight.py @@ -0,0 +1,60 @@ +from model_args_utils import moe_layer_freq + + +MOE_SHARED_EXPERTS = 2 +MOE_FFN_HIDDEN = 1408 +MOE_SHARED_EXPERT_INTERMEDIATE_SIZE = MOE_FFN_HIDDEN * MOE_SHARED_EXPERTS +MOE_ROUTER_TOPK_SCALING_FACTOR = 2.446 +NLAYERS = 27 +FIRST_K_DENSE_REPLACE = 1 + + +def model_args() -> str: + return ( + "--disable-bias-linear " + "--num-layers 27 " + "--hidden-size 2048 " + "--ffn-hidden-size 11264 " + "--num-attention-heads 16 " + "--kv-channels 128 " + "--normalization RMSNorm " + "--position-embedding-type rope " + "--norm-epsilon 1e-5 " + "--rotary-percent 1.0 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--no-masked-softmax-fusion " + "--vocab-size 163840 " + "--multi-latent-attention " + "--kv-lora-rank 512 " + "--qk-head-dim 128 " + "--qk-pos-emb-head-dim 64 " + "--v-head-dim 128 " + "--qk-layernorm " + "--rotary-scaling-factor 1 " + "--rotary-base 50000 " + "--mscale 1.0 " + "--mscale-all-dim 1.0 " + "--attention-softmax-in-fp32 " + "--no-rope-fusion " + # moe + "--num-experts 64 " + f"--moe-layer-freq {moe_layer_freq(nlayers=NLAYERS, first_k_dense_replace=FIRST_K_DENSE_REPLACE)} " + f"--moe-ffn-hidden-size {MOE_FFN_HIDDEN} " + "--moe-router-topk 6 " + f"--moe-shared-expert-intermediate-size {MOE_SHARED_EXPERT_INTERMEDIATE_SIZE} " + "--moe-router-pre-softmax " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-token-dispatcher-type alltoall " + "--moe-aux-loss-coeff 0 " + "--moe-router-bias-update-rate 0 " + "--moe-router-group-topk 1 " + "--moe-router-num-groups 1 " + "--moe-grouped-gemm " + f"--moe-router-topk-scaling-factor {MOE_ROUTER_TOPK_SCALING_FACTOR} " + "--moe-token-drop-policy probs " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + ) diff --git a/scripts/models/moonlight.sh b/scripts/models/moonlight.sh deleted file mode 100644 index bcce99892a7..00000000000 --- a/scripts/models/moonlight.sh +++ /dev/null @@ -1,69 +0,0 @@ -MOE_SHARED_EXPERTS=2 -MOE_FFN_HIDDEN=1408 -MOE_SHARED_EXPERT_INTERMEDIATE_SIZE=$(($MOE_FFN_HIDDEN * $MOE_SHARED_EXPERTS)) -MOE_ROUTER_TOPK_SCALING_FACTOR=2.446 -NLAYERS=27 -FIRST_K_DENSE_REPLACE=1 - -arr=() -for ((i=0; i str: + return ( + "--disable-bias-linear " + "--group-query-attention " + "--num-attention-heads 32 " + "--num-query-groups 2 " + "--kv-channels 128 " + "--num-layers 52 " + "--hidden-size 2688 " + "--ffn-hidden-size 1856 " + "--normalization RMSNorm " + "--position-embedding-type none " + "--vocab-size 131072 " + "--make-vocab-size-divisible-by 128 " + "--untie-embeddings-and-output-weights " + # MoE specifics + "--num-experts 128 " + "--moe-router-topk 6 " + "--moe-ffn-hidden-size 1856 " + "--moe-shared-expert-intermediate-size 3712 " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-grouped-gemm " + "--moe-router-dtype fp32 " + # Routing: config has n_group=1 (MoE groups), topk_group=1, + # routed_scaling_factor=2.5. `n_groups=8` is Mamba groups — unrelated to MoE. + # With n_group=1, group-limited routing is a no-op (single group of 128). + "--moe-router-num-groups 1 " + "--moe-router-group-topk 1 " + "--moe-router-topk-scaling-factor 2.5 " + "--moe-router-pre-softmax " + # Match glm4.7-flash (known-working MoE RL) settings more closely. + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-router-bias-update-rate 0 " + "--moe-aux-loss-coeff 0 " + ) diff --git a/scripts/models/nemotron-3-nano-30b-a3b.sh b/scripts/models/nemotron-3-nano-30b-a3b.sh deleted file mode 100644 index bf10a50b7cc..00000000000 --- a/scripts/models/nemotron-3-nano-30b-a3b.sh +++ /dev/null @@ -1,47 +0,0 @@ -# NVIDIA Nemotron-3-Nano-30B-A3B (BF16, MoE nemotron_h = hybrid Mamba + Attention + MoE). -# HF config (verified 2026-04-21): -# num_hidden_layers=52 hidden_size=2688 num_attention_heads=32 num_key_value_heads=2 -# head_dim=128 intermediate_size=1856 moe_intermediate_size=1856 -# n_routed_experts=128 num_experts_per_tok=6 n_shared_experts=1 -# moe_shared_expert_intermediate_size=3712 sigmoid routing + aux-free expert bias -# The AutoBridge path (--megatron-to-hf-mode bridge) + miles NemotronHBridge MoE shim -# (see miles/backends/megatron_utils/__init__.py) construct the provider and -# HF↔Megatron mapping_registry at load time. Attention-side structural args go -# in MODEL_ARGS for miles' arg parser. - -MODEL_ARGS=( - --disable-bias-linear - --group-query-attention - --num-attention-heads 32 - --num-query-groups 2 - --kv-channels 128 - --num-layers 52 - --hidden-size 2688 - --ffn-hidden-size 1856 - --normalization RMSNorm - --position-embedding-type none - --vocab-size 131072 - --make-vocab-size-divisible-by 128 - --untie-embeddings-and-output-weights - - # MoE specifics - --num-experts 128 - --moe-router-topk 6 - --moe-ffn-hidden-size 1856 - --moe-shared-expert-intermediate-size 3712 - --moe-router-score-function sigmoid - --moe-router-enable-expert-bias - --moe-grouped-gemm - --moe-router-dtype fp32 - # Routing: config has n_group=1 (MoE groups), topk_group=1, - # routed_scaling_factor=2.5. `n_groups=8` is Mamba groups — unrelated to MoE. - # With n_group=1, group-limited routing is a no-op (single group of 128). - --moe-router-num-groups 1 - --moe-router-group-topk 1 - --moe-router-topk-scaling-factor 2.5 - --moe-router-pre-softmax - # Match glm4.7-flash (known-working MoE RL) settings more closely. - --moe-router-load-balancing-type seq_aux_loss - --moe-router-bias-update-rate 0 - --moe-aux-loss-coeff 0 -) diff --git a/scripts/models/nemotron-3-nano-4b.py b/scripts/models/nemotron-3-nano-4b.py new file mode 100644 index 00000000000..2a91c4d7f6f --- /dev/null +++ b/scripts/models/nemotron-3-nano-4b.py @@ -0,0 +1,16 @@ +def model_args() -> str: + return ( + "--disable-bias-linear " + "--group-query-attention " + "--num-attention-heads 40 " + "--num-query-groups 8 " + "--kv-channels 128 " + "--num-layers 42 " + "--hidden-size 3136 " + "--ffn-hidden-size 12544 " + "--normalization RMSNorm " + "--position-embedding-type none " + "--vocab-size 131072 " + "--make-vocab-size-divisible-by 128 " + "--untie-embeddings-and-output-weights " + ) diff --git a/scripts/models/nemotron-3-nano-4b.sh b/scripts/models/nemotron-3-nano-4b.sh deleted file mode 100644 index 9e1a76b5e1d..00000000000 --- a/scripts/models/nemotron-3-nano-4b.sh +++ /dev/null @@ -1,24 +0,0 @@ -# NVIDIA Nemotron-3-Nano-4B (BF16, dense `nemotron_h` = hybrid Mamba + Attention). -# HF config (verified 2026-04-21): -# num_hidden_layers=42 hidden_size=3136 num_attention_heads=40 num_key_value_heads=8 -# vocab_size=131072 max_position_embeddings=262144 no RoPE squared-relu FFN -# The AutoBridge path (--megatron-to-hf-mode bridge) constructs the full Megatron -# provider from the HF config.json at load time, including all Mamba-specific -# fields (mamba_num_heads, mamba_state_dim, hybrid_override_pattern, etc.), so we -# only keep the attention-side structural args here for miles' arg parser. - -MODEL_ARGS=( - --disable-bias-linear - --group-query-attention - --num-attention-heads 40 - --num-query-groups 8 - --kv-channels 128 - --num-layers 42 - --hidden-size 3136 - --ffn-hidden-size 12544 - --normalization RMSNorm - --position-embedding-type none - --vocab-size 131072 - --make-vocab-size-divisible-by 128 - --untie-embeddings-and-output-weights -) diff --git a/scripts/models/nemotron-3-super-120b-a12b.py b/scripts/models/nemotron-3-super-120b-a12b.py new file mode 100644 index 00000000000..83dc71b7e7f --- /dev/null +++ b/scripts/models/nemotron-3-super-120b-a12b.py @@ -0,0 +1,42 @@ +def model_args() -> str: + return ( + "--disable-bias-linear " + "--group-query-attention " + "--num-attention-heads 32 " + "--num-query-groups 2 " + "--kv-channels 128 " + "--num-layers 88 " + "--hidden-size 4096 " + "--ffn-hidden-size 2688 " + "--normalization RMSNorm " + "--position-embedding-type none " + "--vocab-size 131072 " + "--make-vocab-size-divisible-by 128 " + "--untie-embeddings-and-output-weights " + # MoE specifics + "--num-experts 512 " + "--moe-router-topk 22 " + "--moe-ffn-hidden-size 2688 " + "--moe-shared-expert-intermediate-size 5376 " + # Super-120B bottlenecks expert input/output through a 1024-dim latent. + # Routed experts run on moe_latent_size, NOT hidden_size, with two extra + # fc1/fc2 latent projections per MoE layer. The miles NemotronH bridge + # surfaces this from HF config; the CLI arg keeps Megatron's parser happy. + "--moe-latent-size 1024 " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-grouped-gemm " + "--moe-router-dtype fp32 " + # Routing: HF config has n_group=1 (MoE groups), topk_group=1, + # routed_scaling_factor=5.0. With n_group=1, group-limited routing is a + # no-op (single group of 512). `n_groups=8` in HF is Mamba groups — + # unrelated to MoE. + "--moe-router-num-groups 1 " + "--moe-router-group-topk 1 " + "--moe-router-topk-scaling-factor 5.0 " + "--moe-router-pre-softmax " + # Match nano-30b-a3b (known-working MoE RL on nemotron_h) settings. + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-router-bias-update-rate 0 " + "--moe-aux-loss-coeff 0 " + ) diff --git a/scripts/models/nemotron-3-super-120b-a12b.sh b/scripts/models/nemotron-3-super-120b-a12b.sh deleted file mode 100644 index 13e3eceb8bb..00000000000 --- a/scripts/models/nemotron-3-super-120b-a12b.sh +++ /dev/null @@ -1,54 +0,0 @@ -# NVIDIA Nemotron-3-Super-120B-A12B (BF16, MoE nemotron_h = hybrid Mamba + Attention + MoE). -# HF config (verified 2026-05-01): -# num_hidden_layers=88 hidden_size=4096 num_attention_heads=32 num_key_value_heads=2 -# head_dim=128 intermediate_size=2688 moe_intermediate_size=2688 -# n_routed_experts=512 num_experts_per_tok=22 n_shared_experts=1 -# moe_shared_expert_intermediate_size=5376 routed_scaling_factor=5.0 -# n_group=1 topk_group=1 sigmoid routing + aux-free expert bias -# The AutoBridge path (--megatron-to-hf-mode bridge) + miles NemotronHBridge MoE shim -# (see miles_plugins/megatron_bridge/nemotron_h.py) construct the provider and -# HF<->Megatron mapping_registry at load time. Attention-side structural args go -# in MODEL_ARGS for miles' arg parser. - -MODEL_ARGS=( - --disable-bias-linear - --group-query-attention - --num-attention-heads 32 - --num-query-groups 2 - --kv-channels 128 - --num-layers 88 - --hidden-size 4096 - --ffn-hidden-size 2688 - --normalization RMSNorm - --position-embedding-type none - --vocab-size 131072 - --make-vocab-size-divisible-by 128 - --untie-embeddings-and-output-weights - - # MoE specifics - --num-experts 512 - --moe-router-topk 22 - --moe-ffn-hidden-size 2688 - --moe-shared-expert-intermediate-size 5376 - # Super-120B bottlenecks expert input/output through a 1024-dim latent. - # Routed experts run on moe_latent_size, NOT hidden_size, with two extra - # fc1/fc2 latent projections per MoE layer. The miles NemotronH bridge - # surfaces this from HF config; the CLI arg keeps Megatron's parser happy. - --moe-latent-size 1024 - --moe-router-score-function sigmoid - --moe-router-enable-expert-bias - --moe-grouped-gemm - --moe-router-dtype fp32 - # Routing: HF config has n_group=1 (MoE groups), topk_group=1, - # routed_scaling_factor=5.0. With n_group=1, group-limited routing is a - # no-op (single group of 512). `n_groups=8` in HF is Mamba groups — - # unrelated to MoE. - --moe-router-num-groups 1 - --moe-router-group-topk 1 - --moe-router-topk-scaling-factor 5.0 - --moe-router-pre-softmax - # Match nano-30b-a3b (known-working MoE RL on nemotron_h) settings. - --moe-router-load-balancing-type seq_aux_loss - --moe-router-bias-update-rate 0 - --moe-aux-loss-coeff 0 -) diff --git a/scripts/models/nemotron-3-ultra-550b-a55b-4layer.py b/scripts/models/nemotron-3-ultra-550b-a55b-4layer.py new file mode 100644 index 00000000000..c8795f3d414 --- /dev/null +++ b/scripts/models/nemotron-3-ultra-550b-a55b-4layer.py @@ -0,0 +1,48 @@ +# 4-layer slice of NVIDIA Nemotron-3-Ultra-550B-A55B, for single-node (8 GPU) CI. +# +# Built by cluster_scripts/debug_tool_set/checkpoint/prune_nemotron_h.py, which +# keeps source layers 0,1,7,8 and renumbers them 0..3. That selection is the +# cheapest one covering every block type the full 108-layer model has: +# +# layer 0 mamba layer 1 moe layer 2 attention layer 3 moe -> "ME*E" +# MTP head: attention + moe -> "*E" +# +# A prefix cut would need 8 layers to reach the first attention layer and drag +# in 4 MoE layers (~44B params) instead of 2. Everything else (512 experts, +# top-22, moe_latent_size=2048, sigmoid router + expert bias) is unchanged from +# nemotron-3-ultra-550b-a55b.py, so the weight-conversion path is identical. + + +def model_args() -> str: + return ( + "--disable-bias-linear " + "--group-query-attention " + "--num-attention-heads 64 " + "--num-query-groups 2 " + "--kv-channels 128 " + "--num-layers 4 " + "--hidden-size 8192 " + "--ffn-hidden-size 5120 " + "--normalization RMSNorm " + "--position-embedding-type none " + "--vocab-size 131072 " + "--make-vocab-size-divisible-by 128 " + "--untie-embeddings-and-output-weights " + # MoE specifics (identical to the full model) + "--num-experts 512 " + "--moe-router-topk 22 " + "--moe-ffn-hidden-size 5120 " + "--moe-shared-expert-intermediate-size 10240 " + "--moe-latent-size 2048 " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-grouped-gemm " + "--moe-router-dtype fp32 " + "--moe-router-num-groups 1 " + "--moe-router-group-topk 1 " + "--moe-router-topk-scaling-factor 5.0 " + "--moe-router-pre-softmax " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-router-bias-update-rate 0 " + "--moe-aux-loss-coeff 0 " + ) diff --git a/scripts/models/nemotron-3-ultra-550b-a55b-4layer.sh b/scripts/models/nemotron-3-ultra-550b-a55b-4layer.sh deleted file mode 100644 index fc3be08a798..00000000000 --- a/scripts/models/nemotron-3-ultra-550b-a55b-4layer.sh +++ /dev/null @@ -1,47 +0,0 @@ -# 4-layer slice of NVIDIA Nemotron-3-Ultra-550B-A55B, for single-node (8 GPU) CI. -# -# Built by cluster_scripts/debug_tool_set/checkpoint/prune_nemotron_h.py, which -# keeps source layers 0,1,7,8 and renumbers them 0..3. That selection is the -# cheapest one covering every block type the full 108-layer model has: -# -# layer 0 mamba layer 1 moe layer 2 attention layer 3 moe -> "ME*E" -# MTP head: attention + moe -> "*E" -# -# A prefix cut would need 8 layers to reach the first attention layer and drag -# in 4 MoE layers (~44B params) instead of 2. Everything else (512 experts, -# top-22, moe_latent_size=2048, sigmoid router + expert bias) is unchanged from -# nemotron-3-ultra-550b-a55b.sh, so the weight-conversion path is identical. - -MODEL_ARGS=( - --disable-bias-linear - --group-query-attention - --num-attention-heads 64 - --num-query-groups 2 - --kv-channels 128 - --num-layers 4 - --hidden-size 8192 - --ffn-hidden-size 5120 - --normalization RMSNorm - --position-embedding-type none - --vocab-size 131072 - --make-vocab-size-divisible-by 128 - --untie-embeddings-and-output-weights - - # MoE specifics (identical to the full model) - --num-experts 512 - --moe-router-topk 22 - --moe-ffn-hidden-size 5120 - --moe-shared-expert-intermediate-size 10240 - --moe-latent-size 2048 - --moe-router-score-function sigmoid - --moe-router-enable-expert-bias - --moe-grouped-gemm - --moe-router-dtype fp32 - --moe-router-num-groups 1 - --moe-router-group-topk 1 - --moe-router-topk-scaling-factor 5.0 - --moe-router-pre-softmax - --moe-router-load-balancing-type seq_aux_loss - --moe-router-bias-update-rate 0 - --moe-aux-loss-coeff 0 -) diff --git a/scripts/models/nemotron-3-ultra-550b-a55b.py b/scripts/models/nemotron-3-ultra-550b-a55b.py new file mode 100644 index 00000000000..d11d32eeb7d --- /dev/null +++ b/scripts/models/nemotron-3-ultra-550b-a55b.py @@ -0,0 +1,52 @@ +# NVIDIA Nemotron-3-Ultra-550B-A55B (BF16, MoE nemotron_h = hybrid Mamba2 + Attention + MoE). +# HF config: +# num_hidden_layers=108 hidden_size=8192 num_attention_heads=64 num_key_value_heads=2 +# head_dim=128 intermediate_size=5120 moe_intermediate_size=5120 +# n_routed_experts=512 num_experts_per_tok=22 n_shared_experts=1 +# moe_shared_expert_intermediate_size=10240 routed_scaling_factor=5.0 +# moe_latent_size=2048 n_group=1 topk_group=1 sigmoid routing + aux-free expert bias +# num_nextn_predict_layers=1 (MTP head) mamba n_groups=8 +# Same AutoBridge path as Super-120B (--megatron-to-hf-mode bridge) + miles +# NemotronHBridge MoE/latent shim (miles_plugins/megatron_bridge/nemotron_h.py). +# NOTE: Mamba n_groups=8 forces attention/mamba tensor-parallel <= 8 (n_groups % tp == 0). + + +def model_args() -> str: + return ( + "--disable-bias-linear " + "--group-query-attention " + "--num-attention-heads 64 " + "--num-query-groups 2 " + "--kv-channels 128 " + "--num-layers 108 " + "--hidden-size 8192 " + "--ffn-hidden-size 5120 " + "--normalization RMSNorm " + "--position-embedding-type none " + "--vocab-size 131072 " + "--make-vocab-size-divisible-by 128 " + "--untie-embeddings-and-output-weights " + # MoE specifics + "--num-experts 512 " + "--moe-router-topk 22 " + "--moe-ffn-hidden-size 5120 " + "--moe-shared-expert-intermediate-size 10240 " + # Ultra-550B bottlenecks expert input/output through a 2048-dim latent + # (routed experts run on moe_latent_size, not hidden_size; two extra fc1/fc2 + # latent projections per MoE layer). Surfaced from HF config by the miles + # NemotronH bridge; the CLI arg keeps Megatron's parser happy. + "--moe-latent-size 2048 " + "--moe-router-score-function sigmoid " + "--moe-router-enable-expert-bias " + "--moe-grouped-gemm " + "--moe-router-dtype fp32 " + # n_group=1 (MoE groups) -> group-limited routing is a no-op (single group of + # 512). HF n_groups=8 is the Mamba groups, unrelated to MoE. + "--moe-router-num-groups 1 " + "--moe-router-group-topk 1 " + "--moe-router-topk-scaling-factor 5.0 " + "--moe-router-pre-softmax " + "--moe-router-load-balancing-type seq_aux_loss " + "--moe-router-bias-update-rate 0 " + "--moe-aux-loss-coeff 0 " + ) diff --git a/scripts/models/nemotron-3-ultra-550b-a55b.sh b/scripts/models/nemotron-3-ultra-550b-a55b.sh deleted file mode 100644 index 73d99aca60b..00000000000 --- a/scripts/models/nemotron-3-ultra-550b-a55b.sh +++ /dev/null @@ -1,51 +0,0 @@ -# NVIDIA Nemotron-3-Ultra-550B-A55B (BF16, MoE nemotron_h = hybrid Mamba2 + Attention + MoE). -# HF config: -# num_hidden_layers=108 hidden_size=8192 num_attention_heads=64 num_key_value_heads=2 -# head_dim=128 intermediate_size=5120 moe_intermediate_size=5120 -# n_routed_experts=512 num_experts_per_tok=22 n_shared_experts=1 -# moe_shared_expert_intermediate_size=10240 routed_scaling_factor=5.0 -# moe_latent_size=2048 n_group=1 topk_group=1 sigmoid routing + aux-free expert bias -# num_nextn_predict_layers=1 (MTP head) mamba n_groups=8 -# Same AutoBridge path as Super-120B (--megatron-to-hf-mode bridge) + miles -# NemotronHBridge MoE/latent shim (miles_plugins/megatron_bridge/nemotron_h.py). -# NOTE: Mamba n_groups=8 forces attention/mamba tensor-parallel <= 8 (n_groups % tp == 0). - -MODEL_ARGS=( - --disable-bias-linear - --group-query-attention - --num-attention-heads 64 - --num-query-groups 2 - --kv-channels 128 - --num-layers 108 - --hidden-size 8192 - --ffn-hidden-size 5120 - --normalization RMSNorm - --position-embedding-type none - --vocab-size 131072 - --make-vocab-size-divisible-by 128 - --untie-embeddings-and-output-weights - - # MoE specifics - --num-experts 512 - --moe-router-topk 22 - --moe-ffn-hidden-size 5120 - --moe-shared-expert-intermediate-size 10240 - # Ultra-550B bottlenecks expert input/output through a 2048-dim latent - # (routed experts run on moe_latent_size, not hidden_size; two extra fc1/fc2 - # latent projections per MoE layer). Surfaced from HF config by the miles - # NemotronH bridge; the CLI arg keeps Megatron's parser happy. - --moe-latent-size 2048 - --moe-router-score-function sigmoid - --moe-router-enable-expert-bias - --moe-grouped-gemm - --moe-router-dtype fp32 - # n_group=1 (MoE groups) -> group-limited routing is a no-op (single group of - # 512). HF n_groups=8 is the Mamba groups, unrelated to MoE. - --moe-router-num-groups 1 - --moe-router-group-topk 1 - --moe-router-topk-scaling-factor 5.0 - --moe-router-pre-softmax - --moe-router-load-balancing-type seq_aux_loss - --moe-router-bias-update-rate 0 - --moe-aux-loss-coeff 0 -) diff --git a/scripts/models/qwen2.5-0.5B.py b/scripts/models/qwen2.5-0.5B.py new file mode 100644 index 00000000000..d1581103135 --- /dev/null +++ b/scripts/models/qwen2.5-0.5B.py @@ -0,0 +1,17 @@ +def model_args() -> str: + return ( + "--swiglu " + "--num-layers 24 " + "--hidden-size 896 " + "--ffn-hidden-size 4864 " + "--num-attention-heads 14 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--add-qkv-bias " + "--normalization RMSNorm " + "--norm-epsilon 1e-6 " + "--rotary-base 1000000 " + "--group-query-attention " + "--num-query-groups 2 " + "--vocab-size 151936 " + ) diff --git a/scripts/models/qwen2.5-0.5B.sh b/scripts/models/qwen2.5-0.5B.sh deleted file mode 100644 index 66d5b29a024..00000000000 --- a/scripts/models/qwen2.5-0.5B.sh +++ /dev/null @@ -1,16 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 24 - --hidden-size 896 - --ffn-hidden-size 4864 - --num-attention-heads 14 - --use-rotary-position-embeddings - --disable-bias-linear - --add-qkv-bias - --normalization "RMSNorm" - --norm-epsilon 1e-6 - --rotary-base 1000000 - --group-query-attention - --num-query-groups 2 - --vocab-size 151936 -) \ No newline at end of file diff --git a/scripts/models/qwen2.5-1.5B.py b/scripts/models/qwen2.5-1.5B.py new file mode 100644 index 00000000000..19d95e69dbc --- /dev/null +++ b/scripts/models/qwen2.5-1.5B.py @@ -0,0 +1,17 @@ +def model_args() -> str: + return ( + "--swiglu " + "--num-layers 28 " + "--hidden-size 1536 " + "--ffn-hidden-size 8960 " + "--num-attention-heads 12 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--add-qkv-bias " + "--normalization RMSNorm " + "--norm-epsilon 1e-6 " + "--rotary-base 10000 " + "--group-query-attention " + "--num-query-groups 2 " + "--vocab-size 151936 " + ) diff --git a/scripts/models/qwen2.5-1.5B.sh b/scripts/models/qwen2.5-1.5B.sh deleted file mode 100644 index b046a95c66b..00000000000 --- a/scripts/models/qwen2.5-1.5B.sh +++ /dev/null @@ -1,16 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 28 - --hidden-size 1536 - --ffn-hidden-size 8960 - --num-attention-heads 12 - --use-rotary-position-embeddings - --disable-bias-linear - --add-qkv-bias - --normalization "RMSNorm" - --norm-epsilon 1e-6 - --rotary-base 10000 - --group-query-attention - --num-query-groups 2 - --vocab-size 151936 -) \ No newline at end of file diff --git a/scripts/models/qwen2.5-32B.py b/scripts/models/qwen2.5-32B.py new file mode 100644 index 00000000000..bc1251d5ccf --- /dev/null +++ b/scripts/models/qwen2.5-32B.py @@ -0,0 +1,18 @@ +def model_args() -> str: + return ( + "--swiglu " + "--num-layers 64 " + "--hidden-size 5120 " + "--ffn-hidden-size 27648 " + "--num-attention-heads 40 " + "--group-query-attention " + "--num-query-groups 8 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--add-qkv-bias " + "--normalization RMSNorm " + "--norm-epsilon 1e-5 " + "--rotary-base 1000000 " + "--vocab-size 152064 " + "--untie-embeddings-and-output-weights " + ) diff --git a/scripts/models/qwen2.5-32B.sh b/scripts/models/qwen2.5-32B.sh deleted file mode 100644 index 26b49845a41..00000000000 --- a/scripts/models/qwen2.5-32B.sh +++ /dev/null @@ -1,17 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 64 - --hidden-size 5120 - --ffn-hidden-size 27648 - --num-attention-heads 40 - --group-query-attention - --num-query-groups 8 - --use-rotary-position-embeddings - --disable-bias-linear - --add-qkv-bias - --normalization "RMSNorm" - --norm-epsilon 1e-5 - --rotary-base 1000000 - --vocab-size 152064 - --untie-embeddings-and-output-weights -) \ No newline at end of file diff --git a/scripts/models/qwen2.5-3B.py b/scripts/models/qwen2.5-3B.py new file mode 100644 index 00000000000..fe922621ed0 --- /dev/null +++ b/scripts/models/qwen2.5-3B.py @@ -0,0 +1,17 @@ +def model_args() -> str: + return ( + "--swiglu " + "--num-layers 36 " + "--hidden-size 2048 " + "--ffn-hidden-size 11008 " + "--num-attention-heads 16 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--add-qkv-bias " + "--normalization RMSNorm " + "--norm-epsilon 1e-6 " + "--rotary-base 1000000 " + "--group-query-attention " + "--num-query-groups 2 " + "--vocab-size 151936 " + ) diff --git a/scripts/models/qwen2.5-3B.sh b/scripts/models/qwen2.5-3B.sh deleted file mode 100644 index 9da5a9e0339..00000000000 --- a/scripts/models/qwen2.5-3B.sh +++ /dev/null @@ -1,16 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 36 - --hidden-size 2048 - --ffn-hidden-size 11008 - --num-attention-heads 16 - --use-rotary-position-embeddings - --disable-bias-linear - --add-qkv-bias - --normalization "RMSNorm" - --norm-epsilon 1e-6 - --rotary-base 1000000 - --group-query-attention - --num-query-groups 2 - --vocab-size 151936 -) diff --git a/scripts/models/qwen2.5-7B.py b/scripts/models/qwen2.5-7B.py new file mode 100644 index 00000000000..7fcbb618cff --- /dev/null +++ b/scripts/models/qwen2.5-7B.py @@ -0,0 +1,18 @@ +def model_args() -> str: + return ( + "--swiglu " + "--num-layers 28 " + "--hidden-size 3584 " + "--ffn-hidden-size 18944 " + "--num-attention-heads 28 " + "--group-query-attention " + "--num-query-groups 4 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--add-qkv-bias " + "--normalization RMSNorm " + "--norm-epsilon 1e-06 " + "--rotary-base 1000000 " + "--vocab-size 152064 " + "--untie-embeddings-and-output-weights " + ) diff --git a/scripts/models/qwen2.5-7B.sh b/scripts/models/qwen2.5-7B.sh deleted file mode 100644 index eba912b1d7c..00000000000 --- a/scripts/models/qwen2.5-7B.sh +++ /dev/null @@ -1,17 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 28 - --hidden-size 3584 - --ffn-hidden-size 18944 - --num-attention-heads 28 - --group-query-attention - --num-query-groups 4 - --use-rotary-position-embeddings - --disable-bias-linear - --add-qkv-bias - --normalization "RMSNorm" - --norm-epsilon 1e-06 - --rotary-base 1000000 - --vocab-size 152064 - --untie-embeddings-and-output-weights -) \ No newline at end of file diff --git a/scripts/models/qwen3-0.6B.py b/scripts/models/qwen3-0.6B.py new file mode 100644 index 00000000000..781fe2bd437 --- /dev/null +++ b/scripts/models/qwen3-0.6B.py @@ -0,0 +1,18 @@ +def model_args() -> str: + return ( + "--swiglu " + "--num-layers 28 " + "--hidden-size 1024 " + "--ffn-hidden-size 3072 " + "--num-attention-heads 16 " + "--group-query-attention " + "--num-query-groups 8 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--normalization RMSNorm " + "--norm-epsilon 1e-6 " + "--rotary-base 1000000 " + "--vocab-size 151936 " + "--kv-channels 128 " + "--qk-layernorm " + ) diff --git a/scripts/models/qwen3-0.6B.sh b/scripts/models/qwen3-0.6B.sh deleted file mode 100644 index f484ec9519b..00000000000 --- a/scripts/models/qwen3-0.6B.sh +++ /dev/null @@ -1,17 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 28 - --hidden-size 1024 - --ffn-hidden-size 3072 - --num-attention-heads 16 - --group-query-attention - --num-query-groups 8 - --use-rotary-position-embeddings - --disable-bias-linear - --normalization "RMSNorm" - --norm-epsilon 1e-6 - --rotary-base 1000000 - --vocab-size 151936 - --kv-channels 128 - --qk-layernorm -) \ No newline at end of file diff --git a/scripts/models/qwen3-1.7B.py b/scripts/models/qwen3-1.7B.py new file mode 100644 index 00000000000..e74dfac0f56 --- /dev/null +++ b/scripts/models/qwen3-1.7B.py @@ -0,0 +1,22 @@ +import os + + +def model_args(rotary_base: str | None = None) -> str: + rotary_base = rotary_base if rotary_base is not None else os.environ.get("MODEL_ARGS_ROTARY_BASE") or "1000000" + return ( + "--swiglu " + "--num-layers 28 " + "--hidden-size 2048 " + "--ffn-hidden-size 6144 " + "--num-attention-heads 16 " + "--group-query-attention " + "--num-query-groups 8 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--normalization RMSNorm " + "--norm-epsilon 1e-6 " + f"--rotary-base {rotary_base} " + "--vocab-size 151936 " + "--kv-channels 128 " + "--qk-layernorm " + ) diff --git a/scripts/models/qwen3-1.7B.sh b/scripts/models/qwen3-1.7B.sh deleted file mode 100644 index 7435996337e..00000000000 --- a/scripts/models/qwen3-1.7B.sh +++ /dev/null @@ -1,17 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 28 - --hidden-size 2048 - --ffn-hidden-size 6144 - --num-attention-heads 16 - --group-query-attention - --num-query-groups 8 - --use-rotary-position-embeddings - --disable-bias-linear - --normalization "RMSNorm" - --norm-epsilon 1e-6 - --rotary-base "${MODEL_ARGS_ROTARY_BASE:-1000000}" - --vocab-size 151936 - --kv-channels 128 - --qk-layernorm -) \ No newline at end of file diff --git a/scripts/models/qwen3-14B.py b/scripts/models/qwen3-14B.py new file mode 100644 index 00000000000..12b0af4bd78 --- /dev/null +++ b/scripts/models/qwen3-14B.py @@ -0,0 +1,19 @@ +def model_args() -> str: + return ( + "--swiglu " + "--num-layers 40 " + "--hidden-size 5120 " + "--ffn-hidden-size 17408 " + "--num-attention-heads 40 " + "--group-query-attention " + "--num-query-groups 8 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--normalization RMSNorm " + "--norm-epsilon 1e-6 " + "--rotary-base 1000000 " + "--vocab-size 151936 " + "--kv-channels 128 " + "--qk-layernorm " + "--untie-embeddings-and-output-weights " + ) diff --git a/scripts/models/qwen3-14B.sh b/scripts/models/qwen3-14B.sh deleted file mode 100644 index 11b9377da00..00000000000 --- a/scripts/models/qwen3-14B.sh +++ /dev/null @@ -1,18 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 40 - --hidden-size 5120 - --ffn-hidden-size 17408 - --num-attention-heads 40 - --group-query-attention - --num-query-groups 8 - --use-rotary-position-embeddings - --disable-bias-linear - --normalization "RMSNorm" - --norm-epsilon 1e-6 - --rotary-base 1000000 - --vocab-size 151936 - --kv-channels 128 - --qk-layernorm - --untie-embeddings-and-output-weights -) diff --git a/scripts/models/qwen3-235B-A22B.py b/scripts/models/qwen3-235B-A22B.py new file mode 100644 index 00000000000..70c53786ae0 --- /dev/null +++ b/scripts/models/qwen3-235B-A22B.py @@ -0,0 +1,42 @@ +import os + +from model_args_utils import moe_layer_freq + + +NLAYERS = 94 +FIRST_K_DENSE_REPLACE = 0 + + +def model_args(rotary_base: str | None = None) -> str: + rotary_base = rotary_base if rotary_base is not None else os.environ.get("MODEL_ARGS_ROTARY_BASE") or "1000000" + return ( + "--disable-bias-linear " + "--qk-layernorm " + "--group-query-attention " + "--num-attention-heads 64 " + "--num-query-groups 4 " + "--kv-channels 128 " + "--num-layers 94 " + "--hidden-size 4096 " + "--ffn-hidden-size 12288 " + "--normalization RMSNorm " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--rotary-percent 1.0 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 151936 " + f"--rotary-base {rotary_base} " + # moe + "--moe-ffn-hidden-size 1536 " + "--moe-router-score-function softmax " + "--moe-token-dispatcher-type alltoall " + "--moe-router-topk 8 " + f"--moe-layer-freq {moe_layer_freq(nlayers=NLAYERS, first_k_dense_replace=FIRST_K_DENSE_REPLACE)} " + "--num-experts 128 " + "--moe-grouped-gemm " + "--moe-token-drop-policy probs " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + "--moe-aux-loss-coeff 0 " + ) diff --git a/scripts/models/qwen3-235B-A22B.sh b/scripts/models/qwen3-235B-A22B.sh deleted file mode 100644 index 1f663552653..00000000000 --- a/scripts/models/qwen3-235B-A22B.sh +++ /dev/null @@ -1,49 +0,0 @@ -# qwen3-235B-a22B -NLAYERS=94 -FIRST_K_DENSE_REPLACE=0 - -arr=() -for ((i=0; i str: + return load_sibling_model_args(__file__, "qwen3-30B-A3B", nlayers=5) diff --git a/scripts/models/qwen3-30B-A3B-5layer.sh b/scripts/models/qwen3-30B-A3B-5layer.sh deleted file mode 100644 index 449461ae208..00000000000 --- a/scripts/models/qwen3-30B-A3B-5layer.sh +++ /dev/null @@ -1 +0,0 @@ -MODEL_ARGS_NUM_LAYERS=5 source "$(dirname -- "${BASH_SOURCE[0]}")/qwen3-30B-A3B.sh" diff --git a/scripts/models/qwen3-30B-A3B.py b/scripts/models/qwen3-30B-A3B.py new file mode 100644 index 00000000000..4f1a4c33010 --- /dev/null +++ b/scripts/models/qwen3-30B-A3B.py @@ -0,0 +1,42 @@ +import os + +from model_args_utils import moe_layer_freq + + +FIRST_K_DENSE_REPLACE = 0 + + +def model_args(nlayers: int | None = None, rotary_base: str | None = None) -> str: + nlayers = nlayers if nlayers is not None else int(os.environ.get("MODEL_ARGS_NUM_LAYERS") or 48) + rotary_base = rotary_base if rotary_base is not None else os.environ.get("MODEL_ARGS_ROTARY_BASE") or "1000000" + return ( + "--disable-bias-linear " + "--qk-layernorm " + "--group-query-attention " + "--num-attention-heads 32 " + "--num-query-groups 4 " + "--kv-channels 128 " + f"--num-layers {nlayers} " + "--hidden-size 2048 " + "--ffn-hidden-size 6144 " + "--normalization RMSNorm " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--rotary-percent 1.0 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 151936 " + f"--rotary-base {rotary_base} " + # moe + "--moe-ffn-hidden-size 768 " + "--moe-router-score-function softmax " + "--moe-token-dispatcher-type alltoall " + "--moe-router-topk 8 " + f"--moe-layer-freq {moe_layer_freq(nlayers=nlayers, first_k_dense_replace=FIRST_K_DENSE_REPLACE)} " + "--num-experts 128 " + "--moe-grouped-gemm " + "--moe-token-drop-policy probs " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + "--moe-aux-loss-coeff 0 " + ) diff --git a/scripts/models/qwen3-30B-A3B.sh b/scripts/models/qwen3-30B-A3B.sh deleted file mode 100644 index 0221af9bba7..00000000000 --- a/scripts/models/qwen3-30B-A3B.sh +++ /dev/null @@ -1,49 +0,0 @@ -NLAYERS="${MODEL_ARGS_NUM_LAYERS:-48}" -FIRST_K_DENSE_REPLACE=0 - -arr=() -for ((i=0; i str: + return ( + "--swiglu " + "--num-layers 64 " + "--hidden-size 5120 " + "--ffn-hidden-size 25600 " + "--num-attention-heads 64 " + "--group-query-attention " + "--num-query-groups 8 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--normalization RMSNorm " + "--norm-epsilon 1e-6 " + "--rotary-base 1000000 " + "--vocab-size 151936 " + "--kv-channels 128 " + "--qk-layernorm " + "--untie-embeddings-and-output-weights " + ) diff --git a/scripts/models/qwen3-32B.sh b/scripts/models/qwen3-32B.sh deleted file mode 100644 index e7407e327c9..00000000000 --- a/scripts/models/qwen3-32B.sh +++ /dev/null @@ -1,18 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 64 - --hidden-size 5120 - --ffn-hidden-size 25600 - --num-attention-heads 64 - --group-query-attention - --num-query-groups 8 - --use-rotary-position-embeddings - --disable-bias-linear - --normalization "RMSNorm" - --norm-epsilon 1e-6 - --rotary-base 1000000 - --vocab-size 151936 - --kv-channels 128 - --qk-layernorm - --untie-embeddings-and-output-weights -) diff --git a/scripts/models/qwen3-4B-Instruct-2507.py b/scripts/models/qwen3-4B-Instruct-2507.py new file mode 100644 index 00000000000..7c8150db382 --- /dev/null +++ b/scripts/models/qwen3-4B-Instruct-2507.py @@ -0,0 +1,5 @@ +from model_args_utils import load_sibling_model_args + + +def model_args() -> str: + return load_sibling_model_args(__file__, "qwen3-4B", rotary_base=5000000) diff --git a/scripts/models/qwen3-4B-Instruct-2507.sh b/scripts/models/qwen3-4B-Instruct-2507.sh deleted file mode 100644 index 67d13c0c823..00000000000 --- a/scripts/models/qwen3-4B-Instruct-2507.sh +++ /dev/null @@ -1 +0,0 @@ -MODEL_ARGS_ROTARY_BASE=5000000 source "$(dirname -- "${BASH_SOURCE[0]}")/qwen3-4B.sh" \ No newline at end of file diff --git a/scripts/models/qwen3-4B.py b/scripts/models/qwen3-4B.py new file mode 100644 index 00000000000..b1a40729b8b --- /dev/null +++ b/scripts/models/qwen3-4B.py @@ -0,0 +1,22 @@ +import os + + +def model_args(rotary_base: str | None = None) -> str: + rotary_base = rotary_base if rotary_base is not None else os.environ.get("MODEL_ARGS_ROTARY_BASE") or "1000000" + return ( + "--swiglu " + "--num-layers 36 " + "--hidden-size 2560 " + "--ffn-hidden-size 9728 " + "--num-attention-heads 32 " + "--group-query-attention " + "--num-query-groups 8 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--normalization RMSNorm " + "--norm-epsilon 1e-6 " + f"--rotary-base {rotary_base} " + "--vocab-size 151936 " + "--kv-channels 128 " + "--qk-layernorm " + ) diff --git a/scripts/models/qwen3-4B.sh b/scripts/models/qwen3-4B.sh deleted file mode 100644 index 51f9e47581e..00000000000 --- a/scripts/models/qwen3-4B.sh +++ /dev/null @@ -1,17 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 36 - --hidden-size 2560 - --ffn-hidden-size 9728 - --num-attention-heads 32 - --group-query-attention - --num-query-groups 8 - --use-rotary-position-embeddings - --disable-bias-linear - --normalization "RMSNorm" - --norm-epsilon 1e-6 - --rotary-base "${MODEL_ARGS_ROTARY_BASE:-1000000}" - --vocab-size 151936 - --kv-channels 128 - --qk-layernorm -) \ No newline at end of file diff --git a/scripts/models/qwen3-8B.py b/scripts/models/qwen3-8B.py new file mode 100644 index 00000000000..029133f0d8b --- /dev/null +++ b/scripts/models/qwen3-8B.py @@ -0,0 +1,23 @@ +import os + + +def model_args(rotary_base: str | None = None) -> str: + rotary_base = rotary_base if rotary_base is not None else os.environ.get("MODEL_ARGS_ROTARY_BASE") or "1000000" + return ( + "--swiglu " + "--num-layers 36 " + "--hidden-size 4096 " + "--ffn-hidden-size 12288 " + "--num-attention-heads 32 " + "--group-query-attention " + "--num-query-groups 8 " + "--use-rotary-position-embeddings " + "--disable-bias-linear " + "--normalization RMSNorm " + "--norm-epsilon 1e-6 " + f"--rotary-base {rotary_base} " + "--vocab-size 151936 " + "--kv-channels 128 " + "--qk-layernorm " + "--untie-embeddings-and-output-weights " + ) diff --git a/scripts/models/qwen3-8B.sh b/scripts/models/qwen3-8B.sh deleted file mode 100644 index fc573adb37b..00000000000 --- a/scripts/models/qwen3-8B.sh +++ /dev/null @@ -1,18 +0,0 @@ -MODEL_ARGS=( - --swiglu - --num-layers 36 - --hidden-size 4096 - --ffn-hidden-size 12288 - --num-attention-heads 32 - --group-query-attention - --num-query-groups 8 - --use-rotary-position-embeddings - --disable-bias-linear - --normalization "RMSNorm" - --norm-epsilon 1e-6 - --rotary-base "${MODEL_ARGS_ROTARY_BASE:-1000000}" - --vocab-size 151936 - --kv-channels 128 - --qk-layernorm - --untie-embeddings-and-output-weights -) \ No newline at end of file diff --git a/scripts/models/qwen3-next-80B-A3B.py b/scripts/models/qwen3-next-80B-A3B.py new file mode 100644 index 00000000000..402b870c738 --- /dev/null +++ b/scripts/models/qwen3-next-80B-A3B.py @@ -0,0 +1,46 @@ +from model_args_utils import moe_layer_freq + + +NLAYERS = 48 +FIRST_K_DENSE_REPLACE = 0 + + +def model_args() -> str: + return ( + "--spec miles_plugins.models.qwen3_next get_qwen3_next_spec " + "--disable-bias-linear " + "--qk-layernorm " + "--group-query-attention " + "--num-attention-heads 16 " + "--num-query-groups 2 " + "--kv-channels 256 " + "--num-layers 48 " + "--hidden-size 2048 " + "--ffn-hidden-size 5120 " + "--normalization RMSNorm " + "--apply-layernorm-1p " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--rotary-percent 0.25 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 151936 " + "--rotary-base 10000000 " + # moe + "--moe-ffn-hidden-size 512 " + "--moe-shared-expert-intermediate-size 512 " + "--moe-router-score-function softmax " + "--moe-token-dispatcher-type alltoall " + "--moe-router-topk 10 " + f"--moe-layer-freq {moe_layer_freq(nlayers=NLAYERS, first_k_dense_replace=FIRST_K_DENSE_REPLACE)} " + "--num-experts 512 " + "--moe-grouped-gemm " + "--moe-token-drop-policy probs " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + "--moe-aux-loss-coeff 0 " + # qwen3 specific + "--attention-output-gate " + "--moe-shared-expert-gate " + "--mtp-num-layers 1 " + ) diff --git a/scripts/models/qwen3-next-80B-A3B.sh b/scripts/models/qwen3-next-80B-A3B.sh deleted file mode 100644 index e5390854126..00000000000 --- a/scripts/models/qwen3-next-80B-A3B.sh +++ /dev/null @@ -1,58 +0,0 @@ -NLAYERS=48 -FIRST_K_DENSE_REPLACE=0 - -arr=() -for ((i=0; i str: + return ( + "--spec miles_plugins.models.qwen3_5 get_qwen3_5_spec " + "--disable-bias-linear " + "--qk-layernorm " + "--group-query-attention " + "--num-attention-heads 24 " + "--num-query-groups 4 " + "--kv-channels 256 " + "--num-layers 64 " + "--hidden-size 5120 " + "--ffn-hidden-size 17408 " + "--normalization RMSNorm " + "--apply-layernorm-1p " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--rotary-percent 0.25 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 248320 " + "--rotary-base 10000000 " + # qwen3.5 specific + "--attention-output-gate " + ) diff --git a/scripts/models/qwen3.5-27B.sh b/scripts/models/qwen3.5-27B.sh deleted file mode 100644 index 5e76a6d3f9e..00000000000 --- a/scripts/models/qwen3.5-27B.sh +++ /dev/null @@ -1,27 +0,0 @@ -MODEL_ARGS=( - --spec "miles_plugins.models.qwen3_5" "get_qwen3_5_spec" - - --disable-bias-linear - --qk-layernorm - --group-query-attention - --num-attention-heads 24 - --num-query-groups 4 - --kv-channels 256 - --num-layers 64 - --hidden-size 5120 - --ffn-hidden-size 17408 - - --normalization RMSNorm - --apply-layernorm-1p - --position-embedding-type rope - --norm-epsilon 1e-6 - --rotary-percent 0.25 - --swiglu - --untie-embeddings-and-output-weights - --vocab-size 248320 - - --rotary-base 10000000 - - # qwen3.5 specific - --attention-output-gate -) diff --git a/scripts/models/qwen3.5-35B-A3B.py b/scripts/models/qwen3.5-35B-A3B.py new file mode 100644 index 00000000000..e63ef67ee33 --- /dev/null +++ b/scripts/models/qwen3.5-35B-A3B.py @@ -0,0 +1,46 @@ +from model_args_utils import moe_layer_freq + + +NLAYERS = 40 +FIRST_K_DENSE_REPLACE = 0 + + +def model_args() -> str: + return ( + "--spec miles_plugins.models.qwen3_5 get_qwen3_5_spec " + "--disable-bias-linear " + "--qk-layernorm " + "--group-query-attention " + "--num-attention-heads 16 " + "--num-query-groups 2 " + "--kv-channels 256 " + "--num-layers 40 " + "--hidden-size 2048 " + "--ffn-hidden-size 512 " + "--normalization RMSNorm " + "--apply-layernorm-1p " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--rotary-percent 0.25 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 248320 " + "--rotary-base 10000000 " + # moe + "--moe-ffn-hidden-size 512 " + "--moe-shared-expert-intermediate-size 512 " + "--moe-router-score-function softmax " + "--moe-token-dispatcher-type alltoall " + "--moe-router-topk 8 " + f"--moe-layer-freq {moe_layer_freq(nlayers=NLAYERS, first_k_dense_replace=FIRST_K_DENSE_REPLACE)} " + "--num-experts 256 " + "--moe-grouped-gemm " + "--moe-token-drop-policy probs " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + "--moe-aux-loss-coeff 0 " + # qwen3.5 specific + "--attention-output-gate " + "--moe-shared-expert-gate " + "--mtp-num-layers 1 " + ) diff --git a/scripts/models/qwen3.5-35B-A3B.sh b/scripts/models/qwen3.5-35B-A3B.sh deleted file mode 100644 index e6912b17dd8..00000000000 --- a/scripts/models/qwen3.5-35B-A3B.sh +++ /dev/null @@ -1,58 +0,0 @@ -NLAYERS=40 -FIRST_K_DENSE_REPLACE=0 - -arr=() -for ((i=0; i str: + return ( + "--spec miles_plugins.models.qwen3_5 get_qwen3_5_spec " + "--disable-bias-linear " + "--qk-layernorm " + "--group-query-attention " + "--num-attention-heads 16 " + "--num-query-groups 2 " + "--kv-channels 256 " + "--num-layers 40 " + "--hidden-size 2048 " + "--ffn-hidden-size 512 " + "--normalization RMSNorm " + "--apply-layernorm-1p " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--rotary-percent 0.25 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 248320 " + "--rotary-base 10000000 " + # moe + "--moe-ffn-hidden-size 512 " + "--moe-shared-expert-intermediate-size 512 " + "--moe-router-score-function softmax " + "--moe-token-dispatcher-type alltoall " + "--moe-router-topk 8 " + f"--moe-layer-freq {moe_layer_freq(nlayers=NLAYERS, first_k_dense_replace=FIRST_K_DENSE_REPLACE)} " + "--num-experts 256 " + "--moe-grouped-gemm " + "--moe-token-drop-policy probs " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + "--moe-aux-loss-coeff 0 " + # qwen3.5 specific + "--attention-output-gate " + "--moe-shared-expert-gate " + "--mtp-num-layers 1 " + ) diff --git a/scripts/models/qwen3.5-35B-A3B_lora.sh b/scripts/models/qwen3.5-35B-A3B_lora.sh deleted file mode 100644 index 9efa764fffc..00000000000 --- a/scripts/models/qwen3.5-35B-A3B_lora.sh +++ /dev/null @@ -1,62 +0,0 @@ -NLAYERS=40 -FIRST_K_DENSE_REPLACE=0 - -arr=() -for ((i=0; i str: + return ( + "--spec miles_plugins.models.qwen3_5 get_qwen3_5_spec " + "--disable-bias-linear " + "--qk-layernorm " + "--group-query-attention " + "--num-attention-heads 16 " + "--num-query-groups 4 " + "--kv-channels 256 " + "--num-layers 32 " + "--hidden-size 2560 " + "--ffn-hidden-size 9216 " + "--normalization RMSNorm " + "--apply-layernorm-1p " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--rotary-percent 0.25 " + "--swiglu " + "--vocab-size 248320 " + "--rotary-base 10000000 " + # qwen3.5 specific + "--attention-output-gate " + ) diff --git a/scripts/models/qwen3.5-4B.sh b/scripts/models/qwen3.5-4B.sh deleted file mode 100644 index 180ad79763b..00000000000 --- a/scripts/models/qwen3.5-4B.sh +++ /dev/null @@ -1,26 +0,0 @@ -MODEL_ARGS=( - --spec "miles_plugins.models.qwen3_5" "get_qwen3_5_spec" - - --disable-bias-linear - --qk-layernorm - --group-query-attention - --num-attention-heads 16 - --num-query-groups 4 - --kv-channels 256 - --num-layers 32 - --hidden-size 2560 - --ffn-hidden-size 9216 - - --normalization RMSNorm - --apply-layernorm-1p - --position-embedding-type rope - --norm-epsilon 1e-6 - --rotary-percent 0.25 - --swiglu - --vocab-size 248320 - - --rotary-base 10000000 - - # qwen3.5 specific - --attention-output-gate -) diff --git a/scripts/models/qwen3.5-9B.py b/scripts/models/qwen3.5-9B.py new file mode 100644 index 00000000000..38f4e05ad60 --- /dev/null +++ b/scripts/models/qwen3.5-9B.py @@ -0,0 +1,24 @@ +def model_args() -> str: + return ( + "--spec miles_plugins.models.qwen3_5 get_qwen3_5_spec " + "--disable-bias-linear " + "--qk-layernorm " + "--group-query-attention " + "--num-attention-heads 16 " + "--num-query-groups 4 " + "--kv-channels 256 " + "--num-layers 32 " + "--hidden-size 4096 " + "--ffn-hidden-size 12288 " + "--normalization RMSNorm " + "--apply-layernorm-1p " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--rotary-percent 0.25 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 248320 " + "--rotary-base 10000000 " + # qwen3.5 specific + "--attention-output-gate " + ) diff --git a/scripts/models/qwen3.5-9B.sh b/scripts/models/qwen3.5-9B.sh deleted file mode 100644 index 433e730ae64..00000000000 --- a/scripts/models/qwen3.5-9B.sh +++ /dev/null @@ -1,27 +0,0 @@ -MODEL_ARGS=( - --spec "miles_plugins.models.qwen3_5" "get_qwen3_5_spec" - - --disable-bias-linear - --qk-layernorm - --group-query-attention - --num-attention-heads 16 - --num-query-groups 4 - --kv-channels 256 - --num-layers 32 - --hidden-size 4096 - --ffn-hidden-size 12288 - - --normalization RMSNorm - --apply-layernorm-1p - --position-embedding-type rope - --norm-epsilon 1e-6 - --rotary-percent 0.25 - --swiglu - --untie-embeddings-and-output-weights - --vocab-size 248320 - - --rotary-base 10000000 - - # qwen3.5 specific - --attention-output-gate -) diff --git a/scripts/models/qwen3.6-27B.py b/scripts/models/qwen3.6-27B.py new file mode 100644 index 00000000000..3d7404b4b97 --- /dev/null +++ b/scripts/models/qwen3.6-27B.py @@ -0,0 +1,24 @@ +def model_args() -> str: + return ( + "--spec miles_plugins.models.qwen3_5 get_qwen3_5_spec " + "--disable-bias-linear " + "--qk-layernorm " + "--group-query-attention " + "--num-attention-heads 24 " + "--num-query-groups 4 " + "--kv-channels 256 " + "--num-layers 64 " + "--hidden-size 5120 " + "--ffn-hidden-size 17408 " + "--normalization RMSNorm " + "--apply-layernorm-1p " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--rotary-percent 0.25 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 248320 " + "--rotary-base 10000000 " + # qwen3.5-family specific + "--attention-output-gate " + ) diff --git a/scripts/models/qwen3.6-27B.sh b/scripts/models/qwen3.6-27B.sh deleted file mode 100644 index c30e566d0a3..00000000000 --- a/scripts/models/qwen3.6-27B.sh +++ /dev/null @@ -1,27 +0,0 @@ -MODEL_ARGS=( - --spec "miles_plugins.models.qwen3_5" "get_qwen3_5_spec" - - --disable-bias-linear - --qk-layernorm - --group-query-attention - --num-attention-heads 24 - --num-query-groups 4 - --kv-channels 256 - --num-layers 64 - --hidden-size 5120 - --ffn-hidden-size 17408 - - --normalization RMSNorm - --apply-layernorm-1p - --position-embedding-type rope - --norm-epsilon 1e-6 - --rotary-percent 0.25 - --swiglu - --untie-embeddings-and-output-weights - --vocab-size 248320 - - --rotary-base 10000000 - - # qwen3.5-family specific - --attention-output-gate -) diff --git a/scripts/models/qwen3.6-35B-A3B.py b/scripts/models/qwen3.6-35B-A3B.py new file mode 100644 index 00000000000..6dea958668d --- /dev/null +++ b/scripts/models/qwen3.6-35B-A3B.py @@ -0,0 +1,46 @@ +from model_args_utils import moe_layer_freq + + +NLAYERS = 40 +FIRST_K_DENSE_REPLACE = 0 + + +def model_args() -> str: + return ( + "--spec miles_plugins.models.qwen3_5 get_qwen3_5_spec " + "--disable-bias-linear " + "--qk-layernorm " + "--group-query-attention " + "--num-attention-heads 16 " + "--num-query-groups 2 " + "--kv-channels 256 " + "--num-layers 40 " + "--hidden-size 2048 " + "--ffn-hidden-size 512 " + "--normalization RMSNorm " + "--apply-layernorm-1p " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--rotary-percent 0.25 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 248320 " + "--rotary-base 10000000 " + # moe + "--moe-ffn-hidden-size 512 " + "--moe-shared-expert-intermediate-size 512 " + "--moe-router-score-function softmax " + "--moe-token-dispatcher-type alltoall " + "--moe-router-topk 8 " + f"--moe-layer-freq {moe_layer_freq(nlayers=NLAYERS, first_k_dense_replace=FIRST_K_DENSE_REPLACE)} " + "--num-experts 256 " + "--moe-grouped-gemm " + "--moe-token-drop-policy probs " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + "--moe-aux-loss-coeff 0 " + # qwen3.6 specific (same architecture as qwen3.5) + "--attention-output-gate " + "--moe-shared-expert-gate " + "--mtp-num-layers 1 " + ) diff --git a/scripts/models/qwen3.6-35B-A3B.sh b/scripts/models/qwen3.6-35B-A3B.sh deleted file mode 100644 index 48d323213a6..00000000000 --- a/scripts/models/qwen3.6-35B-A3B.sh +++ /dev/null @@ -1,58 +0,0 @@ -NLAYERS=40 -FIRST_K_DENSE_REPLACE=0 - -arr=() -for ((i=0; i str: + return ( + "--spec miles_plugins.models.qwen3_5 get_qwen3_5_spec " + "--disable-bias-linear " + "--qk-layernorm " + "--group-query-attention " + "--num-attention-heads 16 " + "--num-query-groups 2 " + "--kv-channels 256 " + "--num-layers 40 " + "--hidden-size 2048 " + "--ffn-hidden-size 512 " + "--normalization RMSNorm " + "--apply-layernorm-1p " + "--position-embedding-type rope " + "--norm-epsilon 1e-6 " + "--rotary-percent 0.25 " + "--swiglu " + "--untie-embeddings-and-output-weights " + "--vocab-size 248320 " + "--rotary-base 10000000 " + # moe + "--moe-ffn-hidden-size 512 " + "--moe-shared-expert-intermediate-size 512 " + "--moe-router-score-function softmax " + "--moe-token-dispatcher-type alltoall " + "--moe-router-topk 8 " + f"--moe-layer-freq {moe_layer_freq(nlayers=NLAYERS, first_k_dense_replace=FIRST_K_DENSE_REPLACE)} " + "--num-experts 256 " + "--moe-grouped-gemm " + "--moe-token-drop-policy probs " + "--moe-router-dtype fp32 " + "--moe-permute-fusion " + "--moe-aux-loss-coeff 0 " + # qwen3.6 specific (same architecture as qwen3.5) + "--attention-output-gate " + "--moe-shared-expert-gate " + "--mtp-num-layers 1 " + ) diff --git a/scripts/models/qwen3.6-35B-A3B_lora.sh b/scripts/models/qwen3.6-35B-A3B_lora.sh deleted file mode 100644 index cca1ee11399..00000000000 --- a/scripts/models/qwen3.6-35B-A3B_lora.sh +++ /dev/null @@ -1,62 +0,0 @@ -NLAYERS=40 -FIRST_K_DENSE_REPLACE=0 - -arr=() -for ((i=0; i/dev/null && pwd)" -source "${SCRIPT_DIR}/models/deepseek-v3.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "deepseek-v3")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint $BASE_DIR/DeepSeek-R1/ #--hf-checkpoint $BASE_DIR/DeepSeek-R1-bf16/ diff --git a/scripts/run-glm4-9B-4xgpu-radixtree.sh b/scripts/run-glm4-9B-4xgpu-radixtree.sh index dbebcd3782b..2d0b0f2a179 100755 --- a/scripts/run-glm4-9B-4xgpu-radixtree.sh +++ b/scripts/run-glm4-9B-4xgpu-radixtree.sh @@ -26,8 +26,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/glm4-9B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "glm4-9B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/GLM-Z1-9B-0414/ --ref-load /root/GLM-Z1-9B-0414_torch_dist diff --git a/scripts/run-glm4-9B.sh b/scripts/run-glm4-9B.sh index 84080ae63bb..b45ecde9885 100644 --- a/scripts/run-glm4-9B.sh +++ b/scripts/run-glm4-9B.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/glm4-9B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "glm4-9B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/GLM-Z1-9B-0414/ --ref-load /root/GLM-Z1-9B-0414_torch_dist diff --git a/scripts/run-glm4.5-355B-A32B.sh b/scripts/run-glm4.5-355B-A32B.sh index 36e3366e0ca..dc0c1a69441 100644 --- a/scripts/run-glm4.5-355B-A32B.sh +++ b/scripts/run-glm4.5-355B-A32B.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/glm4.5-355B-A32B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "glm4.5-355B-A32B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint $BASE_DIR/GLM-4.5-355B-A32B --ref-load $BASE_DIR/GLM-4.5-355B-A32B_torch_dist/ diff --git a/scripts/run-glm4.7-flash.sh b/scripts/run-glm4.7-flash.sh index 18e58fa88e1..06a8192fad4 100644 --- a/scripts/run-glm4.7-flash.sh +++ b/scripts/run-glm4.7-flash.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/../scripts/models/glm4.7-flash.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "glm4.7-flash")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" BASE_DIR=/root/shared CKPT_ARGS=( diff --git a/scripts/run-gpt-oss-20b-bf16.sh b/scripts/run-gpt-oss-20b-bf16.sh index 6ad71ce4c26..4cf823ee1b4 100644 --- a/scripts/run-gpt-oss-20b-bf16.sh +++ b/scripts/run-gpt-oss-20b-bf16.sh @@ -18,8 +18,8 @@ export HF_HOME=/workspace/hf_cache # Load model architecture config SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/gpt-oss-20b.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "gpt-oss-20b")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" BASE_DIR=/root/shared CKPT_ARGS=( diff --git a/scripts/run-kimi-k2-Instruct.sh b/scripts/run-kimi-k2-Instruct.sh index 525f63c6a71..f2e64742034 100644 --- a/scripts/run-kimi-k2-Instruct.sh +++ b/scripts/run-kimi-k2-Instruct.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/kimi-k2.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "kimi-k2")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint $BASE_DIR/Kimi-K2-Instruct/ # --hf-checkpoint $BASE_DIR/Kimi-K2-bf16/ diff --git a/scripts/run-kimi-k2-Thinking.sh b/scripts/run-kimi-k2-Thinking.sh index d603fedb472..da6b0371e15 100644 --- a/scripts/run-kimi-k2-Thinking.sh +++ b/scripts/run-kimi-k2-Thinking.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/kimi-k2-thinking.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "kimi-k2-thinking")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( # --hf-checkpoint $BASE_DIR/Kimi-K2-Thinking-bf16/ --hf-checkpoint $BASE_DIR/Kimi-K2-Thinking-fp8/ diff --git a/scripts/run-kimi-k25.sh b/scripts/run-kimi-k25.sh index e0ec3cccccc..01c82e7e1c4 100755 --- a/scripts/run-kimi-k25.sh +++ b/scripts/run-kimi-k25.sh @@ -27,8 +27,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/kimi-k2-thinking.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "kimi-k2-thinking")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint $BASE_DIR/Kimi-K2.5-int4 --ref-load $BASE_DIR/Kimi-K2.5-bf16 diff --git a/scripts/run-mimo-7B-rl-eagle.sh b/scripts/run-mimo-7B-rl-eagle.sh index 2efbc2d6be8..3d7fa6f7711 100644 --- a/scripts/run-mimo-7B-rl-eagle.sh +++ b/scripts/run-mimo-7B-rl-eagle.sh @@ -25,8 +25,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/mimo-7B-rl.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "mimo-7B-rl")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/MiMo-7B-RL #--hf-checkpoint /root/Qwen3-4B-FP8 diff --git a/scripts/run-moonlight-16B-A3B.sh b/scripts/run-moonlight-16B-A3B.sh index 69a66fdfc4d..772a7c279f0 100644 --- a/scripts/run-moonlight-16B-A3B.sh +++ b/scripts/run-moonlight-16B-A3B.sh @@ -25,8 +25,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/moonlight.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "moonlight")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Moonlight-16B-A3B --ref-load /root/Moonlight-16B-A3B_torch_dist diff --git a/scripts/run-nemotron-3-nano-30b-a3b.sh b/scripts/run-nemotron-3-nano-30b-a3b.sh index 4123adb8820..c405eead797 100755 --- a/scripts/run-nemotron-3-nano-30b-a3b.sh +++ b/scripts/run-nemotron-3-nano-30b-a3b.sh @@ -21,8 +21,8 @@ if [ "$NVLINK_COUNT" -gt 0 ]; then HAS_NVLINK=1; else HAS_NVLINK=0; fi echo "HAS_NVLINK: $HAS_NVLINK" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/nemotron-3-nano-30b-a3b.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "nemotron-3-nano-30b-a3b")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint $BASE_DIR/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16 --ref-load $BASE_DIR/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16 diff --git a/scripts/run-nemotron-3-nano-4b.sh b/scripts/run-nemotron-3-nano-4b.sh index dfdfe7743b6..3316085ba6c 100644 --- a/scripts/run-nemotron-3-nano-4b.sh +++ b/scripts/run-nemotron-3-nano-4b.sh @@ -26,8 +26,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/nemotron-3-nano-4b.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "nemotron-3-nano-4b")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint $BASE_DIR/NVIDIA-Nemotron-3-Nano-4B-BF16 --ref-load $BASE_DIR/NVIDIA-Nemotron-3-Nano-4B-BF16 diff --git a/scripts/run-nemotron-3-super-120b-a12b.sh b/scripts/run-nemotron-3-super-120b-a12b.sh index fd1e31f1461..54f77e66b84 100755 --- a/scripts/run-nemotron-3-super-120b-a12b.sh +++ b/scripts/run-nemotron-3-super-120b-a12b.sh @@ -45,8 +45,8 @@ if [[ "$ROLE" == "worker" ]]; then fi SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/nemotron-3-super-120b-a12b.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "nemotron-3-super-120b-a12b")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" MODELS_DIR=${MODELS_DIR:-/cluster_public/miles_data/models} DATASETS_DIR=${DATASETS_DIR:-/cluster_public/miles_data/datasets} diff --git a/scripts/run-qwen3-235B-A22B-sft.sh b/scripts/run-qwen3-235B-A22B-sft.sh index a5a801c4c8f..3233241189c 100644 --- a/scripts/run-qwen3-235B-A22B-sft.sh +++ b/scripts/run-qwen3-235B-A22B-sft.sh @@ -35,8 +35,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3-235B-A22B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3-235B-A22B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint ${BASE_FOLDER}/Qwen3-235B-A22B --ref-load ${BASE_FOLDER}/Qwen3-235B-A22B_torch_dist diff --git a/scripts/run-qwen3-235B-A22B.sh b/scripts/run-qwen3-235B-A22B.sh index 45067036fb2..51cb60afbd0 100644 --- a/scripts/run-qwen3-235B-A22B.sh +++ b/scripts/run-qwen3-235B-A22B.sh @@ -35,8 +35,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3-235B-A22B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3-235B-A22B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint ${BASE_FOLDER}/Qwen3-235B-A22B-FP8 --ref-load ${BASE_FOLDER}/Qwen3-235B-A22B_torch_dist diff --git a/scripts/run-qwen3-32B.sh b/scripts/run-qwen3-32B.sh index 156bcf5d03a..92b5f6ce30f 100644 --- a/scripts/run-qwen3-32B.sh +++ b/scripts/run-qwen3-32B.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3-32B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3-32B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-32B --ref-load /root/Qwen3-32B_torch_dist/ diff --git a/scripts/run-qwen3-4B-base-sft.sh b/scripts/run-qwen3-4B-base-sft.sh index a30209f750f..e8acefd9ea4 100644 --- a/scripts/run-qwen3-4B-base-sft.sh +++ b/scripts/run-qwen3-4B-base-sft.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3-4B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3-4B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-4B-Base/ --ref-load /root/Qwen3-4B-Base_torch_dist diff --git a/scripts/run-qwen3-4B.sh b/scripts/run-qwen3-4B.sh index 2285cf57c0a..d52e732fbe2 100644 --- a/scripts/run-qwen3-4B.sh +++ b/scripts/run-qwen3-4B.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3-4B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3-4B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-4B #--hf-checkpoint /root/Qwen3-4B-FP8 diff --git a/scripts/run-qwen3-4B_4xgpu.sh b/scripts/run-qwen3-4B_4xgpu.sh index 085266c7dd8..54a268e657b 100755 --- a/scripts/run-qwen3-4B_4xgpu.sh +++ b/scripts/run-qwen3-4B_4xgpu.sh @@ -26,8 +26,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3-4B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3-4B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3-4B #--hf-checkpoint /root/Qwen3-4B-FP8 diff --git a/scripts/run-qwen3-next-80B-A3B-8gpus.sh b/scripts/run-qwen3-next-80B-A3B-8gpus.sh index bc110cf43b7..e8e2aee2c8d 100644 --- a/scripts/run-qwen3-next-80B-A3B-8gpus.sh +++ b/scripts/run-qwen3-next-80B-A3B-8gpus.sh @@ -35,8 +35,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3-next-80B-A3B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3-next-80B-A3B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint ${BASE_FOLDER}/Qwen3-Next-80B-A3B-Thinking --ref-load ${BASE_FOLDER}/Qwen3-Next-80B-A3B-Thinking_torch_dist diff --git a/scripts/run-qwen3-next-80B-A3B.sh b/scripts/run-qwen3-next-80B-A3B.sh index 545c8a1309a..6973de821f2 100644 --- a/scripts/run-qwen3-next-80B-A3B.sh +++ b/scripts/run-qwen3-next-80B-A3B.sh @@ -35,8 +35,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3-next-80B-A3B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3-next-80B-A3B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint ${BASE_FOLDER}/Qwen3-Next-80B-A3B-Thinking --ref-load ${BASE_FOLDER}/Qwen3-Next-80B-A3B-Thinking_torch_dist diff --git a/scripts/run-qwen3.5-27B.sh b/scripts/run-qwen3.5-27B.sh index 3eab260fb18..4221e996caa 100644 --- a/scripts/run-qwen3.5-27B.sh +++ b/scripts/run-qwen3.5-27B.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3.5-27B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3.5-27B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3.5-27B --ref-load /root/Qwen3.5-27B_torch_dist diff --git a/scripts/run-qwen3.5-35B-A3B-mtp.sh b/scripts/run-qwen3.5-35B-A3B-mtp.sh index 062d99686e5..cc8122645ad 100755 --- a/scripts/run-qwen3.5-35B-A3B-mtp.sh +++ b/scripts/run-qwen3.5-35B-A3B-mtp.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3.5-35B-A3B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3.5-35B-A3B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3.5-35B-A3B --ref-load /root/Qwen3.5-35B-A3B_torch_dist diff --git a/scripts/run-qwen3.5-4B.sh b/scripts/run-qwen3.5-4B.sh index 7fce9bdae9e..c9278f6ef8f 100644 --- a/scripts/run-qwen3.5-4B.sh +++ b/scripts/run-qwen3.5-4B.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3.5-4B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3.5-4B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3.5-4B --ref-load /root/Qwen3.5-4B_torch_dist diff --git a/scripts/run-qwen3.5-9B.sh b/scripts/run-qwen3.5-9B.sh index 7664feb47b8..66db36dfaf4 100644 --- a/scripts/run-qwen3.5-9B.sh +++ b/scripts/run-qwen3.5-9B.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3.5-9B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3.5-9B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" CKPT_ARGS=( --hf-checkpoint /root/Qwen3.5-9B --ref-load /root/Qwen3.5-9B_torch_dist diff --git a/scripts/run-qwen3.6-27B.sh b/scripts/run-qwen3.6-27B.sh index dbe7c7c2c07..c7b66f2ad12 100755 --- a/scripts/run-qwen3.6-27B.sh +++ b/scripts/run-qwen3.6-27B.sh @@ -24,8 +24,8 @@ fi echo "HAS_NVLINK: $HAS_NVLINK (detected $NVLINK_COUNT NVLink references)" SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)" -source "${SCRIPT_DIR}/models/qwen3.6-27B.sh" - +MODEL_ARGS_LINE="$(python3 "${SCRIPT_DIR}/../miles/utils/external_utils/model_args_utils.py" "qwen3.6-27B")" || exit 1 +read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" MODEL_DIR="${MODEL_DIR:-/cluster_public/miles_data/models}" DATA_DIR="${DATA_DIR:-/cluster_public/miles_data/datasets}" OUTPUT_DIR="${OUTPUT_DIR:?set OUTPUT_DIR to a writable checkpoint directory}" diff --git a/scripts/run_gemma_4_26b_a4b.py b/scripts/run_gemma_4_26b_a4b.py index f654badc13e..c87b86438ad 100644 --- a/scripts/run_gemma_4_26b_a4b.py +++ b/scripts/run_gemma_4_26b_a4b.py @@ -4,7 +4,7 @@ Trained via the HF<->Megatron bridge (`--megatron-to-hf-mode bridge`) on the base VLM checkpoint directly — sglang runs Gemma4ForConditionalGeneration (hybrid swa), which loads gemma-4's hybrid head_dim weights correctly. MODEL_ARGS come from -scripts/models/gemma-4-26b-a4b-it.sh. +scripts/models/gemma-4-26b-a4b-it.py. Single-node smoke test: python scripts/run_gemma_4_26b_a4b.py full-train --num-nodes 1 diff --git a/scripts/run_gemma_4_31b.py b/scripts/run_gemma_4_31b.py index 8e8f40a67a1..81d83739de1 100644 --- a/scripts/run_gemma_4_31b.py +++ b/scripts/run_gemma_4_31b.py @@ -5,7 +5,7 @@ parallelism. Trained via the HF<->Megatron bridge (`--megatron-to-hf-mode bridge`); the dense config is driven directly through Gemma4VLBridge, so there is no LLM-view rewrite or offline conversion — `prepare` only downloads. -MODEL_ARGS come from scripts/models/gemma-4-31b-it.sh. +MODEL_ARGS come from scripts/models/gemma-4-31b-it.py. Requires the radixark/Megatron-Bridge gemma4-dense branch. diff --git a/scripts/run_inkling.py b/scripts/run_inkling.py index df3c59e8d0b..a59a15952c2 100644 --- a/scripts/run_inkling.py +++ b/scripts/run_inkling.py @@ -68,7 +68,7 @@ app = typer.Typer() -# model name -> scripts/models/.sh; the 4-layer slices reuse the base +# model name -> scripts/models/.py; the 4-layer slices reuse the base # definition with MODEL_ARGS_NUM_LAYERS=4 (set in ScriptArgs.__post_init__) _MODEL_REGISTRY = { "Inkling": "inkling", diff --git a/scripts/run_kimi_k25.py b/scripts/run_kimi_k25.py index 589c540f20b..00e92821c33 100644 --- a/scripts/run_kimi_k25.py +++ b/scripts/run_kimi_k25.py @@ -8,7 +8,7 @@ weights for the SGLang rollout while Megatron loads a BF16 reference via the HF<->Megatron bridge (`--megatron-to-hf-mode bridge`), so there is no offline `torch_dist` conversion step. The architecture is shared with Kimi-K2-Thinking, -whose Megatron MODEL_ARGS we reuse (`scripts/models/kimi-k2-thinking.sh`). +whose Megatron MODEL_ARGS we reuse (`scripts/models/kimi-k2-thinking.py`). ===================== diff --git a/tests/e2e/sglang/test_r3_router_equivalence.py b/tests/e2e/sglang/test_r3_router_equivalence.py index 5ade9d0d43a..3334cdfe656 100644 --- a/tests/e2e/sglang/test_r3_router_equivalence.py +++ b/tests/e2e/sglang/test_r3_router_equivalence.py @@ -32,7 +32,7 @@ Backend / checkpoint ~~~~~~~~~~~~~~~~~~~~ Megatron backend (same as the sibling ``tests/e2e/megatron/*_r3.py`` -tests) — sourcing ``scripts/models/{type}.sh`` populates +tests) — loading ``scripts/models/{type}.py`` populates ``args.num_layers`` / ``args.moe_router_topk`` that the rollout-side reshape of ``routed_experts`` depends on. We do *not* set ``--use-kl-loss`` or ``--kl-coef`` > 0, which is what gates the diff --git a/tests/fast/launch_scripts/model_args_harness.py b/tests/fast/launch_scripts/model_args_harness.py index ceaa42a47d3..516927958e9 100644 --- a/tests/fast/launch_scripts/model_args_harness.py +++ b/tests/fast/launch_scripts/model_args_harness.py @@ -1,31 +1,14 @@ -import subprocess - from tests.fast.launch_scripts.sh_harness import REPO_ROOT -MODEL_SCRIPT_DIR = REPO_ROOT / "scripts" / "models" +from miles.utils.external_utils.model_args_utils import load_model_args -_ENV_WITHOUT_THE_MODEL_ARGS_KNOBS = { - "PATH": "/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin", - "HOME": "/root", - "LANG": "C", - "LC_ALL": "C", -} +MODEL_SCRIPT_DIR = REPO_ROOT / "scripts" / "models" def iter_model_types() -> list[str]: - return sorted(path.stem for path in MODEL_SCRIPT_DIR.glob("*.sh")) + return sorted(path.stem for path in MODEL_SCRIPT_DIR.glob("*.py")) def expand_model_args(model_type: str) -> list[str]: - """The golden files are taken from this shell expansion; whatever replaces it must reproduce them.""" - script = MODEL_SCRIPT_DIR / f"{model_type}.sh" - result = subprocess.run( - f'source "{script}" && printf "%s\\n" "${{MODEL_ARGS[@]}}"', - shell=True, - executable="/bin/bash", - env=_ENV_WITHOUT_THE_MODEL_ARGS_KNOBS, - capture_output=True, - text=True, - check=True, - ) - return result.stdout.splitlines() + """Only the producer changed here; the golden files still hold what the shell era expanded to.""" + return load_model_args(model_type).split() diff --git a/tests/fast/launch_scripts/py_harness.py b/tests/fast/launch_scripts/py_harness.py index d50baf9e67d..9690989669c 100644 --- a/tests/fast/launch_scripts/py_harness.py +++ b/tests/fast/launch_scripts/py_harness.py @@ -1,10 +1,8 @@ import ast -import importlib.util import inspect import os import re import subprocess -import sys import time from collections.abc import Iterator from contextlib import contextmanager @@ -16,6 +14,7 @@ from tests.fast.utils.command_recorder import record_commands import miles.utils.external_utils.command_utils as command_utils +from miles.utils.external_utils.model_args_utils import import_module_from_path FROZEN_RUN_ID = "260101-000000-000" @@ -126,15 +125,7 @@ def fake_encode_pseudo_file(text: str) -> str: def import_launch_script(path: Path) -> ModuleType: name = "miles_launch_script_" + path.relative_to(REPO_ROOT).with_suffix("").as_posix().replace("/", "_") - spec = importlib.util.spec_from_file_location(name, path) - assert spec is not None and spec.loader is not None - module = importlib.util.module_from_spec(spec) - sys.modules[name] = module - try: - spec.loader.exec_module(module) - finally: - del sys.modules[name] - return module + return import_module_from_path(path, name) @contextmanager diff --git a/tests/fast/launch_scripts/sh_harness.py b/tests/fast/launch_scripts/sh_harness.py index c3069e8f70d..16bbbfbab32 100644 --- a/tests/fast/launch_scripts/sh_harness.py +++ b/tests/fast/launch_scripts/sh_harness.py @@ -66,6 +66,7 @@ } _PYTHON_SHIM_BODY = """case "${1:-}" in +*/model_args_utils.py) "$MILES_SH_HARNESS_REAL_PYTHON" "$@" ;; -c) case "$2" in *cluster_resources*) printf '%s\\n' 'REPLACE_GPU_COUNT' ;; diff --git a/tests/fast/launch_scripts/test_sh_harness.py b/tests/fast/launch_scripts/test_sh_harness.py index c0de9fc1b07..f91a3163b7c 100644 --- a/tests/fast/launch_scripts/test_sh_harness.py +++ b/tests/fast/launch_scripts/test_sh_harness.py @@ -60,7 +60,7 @@ def test_ray_start_is_recorded_with_the_frozen_master_addr(self, run): assert ray_start[ray_start.index("--node-ip-address") + 1] == "127.0.0.1" def test_ray_job_submit_argv_contains_the_expanded_model_args(self, run): - """`source scripts/models/*.sh` expansion must be visible in the captured argv.""" + """The scripts/models/*.py expansion must be visible in the captured argv.""" argv = run.ray_job_submit_argv() assert argv[:3] == ["ray", "job", "submit"] assert "--num-layers" in argv diff --git a/tests/fast/launch_scripts/test_shell_script_hygiene.py b/tests/fast/launch_scripts/test_shell_script_hygiene.py index 22c4502e75f..501012d41d4 100644 --- a/tests/fast/launch_scripts/test_shell_script_hygiene.py +++ b/tests/fast/launch_scripts/test_shell_script_hygiene.py @@ -6,6 +6,10 @@ _REMOVED_COMMAND_HELPERS = re.compile(r"(? None: class TestRunImplExecCommand: """Only mock exec_command_gpu, generate_token_ids, write_token_ids_to_tmpfile, - and resolve_model_script — let the rest (build_worker_args, build_dumper_env, + and load_model_args — let the rest (build_worker_args, build_dumper_env, build_torchrun_cmd, ParallelConfig, WorkerScriptArgs) run for real.""" @pytest.fixture(autouse=True) @@ -58,8 +58,8 @@ def _patch_externals(self) -> Generator[None, None, None]: return_value=Path("/tmp/tokens.json"), ), patch( - "miles.utils.debug_utils.run_megatron.cli.worker_executor.resolve_model_script", - return_value=Path("/repo/scripts/models/deepseek_v3.sh"), + "miles.utils.debug_utils.run_megatron.cli.worker_executor.load_model_args", + return_value="--num-layers 61", ), ): self.mock_exec = mock_exec diff --git a/tests/fast/utils/debug_utils/run_megatron/cli/test_path_utils.py b/tests/fast/utils/debug_utils/run_megatron/cli/test_path_utils.py index 3858d58d24a..7c9790ec972 100644 --- a/tests/fast/utils/debug_utils/run_megatron/cli/test_path_utils.py +++ b/tests/fast/utils/debug_utils/run_megatron/cli/test_path_utils.py @@ -29,7 +29,7 @@ class TestResolveModelScript: def test_returns_path_when_exists(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: scripts_dir = tmp_path / "scripts" / "models" scripts_dir.mkdir(parents=True) - script_file = scripts_dir / "deepseek_v3.sh" + script_file = scripts_dir / "deepseek_v3.py" script_file.touch() monkeypatch.setattr( diff --git a/tests/fast/utils/debug_utils/run_megatron/cli/test_worker_executor.py b/tests/fast/utils/debug_utils/run_megatron/cli/test_worker_executor.py index ea0aa24d8d6..27f4c3929aa 100644 --- a/tests/fast/utils/debug_utils/run_megatron/cli/test_worker_executor.py +++ b/tests/fast/utils/debug_utils/run_megatron/cli/test_worker_executor.py @@ -182,9 +182,9 @@ def test_no_routing_replay(self) -> None: class TestBuildTorchrunCmd: - @patch("miles.utils.debug_utils.run_megatron.cli.worker_executor.resolve_model_script") - def test_basic_structure(self, mock_resolve: object) -> None: - mock_resolve.return_value = Path("/repo/scripts/models/deepseek_v3.sh") # type: ignore[union-attr] + @patch("miles.utils.debug_utils.run_megatron.cli.worker_executor.load_model_args") + def test_basic_structure(self, mock_load: object) -> None: + mock_load.return_value = "--num-layers 61" # type: ignore[union-attr] cmd = build_torchrun_cmd( model_type="deepseek_v3", megatron_path=Path("/megatron"), @@ -192,12 +192,12 @@ def test_basic_structure(self, mock_resolve: object) -> None: worker_args="--foo bar", ) assert "torchrun" in cmd - assert "source" in cmd + assert "--num-layers 61" in cmd assert "PYTHONPATH" in cmd - @patch("miles.utils.debug_utils.run_megatron.cli.worker_executor.resolve_model_script") - def test_nproc(self, mock_resolve: object) -> None: - mock_resolve.return_value = Path("/repo/scripts/models/test.sh") # type: ignore[union-attr] + @patch("miles.utils.debug_utils.run_megatron.cli.worker_executor.load_model_args") + def test_nproc(self, mock_load: object) -> None: + mock_load.return_value = "--num-layers 61" # type: ignore[union-attr] cmd = build_torchrun_cmd( model_type="test", megatron_path=Path("/megatron"), @@ -206,9 +206,9 @@ def test_nproc(self, mock_resolve: object) -> None: ) assert "--nproc-per-node 8" in cmd - @patch("miles.utils.debug_utils.run_megatron.cli.worker_executor.resolve_model_script") - def test_worker_args_in_cmd(self, mock_resolve: object) -> None: - mock_resolve.return_value = Path("/repo/scripts/models/test.sh") # type: ignore[union-attr] + @patch("miles.utils.debug_utils.run_megatron.cli.worker_executor.load_model_args") + def test_worker_args_in_cmd(self, mock_load: object) -> None: + mock_load.return_value = "--num-layers 61" # type: ignore[union-attr] cmd = build_torchrun_cmd( model_type="test", megatron_path=Path("/megatron"), @@ -217,9 +217,9 @@ def test_worker_args_in_cmd(self, mock_resolve: object) -> None: ) assert "--my-flag 42" in cmd - @patch("miles.utils.debug_utils.run_megatron.cli.worker_executor.resolve_model_script") - def test_megatron_in_pythonpath(self, mock_resolve: object) -> None: - mock_resolve.return_value = Path("/repo/scripts/models/test.sh") # type: ignore[union-attr] + @patch("miles.utils.debug_utils.run_megatron.cli.worker_executor.load_model_args") + def test_megatron_in_pythonpath(self, mock_load: object) -> None: + mock_load.return_value = "--num-layers 61" # type: ignore[union-attr] cmd = build_torchrun_cmd( model_type="test", megatron_path=Path("/my/megatron"), diff --git a/tests/fast/utils/external_utils/test_model_args_utils.py b/tests/fast/utils/external_utils/test_model_args_utils.py new file mode 100644 index 00000000000..3b6fece1c6f --- /dev/null +++ b/tests/fast/utils/external_utils/test_model_args_utils.py @@ -0,0 +1,233 @@ +import os +import subprocess +import sys +from pathlib import Path + +import pytest + +import miles.utils.external_utils.model_args_utils as model_args_utils +from miles.utils.external_utils.model_args_utils import load_model_args, load_sibling_model_args, moe_layer_freq + +_MODEL_ARGS_CLI = Path(model_args_utils.__file__).resolve() + +_SCRIPT_BODY = """ +import os + +from model_args_utils import moe_layer_freq + + +def model_args(nlayers: int | None = None) -> str: + nlayers = nlayers if nlayers is not None else int(os.environ.get("MODEL_ARGS_NUM_LAYERS") or 61) + return ( + "--swiglu " + f"--num-layers {nlayers} " + f"--moe-layer-freq {moe_layer_freq(nlayers=nlayers, first_k_dense_replace=3)} " + ) +""" + +_WRAPPER_BODY = """ +from model_args_utils import load_sibling_model_args + + +def model_args() -> str: + return load_sibling_model_args(__file__, "fake-model.4layer", nlayers=4) +""" + + +@pytest.fixture +def model_script(monkeypatch, tmp_path): + path = tmp_path / "fake-model.4layer.py" + path.write_text(_SCRIPT_BODY) + (tmp_path / "fake-wrapper.py").write_text(_WRAPPER_BODY) + monkeypatch.setattr("miles.utils.external_utils.model_args_utils.MODEL_SCRIPT_DIR", tmp_path) + monkeypatch.delenv("MODEL_ARGS_NUM_LAYERS", raising=False) + return path + + +class TestLoadModelArgsSplitting: + def test_splits_each_line_the_way_read_ra_would(self, model_script): + """One source line per flag must expand to the same argv the shell array held.""" + assert load_model_args("fake-model.4layer").split()[:3] == ["--swiglu", "--num-layers", "61"] + + def test_collapses_a_declaration_that_spans_several_lines(self, model_script): + """A newline makes the launcher's `read -ra ... <<< "$(...)"` stop after the first line, silently.""" + model_script.write_text("def model_args() -> str:\n return '--a 1\\n--b 2'\n") + + assert load_model_args("fake-model.4layer") == "--a 1 --b 2" + + def test_rejects_a_model_script_that_declares_nothing(self, model_script): + """An all-whitespace declaration is a generator bug, not an empty argument.""" + model_script.write_text("def model_args() -> str:\n return ' '\n") + + with pytest.raises(AssertionError): + load_model_args("fake-model.4layer") + + def test_keeps_the_bracket_patterns_megatron_expects(self, model_script): + """--moe-layer-freq values contain brackets and stars, which must survive as one token.""" + model_script.write_text("def model_args() -> str:\n return '--moe-layer-freq [0]*3+[1]*75'\n") + + assert load_model_args("fake-model.4layer") == "--moe-layer-freq [0]*3+[1]*75" + + +class TestMoeLayerFreq: + def test_renders_the_dense_prefix_then_moe_layers(self): + """The mask must match `arr+=(0)` for the first K layers and `arr+=(1)` after.""" + assert moe_layer_freq(nlayers=5, first_k_dense_replace=2) == "[0,0,1,1,1]" + + def test_renders_only_dense_layers_when_the_model_is_shorter_than_the_dense_prefix(self): + """The shell loop ran over the layer count, so a 2-layer deepseek-v3 got [0,0], not [0,0,0].""" + assert moe_layer_freq(nlayers=2, first_k_dense_replace=3) == "[0,0]" + + def test_renders_an_all_moe_mask_when_no_dense_layers(self): + """DeepSeek V4 has no dense prefix, so every entry is a MoE layer.""" + assert moe_layer_freq(nlayers=3, first_k_dense_replace=0) == "[1,1,1]" + + +class TestLoadModelArgs: + def test_returns_the_declared_argv(self, model_script): + """A python consumer gets argv tokens directly instead of sourcing a shell script.""" + assert load_model_args("fake-model.4layer").split() == [ + "--swiglu", + "--num-layers", + "61", + "--moe-layer-freq", + moe_layer_freq(nlayers=61, first_k_dense_replace=3), + ] + + def test_forwards_keyword_overrides(self, model_script): + """Layer-count variants are the same script called with a different argument.""" + assert load_model_args("fake-model.4layer", nlayers=4).split()[2] == "4" + + def test_rejects_an_unknown_model_type(self, model_script): + """A typo must fail loudly rather than silently produce an argument-less run.""" + with pytest.raises(AssertionError): + load_model_args("no-such-model") + + def test_a_dotted_filename_is_importable(self, model_script): + """Model names like glm4.5-106B-A12B cannot be imported by module path.""" + assert "." in model_script.stem + assert load_model_args(model_script.stem) + + def test_reads_the_model_scripts_of_the_requested_checkout(self, model_script, tmp_path_factory): + """A launcher must get the model definition of its own checkout, not of the installed package.""" + other = tmp_path_factory.mktemp("other-checkout") + (other / model_script.name).write_text( + _SCRIPT_BODY.replace('"MODEL_ARGS_NUM_LAYERS") or 61', '"UNUSED") or 7') + ) + + assert load_model_args(model_script.stem, model_script_dir=other).split()[2] == "7" + + def test_a_wrapper_stays_inside_the_checkout_it_was_loaded_from(self, model_script, tmp_path_factory): + """A variant script must reach the base script next to it, not the one of the installed package.""" + other = tmp_path_factory.mktemp("other-checkout") + (other / "fake-wrapper.py").write_text(_WRAPPER_BODY) + (other / model_script.name).write_text( + _SCRIPT_BODY.replace('"MODEL_ARGS_NUM_LAYERS") or 61', '"UNUSED") or 7') + ) + + assert load_model_args("fake-wrapper", model_script_dir=other).split() == ["--swiglu", "--num-layers", "4"] + [ + "--moe-layer-freq", + moe_layer_freq(nlayers=4, first_k_dense_replace=3), + ] + + def test_still_honours_the_environment_override_the_shell_scripts_read(self, model_script, monkeypatch): + """MODEL_ARGS_NUM_LAYERS used to reach the sourced .sh, so it must reach the .py too.""" + monkeypatch.setenv("MODEL_ARGS_NUM_LAYERS", "9") + + assert load_model_args("fake-model.4layer").split()[2] == "9" + + def test_ignores_an_empty_environment_override(self, model_script, monkeypatch): + """`${VAR:-default}` falls back to the default when the variable is set but empty.""" + monkeypatch.setenv("MODEL_ARGS_NUM_LAYERS", "") + + assert load_model_args("fake-model.4layer").split()[2] == "61" + + def test_an_explicit_override_beats_the_environment(self, model_script, monkeypatch): + """`MODEL_ARGS_NUM_LAYERS=5 source x.sh` let the caller's assignment win; keyword arguments must too.""" + monkeypatch.setenv("MODEL_ARGS_NUM_LAYERS", "9") + + assert load_model_args("fake-model.4layer", nlayers=4).split()[2] == "4" + + def test_honours_a_zero_override(self, model_script): + """A dense-layer count of zero is a real value; `x or default` would silently restore the default.""" + model_script.write_text(_SCRIPT_BODY.replace("first_k_dense_replace=3", "first_k_dense_replace=nlayers")) + + assert load_model_args("fake-model.4layer", nlayers=0) == "--swiglu --num-layers 0 --moe-layer-freq []" + + def test_ignores_an_environment_override_the_model_does_not_declare(self, model_script, monkeypatch): + """A model without a rotary base must not fail because some other model's variable is exported.""" + monkeypatch.setenv("MODEL_ARGS_ROTARY_BASE", "5000000") + + assert load_model_args("fake-model.4layer").split()[2] == "61" + + def test_rejects_an_override_the_model_does_not_declare(self, model_script): + """The keyword reaches model_args() directly, so a misspelling is a TypeError rather than a silent no-op.""" + with pytest.raises(TypeError): + load_model_args("fake-model.4layer", n_layers=4) + + +class TestLoadSiblingModelArgs: + def test_resolves_the_base_next_to_the_variant_script(self, model_script, tmp_path_factory): + """The variant knows where it lives; nothing else in the process does.""" + other = tmp_path_factory.mktemp("sibling-checkout") + (other / model_script.name).write_text( + _SCRIPT_BODY.replace('"MODEL_ARGS_NUM_LAYERS") or 61', '"UNUSED") or 7') + ) + + assert load_sibling_model_args(str(other / "anything.py"), model_script.stem).split()[2] == "7" + + +class TestModelArgsScript: + def test_shell_consumers_recover_the_original_tokens(self): + """Bracket patterns must survive read -ra without being glob-expanded.""" + script = ( + f'set -e; MODEL_ARGS_LINE="$({sys.executable} {_MODEL_ARGS_CLI} qwen3-4B)" || exit 1; ' + 'read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}"; printf "%s\\n" "${MODEL_ARGS[@]}"' + ) + result = subprocess.run(["bash", "-c", script], capture_output=True, text=True, check=True) + + assert result.stdout.splitlines() == load_model_args("qwen3-4B").split() + + def test_an_unknown_model_type_stops_the_launcher(self): + """A bare `read -ra ... <<< "$(...)"` swallows the failure and trains with no architecture flags.""" + script = ( + f'MODEL_ARGS_LINE="$({sys.executable} {_MODEL_ARGS_CLI} no-such-model 2>/dev/null)" || exit 1; ' + 'echo "the launcher kept going"' + ) + result = subprocess.run(["bash", "-c", script], capture_output=True, text=True) + + assert result.returncode == 1 + assert result.stdout == "" + + def test_a_here_string_read_stops_at_the_first_line(self): + """Why load_model_args() collapses its result: read -ra drops the rest of a multi-line value silently.""" + script = 'read -ra MODEL_ARGS <<< "$1"; printf "%s\\n" "${MODEL_ARGS[@]}"' + result = subprocess.run( + ["bash", "-c", script, "_", "--a 1\n--b 2"], capture_output=True, text=True, check=True + ) + + assert result.stdout.split() == ["--a", "1"] + + def test_runs_from_a_checkout_whose_package_is_not_installed(self): + """Executed by path with no site-packages, a model script must still reach the loader's helpers.""" + result = subprocess.run( + [sys.executable, "-S", "-E", str(_MODEL_ARGS_CLI), "qwen3-30B-A3B"], + capture_output=True, + text=True, + check=True, + cwd="/", + ) + + assert "--num-layers" in result.stdout + + def test_forwards_the_rotary_base_override(self): + """The geo3k launcher prefixes the command with MODEL_ARGS_ROTARY_BASE, as the shell scripts did.""" + result = subprocess.run( + [sys.executable, str(_MODEL_ARGS_CLI), "qwen3-4B"], + capture_output=True, + text=True, + check=True, + env={**os.environ, "MODEL_ARGS_ROTARY_BASE": "5000000"}, + ) + + assert "--rotary-base 5000000" in result.stdout diff --git a/tests/snapshots/launch_scripts/self_executing/examples/experimental/formal_math/single_round/run_minimal.py/import.txt b/tests/snapshots/launch_scripts/self_executing/examples/experimental/formal_math/single_round/run_minimal.py/import.txt index ad94fb21e00..04337367e57 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/experimental/formal_math/single_round/run_minimal.py/import.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/experimental/formal_math/single_round/run_minimal.py/import.txt @@ -1,8 +1,24 @@ ### 0 -bash -c export PYTHONUNBUFFERED=1 && source "/scripts/models/qwen3-8B.sh" && ray job submit +bash -c export PYTHONUNBUFFERED=1 && ray job submit --address="http://127.0.0.1:8265" --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "1"}}' - -- python3 train.py ${MODEL_ARGS[@]} + -- python3 train.py + --swiglu + --num-layers 36 + --hidden-size 4096 + --ffn-hidden-size 12288 + --num-attention-heads 32 + --group-query-attention + --num-query-groups 8 + --use-rotary-position-embeddings + --disable-bias-linear + --normalization RMSNorm + --norm-epsilon 1e-6 + --rotary-base 1000000 + --vocab-size 151936 + --kv-channels 128 + --qk-layernorm + --untie-embeddings-and-output-weights --hf-checkpoint /root/models/Qwen3-8B/ --ref-load /root/models/Qwen3-8B_torch_dist --save-interval 20 diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.5-Air/broadcast.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.5-Air/broadcast.txt index 4a24915d1fb..316ebbc2213 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.5-Air/broadcast.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.5-Air/broadcast.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -MODEL_ARGS_ROTARY_BASE=1000000 source "/scripts/models/glm4.5-106B-A12B.sh" && ray job submit +MODEL_ARGS_LINE="$(MODEL_ARGS_ROTARY_BASE=1000000 python3 "/miles/utils/external_utils/model_args_utils.py" glm4.5-106B-A12B)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "0", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "300", "MODEL_ARGS_ROTARY_BASE": "1000000"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.5-Air/p2p.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.5-Air/p2p.txt index b992f747747..a839b640b95 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.5-Air/p2p.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.5-Air/p2p.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -MODEL_ARGS_ROTARY_BASE=1000000 source "/scripts/models/glm4.5-106B-A12B.sh" && ray job submit +MODEL_ARGS_LINE="$(MODEL_ARGS_ROTARY_BASE=1000000 python3 "/miles/utils/external_utils/model_args_utils.py" glm4.5-106B-A12B)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "0", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "300", "MODEL_ARGS_ROTARY_BASE": "1000000"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.7-Flash/broadcast.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.7-Flash/broadcast.txt index 433f05bf261..c4dec6d382e 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.7-Flash/broadcast.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.7-Flash/broadcast.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -source "/scripts/models/glm4.7-flash.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" glm4.7-flash)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "0", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "300"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.7-Flash/p2p.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.7-Flash/p2p.txt index fd2c92f05f4..0758e1a264e 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.7-Flash/p2p.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-4.7-Flash/p2p.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -source "/scripts/models/glm4.7-flash.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" glm4.7-flash)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "0", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "300"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5/broadcast.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5/broadcast.txt index 878aae279ca..6a1821a5298 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5/broadcast.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5/broadcast.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -source "/scripts/models/glm5-744B-A40B.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" glm5-744B-A40B)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "1", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "600", "INDEXER_ROPE_NEOX_STYLE": "0", "NVSHMEM_DISABLE_NCCL": "1"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5/p2p.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5/p2p.txt index 84ba31d3922..8a9ec08070c 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5/p2p.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5/p2p.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -source "/scripts/models/glm5-744B-A40B.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" glm5-744B-A40B)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "1", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "600", "INDEXER_ROPE_NEOX_STYLE": "0", "NVSHMEM_DISABLE_NCCL": "1"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_20layer/broadcast.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_20layer/broadcast.txt index f2e0345b5d0..66c44a8f745 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_20layer/broadcast.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_20layer/broadcast.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -source "/scripts/models/glm5-744B-A40B_20layer.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" glm5-744B-A40B_20layer)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "1", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "600", "INDEXER_ROPE_NEOX_STYLE": "0", "NVSHMEM_DISABLE_NCCL": "1"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_20layer/p2p.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_20layer/p2p.txt index 060b1c407c5..8985ffe0ed3 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_20layer/p2p.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_20layer/p2p.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -source "/scripts/models/glm5-744B-A40B_20layer.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" glm5-744B-A40B_20layer)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "1", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "600", "INDEXER_ROPE_NEOX_STYLE": "0", "NVSHMEM_DISABLE_NCCL": "1"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_4layer/broadcast.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_4layer/broadcast.txt index 76fdfe325cd..1ee2b15c8e1 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_4layer/broadcast.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_4layer/broadcast.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -source "/scripts/models/glm5-744B-A40B_4layer.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" glm5-744B-A40B_4layer)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "1", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "600", "INDEXER_ROPE_NEOX_STYLE": "0", "NVSHMEM_DISABLE_NCCL": "1"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_4layer/p2p.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_4layer/p2p.txt index d1012211f08..51b11f3c7c5 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_4layer/p2p.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-5_4layer/p2p.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -source "/scripts/models/glm5-744B-A40B_4layer.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" glm5-744B-A40B_4layer)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "1", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "600", "INDEXER_ROPE_NEOX_STYLE": "0", "NVSHMEM_DISABLE_NCCL": "1"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-Z1-9B-0414/broadcast.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-Z1-9B-0414/broadcast.txt index 560206d4205..98d9bec5996 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-Z1-9B-0414/broadcast.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-Z1-9B-0414/broadcast.txt @@ -35,7 +35,7 @@ ray start --dashboard-port=8265 ### 10 -source "/scripts/models/glm4-9B.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" glm4-9B)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "0", "MILES_LOG_DIR": ""}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-Z1-9B-0414/p2p.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-Z1-9B-0414/p2p.txt index cb2a9a61cbc..f05548c663d 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-Z1-9B-0414/p2p.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/GLM-Z1-9B-0414/p2p.txt @@ -35,7 +35,7 @@ ray start --dashboard-port=8265 ### 10 -source "/scripts/models/glm4-9B.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" glm4-9B)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "0", "MILES_LOG_DIR": ""}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Kimi-K2-Instruct/broadcast.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Kimi-K2-Instruct/broadcast.txt index 115505f9e0f..335c01146d1 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Kimi-K2-Instruct/broadcast.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Kimi-K2-Instruct/broadcast.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -source "/scripts/models/kimi-k2.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" kimi-k2)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "1", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "300"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Kimi-K2-Instruct/p2p.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Kimi-K2-Instruct/p2p.txt index 52de614a9da..31061377330 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Kimi-K2-Instruct/p2p.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Kimi-K2-Instruct/p2p.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -source "/scripts/models/kimi-k2.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" kimi-k2)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "1", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "300"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Moonlight-16B-A3B-Instruct/broadcast.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Moonlight-16B-A3B-Instruct/broadcast.txt index ee975b6212c..4b5032c1c73 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Moonlight-16B-A3B-Instruct/broadcast.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Moonlight-16B-A3B-Instruct/broadcast.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -source "/scripts/models/moonlight.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" moonlight)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "0", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "300"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Moonlight-16B-A3B-Instruct/p2p.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Moonlight-16B-A3B-Instruct/p2p.txt index 5886226820a..5fa21e2e528 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Moonlight-16B-A3B-Instruct/p2p.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Moonlight-16B-A3B-Instruct/p2p.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -source "/scripts/models/moonlight.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" moonlight)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "0", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "300"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-235B-A22B-Instruct-2507/broadcast.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-235B-A22B-Instruct-2507/broadcast.txt index 225afb2058d..19721090d74 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-235B-A22B-Instruct-2507/broadcast.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-235B-A22B-Instruct-2507/broadcast.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -MODEL_ARGS_ROTARY_BASE=5000000 source "/scripts/models/qwen3-235B-A22B.sh" && ray job submit +MODEL_ARGS_LINE="$(MODEL_ARGS_ROTARY_BASE=5000000 python3 "/miles/utils/external_utils/model_args_utils.py" qwen3-235B-A22B)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "1", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "300", "MODEL_ARGS_ROTARY_BASE": "5000000"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-235B-A22B-Instruct-2507/p2p.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-235B-A22B-Instruct-2507/p2p.txt index 0a96a138158..274ec400820 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-235B-A22B-Instruct-2507/p2p.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-235B-A22B-Instruct-2507/p2p.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -MODEL_ARGS_ROTARY_BASE=5000000 source "/scripts/models/qwen3-235B-A22B.sh" && ray job submit +MODEL_ARGS_LINE="$(MODEL_ARGS_ROTARY_BASE=5000000 python3 "/miles/utils/external_utils/model_args_utils.py" qwen3-235B-A22B)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "1", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "300", "MODEL_ARGS_ROTARY_BASE": "5000000"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-30B-A3B/broadcast.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-30B-A3B/broadcast.txt index 74ab1db0cac..1bcdc16346d 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-30B-A3B/broadcast.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-30B-A3B/broadcast.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -MODEL_ARGS_ROTARY_BASE=1000000 source "/scripts/models/qwen3-30B-A3B.sh" && ray job submit +MODEL_ARGS_LINE="$(MODEL_ARGS_ROTARY_BASE=1000000 python3 "/miles/utils/external_utils/model_args_utils.py" qwen3-30B-A3B)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "1", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "300", "MODEL_ARGS_ROTARY_BASE": "1000000"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-30B-A3B/p2p.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-30B-A3B/p2p.txt index 259940107b6..a6e57018a74 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-30B-A3B/p2p.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-30B-A3B/p2p.txt @@ -39,7 +39,7 @@ RAY_memory_monitor_refresh_ms=0 ray start python3 -c "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" ### 11 -MODEL_ARGS_ROTARY_BASE=1000000 source "/scripts/models/qwen3-30B-A3B.sh" && ray job submit +MODEL_ARGS_LINE="$(MODEL_ARGS_ROTARY_BASE=1000000 python3 "/miles/utils/external_utils/model_args_utils.py" qwen3-30B-A3B)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "1", "MILES_LOG_DIR": "", "MC_TRANSFER_TIMEOUT": "300", "MODEL_ARGS_ROTARY_BASE": "1000000"}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-4B/broadcast.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-4B/broadcast.txt index 40fc6151875..5a8149a76ff 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-4B/broadcast.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-4B/broadcast.txt @@ -35,7 +35,7 @@ ray start --dashboard-port=8265 ### 10 -source "/scripts/models/qwen3-4B.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" qwen3-4B)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "0", "MILES_LOG_DIR": ""}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-4B/p2p.txt b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-4B/p2p.txt index c89b79aaf14..7480814ef17 100644 --- a/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-4B/p2p.txt +++ b/tests/snapshots/launch_scripts/self_executing/examples/infra_features/p2p_weight_transfer/run.py/run/Qwen3-4B/p2p.txt @@ -35,7 +35,7 @@ ray start --dashboard-port=8265 ### 10 -source "/scripts/models/qwen3-4B.sh" && ray job submit +MODEL_ARGS_LINE="$(python3 "/miles/utils/external_utils/model_args_utils.py" qwen3-4B)" || exit 1; read -ra MODEL_ARGS <<< "${MODEL_ARGS_LINE}" && ray job submit --address='http://127.0.0.1:8265' --runtime-env-json='{"env_vars": {"PYTHONUNBUFFERED": "1", "RAY_DEBUG": "1", "PYTHONPATH": "/root/Megatron-LM/", "CUDA_DEVICE_MAX_CONNECTIONS": "1", "NCCL_NVLS_ENABLE": "0", "MILES_LOG_DIR": ""}}' -- python3 "/train.py" ${MODEL_ARGS[@]} diff --git a/tests/snapshots/launch_scripts/sh/examples/experimental/eval/scripts/run-qwen3-32B.sh.txt b/tests/snapshots/launch_scripts/sh/examples/experimental/eval/scripts/run-qwen3-32B.sh.txt index 4ce3827918b..1b4a112ce30 100644 --- a/tests/snapshots/launch_scripts/sh/examples/experimental/eval/scripts/run-qwen3-32B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/experimental/eval/scripts/run-qwen3-32B.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/miles/utils/external_utils/model_args_utils.py" +"qwen3-32B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/experimental/eval/scripts/run-qwen3-4B.sh.txt b/tests/snapshots/launch_scripts/sh/examples/experimental/eval/scripts/run-qwen3-4B.sh.txt index 24bd9216b60..14392971a53 100644 --- a/tests/snapshots/launch_scripts/sh/examples/experimental/eval/scripts/run-qwen3-4B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/experimental/eval/scripts/run-qwen3-4B.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/miles/utils/external_utils/model_args_utils.py" +"qwen3-4B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/experimental/eval_multi_task/multi_task.sh.txt b/tests/snapshots/launch_scripts/sh/examples/experimental/eval_multi_task/multi_task.sh.txt index 7217c79d1ff..a0a0c938015 100644 --- a/tests/snapshots/launch_scripts/sh/examples/experimental/eval_multi_task/multi_task.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/experimental/eval_multi_task/multi_task.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/miles/utils/external_utils/model_args_utils.py" +"qwen3-4B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/experimental/multi_agent/run-qwen3-30B-A3B-multi-agent.sh.txt b/tests/snapshots/launch_scripts/sh/examples/experimental/multi_agent/run-qwen3-30B-A3B-multi-agent.sh.txt index 797fee6c211..6cc7ba51340 100644 --- a/tests/snapshots/launch_scripts/sh/examples/experimental/multi_agent/run-qwen3-30B-A3B-multi-agent.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/experimental/multi_agent/run-qwen3-30B-A3B-multi-agent.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/examples/experimental/multi_agent/../../../miles/utils/external_utils/model_args_utils.py" +"qwen3-30B-A3B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/experimental/reproducibility/run-qwen2.5-0.5B-gsm8k.sh.txt b/tests/snapshots/launch_scripts/sh/examples/experimental/reproducibility/run-qwen2.5-0.5B-gsm8k.sh.txt index 132b497801b..a10b60aa56a 100644 --- a/tests/snapshots/launch_scripts/sh/examples/experimental/reproducibility/run-qwen2.5-0.5B-gsm8k.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/experimental/reproducibility/run-qwen2.5-0.5B-gsm8k.sh.txt @@ -39,6 +39,11 @@ "python" ### 8 +"python3" +"/examples/experimental/reproducibility/../../../miles/utils/external_utils/model_args_utils.py" +"qwen2.5-0.5B" + +### 9 "ray" "start" "--head" @@ -48,7 +53,7 @@ "8" "--disable-usage-stats" -### 9 +### 10 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/experimental/search-r1/run_qwen2.5_3B.sh.txt b/tests/snapshots/launch_scripts/sh/examples/experimental/search-r1/run_qwen2.5_3B.sh.txt index abbfec1ecad..88f140e55ed 100644 --- a/tests/snapshots/launch_scripts/sh/examples/experimental/search-r1/run_qwen2.5_3B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/experimental/search-r1/run_qwen2.5_3B.sh.txt @@ -39,6 +39,11 @@ "python" ### 8 +"python3" +"/examples/experimental/search-r1/../../../miles/utils/external_utils/model_args_utils.py" +"qwen2.5-3B" + +### 9 "ray" "start" "--head" @@ -48,7 +53,7 @@ "8" "--disable-usage-stats" -### 9 +### 10 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/experimental/strands_sglang/strands_qwen3_8b.sh.txt b/tests/snapshots/launch_scripts/sh/examples/experimental/strands_sglang/strands_qwen3_8b.sh.txt index b7344a73b12..b11aa347375 100644 --- a/tests/snapshots/launch_scripts/sh/examples/experimental/strands_sglang/strands_qwen3_8b.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/experimental/strands_sglang/strands_qwen3_8b.sh.txt @@ -44,10 +44,15 @@ "-m" ### 9 +"python3" +"/examples/experimental/strands_sglang/../../../miles/utils/external_utils/model_args_utils.py" +"qwen3-8B" + +### 10 "date" "+%Y%m%d_%H%M%S" -### 10 +### 11 "ray" "start" "--head" @@ -59,7 +64,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 11 +### 12 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/experimental/tau-bench/run_qwen3_4B.sh.txt b/tests/snapshots/launch_scripts/sh/examples/experimental/tau-bench/run_qwen3_4B.sh.txt index d0fb68d3d7e..51b208db418 100644 --- a/tests/snapshots/launch_scripts/sh/examples/experimental/tau-bench/run_qwen3_4B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/experimental/tau-bench/run_qwen3_4B.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/examples/experimental/tau-bench/../../../miles/utils/external_utils/model_args_utils.py" +"qwen3-4B-Instruct-2507" + +### 10 "ray" "start" "--head" @@ -57,7 +62,7 @@ "--temp-dir" "/root/shared/ray_temp" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/fully_async/run-qwen3-4b-fully_async.sh.txt b/tests/snapshots/launch_scripts/sh/examples/fully_async/run-qwen3-4b-fully_async.sh.txt index 4e35206c448..cf90b6930ee 100644 --- a/tests/snapshots/launch_scripts/sh/examples/fully_async/run-qwen3-4b-fully_async.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/fully_async/run-qwen3-4b-fully_async.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/examples/fully_async/../../miles/utils/external_utils/model_args_utils.py" +"qwen3-4B" + +### 10 "ray" "start" "--head" @@ -53,7 +58,7 @@ "8" "--disable-usage-stats" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/geo3k_vlm/run_geo3k_vlm.sh.txt b/tests/snapshots/launch_scripts/sh/examples/geo3k_vlm/run_geo3k_vlm.sh.txt index e00be714548..dd9a072c6b1 100644 --- a/tests/snapshots/launch_scripts/sh/examples/geo3k_vlm/run_geo3k_vlm.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/geo3k_vlm/run_geo3k_vlm.sh.txt @@ -71,6 +71,11 @@ "/root/datasets/geo3k_imgurl" ### 13 +"python3" +"/miles/utils/external_utils/model_args_utils.py" +"qwen3-8B" + +### 14 "ray" "start" "--head" @@ -82,7 +87,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 14 +### 15 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/geo3k_vlm/run_geo3k_vlm_sft.sh.txt b/tests/snapshots/launch_scripts/sh/examples/geo3k_vlm/run_geo3k_vlm_sft.sh.txt index a2355b5e92e..55d09629604 100644 --- a/tests/snapshots/launch_scripts/sh/examples/geo3k_vlm/run_geo3k_vlm_sft.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/geo3k_vlm/run_geo3k_vlm_sft.sh.txt @@ -71,6 +71,11 @@ "/root/datasets/geo3k_imgurl" ### 13 +"python3" +"/miles/utils/external_utils/model_args_utils.py" +"qwen3-8B" + +### 14 "ray" "start" "--head" @@ -82,7 +87,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 14 +### 15 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-kimi-k2-Thinking-int4.sh.txt b/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-kimi-k2-Thinking-int4.sh.txt index c32efbf09fd..cf1878c66a6 100644 --- a/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-kimi-k2-Thinking-int4.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-kimi-k2-Thinking-int4.sh.txt @@ -42,6 +42,11 @@ "nvidia-smi" ### 9 +"python3" +"/examples/infra_features/low_precision/../../../miles/utils/external_utils/model_args_utils.py" +"kimi-k2-thinking" + +### 10 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-moonlight-16B-A3B-int4.sh.txt b/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-moonlight-16B-A3B-int4.sh.txt index 3c647619e0a..6df380bb002 100644 --- a/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-moonlight-16B-A3B-int4.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-moonlight-16B-A3B-int4.sh.txt @@ -49,6 +49,11 @@ "-m" ### 10 +"python3" +"/examples/infra_features/low_precision/../../../miles/utils/external_utils/model_args_utils.py" +"moonlight" + +### 11 "ray" "start" "--head" @@ -60,7 +65,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 11 +### 12 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-235B-A22B-int4.sh.txt b/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-235B-A22B-int4.sh.txt index c5f34c7d39e..1b05aa7262f 100644 --- a/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-235B-A22B-int4.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-235B-A22B-int4.sh.txt @@ -42,6 +42,11 @@ "nvidia-smi" ### 9 +"python3" +"/examples/infra_features/low_precision/../../../miles/utils/external_utils/model_args_utils.py" +"qwen3-235B-A22B" + +### 10 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-30B-A3B-int4.sh.txt b/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-30B-A3B-int4.sh.txt index d0b98417729..62fe888f50f 100644 --- a/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-30B-A3B-int4.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-30B-A3B-int4.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/examples/infra_features/low_precision/../../../miles/utils/external_utils/model_args_utils.py" +"qwen3-30B-A3B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-30b-a3b-fp8-two-nodes.sh.txt b/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-30b-a3b-fp8-two-nodes.sh.txt index a89c5e24a69..7506aab4744 100644 --- a/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-30b-a3b-fp8-two-nodes.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-30b-a3b-fp8-two-nodes.sh.txt @@ -6,14 +6,19 @@ "-m" ### 1 -"ps" -"aux" +"python3" +"/examples/infra_features/low_precision/../../../miles/utils/external_utils/model_args_utils.py" +"qwen3-30B-A3B" ### 2 "ps" "aux" ### 3 +"ps" +"aux" + +### 4 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-4b-fp8.sh.txt b/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-4b-fp8.sh.txt index 3580be18de6..3877f48f96c 100644 --- a/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-4b-fp8.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/infra_features/low_precision/run-qwen3-4b-fp8.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/examples/infra_features/low_precision/../../../miles/utils/external_utils/model_args_utils.py" +"qwen3-4B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-glm4.5-air-8node-profile.sh.txt b/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-glm4.5-air-8node-profile.sh.txt index 433a99f4522..a2f17d691da 100644 --- a/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-glm4.5-air-8node-profile.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-glm4.5-air-8node-profile.sh.txt @@ -45,10 +45,15 @@ ### 9 "python3" +"/miles/utils/external_utils/model_args_utils.py" +"glm4.5-106B-A12B" + +### 10 +"python3" "-c" "print(int(1.0 * 1024 * 1024 * 1024))" -### 10 +### 11 "ray" "start" "--head" @@ -60,22 +65,22 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 11 +### 12 "python3" "-c" "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" -### 12 +### 13 "mkdir" "-p" "/workdir" -### 13 +### 14 "rm" "-f" "/workdir/job_done_p2p" -### 14 +### 15 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-glm4.7-flash-2node-profile.sh.txt b/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-glm4.7-flash-2node-profile.sh.txt index 2fcae95126e..033bacb2cc7 100644 --- a/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-glm4.7-flash-2node-profile.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-glm4.7-flash-2node-profile.sh.txt @@ -49,6 +49,11 @@ "-m" ### 10 +"python3" +"/miles/utils/external_utils/model_args_utils.py" +"glm4.7-flash" + +### 11 "ray" "start" "--head" @@ -60,12 +65,12 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 11 +### 12 "python3" "-c" "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" -### 12 +### 13 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-glm5-disagg-profile.sh.txt b/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-glm5-disagg-profile.sh.txt index 03d6e0dd41c..23363db1791 100644 --- a/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-glm5-disagg-profile.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-glm5-disagg-profile.sh.txt @@ -44,6 +44,11 @@ "redis" ### 9 +"python3" +"/miles/utils/external_utils/model_args_utils.py" +"glm5-744B-A40B" + +### 10 "ray" "start" "--head" @@ -55,22 +60,22 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "python3" "-c" "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" -### 11 +### 12 "mkdir" "-p" "/workdir" -### 12 +### 13 "rm" "-f" "/workdir/job_done_p2p" -### 13 +### 14 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-kimi-k2-64node-profile.sh.txt b/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-kimi-k2-64node-profile.sh.txt index b53d6ae242a..724c1f63b9c 100644 --- a/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-kimi-k2-64node-profile.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-kimi-k2-64node-profile.sh.txt @@ -49,6 +49,11 @@ "-m" ### 10 +"python3" +"/miles/utils/external_utils/model_args_utils.py" +"kimi-k2" + +### 11 "ray" "start" "--head" @@ -60,12 +65,12 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 11 +### 12 "python3" "-c" "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" -### 12 +### 13 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-qwen3-235B-A22B-16node-profile.sh.txt b/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-qwen3-235B-A22B-16node-profile.sh.txt index 6576a46dad4..36d913e8721 100644 --- a/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-qwen3-235B-A22B-16node-profile.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-qwen3-235B-A22B-16node-profile.sh.txt @@ -50,10 +50,15 @@ ### 10 "python3" +"/miles/utils/external_utils/model_args_utils.py" +"qwen3-235B-A22B" + +### 11 +"python3" "-c" "print(int(1.0 * 1024 * 1024 * 1024))" -### 11 +### 12 "ray" "start" "--head" @@ -65,12 +70,12 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 12 +### 13 "python3" "-c" "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" -### 13 +### 14 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-qwen3-30B-A3B-4node-profile.sh.txt b/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-qwen3-30B-A3B-4node-profile.sh.txt index d04fe971f32..42937904913 100644 --- a/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-qwen3-30B-A3B-4node-profile.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/infra_features/p2p_weight_transfer/run-qwen3-30B-A3B-4node-profile.sh.txt @@ -49,6 +49,11 @@ "-m" ### 10 +"python3" +"/miles/utils/external_utils/model_args_utils.py" +"qwen3-30B-A3B" + +### 11 "ray" "start" "--head" @@ -60,12 +65,12 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 11 +### 12 "python3" "-c" "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" -### 12 +### 13 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/infra_features/train_infer_mismatch_helper/run-qwen3-4b-mis.sh.txt b/tests/snapshots/launch_scripts/sh/examples/infra_features/train_infer_mismatch_helper/run-qwen3-4b-mis.sh.txt index 5d0cf0562d9..0b47754d4c5 100644 --- a/tests/snapshots/launch_scripts/sh/examples/infra_features/train_infer_mismatch_helper/run-qwen3-4b-mis.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/infra_features/train_infer_mismatch_helper/run-qwen3-4b-mis.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/examples/infra_features/train_infer_mismatch_helper/../../../miles/utils/external_utils/model_args_utils.py" +"qwen3-4B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/lora/dev.sh.txt b/tests/snapshots/launch_scripts/sh/examples/lora/dev.sh.txt index dda91e58a86..d23df4cf680 100644 --- a/tests/snapshots/launch_scripts/sh/examples/lora/dev.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/lora/dev.sh.txt @@ -29,6 +29,11 @@ "python" ### 6 +"python3" +"/examples/lora/../../miles/utils/external_utils/model_args_utils.py" +"qwen2.5-3B" + +### 7 "ray" "start" "--head" @@ -38,7 +43,7 @@ "1" "--disable-usage-stats" -### 7 +### 8 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/lora/run-gpt-oss-20B-megatron-moe-lora.sh.txt b/tests/snapshots/launch_scripts/sh/examples/lora/run-gpt-oss-20B-megatron-moe-lora.sh.txt index a1025b6d466..cfc63b109a2 100644 --- a/tests/snapshots/launch_scripts/sh/examples/lora/run-gpt-oss-20B-megatron-moe-lora.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/lora/run-gpt-oss-20B-megatron-moe-lora.sh.txt @@ -39,6 +39,11 @@ "python" ### 8 +"python3" +"/examples/lora/../../miles/utils/external_utils/model_args_utils.py" +"gpt-oss-20b" + +### 9 "ray" "start" "--head" @@ -50,7 +55,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 9 +### 10 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/lora/run-kimi-k25-megatron-lora.sh.txt b/tests/snapshots/launch_scripts/sh/examples/lora/run-kimi-k25-megatron-lora.sh.txt index 48d1e0ee363..a92c64589ef 100644 --- a/tests/snapshots/launch_scripts/sh/examples/lora/run-kimi-k25-megatron-lora.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/lora/run-kimi-k25-megatron-lora.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/examples/lora/../../miles/utils/external_utils/model_args_utils.py" +"kimi-k2-thinking" + +### 10 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen2.5-0.5B-megatron-lora.sh.txt b/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen2.5-0.5B-megatron-lora.sh.txt index 87c29860d42..8ffdd1560aa 100644 --- a/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen2.5-0.5B-megatron-lora.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen2.5-0.5B-megatron-lora.sh.txt @@ -29,6 +29,11 @@ "python" ### 6 +"python3" +"/examples/lora/../../miles/utils/external_utils/model_args_utils.py" +"qwen2.5-0.5B" + +### 7 "ray" "start" "--head" @@ -38,7 +43,7 @@ "8" "--disable-usage-stats" -### 7 +### 8 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated-multi-node.sh.txt b/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated-multi-node.sh.txt index a0c726c5143..69deec869e8 100644 --- a/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated-multi-node.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated-multi-node.sh.txt @@ -53,6 +53,11 @@ "python" ### 10 +"python3" +"/examples/lora/../../miles/utils/external_utils/model_args_utils.py" +"qwen2.5-3B" + +### 11 "ray" "start" "--head" @@ -64,12 +69,12 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 11 +### 12 "python3" "-c" "import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()" -### 12 +### 13 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated.sh.txt b/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated.sh.txt index f6c62b12adc..ef8dee4b7ae 100644 --- a/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen2.5-3B-megatron-lora-disaggregated.sh.txt @@ -29,6 +29,11 @@ "python" ### 6 +"python3" +"/examples/lora/../../miles/utils/external_utils/model_args_utils.py" +"qwen2.5-3B" + +### 7 "ray" "start" "--head" @@ -38,7 +43,7 @@ "2" "--disable-usage-stats" -### 7 +### 8 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen3-4B-megatron-lora.sh.txt b/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen3-4B-megatron-lora.sh.txt index e70fe24e1c3..6864acd787d 100644 --- a/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen3-4B-megatron-lora.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen3-4B-megatron-lora.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/miles/utils/external_utils/model_args_utils.py" +"qwen3-4B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen3-4b-megatron-lora-result.sh.txt b/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen3-4b-megatron-lora-result.sh.txt index c7444461f45..82a54662016 100644 --- a/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen3-4b-megatron-lora-result.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/lora/run-qwen3-4b-megatron-lora-result.sh.txt @@ -42,6 +42,11 @@ "nvidia-smi" ### 9 +"python3" +"/examples/lora/../../miles/utils/external_utils/model_args_utils.py" +"qwen3-4B" + +### 10 "ray" "start" "--head" @@ -51,7 +56,7 @@ "4" "--disable-usage-stats" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/examples/on_policy_distillation/run-qwen3-8B-opd-megatron.sh.txt b/tests/snapshots/launch_scripts/sh/examples/on_policy_distillation/run-qwen3-8B-opd-megatron.sh.txt index d4a648418ab..d746325862c 100644 --- a/tests/snapshots/launch_scripts/sh/examples/on_policy_distillation/run-qwen3-8B-opd-megatron.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/on_policy_distillation/run-qwen3-8B-opd-megatron.sh.txt @@ -6,6 +6,11 @@ "-m" ### 1 +"python3" +"/examples/on_policy_distillation/../../miles/utils/external_utils/model_args_utils.py" +"qwen3-8B" + +### 2 "ray" "start" "--head" @@ -17,7 +22,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 2 +### 3 "ray" "job" "submit" @@ -150,40 +155,40 @@ "--rm-type" "math" -### 3 +### 4 "pkill" "-9" "sglang" -### 4 +### 5 "sleep" "3" -### 5 +### 6 "ray" "stop" "--force" -### 6 +### 7 "pkill" "-9" "ray" -### 7 +### 8 "pkill" "-9" "python" -### 8 +### 9 "sleep" "3" -### 9 +### 10 "pkill" "-9" "ray" -### 10 +### 11 "pkill" "-9" "python" diff --git a/tests/snapshots/launch_scripts/sh/examples/on_policy_distillation/run-qwen3-8B-opd-multi-teacher.sh.txt b/tests/snapshots/launch_scripts/sh/examples/on_policy_distillation/run-qwen3-8B-opd-multi-teacher.sh.txt index 590060a19f6..2ee5f94badc 100644 --- a/tests/snapshots/launch_scripts/sh/examples/on_policy_distillation/run-qwen3-8B-opd-multi-teacher.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/on_policy_distillation/run-qwen3-8B-opd-multi-teacher.sh.txt @@ -71,6 +71,11 @@ "-m" ### 10 +"python3" +"/examples/on_policy_distillation/../../miles/utils/external_utils/model_args_utils.py" +"qwen3-8B" + +### 11 "ray" "start" "--head" @@ -82,7 +87,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 11 +### 12 "ray" "job" "submit" @@ -227,40 +232,40 @@ "--opd-teacher-key" "opd_teacher" -### 12 +### 13 "pkill" "-9" "sglang" -### 13 +### 14 "sleep" "3" -### 14 +### 15 "ray" "stop" "--force" -### 15 +### 16 "pkill" "-9" "ray" -### 16 +### 17 "pkill" "-9" "python" -### 17 +### 18 "sleep" "3" -### 18 +### 19 "pkill" "-9" "ray" -### 19 +### 20 "pkill" "-9" "python" diff --git a/tests/snapshots/launch_scripts/sh/examples/on_policy_distillation/run-qwen3-8B-opd.sh.txt b/tests/snapshots/launch_scripts/sh/examples/on_policy_distillation/run-qwen3-8B-opd.sh.txt index 7e32ebd59d6..33b048a20a4 100644 --- a/tests/snapshots/launch_scripts/sh/examples/on_policy_distillation/run-qwen3-8B-opd.sh.txt +++ b/tests/snapshots/launch_scripts/sh/examples/on_policy_distillation/run-qwen3-8B-opd.sh.txt @@ -36,6 +36,11 @@ "-m" ### 5 +"python3" +"/examples/on_policy_distillation/../../miles/utils/external_utils/model_args_utils.py" +"qwen3-8B" + +### 6 "ray" "start" "--head" @@ -47,7 +52,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 6 +### 7 "ray" "job" "submit" @@ -188,40 +193,40 @@ "--rm-url" "http://127.0.0.1:13141/generate" -### 7 +### 8 "pkill" "-9" "sglang" -### 8 +### 9 "sleep" "3" -### 9 +### 10 "ray" "stop" "--force" -### 10 +### 11 "pkill" "-9" "ray" -### 11 +### 12 "pkill" "-9" "python" -### 12 +### 13 "sleep" "3" -### 13 +### 14 "pkill" "-9" "ray" -### 14 +### 15 "pkill" "-9" "python" diff --git a/tests/snapshots/launch_scripts/sh/scripts/amd/run-qwen3-4B-amd.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/amd/run-qwen3-4B-amd.sh.txt index 05e1b71c70a..553ca0e6315 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/amd/run-qwen3-4B-amd.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/amd/run-qwen3-4B-amd.sh.txt @@ -39,6 +39,11 @@ "python" ### 8 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3-4B" + +### 9 "ray" "start" "--head" @@ -50,7 +55,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 9 +### 10 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-deepseek-r1.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-deepseek-r1.sh.txt index c3bf834fc9e..ee964dfc2f3 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-deepseek-r1.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-deepseek-r1.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"deepseek-v3" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-glm4-9B-4xgpu-radixtree.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-glm4-9B-4xgpu-radixtree.sh.txt index 9e5b0b4f7f0..a6e70ac09f7 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-glm4-9B-4xgpu-radixtree.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-glm4-9B-4xgpu-radixtree.sh.txt @@ -42,6 +42,11 @@ "nvidia-smi" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"glm4-9B" + +### 10 "ray" "start" "--head" @@ -53,7 +58,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-glm4-9B.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-glm4-9B.sh.txt index 26f5eddcd26..36fe469186c 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-glm4-9B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-glm4-9B.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"glm4-9B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-glm4.5-355B-A32B.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-glm4.5-355B-A32B.sh.txt index fb0cb552353..c7af53a0b7c 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-glm4.5-355B-A32B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-glm4.5-355B-A32B.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"glm4.5-355B-A32B" + +### 10 "ray" "start" "--head" @@ -54,12 +59,12 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "awk" "{print $1}" "/root/mpi_rack_hostfile" -### 11 +### 12 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-glm4.7-flash.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-glm4.7-flash.sh.txt index b1fd672585e..9262d58117d 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-glm4.7-flash.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-glm4.7-flash.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"glm4.7-flash" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-gpt-oss-20b-bf16.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-gpt-oss-20b-bf16.sh.txt index 577a4b430c2..093da50e7a6 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-gpt-oss-20b-bf16.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-gpt-oss-20b-bf16.sh.txt @@ -39,6 +39,11 @@ "python" ### 8 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"gpt-oss-20b" + +### 9 "ray" "start" "--head" @@ -50,7 +55,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 9 +### 10 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-kimi-k2-Instruct.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-kimi-k2-Instruct.sh.txt index 66dad480f75..11a38513a36 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-kimi-k2-Instruct.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-kimi-k2-Instruct.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"kimi-k2" + +### 10 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-kimi-k2-Thinking.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-kimi-k2-Thinking.sh.txt index 88a8cd890fc..fc6e3a6def9 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-kimi-k2-Thinking.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-kimi-k2-Thinking.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"kimi-k2-thinking" + +### 10 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-kimi-k25.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-kimi-k25.sh.txt index 909108edc3b..47cb8d8e819 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-kimi-k25.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-kimi-k25.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"kimi-k2-thinking" + +### 10 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-mimo-7B-rl-eagle.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-mimo-7B-rl-eagle.sh.txt index b8b62f49951..9fa1648672f 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-mimo-7B-rl-eagle.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-mimo-7B-rl-eagle.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"mimo-7B-rl" + +### 10 "ray" "start" "--head" @@ -53,7 +58,7 @@ "8" "--disable-usage-stats" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-moonlight-16B-A3B.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-moonlight-16B-A3B.sh.txt index b71b7296deb..278c7479a23 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-moonlight-16B-A3B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-moonlight-16B-A3B.sh.txt @@ -49,6 +49,11 @@ "-m" ### 10 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"moonlight" + +### 11 "ray" "start" "--head" @@ -60,7 +65,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 11 +### 12 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-nemotron-3-nano-30b-a3b.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-nemotron-3-nano-30b-a3b.sh.txt index 3e128f7144b..fa7ca0d683d 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-nemotron-3-nano-30b-a3b.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-nemotron-3-nano-30b-a3b.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"nemotron-3-nano-30b-a3b" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-nemotron-3-nano-4b.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-nemotron-3-nano-4b.sh.txt index 25060e59960..908fb643bb9 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-nemotron-3-nano-4b.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-nemotron-3-nano-4b.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"nemotron-3-nano-4b" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-nemotron-3-super-120b-a12b.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-nemotron-3-super-120b-a12b.sh.txt index 97010455b1c..d19fef80bc5 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-nemotron-3-super-120b-a12b.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-nemotron-3-super-120b-a12b.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"nemotron-3-super-120b-a12b" + +### 10 "ray" "start" "--head" @@ -55,971 +60,971 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "status" -### 11 +### 12 "sleep" "5" -### 12 +### 13 "ray" "status" -### 13 +### 14 "sleep" "5" -### 14 +### 15 "ray" "status" -### 15 +### 16 "sleep" "5" -### 16 +### 17 "ray" "status" -### 17 +### 18 "sleep" "5" -### 18 +### 19 "ray" "status" -### 19 +### 20 "sleep" "5" -### 20 +### 21 "ray" "status" -### 21 +### 22 "sleep" "5" -### 22 +### 23 "ray" "status" -### 23 +### 24 "sleep" "5" -### 24 +### 25 "ray" "status" -### 25 +### 26 "sleep" "5" -### 26 +### 27 "ray" "status" -### 27 +### 28 "sleep" "5" -### 28 +### 29 "ray" "status" -### 29 +### 30 "sleep" "5" -### 30 +### 31 "ray" "status" -### 31 +### 32 "sleep" "5" -### 32 +### 33 "ray" "status" -### 33 +### 34 "sleep" "5" -### 34 +### 35 "ray" "status" -### 35 +### 36 "sleep" "5" -### 36 +### 37 "ray" "status" -### 37 +### 38 "sleep" "5" -### 38 +### 39 "ray" "status" -### 39 +### 40 "sleep" "5" -### 40 +### 41 "ray" "status" -### 41 +### 42 "sleep" "5" -### 42 +### 43 "ray" "status" -### 43 +### 44 "sleep" "5" -### 44 +### 45 "ray" "status" -### 45 +### 46 "sleep" "5" -### 46 +### 47 "ray" "status" -### 47 +### 48 "sleep" "5" -### 48 +### 49 "ray" "status" -### 49 +### 50 "sleep" "5" -### 50 +### 51 "ray" "status" -### 51 +### 52 "sleep" "5" -### 52 +### 53 "ray" "status" -### 53 +### 54 "sleep" "5" -### 54 +### 55 "ray" "status" -### 55 +### 56 "sleep" "5" -### 56 +### 57 "ray" "status" -### 57 +### 58 "sleep" "5" -### 58 +### 59 "ray" "status" -### 59 +### 60 "sleep" "5" -### 60 +### 61 "ray" "status" -### 61 +### 62 "sleep" "5" -### 62 +### 63 "ray" "status" -### 63 +### 64 "sleep" "5" -### 64 +### 65 "ray" "status" -### 65 +### 66 "sleep" "5" -### 66 +### 67 "ray" "status" -### 67 +### 68 "sleep" "5" -### 68 +### 69 "ray" "status" -### 69 +### 70 "sleep" "5" -### 70 +### 71 "ray" "status" -### 71 +### 72 "sleep" "5" -### 72 +### 73 "ray" "status" -### 73 +### 74 "sleep" "5" -### 74 +### 75 "ray" "status" -### 75 +### 76 "sleep" "5" -### 76 +### 77 "ray" "status" -### 77 +### 78 "sleep" "5" -### 78 +### 79 "ray" "status" -### 79 +### 80 "sleep" "5" -### 80 +### 81 "ray" "status" -### 81 +### 82 "sleep" "5" -### 82 +### 83 "ray" "status" -### 83 +### 84 "sleep" "5" -### 84 +### 85 "ray" "status" -### 85 +### 86 "sleep" "5" -### 86 +### 87 "ray" "status" -### 87 +### 88 "sleep" "5" -### 88 +### 89 "ray" "status" -### 89 +### 90 "sleep" "5" -### 90 +### 91 "ray" "status" -### 91 +### 92 "sleep" "5" -### 92 +### 93 "ray" "status" -### 93 +### 94 "sleep" "5" -### 94 +### 95 "ray" "status" -### 95 +### 96 "sleep" "5" -### 96 +### 97 "ray" "status" -### 97 +### 98 "sleep" "5" -### 98 +### 99 "ray" "status" -### 99 +### 100 "sleep" "5" -### 100 +### 101 "ray" "status" -### 101 +### 102 "sleep" "5" -### 102 +### 103 "ray" "status" -### 103 +### 104 "sleep" "5" -### 104 +### 105 "ray" "status" -### 105 +### 106 "sleep" "5" -### 106 +### 107 "ray" "status" -### 107 +### 108 "sleep" "5" -### 108 +### 109 "ray" "status" -### 109 +### 110 "sleep" "5" -### 110 +### 111 "ray" "status" -### 111 +### 112 "sleep" "5" -### 112 +### 113 "ray" "status" -### 113 +### 114 "sleep" "5" -### 114 +### 115 "ray" "status" -### 115 +### 116 "sleep" "5" -### 116 +### 117 "ray" "status" -### 117 +### 118 "sleep" "5" -### 118 +### 119 "ray" "status" -### 119 +### 120 "sleep" "5" -### 120 +### 121 "ray" "status" -### 121 +### 122 "sleep" "5" -### 122 +### 123 "ray" "status" -### 123 +### 124 "sleep" "5" -### 124 +### 125 "ray" "status" -### 125 +### 126 "sleep" "5" -### 126 +### 127 "ray" "status" -### 127 +### 128 "sleep" "5" -### 128 +### 129 "ray" "status" -### 129 +### 130 "sleep" "5" -### 130 +### 131 "ray" "status" -### 131 +### 132 "sleep" "5" -### 132 +### 133 "ray" "status" -### 133 +### 134 "sleep" "5" -### 134 +### 135 "ray" "status" -### 135 +### 136 "sleep" "5" -### 136 +### 137 "ray" "status" -### 137 +### 138 "sleep" "5" -### 138 +### 139 "ray" "status" -### 139 +### 140 "sleep" "5" -### 140 +### 141 "ray" "status" -### 141 +### 142 "sleep" "5" -### 142 +### 143 "ray" "status" -### 143 +### 144 "sleep" "5" -### 144 +### 145 "ray" "status" -### 145 +### 146 "sleep" "5" -### 146 +### 147 "ray" "status" -### 147 +### 148 "sleep" "5" -### 148 +### 149 "ray" "status" -### 149 +### 150 "sleep" "5" -### 150 +### 151 "ray" "status" -### 151 +### 152 "sleep" "5" -### 152 +### 153 "ray" "status" -### 153 +### 154 "sleep" "5" -### 154 +### 155 "ray" "status" -### 155 +### 156 "sleep" "5" -### 156 +### 157 "ray" "status" -### 157 +### 158 "sleep" "5" -### 158 +### 159 "ray" "status" -### 159 +### 160 "sleep" "5" -### 160 +### 161 "ray" "status" -### 161 +### 162 "sleep" "5" -### 162 +### 163 "ray" "status" -### 163 +### 164 "sleep" "5" -### 164 +### 165 "ray" "status" -### 165 +### 166 "sleep" "5" -### 166 +### 167 "ray" "status" -### 167 +### 168 "sleep" "5" -### 168 +### 169 "ray" "status" -### 169 +### 170 "sleep" "5" -### 170 +### 171 "ray" "status" -### 171 +### 172 "sleep" "5" -### 172 +### 173 "ray" "status" -### 173 +### 174 "sleep" "5" -### 174 +### 175 "ray" "status" -### 175 +### 176 "sleep" "5" -### 176 +### 177 "ray" "status" -### 177 +### 178 "sleep" "5" -### 178 +### 179 "ray" "status" -### 179 +### 180 "sleep" "5" -### 180 +### 181 "ray" "status" -### 181 +### 182 "sleep" "5" -### 182 +### 183 "ray" "status" -### 183 +### 184 "sleep" "5" -### 184 +### 185 "ray" "status" -### 185 +### 186 "sleep" "5" -### 186 +### 187 "ray" "status" -### 187 +### 188 "sleep" "5" -### 188 +### 189 "ray" "status" -### 189 +### 190 "sleep" "5" -### 190 +### 191 "ray" "status" -### 191 +### 192 "sleep" "5" -### 192 +### 193 "ray" "status" -### 193 +### 194 "sleep" "5" -### 194 +### 195 "ray" "status" -### 195 +### 196 "sleep" "5" -### 196 +### 197 "ray" "status" -### 197 +### 198 "sleep" "5" -### 198 +### 199 "ray" "status" -### 199 +### 200 "sleep" "5" -### 200 +### 201 "ray" "status" -### 201 +### 202 "sleep" "5" -### 202 +### 203 "ray" "status" -### 203 +### 204 "sleep" "5" -### 204 +### 205 "ray" "status" -### 205 +### 206 "sleep" "5" -### 206 +### 207 "ray" "status" -### 207 +### 208 "sleep" "5" -### 208 +### 209 "ray" "status" -### 209 +### 210 "sleep" "5" -### 210 +### 211 "ray" "status" -### 211 +### 212 "sleep" "5" -### 212 +### 213 "ray" "status" -### 213 +### 214 "sleep" "5" -### 214 +### 215 "ray" "status" -### 215 +### 216 "sleep" "5" -### 216 +### 217 "ray" "status" -### 217 +### 218 "sleep" "5" -### 218 +### 219 "ray" "status" -### 219 +### 220 "sleep" "5" -### 220 +### 221 "ray" "status" -### 221 +### 222 "sleep" "5" -### 222 +### 223 "ray" "status" -### 223 +### 224 "sleep" "5" -### 224 +### 225 "ray" "status" -### 225 +### 226 "sleep" "5" -### 226 +### 227 "ray" "status" -### 227 +### 228 "sleep" "5" -### 228 +### 229 "ray" "status" -### 229 +### 230 "sleep" "5" -### 230 +### 231 "ray" "status" -### 231 +### 232 "sleep" "5" -### 232 +### 233 "ray" "status" -### 233 +### 234 "sleep" "5" -### 234 +### 235 "ray" "status" -### 235 +### 236 "sleep" "5" -### 236 +### 237 "ray" "status" -### 237 +### 238 "sleep" "5" -### 238 +### 239 "ray" "status" -### 239 +### 240 "sleep" "5" -### 240 +### 241 "ray" "status" -### 241 +### 242 "sleep" "5" -### 242 +### 243 "ray" "status" -### 243 +### 244 "sleep" "5" -### 244 +### 245 "ray" "status" -### 245 +### 246 "sleep" "5" -### 246 +### 247 "ray" "status" -### 247 +### 248 "sleep" "5" -### 248 +### 249 "ray" "status" -### 249 +### 250 "sleep" "5" -### 250 +### 251 "ray" "status" -### 251 +### 252 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-235B-A22B-sft.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-235B-A22B-sft.sh.txt index e91727a97db..3915b2df14f 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-235B-A22B-sft.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-235B-A22B-sft.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3-235B-A22B" + +### 10 "ray" "start" "--head" @@ -55,12 +60,12 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "awk" "{print $1}" "/root/mpi_rack_hostfile" -### 11 +### 12 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-235B-A22B.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-235B-A22B.sh.txt index 2387b134bae..801704f8d08 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-235B-A22B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-235B-A22B.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3-235B-A22B" + +### 10 "ray" "start" "--head" @@ -55,12 +60,12 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "awk" "{print $1}" "/root/mpi_rack_hostfile" -### 11 +### 12 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-32B.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-32B.sh.txt index 9fab9236ead..8adb2d846f7 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-32B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-32B.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3-32B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-4B-base-sft.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-4B-base-sft.sh.txt index 3563312a8c8..de7ee3f9dc6 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-4B-base-sft.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-4B-base-sft.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3-4B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-4B.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-4B.sh.txt index d4c1bb37af7..62832339fc2 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-4B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-4B.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3-4B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-4B_4xgpu.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-4B_4xgpu.sh.txt index 5aca51b031f..0ac3ee9d6ad 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-4B_4xgpu.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-4B_4xgpu.sh.txt @@ -42,6 +42,11 @@ "nvidia-smi" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3-4B" + +### 10 "ray" "start" "--head" @@ -53,7 +58,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-next-80B-A3B-8gpus.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-next-80B-A3B-8gpus.sh.txt index fa31c876848..48497e0a2f3 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-next-80B-A3B-8gpus.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-next-80B-A3B-8gpus.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3-next-80B-A3B" + +### 10 "ray" "start" "--head" @@ -55,12 +60,12 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "awk" "{print $1}" "/root/mpi_rack_hostfile" -### 11 +### 12 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-next-80B-A3B.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-next-80B-A3B.sh.txt index 87689bcd220..d579cbfe6a5 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-next-80B-A3B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3-next-80B-A3B.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3-next-80B-A3B" + +### 10 "ray" "start" "--head" @@ -55,12 +60,12 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "awk" "{print $1}" "/root/mpi_rack_hostfile" -### 11 +### 12 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-27B.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-27B.sh.txt index 53c86221e6a..58ee19fba5c 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-27B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-27B.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3.5-27B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-35B-A3B-mtp.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-35B-A3B-mtp.sh.txt index 042f63869fa..e9e4f080a3d 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-35B-A3B-mtp.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-35B-A3B-mtp.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3.5-35B-A3B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-4B.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-4B.sh.txt index 49b84e687bb..4c4cb5787f2 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-4B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-4B.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3.5-4B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-9B.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-9B.sh.txt index b44019d4a58..55e5bcccddf 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-9B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.5-9B.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3.5-9B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit" diff --git a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.6-27B.sh.txt b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.6-27B.sh.txt index e3ec1d82697..a8a67450fd3 100644 --- a/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.6-27B.sh.txt +++ b/tests/snapshots/launch_scripts/sh/scripts/run-qwen3.6-27B.sh.txt @@ -44,6 +44,11 @@ "-m" ### 9 +"python3" +"/scripts/../miles/utils/external_utils/model_args_utils.py" +"qwen3.6-27B" + +### 10 "ray" "start" "--head" @@ -55,7 +60,7 @@ "--dashboard-host=0.0.0.0" "--dashboard-port=8265" -### 10 +### 11 "ray" "job" "submit"