diff --git a/.github/workflows/pr-test.yml b/.github/workflows/pr-test.yml index f88e7cf451..3c663db0b1 100644 --- a/.github/workflows/pr-test.yml +++ b/.github/workflows/pr-test.yml @@ -201,7 +201,7 @@ jobs: strategy: fail-fast: false matrix: - info: [{"num_gpus": 8, "test_file": "test_quick_start_glm4_9B.py"}, {"num_gpus": 8, "test_file": "test_glm4.7_30B_A3B_pd_mooncake.py"}, {"num_gpus": 8, "test_file": "test_qwen3_30B_A3B.py", "use_deepep": "1", "use_fp8_rollout": "1"}, {"num_gpus": 8, "test_file": "test_qwen3_30B_A3B_pd_mooncake.py", "use_deepep": "1"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "test_qwen3_30B_A3B_r3.py", "use_deepep": "1", "use_fp8_rollout": "1"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "test_qwen3_30B_A3B_r3.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ppo.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ppo_train_critic_only.py"}, {"num_gpus": 8, "test_file": "test_moonlight_16B_A3B.py"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "test_moonlight_16B_A3B_r3.py"}, {"num_gpus": 8, "test_file": "test_mimo_7B_mtp_only_grad.py"}, {"num_gpus": 8, "test_file": "test_qwen2.5_0.5B_debug_rollout_then_train.py"}, {"num_gpus": 8, "test_file": "test_qwen2.5_0.5B_opd_sglang.py"}] + info: [{"num_gpus": 8, "test_file": "test_quick_start_glm4_9B.py"}, {"num_gpus": 8, "test_file": "test_glm4.7_30B_A3B_pd_mooncake.py"}, {"num_gpus": 8, "test_file": "test_qwen3_30B_A3B.py", "use_deepep": "1", "use_fp8_rollout": "1"}, {"num_gpus": 8, "test_file": "test_qwen3.6_35B_A3B_pd_mooncake.py", "use_deepep": "1"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "test_qwen3_30B_A3B_r3.py", "use_deepep": "1", "use_fp8_rollout": "1"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "test_qwen3_30B_A3B_r3.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ppo.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ppo_train_critic_only.py"}, {"num_gpus": 8, "test_file": "test_moonlight_16B_A3B.py"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "test_moonlight_16B_A3B_r3.py"}, {"num_gpus": 8, "test_file": "test_mimo_7B_mtp_only_grad.py"}, {"num_gpus": 8, "test_file": "test_qwen2.5_0.5B_debug_rollout_then_train.py"}, {"num_gpus": 8, "test_file": "test_qwen2.5_0.5B_opd_sglang.py"}] defaults: run: working-directory: ${{ github.workspace }} @@ -515,7 +515,7 @@ jobs: strategy: fail-fast: false matrix: - info: [{"num_gpus": 4, "test_file": "test_qwen3.5_0.8B_gsm8k_async_short.py"}, {"num_gpus": 4, "test_file": "test_qwen3.5_0.8B_gsm8k_short.py"}, {"num_gpus": 8, "test_file": "test_quick_start_glm4_9B.py"}, {"num_gpus": 8, "test_file": "test_glm4.7_30B_A3B_pd_mooncake.py"}, {"num_gpus": 8, "test_file": "test_qwen3_30B_A3B.py"}, {"num_gpus": 8, "test_file": "test_qwen3_30B_A3B_pd_mooncake.py", "use_deepep": "1"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ppo.py"}, {"num_gpus": 8, "test_file": "test_moonlight_16B_A3B.py"}, {"num_gpus": 8, "test_file": "test_mimo_7B_mtp_only_grad.py"}, {"num_gpus": 8, "test_file": "test_qwen3_0.6B_parallel_check.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ckpt.py"}, {"num_gpus": 8, "test_args": "--async-save", "test_file": "test_qwen3_4B_ckpt.py"}, {"num_gpus": 8, "test_file": "test_qwen2.5_0.5B_debug_rollout_then_train.py"}, {"num_gpus": 8, "test_file": "test_qwen2.5_0.5B_opd_sglang.py"}] + info: [{"num_gpus": 4, "test_file": "test_qwen3.5_0.8B_gsm8k_async_short.py"}, {"num_gpus": 4, "test_file": "test_qwen3.5_0.8B_gsm8k_short.py"}, {"num_gpus": 8, "test_file": "test_quick_start_glm4_9B.py"}, {"num_gpus": 8, "test_file": "test_glm4.7_30B_A3B_pd_mooncake.py"}, {"num_gpus": 8, "test_file": "test_qwen3_30B_A3B.py"}, {"num_gpus": 8, "test_file": "test_qwen3.6_35B_A3B_pd_mooncake.py", "use_deepep": "1"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ppo.py"}, {"num_gpus": 8, "test_file": "test_moonlight_16B_A3B.py"}, {"num_gpus": 8, "test_file": "test_mimo_7B_mtp_only_grad.py"}, {"num_gpus": 8, "test_file": "test_qwen3_0.6B_parallel_check.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ckpt.py"}, {"num_gpus": 8, "test_args": "--async-save", "test_file": "test_qwen3_4B_ckpt.py"}, {"num_gpus": 8, "test_file": "test_qwen2.5_0.5B_debug_rollout_then_train.py"}, {"num_gpus": 8, "test_file": "test_qwen2.5_0.5B_opd_sglang.py"}] defaults: run: working-directory: ${{ github.workspace }} diff --git a/.github/workflows/pr-test.yml.j2 b/.github/workflows/pr-test.yml.j2 index 16cc38bd32..2314e01861 100644 --- a/.github/workflows/pr-test.yml.j2 +++ b/.github/workflows/pr-test.yml.j2 @@ -22,7 +22,7 @@ {'test_file': 'test_quick_start_glm4_9B.py', 'num_gpus': 8}, {'test_file': 'test_glm4.7_30B_A3B_pd_mooncake.py', 'num_gpus': 8}, {'test_file': 'test_qwen3_30B_A3B.py', 'num_gpus': 8, 'use_deepep': '1', 'use_fp8_rollout': '1'}, - {'test_file': 'test_qwen3_30B_A3B_pd_mooncake.py', 'num_gpus': 8, 'use_deepep': '1'}, + {'test_file': 'test_qwen3.6_35B_A3B_pd_mooncake.py', 'num_gpus': 8, 'use_deepep': '1'}, {'test_file': 'test_qwen3_30B_A3B_r3.py', 'num_gpus': 8, 'use_deepep': '1', 'use_fp8_rollout': '1', 'enable_eval': '0'}, {'test_file': 'test_qwen3_30B_A3B_r3.py', 'num_gpus': 8, 'enable_eval': '0'}, {'test_file': 'test_qwen3_4B_ppo.py', 'num_gpus': 8}, @@ -69,7 +69,7 @@ {'test_file': 'test_quick_start_glm4_9B.py', 'num_gpus': 8}, {'test_file': 'test_glm4.7_30B_A3B_pd_mooncake.py', 'num_gpus': 8}, {'test_file': 'test_qwen3_30B_A3B.py', 'num_gpus': 8}, - {'test_file': 'test_qwen3_30B_A3B_pd_mooncake.py', 'num_gpus': 8, 'use_deepep': '1'}, + {'test_file': 'test_qwen3.6_35B_A3B_pd_mooncake.py', 'num_gpus': 8, 'use_deepep': '1'}, {'test_file': 'test_qwen3_4B_ppo.py', 'num_gpus': 8}, {'test_file': 'test_moonlight_16B_A3B.py', 'num_gpus': 8}, {'test_file': 'test_mimo_7B_mtp_only_grad.py', 'num_gpus': 8}, diff --git a/README.md b/README.md index e48165c5c6..6d0232c344 100644 --- a/README.md +++ b/README.md @@ -11,7 +11,7 @@ 2. **Flexible Data Generation**: Enables arbitrary training data generation workflows through custom data generation interfaces and server-based engines. slime is the RL-framework behind [GLM-5.1](https://z.ai/blog/glm-5.1), [GLM-5](https://z.ai/blog/glm-5), [GLM-4.7](https://z.ai/blog/glm-4.7), [GLM-4.6](https://z.ai/blog/glm-4.6), [GLM-4.5](https://z.ai/blog/glm-4.5) and apart from models from Z.ai, we also supports the following models: -- Qwen3 series (Qwen3Next, Qwen3MoE, Qwen3), Qwen2.5 series; +- Qwen series (Qwen3.6, Qwen3.5, Qwen3Next, Qwen3MoE, Qwen3, Qwen2.5); - DeepSeek V3 series (DeepSeek V3, V3.1, DeepSeek R1); - Llama 3. diff --git a/README_zh.md b/README_zh.md index 12c551e7b3..0c00795dbf 100644 --- a/README_zh.md +++ b/README_zh.md @@ -11,7 +11,7 @@ 2. **灵活的数据生成**:通过自定义数据生成接口以及 server based engine,实现任意的数据训练数据生成流程。 slime 是 [GLM-5.1](https://z.ai/blog/glm-5.1)、[GLM-5](https://z.ai/blog/glm-5)、[GLM-4.7](https://z.ai/blog/glm-4.7)、[GLM-4.6](https://z.ai/blog/glm-4.6)、[GLM-4.5](https://z.ai/blog/glm-4.5) 背后的 RL 训练框架,除此之外,slime 还支持: -- Qwen3 系列 (Qwen3Next, Qwen3MoE, Qwen3), Qwen2.5 系列; +- Qwen 系列 (Qwen3.6、Qwen3.5、Qwen3Next、Qwen3MoE、Qwen3、Qwen2.5); - DeepSeek V3 系列 (DeepSeek V3, V3.1, DeepSeek R1); - Llama 3。 diff --git a/docker/Dockerfile b/docker/Dockerfile index a4791f2548..3876daaf07 100644 --- a/docker/Dockerfile +++ b/docker/Dockerfile @@ -54,8 +54,8 @@ RUN git clone https://github.com/NVIDIA/Megatron-LM.git --recursive && \ cd Megatron-LM && git checkout ${MEGATRON_COMMIT} && \ pip install -e . -RUN pip install git+https://github.com/fzyzcjy/torch_memory_saver.git@dc6876905830430b5054325fa4211ff302169c6b --no-cache-dir --force-reinstall -RUN pip install git+https://github.com/fzyzcjy/Megatron-Bridge.git@dev_rl --no-build-isolation +RUN pip install git+https://github.com/fzyzcjy/torch_memory_saver.git@d64a639 --no-cache-dir --force-reinstall +RUN pip install git+https://github.com/radixark/Megatron-Bridge.git@bridge --no-deps --no-build-isolation RUN pip install nvidia-modelopt[torch]>=0.37.0 --no-build-isolation # This patch from masahi will be included in later Triton releases diff --git a/docker/version.txt b/docker/version.txt index ba2be939c8..54428a6001 100644 --- a/docker/version.txt +++ b/docker/version.txt @@ -1 +1 @@ -nightly-dev-20260430a +nightly-dev-20260430b diff --git a/docs/en/developer_guide/ci.md b/docs/en/developer_guide/ci.md index 1ffaac40ec..f84f46879a 100644 --- a/docs/en/developer_guide/ci.md +++ b/docs/en/developer_guide/ci.md @@ -6,7 +6,7 @@ slime uses GitHub Actions for CI. Tests are triggered by **PR labels** — addin The workflow is defined in `.github/workflows/pr-test.yml` (auto-generated from `pr-test.yml.j2`). Each CI job: -1. Runs on a self-hosted GPU runner inside a Docker container (`slimerl/slime:latest`). +1. Runs on a self-hosted GPU runner via `docker run`; most tests use `slimerl/slime:latest`, while image validation uses `slimerl/slime-test:latest`. 2. Installs slime with `pip install -e . --no-deps`. 3. Acquires the required GPUs via `tests/ci/gpu_lock_exec.py --count `. 4. Executes the test file: `python .py` or `python tests/.py`, depending on whether the test lives under `tests/` or a subdirectory such as `tests/plugin_contracts/`. @@ -56,7 +56,7 @@ Since this includes every test, it consumes significant GPU time — use it spar This is the primary label for validating Megatron-backend changes. It covers: - Dense models: GLM4-9B, Qwen3-4B (PPO) -- MoE models: Qwen3-30B-A3B (with/without DeepEP + FP8), Moonlight-16B-A3B +- MoE models: Qwen3-30B-A3B (with DeepEP + FP8), Qwen3.6-35B-A3B PD + Mooncake, Moonlight-16B-A3B - Specialized: MiMo-7B MTP, Qwen2.5-0.5B debug rollout-then-train, OPD with sglang teacher All tests use 8 GPUs. If you are modifying Megatron training logic, loss computation, or checkpoint conversion, this is the label to use. diff --git a/docs/en/get_started/quick_start.md b/docs/en/get_started/quick_start.md index 4b21f0c230..5d3856fdeb 100644 --- a/docs/en/get_started/quick_start.md +++ b/docs/en/get_started/quick_start.md @@ -71,7 +71,7 @@ hf download --repo-type dataset zhuzilin/aime-2024 \ When using Megatron as the training backend, you need to first convert Hugging Face format model weights to Megatron `torch_dist` format. -First, load the configuration file of the target model. The `slime/scripts/models` directory contains configuration files for supported models. You need to `source` the corresponding model script to load the configuration parameters into the current environment. Here we use GLM4-9B model as an example, and it's similar for Qwen3-4B, GLM-4.7-Flash, Qwen3-30B-A3B, etc. +First, load the configuration file of the target model. The `slime/scripts/models` directory contains configuration files for supported models. You need to `source` the corresponding model script to load the configuration parameters into the current environment. Here we use GLM4-9B model as an example, and it's similar for Qwen3-4B, Qwen3.5, Qwen3.6, GLM-4.7-Flash, Qwen3-30B-A3B, etc. ```bash cd /root/slime diff --git a/docs/zh/developer_guide/ci.md b/docs/zh/developer_guide/ci.md index 603b9a1734..b8830ecc77 100644 --- a/docs/zh/developer_guide/ci.md +++ b/docs/zh/developer_guide/ci.md @@ -6,7 +6,7 @@ slime 使用 GitHub Actions 进行 CI。测试通过 **PR label** 触发—— 工作流定义在 `.github/workflows/pr-test.yml`(由 `pr-test.yml.j2` 自动生成)。每个 CI 任务会: -1. 在自托管 GPU runner 上以 Docker 容器(`slimerl/slime:latest`)运行。 +1. 在自托管 GPU runner 上通过 `docker run` 运行;大多数测试使用 `slimerl/slime:latest`,镜像验证使用 `slimerl/slime-test:latest`。 2. 通过 `pip install -e . --no-deps` 安装 slime。 3. 通过 `tests/ci/gpu_lock_exec.py --count ` 获取所需数量的 GPU。 4. 执行测试文件:`python .py` 或 `python tests/.py`。如果测试位于 `tests/plugin_contracts/` 这样的子目录,CI 也会自动处理。 @@ -56,7 +56,7 @@ slime 使用 GitHub Actions 进行 CI。测试通过 **PR label** 触发—— 这是验证 Megatron 后端改动的主要 label,覆盖: - Dense 模型:GLM4-9B、Qwen3-4B(PPO) -- MoE 模型:Qwen3-30B-A3B(有/无 DeepEP + FP8)、Moonlight-16B-A3B +- MoE 模型:Qwen3-30B-A3B(DeepEP + FP8)、Qwen3.6-35B-A3B PD + Mooncake、Moonlight-16B-A3B - 特殊场景:MiMo-7B MTP、Qwen2.5-0.5B debug rollout-then-train、OPD(sglang teacher 模式) 所有测试使用 8 张 GPU。如果你正在修改 Megatron 训练逻辑、loss 计算或 checkpoint 转换,应该使用这个 label。 diff --git a/docs/zh/get_started/quick_start.md b/docs/zh/get_started/quick_start.md index 98fe2e16dc..463af729f4 100644 --- a/docs/zh/get_started/quick_start.md +++ b/docs/zh/get_started/quick_start.md @@ -70,7 +70,7 @@ hf download --repo-type dataset zhuzilin/aime-2024 \ 当使用 Megatron 作为训练后端时,需要先将 Hugging Face 格式的模型权重转换为 Megatron `torch_dist` 格式。 -首先,加载目标模型的配置文件。`slime/scripts/models` 目录下包含了支持模型的配置文件。需要 `source` 对应模型的脚本,将配置参数加载到当前环境中。此处我们以 GLM4-9B 模型为例子,对于 Qwen3-4B,GLM-4.7-Flash,Qwen3-30B-A3B,是类似的。 +首先,加载目标模型的配置文件。`slime/scripts/models` 目录下包含了支持模型的配置文件。需要 `source` 对应模型的脚本,将配置参数加载到当前环境中。此处我们以 GLM4-9B 模型为例子,对于 Qwen3-4B、Qwen3.5、Qwen3.6、GLM-4.7-Flash、Qwen3-30B-A3B,是类似的。 ```bash cd /root/slime diff --git a/slime/ray/rollout.py b/slime/ray/rollout.py index 83c6600a72..c40f0fc33b 100644 --- a/slime/ray/rollout.py +++ b/slime/ray/rollout.py @@ -123,7 +123,6 @@ def start_engines(self, port_cursors: dict[int, int] | None = None) -> tuple[lis "SLIME_ENABLE_PROFILING": "true", }.items() } - rollout_engine = RolloutRayActor.options( num_cpus=num_cpus, num_gpus=num_gpus, diff --git a/tests/test_glm4.7_30B_A3B_pd_mooncake.py b/tests/test_glm4.7_30B_A3B_pd_mooncake.py index 7048c34c69..9ccbf2b215 100644 --- a/tests/test_glm4.7_30B_A3B_pd_mooncake.py +++ b/tests/test_glm4.7_30B_A3B_pd_mooncake.py @@ -122,6 +122,10 @@ def execute(): "--sglang-cuda-graph-max-bs 8 " "--sglang-max-running-requests 16 " "--sglang-disaggregation-transfer-backend mooncake " + "--sglang-speculative-algorithm EAGLE " + "--sglang-speculative-num-steps 3 " + "--sglang-speculative-eagle-topk 1 " + "--sglang-speculative-num-draft-tokens 4 " "--sglang-watchdog-timeout 1200 " "--sglang-router-request-timeout-secs 1200 " "--sglang-enable-metrics " diff --git a/tests/test_qwen3_30B_A3B_pd_mooncake.py b/tests/test_qwen3.6_35B_A3B_pd_mooncake.py similarity index 87% rename from tests/test_qwen3_30B_A3B_pd_mooncake.py rename to tests/test_qwen3.6_35B_A3B_pd_mooncake.py index 7ff1ed652f..83e2449dae 100644 --- a/tests/test_qwen3_30B_A3B_pd_mooncake.py +++ b/tests/test_qwen3.6_35B_A3B_pd_mooncake.py @@ -4,9 +4,10 @@ import slime.utils.external_utils.command_utils as U -MODEL_NAME = "Qwen3-30B-A3B" -MODEL_TYPE = "qwen3-30B-A3B" +MODEL_NAME = "Qwen3.6-35B-A3B" +MODEL_TYPE = "qwen3.5-35B-A3B" NUM_GPUS = 8 +TORCH_DIST_CKPT = f"/root/models/{MODEL_NAME}_torch_dist" def prepare(): @@ -18,12 +19,13 @@ def prepare(): model_name=MODEL_NAME, megatron_model_type=MODEL_TYPE, num_gpus_per_node=NUM_GPUS, + dir_dst="/root/models", ) def execute(): debug_data_path = os.environ.get("DEBUG_ROLLOUT_DATA") or tempfile.mktemp( - prefix="qwen3_30b_a3b_pd_rollout_", suffix=".pt" + prefix="qwen3_6_35b_a3b_pd_rollout_", suffix=".pt" ) try: os.remove(debug_data_path) @@ -31,7 +33,7 @@ def execute(): pass print(f"Saving debug rollout data to {debug_data_path}") - ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} " f"--ref-load /root/{MODEL_NAME}_torch_dist " + ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} " f"--ref-load {TORCH_DIST_CKPT} " rollout_args = ( "--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl " @@ -57,7 +59,7 @@ def execute(): ) perf_args = ( - "--tensor-model-parallel-size 4 " + "--tensor-model-parallel-size 2 " "--sequence-parallel " "--pipeline-model-parallel-size 1 " "--context-parallel-size 2 " @@ -100,6 +102,12 @@ def execute(): "--sglang-cuda-graph-bs 1 2 4 8 16 24 32 " "--sglang-max-running-requests 512 " "--prefill-num-servers 1 " + "--sglang-disaggregation-transfer-backend mooncake " + "--sglang-speculative-algorithm EAGLE " + "--sglang-speculative-num-steps 3 " + "--sglang-speculative-eagle-topk 1 " + "--sglang-speculative-num-draft-tokens 4 " + "--sglang-mamba-scheduler-strategy extra_buffer " "--sglang-enable-metrics " )