diff --git a/skills/create-issue/SKILL.md b/skills/create-issue/SKILL.md index 1a6729d4636..1370f5ed09d 100644 --- a/skills/create-issue/SKILL.md +++ b/skills/create-issue/SKILL.md @@ -1,9 +1,10 @@ --- name: create-issue description: Investigate a failing GitHub Actions run or job and create a GitHub issue for the failure. +license: Apache-2.0 when_to_use: User shares a GitHub Actions URL and wants to file a bug report; 'create an issue for this failure', 'file a bug for this CI run', 'triage this GitHub Actions failure'. user_invocable: true -argument: "" +argument: "GitHub Actions run or job URL" --- # Triage CI Failure into a GitHub Issue diff --git a/skills/create-issue/evals/evals.json b/skills/create-issue/evals/evals.json new file mode 100644 index 00000000000..17221c20f85 --- /dev/null +++ b/skills/create-issue/evals/evals.json @@ -0,0 +1,13 @@ +[ + { + "id": "create-issue-skill-specific-001", + "question": "I'm filing a GitHub issue for a failing Megatron-LM CI test using the create-issue skill workflow. According to the skill: (a) what is the exact `gh issue list` command to check whether an open duplicate already exists for the failed test (include every flag the skill prescribes: --repo, --state, --search, --json, --limit), and (b) what is the exact issue title format the skill prescribes (use `` as a placeholder for the pytest node ID)? Reply with exactly these two lines and nothing else:\nDuplicate check: \nTitle: ", + "expected_skill": "create-issue", + "expected_script": null, + "ground_truth": "Duplicate check: gh issue list --repo NVIDIA/Megatron-LM --state open --search \"<failed-test-filename>\" --json number,title,url --limit 10\nTitle: 🐛 CI failure: <failed-test-node-id>", + "expected_behavior": [ + "Names the gh issue list command with --repo NVIDIA/Megatron-LM --state open --search \"<failed-test-filename>\" --json number,title,url --limit 10 (all five flags)", + "Names the 🐛 CI failure: <failed-test-node-id> title format" + ] + } +] diff --git a/skills/create-issue/skill-card.md b/skills/create-issue/skill-card.md new file mode 100644 index 00000000000..e853a5c0c35 --- /dev/null +++ b/skills/create-issue/skill-card.md @@ -0,0 +1,38 @@ +## Description: <br> +Investigate a failing GitHub Actions run or job and create a GitHub issue for the failure. <br> + +This skill is ready for commercial/non-commercial use. <br> + +## Owner: <span style="color:#d73a49">NVIDIA</span> <!-- VERIFY: inferred from repo remote (github.com/NVIDIA/Megatron-LM); no explicit owner key in frontmatter --> <br> + +### License/Terms of Use: <br> +Apache 2.0 <br> +## Use Case: <br> +Developers and CI engineers use this skill to automatically triage failing GitHub Actions jobs and file well-structured bug issues against the repository. <br> + +### Deployment Geography for Use: <br> +Global <br> + +## Known Risks and Mitigations: <br> +Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills. <br> +Mitigation: Review and scan skill before deployment. <br> + +## Reference(s): <br> +- [NVIDIA Megatron-LM Repository](https://github.com/NVIDIA/Megatron-LM) <br> +- [Contributing to Megatron-LM](https://docs.nvidia.com/megatron-core/developer-guide/latest/developer/contribute.html) <br> + + +## Skill Output: <br> +**Output Type(s):** [API Calls, Shell commands] <br> +**Output Format:** [Markdown with inline bash code blocks] <br> +**Output Parameters:** [1D] <br> +**Other Properties Related to Output:** [None] <br> + +## Skill Version(s): <br> +f3431cbec (source: git SHA, committed 2026-05-28) <br> + +## Ethical Considerations: <br> +NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse. <br> + +(For Release on NVIDIA Platforms Only) <br> +Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail). <br> diff --git a/skills/create-issue/skill.oms.sig b/skills/create-issue/skill.oms.sig new file mode 100644 index 00000000000..279c5f69d36 --- /dev/null +++ b/skills/create-issue/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAiY3JlYXRlLWlzc3VlIiwKICAgICAgImRpZ2VzdCI6IHsKICAgICAgICAic2hhMjU2IjogIjYzMGY3NDMyYTAyOGJkMjdjNTZmZGI1NDQzNzllMjZmNTdkYTlhZTJjZGQ0M2ExMzYwNTExMWQ4OGVhYTk4ODciCiAgICAgIH0KICAgIH0KICBdLAogICJwcmVkaWNhdGVUeXBlIjogImh0dHBzOi8vbW9kZWxfc2lnbmluZy9zaWduYXR1cmUvdjEuMCIsCiAgInByZWRpY2F0ZSI6IHsKICAgICJyZXNvdXJjZXMiOiBbCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJkNGUwYzQzNTMxYmM1ZmE2N2RiMjk5MTkwMzdhOTNiNTkyMmM3ZjhhNmEwODEwZjI3OWQxOGNjZjdkMTM1Y2Y4IiwKICAgICAgICAibmFtZSI6ICJTS0lMTC5tZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogImRjZDE1ZmJmZDM1NjYzOTE2MDM0NDVlNjM5MTkzZjc4NDhjYTQwYWVlOWZhODE5ODgwYTdhMDBjNDVhOTZjZTgiLAogICAgICAgICJuYW1lIjogInNraWxsLWNhcmQubWQiCiAgICAgIH0KICAgIF0sCiAgICAic2VyaWFsaXphdGlvbiI6IHsKICAgICAgIm1ldGhvZCI6ICJmaWxlcyIsCiAgICAgICJpZ25vcmVfcGF0aHMiOiBbCiAgICAgICAgIi5naXRpZ25vcmUiLAogICAgICAgICIuZ2l0YXR0cmlidXRlcyIsCiAgICAgICAgIi5naXQiLAogICAgICAgICIuZ2l0aHViIgogICAgICBdLAogICAgICAiaGFzaF90eXBlIjogInNoYTI1NiIsCiAgICAgICJhbGxvd19zeW1saW5rcyI6IGZhbHNlCiAgICB9CiAgfQp9","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGYCMQD+c4rz7FZUR75nSEXVOiUmGQ7u7V1C7qUoLztxdcih2NUEEfxDhY1SbPraqA6EGt8CMQC1xrTznVzWGlKzAtj4GZi+OdM79T9qXy1lTJvQo8arTIxuNGuO4t2t5rx1p06iCpo=","keyid":""}]}} \ No newline at end of file diff --git a/skills/linting-and-formatting/SKILL.md b/skills/linting-and-formatting/SKILL.md index 00ab01a1342..d07485eee00 100644 --- a/skills/linting-and-formatting/SKILL.md +++ b/skills/linting-and-formatting/SKILL.md @@ -1,6 +1,7 @@ --- name: linting-and-formatting description: Linting and formatting for Megatron-LM. Covers running autoformat.sh, tools (ruff, black, isort, pylint, mypy), and code style rules. +license: Apache-2.0 when_to_use: Running linting or autoformat; fixing style violations before a PR; 'pre-commit fails', 'ruff error', 'isort', 'mypy', 'style violation', 'how do I format', 'autoformat.sh'. --- diff --git a/skills/linting-and-formatting/evals/evals.json b/skills/linting-and-formatting/evals/evals.json new file mode 100644 index 00000000000..f129340e37c --- /dev/null +++ b/skills/linting-and-formatting/evals/evals.json @@ -0,0 +1,13 @@ +[ + { + "id": "linting-and-formatting-skill-specific-001", + "question": "I'm preparing a PR to Megatron-LM. According to the linting-and-formatting skill: (a) what is the exact bash command that runs autoformat.sh in check-only mode (without rewriting files), and (b) what is the configured maximum line length in characters (from the skill's Code Style Rules section)? Reply with exactly these two lines and nothing else:\nCheck command: <full bash command>\nLine length: <number>", + "expected_skill": "linting-and-formatting", + "expected_script": "tools/autoformat.sh", + "ground_truth": "Check command: BASE_REF=main CHECK_ONLY=true SKIP_DOCS=false bash tools/autoformat.sh\nLine length: 119", + "expected_behavior": [ + "Names the autoformat.sh check command with BASE_REF=main, CHECK_ONLY=true, and SKIP_DOCS=false env vars", + "Names the 119-character line length rule (not 80, not 100, not 120)" + ] + } +] diff --git a/skills/linting-and-formatting/skill-card.md b/skills/linting-and-formatting/skill-card.md new file mode 100644 index 00000000000..6f89740082b --- /dev/null +++ b/skills/linting-and-formatting/skill-card.md @@ -0,0 +1,37 @@ +## Description: <br> +Linting and formatting for Megatron-LM. Covers running autoformat.sh, tools (ruff, black, isort, pylint, mypy), and code style rules. <br> + +This skill is ready for commercial/non-commercial use. <br> + +## Owner: NVIDIA <br> + +### License/Terms of Use: <br> +Apache 2.0 <br> +## Use Case: <br> +Developers and engineers use this skill to run linting and autoformatting tools on Megatron-LM code, fix style violations, and ensure compliance with project code style rules before submitting pull requests. <br> + +### Deployment Geography for Use: <br> +Global <br> + +## Known Risks and Mitigations: <br> +Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills. <br> +Mitigation: Review and scan skill before deployment. <br> + +## Reference(s): <br> +- [Megatron-LM Contributing Guide](../../docs/developer/contribute.md) <br> + + +## Skill Output: <br> +**Output Type(s):** [Shell commands, Configuration instructions] <br> +**Output Format:** [Markdown with inline bash code blocks] <br> +**Output Parameters:** [1D] <br> +**Other Properties Related to Output:** [None] <br> + +## Skill Version(s): <br> +core_v0.15.0rc7-1642-gf3431cbec (source: git tag) <br> + +## Ethical Considerations: <br> +NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse. <br> + +(For Release on NVIDIA Platforms Only) <br> +Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail). <br> diff --git a/skills/linting-and-formatting/skill.oms.sig b/skills/linting-and-formatting/skill.oms.sig new file mode 100644 index 00000000000..ef9f8c62255 --- /dev/null +++ b/skills/linting-and-formatting/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAibGludGluZy1hbmQtZm9ybWF0dGluZyIsCiAgICAgICJkaWdlc3QiOiB7CiAgICAgICAgInNoYTI1NiI6ICIxYmY5ODgyYzBmYTcwYWFhNzZlNjc1OTI3ODRhOWViMDQ5MzBiYWNmMjhmZTE4YmIyOGEyZjRjNDAxM2MwNTZlIgogICAgICB9CiAgICB9CiAgXSwKICAicHJlZGljYXRlVHlwZSI6ICJodHRwczovL21vZGVsX3NpZ25pbmcvc2lnbmF0dXJlL3YxLjAiLAogICJwcmVkaWNhdGUiOiB7CiAgICAic2VyaWFsaXphdGlvbiI6IHsKICAgICAgIm1ldGhvZCI6ICJmaWxlcyIsCiAgICAgICJpZ25vcmVfcGF0aHMiOiBbCiAgICAgICAgIi5naXQiLAogICAgICAgICIuZ2l0YXR0cmlidXRlcyIsCiAgICAgICAgIi5naXRpZ25vcmUiLAogICAgICAgICIuZ2l0aHViIgogICAgICBdLAogICAgICAiaGFzaF90eXBlIjogInNoYTI1NiIsCiAgICAgICJhbGxvd19zeW1saW5rcyI6IGZhbHNlCiAgICB9LAogICAgInJlc291cmNlcyI6IFsKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJTS0lMTC5tZCIsCiAgICAgICAgImRpZ2VzdCI6ICI1ZTU3ZWQ4ZWMzMjJjZTk5ZGEzNzRiMTQzOWI3OWRmZDFkM2Y3YWE3OTA0MzkwNDIwZDU4NmE2N2VmMGY2MmEyIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJuYW1lIjogInNraWxsLWNhcmQubWQiLAogICAgICAgICJkaWdlc3QiOiAiZWI3NTI3ZWU5MmVlM2ExMzcwYThhY2JmYjRlZTEzNTc3MTM3ODJlYTk0YzQ4Mzk2NjAyZDgzMWIyMjFlY2M2NCIKICAgICAgfQogICAgXQogIH0KfQ==","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGYCMQCFWzq3Csti0C0oYSWpAoVrof7PeWf6hOAfQi550cOcqQuqr3/dMSsSCiWd94x+xt8CMQCKV8DPl0zMJkm5YaNbtZztmeH9mrzzU417ym4nwplISvVIZR/0fA7sQQYw75/5CuY=","keyid":""}]}} \ No newline at end of file diff --git a/skills/nightly-sync/SKILL.md b/skills/nightly-sync/SKILL.md index d350d4a7a6f..0d05d102008 100644 --- a/skills/nightly-sync/SKILL.md +++ b/skills/nightly-sync/SKILL.md @@ -1,6 +1,7 @@ --- name: nightly-sync description: Domain knowledge for the nightly main-to-dev sync workflow. Covers merge strategy, CI architecture, failure investigation, and known issues. +license: Apache-2.0 when_to_use: Working on the nightly sync PR; investigating a nightly sync failure; resolving merge conflicts between main and dev; 'nightly sync failed', 'main-to-dev merge', 'sync bot'. --- @@ -10,6 +11,11 @@ This skill is read by the automated sync bot during the nightly-sync-main-to-dev workflow. It contains all domain knowledge for merging main into dev, resolving conflicts, iterating on CI, and shipping the PR. +Detailed shell templates and historical regression examples live in +[detailed procedures](references/detailed-procedures.md). Read that reference +when resolving non-trivial merge conflicts, running pre-push invariant checks, +or executing the Phase 3 CI polling loop. + --- ## Phase 1: Create the Sync Branch and Merge @@ -39,37 +45,12 @@ builds on the last. When PR1 is squash-merged to main, git sees main's squashed version and dev's original commits as unrelated changes. A conflict resolution that blindly picks main can silently discard PR2/PR3's improvements on dev. -After the merge, check for this pattern: - -1. For each conflicted file, run `git log --oneline origin/dev -- <file>` to - see if dev has commits that came AFTER the code main is bringing in. -2. If dev has follow-up commits (bug fixes, refactors, extensions), **favor - dev's version** for those sections. -3. If the conflict is just main bringing in a clean copy of what dev already - has (no follow-ups), main's version is fine. - -Practical check: run `git diff origin/dev -- <file>` on conflicted files. If -dev's code was removed or reverted, investigate whether dev's version is the -more evolved one. - -Real examples from PR #4291: -- `emerging_optimizers.py`: Main's version was MORE complete — it squash-merged - dev's PRs plus added more. Taking main for that section was correct. -- `distrib_optimizer.py`: Main overwrote dev's `GroupedQuantizedTensor` support. - Had to restore `_is_distopt_quantized_param` and the expanded - `_expand_quantized_param_shard_for_cast` loop while keeping main's NVFP4 - additions. This required a surgical merge combining sections from both. - -Key insight: squash-merge chains can go in EITHER direction. Sometimes main -is ahead (it squash-merged dev's work + more), sometimes dev is ahead (it has -follow-up PRs). Always diff both ways before deciding which version to favor. - -Real example from PR #4882 / PR #4318: -- `transformer_engine.py`: main had unrelated `TEFusedMLP` refactors, while dev - had the new `TEFusedDenseMLP` class. The sync kept the config flag and test - but dropped the class and `gpt_layer_specs.py` selection. That is a merge - accident: restore the dev class and selection while preserving main's - `TEFusedMLP.as_mlp_submodule` refactor. +For conflicted files, compare `git log --oneline origin/dev -- <file>` and +`git diff origin/dev -- <file>` before choosing either side. Favor the version +that contains later follow-up work, or combine both sides when each branch added +distinct behavior. See +[merge conflict examples](references/detailed-procedures.md#merge-conflict-examples) +for known recurring cases. ### Files to Override from Main @@ -128,24 +109,9 @@ merged code may import from packages only available at specific git revisions. 3. For sources in both but at different revisions, check whether dev's revision works. If dev's revision is broken (TOML parse errors, missing classes main's code imports), take main's revision instead. - -Real examples from PR #4291: -- `nvidia-resiliency-ext`: Main's `torch.py` imports `get_write_results_queue` - which only existed in main's pinned git revision, not on PyPI. Had to add - main's git source to dev's pyproject.toml. -- `nemo-run`: Dev's pinned revision had a TOML parse error with uv 0.7.2. - Had to swap to main's revision. - -After any changes to `pyproject.toml`, regenerate `uv.lock` inside a CUDA -container: -```bash -docker run --rm -v $(pwd):/workspace nvcr.io/nvidia/pytorch:26.02-py3 \ - bash -c "pip install uv==0.7.2 && cd /workspace && \ - uv venv .venv --system-site-packages && uv sync --only-group build && uv lock" -# Clean up root-owned .venv: -docker run --rm -v $(pwd):/workspace nvcr.io/nvidia/pytorch:26.02-py3 \ - bash -c "rm -rf /workspace/.venv" -``` +4. If `pyproject.toml` changes, regenerate `uv.lock` inside a CUDA container. + Use the exact command in + [dependency reconciliation](references/detailed-procedures.md#dependency-reconciliation). ### API Mismatch Detection (Post-Merge Audit) @@ -165,38 +131,17 @@ After the merge, audit cross-boundary call sites: — if main and dev evolved the interface differently, every caller and implementer must agree -Real examples from PR #4291: -- `multi_latent_attention.py` (main) called `off_interface.group_commit()` - but dev's interface only had `group_offload()` — method renamed -- `mamba_model.py` (main) called `init_chunk_handler(3 params)` but dev's - interface required 6 params — signature expanded on dev -- `mamba_model.py` called `mark_not_offloadable()` but dev had - `mark_not_offload()` — method renamed -- `bulk_offload()` did `.remove()` after `bulk_offload_group()` already - `.pop()`d the same item — double-removal from a list - -Practical detection: -```bash -# For each file taken from main, find what it imports and calls -grep -rn "from <module> import\|<module>\." megatron/ -# Cross-reference with the actual implementations in the merged tree -``` +Practical detection: for each file taken from main, find imported symbols and +method calls, then compare them with implementations in the merged tree. +Recurring examples are listed in +[API mismatch examples](references/detailed-procedures.md#api-mismatch-examples). ### File-Specific Merge Lessons -These lessons were learned from PR #4291. They may recur if the same files -continue to diverge: - -- `gated_delta_net.py`: If the merge creates code calling non-existent helper - methods (e.g. `_resolve_cu_seqlens`), take dev's version wholesale. -- `model_chunk_schedule_plan.py`: Watch for missing imports (e.g. - `CudaGraphScope`) silently dropped during conflict resolution. -- `fine_grained_activation_offload.py`: Critical interface file used by many - callers. If main and dev have divergent method names/signatures, prefer - dev's implementation and patch main-originated callers to match. -- `distrib_optimizer.py`: Dev may have broader type abstractions (e.g. - `_is_distopt_quantized_param` covering both FP8 and GroupedQuantizedTensor). - Main may simplify to explicit type checks. Restore dev's abstractions. +Known recurring conflict patterns are maintained in +[file-specific lessons](references/detailed-procedures.md#file-specific-lessons). +Load them before resolving conflicts in optimizer, offload, schedule, or +TransformerEngine files. ### Special Handling: data_schedule.py @@ -234,123 +179,11 @@ Run on ALL changed Python files (relative to `origin/dev`), in this order: ### Pre-push invariant checks -Before every `git push` in this workflow (the initial push in Phase 1 -AND every fix-push in Phase 3), run these bash checks. If any fails, -fix the condition and re-check before pushing: - -```bash -MERGE_COMMIT=$(git rev-list --min-parents=2 --max-count=1 HEAD || true) -if [ -n "$MERGE_COMMIT" ]; then - DEV_REF="${MERGE_COMMIT}^1" - MAIN_REF="${MERGE_COMMIT}^2" -else - DEV_REF="origin/dev" - MAIN_REF="origin/main" -fi - -# 1. CODEOWNERS must be identical to dev's. -if ! git diff --quiet "$DEV_REF" HEAD -- .github/CODEOWNERS; then - echo "ABORT: .github/CODEOWNERS differs from dev. Restore with:" - echo " git checkout $DEV_REF -- .github/CODEOWNERS" - exit 1 -fi - -# 2. Dependency-management triple must be identical to dev's. -for f in pyproject.toml uv.lock docker/Dockerfile.ci.dev; do - if ! git diff --quiet "$DEV_REF" HEAD -- "$f"; then - # pyproject.toml is allowed to differ ONLY for git source reconciliation - # (new [tool.uv.sources] entries from main). If you intentionally edited - # it for that reason, bypass this check by re-running with $f skipped. - echo "WARNING: $f differs from dev" - fi -done - -# 3. Dev-feature preservation audit. -# -# The most common sync regression is silently dropping a dev-only feature -# that main does not have yet. Pattern: -# T0: a feature lands on dev -# T1 > T0: the same feature lands on main (possibly reformatted) -# The sync runs between T0 and T1. Blindly resolving a conflict in -# main's favour drops dev's addition wherever main happened to touch a -# nearby line for an unrelated reason. -# -# For each file the sync touched (modulo skill-sanctioned overrides and -# the dependency triple), find every line that satisfies ALL of: -# line is on origin/dev (dev had it) -# line is NOT on origin/main (main never owned it) -# line is NOT in the merged tree (the merge dropped it) -# Filter out whitespace-only lines and bracket-only lines (they -# frequently differ for cosmetic reasons). -# -# Files in the "Files to Override from Main" list (training.py, -# initialize.py, utils.py, data_samplers.py, layer_wise_optimizer.py) -# are exempt by skill convention — main may legitimately win there. -# CODEOWNERS and the dep triple are checked above; skip them here. - -INTENTIONAL_OVERRIDE_REGEX='^(megatron/training/training\.py|megatron/training/initialize\.py|megatron/training/utils\.py|megatron/training/datasets/data_samplers\.py|megatron/core/optimizer/layer_wise_optimizer\.py)$' -SKIP_REGEX='^(pyproject\.toml|uv\.lock|docker/Dockerfile\.ci\.dev|\.github/CODEOWNERS)$' - -VIOLATIONS=0 -for f in $(git diff --name-only "$DEV_REF"..HEAD \ - -- '*.py' '*.md' '*.yaml' '*.yml' '*.toml' \ - '*.sh' '*.cpp' '*.cu' '*.h' \ - | sort -u); do - [[ "$f" =~ $SKIP_REGEX ]] && continue - [[ "$f" =~ $INTENTIONAL_OVERRIDE_REGEX ]] && continue - git cat-file -e "HEAD:$f" 2>/dev/null || continue - - missing=$(comm -23 \ - <(git show "$DEV_REF:$f" 2>/dev/null | sort -u) \ - <(git show "$MAIN_REF:$f" 2>/dev/null | sort -u) \ - | comm -23 - <(git show "HEAD:$f" 2>/dev/null | sort -u) \ - | grep -E '[[:alnum:]_]' \ - || true) - - if [ -n "$missing" ]; then - echo "=== $f ===" - printf '%s\n' "$missing" - VIOLATIONS=$((VIOLATIONS + $(printf '%s\n' "$missing" | grep -c .))) - fi -done - -if [ "$VIOLATIONS" -gt 0 ]; then - echo "ABORT: $VIOLATIONS dev-only line(s) dropped by the merge. For each:" - echo " (a) MAIN INTENTIONALLY REMOVED — find the specific commit in" - echo " 'git log origin/main -- <file>' that removed it; document the" - echo " SHA in the PR body, then the drop is acceptable." - echo " (b) MERGE ACCIDENT — main never explicitly touched that line." - echo " RESTORE the dev line (Edit/Write to put it back)." - echo "Default to (b); only declare (a) with a specific main commit as evidence." - exit 1 -fi -``` - -The CODEOWNERS check and the dev-feature preservation audit are HARD -aborts — never push if either fails. The dep-triple check is a warning -because git-source reconciliation can produce legitimate diffs there. - -Recent regressions the dev-feature audit would have flagged (all -"merge accident" type from #4659 and #4716): - -- `transformer_layer.py` lost `_forward_mlp_router(input_ids=None)` -- `token_dispatcher.py` lost the - `num_sms_preprocessing_api=...` kwarg on the `_HybridEPManager` call -- `moe_layer.py` lost `self._maybe_record_overload_factor(...)` -- `gpt_dynamic_inference_with_coordinator.py` lost - `from megatron.training.arguments import parse_and_validate_args` -- `datasets/readme.md` lost the dev-only "Packing Scheduler" section -- PR #4882 / PR #4318 dropped the `TEFusedDenseMLP` implementation and - `gpt_layer_specs.py` selection while leaving the config flag and unit test -- `data_samplers.py` / `utils.py` / `training.py` kept main's - `args.hybrid_context_parallel` instead of dev's - `args.dynamic_context_parallel` (counts as a MERGE ACCIDENT — dev's - reference is present, main's is the deprecated alias that's False - when callers pass `--dynamic-context-parallel`). These files are on - the override list so the audit treats them as "advisory", but you - should still rename `args.hybrid_context_parallel` → - `args.dynamic_context_parallel` on every reference after taking - main's version of these files. +Before every `git push` in this workflow, run the invariant script in +[detailed procedures](references/detailed-procedures.md#pre-push-invariant-checks). +The CODEOWNERS check and dev-feature preservation audit are hard aborts; never +push if either fails. The dependency-triple check is a warning because git-source +reconciliation can produce legitimate diffs. ### Commit and Push @@ -389,7 +222,7 @@ Phase 3 step 4 and the two-commit policy in Rules). }' ``` - Include the exact line (e.g. `Python lines: +1234 / -567 across 42 files`) + Include the exact output line in the PR body in the PR body so reviewers see it at a glance. 3. List of files where main's version was taken over the merge 4. List of files that were deleted in dev but restored (and why) @@ -440,159 +273,23 @@ tool call that blocks inline until the wait is resolved. ### The Fix-Then-Retrigger Loop -Two nested loops. Do NOT conflate them: - -- The **outer loop** is YOUR sequence of tool calls (each iteration: one - `/ok to test`, one blocking poll, maybe one fix-and-push). It is NOT a - Bash loop. It advances because you make new tool calls. -- The **inner loop** is a single blocking Bash tool call using - `while true; do ... sleep 120; done`. It runs during one iteration of - the outer loop and ends when CI reaches a terminal state for that - iteration. - -The outer loop terminates ONLY when Phase 4's gate is satisfied. - -**Source of truth:** `gh pr view <PR_NUMBER> --repo $REPO --json statusCheckRollup`. -This lists every required check, including external status contexts -(GitLab CI, `copy-pr-bot`, etc.) that `gh api .../actions/runs/.../jobs` -does NOT show. - -**Outer-loop iteration (each iteration is a few tool calls):** - -1. `latest_sha=$(git rev-parse HEAD)` (one Bash call). -2. Post `/ok to test $latest_sha` on the PR: - `gh pr comment <PR_NUMBER> --repo $REPO --body "/ok to test $latest_sha"` -3. ONE blocking Bash tool call. This is the inner loop. Copy this - template verbatim, only changing `REPO` and `PR`: - - ```bash - REPO='NVIDIA/Megatron-LM' - PR='<PR_NUMBER>' - # Names matched case-insensitively, anchored to the START of the name. - EXEMPT='copy-pr-bot|is-not-external-contributor|greptile|coderabbit|codeowners|.*review|.*approval|codecov|coverage|build-docs|doc-build|readthedocs|sphinx' - # Sentinel check that tells us CI has fully run. Update this if the - # aggregate gate job is renamed. - SENTINEL='Nemo_CICD_Test' - - while true; do - # Normalize both CheckRun (.status / .conclusion) and StatusContext - # (.state) entries into the same {name, status, conclusion} shape. - rollup=$(gh pr view "$PR" --repo "$REPO" --json statusCheckRollup --jq ' - .statusCheckRollup[] | [ - (.name // .context // "?"), - (if .__typename == "StatusContext" then - (if (.state == "PENDING" or .state == "EXPECTED") then "IN_PROGRESS" - else "COMPLETED" end) - else (.status // "UNKNOWN") end), - (if .__typename == "StatusContext" then - (if .state == "SUCCESS" then "SUCCESS" - elif (.state == "FAILURE" or .state == "ERROR") then "FAILURE" - else "NEUTRAL" end) - else (.conclusion // "UNKNOWN") end) - ] | @tsv') - - # Sentinel: do NOT declare green until the CI aggregate gate has - # reached a terminal state. Before /ok to test triggers the run, - # the sentinel is absent; while CI is running, it's IN_PROGRESS. - sentinel_line=$(printf '%s\n' "$rollup" | awk -F'\t' -v s="$SENTINEL" '$1 == s') - sentinel_status=$(printf '%s\n' "$sentinel_line" | awk -F'\t' 'NR==1 {print $2}') - if [ "$sentinel_status" != "COMPLETED" ]; then - echo "=== $(date -u) waiting for $SENTINEL (status: ${sentinel_status:-absent}) ===" - sleep 120 - continue - fi - - # Classify non-exempt checks (exempt list applied to the NAME only). - non_exempt=$(printf '%s\n' "$rollup" | awk -F'\t' -v p="^($EXEMPT)" 'tolower($1) !~ tolower(p)') - failed=$(printf '%s\n' "$non_exempt" | awk -F'\t' '$2 == "COMPLETED" && $3 !~ /^(SUCCESS|SKIPPED|NEUTRAL)$/') - pending=$(printf '%s\n' "$non_exempt" | awk -F'\t' '$2 != "COMPLETED"') - - if [ -n "$failed" ]; then - echo "=== NON-EXEMPT FAILURES ===" - printf '%s\n' "$failed" - echo "RESULT=FAILURE" - exit 0 - fi - if [ -n "$pending" ]; then - # Sentinel is COMPLETED but a non-exempt check is still pending — - # rare but possible. Keep waiting; do NOT ship. - echo "=== $(date -u) sentinel done but non-exempt checks still pending ===" - printf '%s\n' "$pending" - sleep 120 - continue - fi - - echo "=== ALL NON-EXEMPT CHECKS COMPLETED GREEN ===" - printf '%s\n' "$non_exempt" - echo "RESULT=GREEN" - exit 0 - done - ``` - - This Bash call blocks for as long as CI takes (minutes to hours). Do - NOT split it into many short polls interleaved with other tool calls - — that wastes `--max-turns` and creates windows where you could lose - track of the loop state. - -4. Read the tool output: - - If `RESULT=FAILURE`: diagnose via - `gh api repos/$REPO/actions/jobs/<JOB_ID>/logs` (or the - external-context equivalent) and fix the code. The Phase 1 - commit is immutable; fixes accumulate in a single rolling fix - commit on top of it: - ```bash - git add -A - if git rev-parse --verify HEAD^2 >/dev/null 2>&1; then - # HEAD has two parents → still the Phase 1 merge commit. - # First failure of this run: create the fix commit. - git commit -m "fix: post-CI corrections" - git push origin "$BRANCH" - else - # HEAD is the existing fix commit → amend it. - git commit --amend --no-edit - git push --force-with-lease origin "$BRANCH" - fi - ``` - `--force-with-lease` (not `--force`): if a human pushed onto the - branch since the bot last fetched, the lease aborts the push - instead of clobbering them — fetch and decide what to do. - Start a new outer-loop iteration at step 1 with the new HEAD SHA. - - If `RESULT=GREEN`: outer loop is done. Proceed to Phase 4. - -**Why not wait-for-run-to-register first?** `gh pr comment` with -`/ok to test <sha>` is handled by `copy-pr-bot`, which takes a few -seconds to trigger the CI run. The `statusCheckRollup` poll in step 3 -will initially show checks in `PENDING` / `QUEUED`; that's fine — the -inner loop treats those as "keep waiting" and will see them advance as -CI progresses. No separate registration poll needed. +Use two nested loops: an outer sequence of tool calls and one inner blocking +Bash poll per CI iteration. The source of truth is +`gh pr view <PR_NUMBER> --repo $REPO --json statusCheckRollup`, because it +includes external contexts such as GitLab CI and `copy-pr-bot`. + +For each iteration, get the current SHA, post `/ok to test <sha>`, then run the +blocking polling template in +[detailed procedures](references/detailed-procedures.md#ci-polling-template). +If it reports `RESULT=FAILURE`, fix the code and update the single rolling fix +commit. If it reports `RESULT=GREEN`, proceed to Phase 4. ### Anti-Patterns (what went wrong on run 24800621116) -- **Do NOT classify a queued/in-progress job as "infrastructure- - blocked" and ship.** A stuck queue drains eventually — wait. If the - job eventually passes, great; if it fails, go fix it. -- **Do NOT mark ready while any required check is `PENDING` / - `QUEUED` / `IN_PROGRESS` on the HEAD SHA.** A push is not a pass; - only a `COMPLETED` + green status is. -- **Do NOT declare an untested job "pre-existing."** Pre-existing - means the test ran to completion and failed the same way on recent - dev CI. A job that never ran on your PR cannot be pre-existing. -- **Do NOT use `gh api .../actions/runs/.../jobs` alone** as the gate - signal. External status contexts (GitLab CI pipelines, copy-pr-bot - status, etc.) do NOT appear there. Use `statusCheckRollup`. -- **Do NOT start any background process.** No `&`, no `nohup`, no - `run_in_background: true`, no `ScheduleWakeup`. The GitHub Actions - step owns your shell; when the step ends, every background process - is killed and cannot resume. -- **Do NOT push directly to `pull-request/<PR_NUMBER>` branches.** - The community bot manages those branches when it processes - `/ok to test`. Pushing to them directly breaks the CI trigger - mechanism. Always push to your own sync branch (e.g. - `main2dev/<DATE>`) instead. -- **Do NOT forget the `Run functional tests` and `Run MBridge tests` - labels.** Without `Run functional tests`, the internal GitLab - functional tests do not run; without `Run MBridge tests`, the - MBridge test suite does not run. +Read [CI anti-patterns](references/detailed-procedures.md#ci-anti-patterns). +Key rules: wait for queued/in-progress jobs, gate on `statusCheckRollup`, never +push to `pull-request/<PR_NUMBER>` branches, and do not mark ready until every +non-exempt required check is completed green or documented as pre-existing. ### Failure Investigation @@ -644,9 +341,6 @@ H100/GB200 hardware that may reveal issues GitHub CI does not catch. These surface in `statusCheckRollup` as external status contexts (the bash template already handles them via the `__typename == "StatusContext"` branch). - -- Fine-grained activation offloading failures, for example, only showed - up in GitLab functional tests during PR #4291 - If GitHub CI passes but a reviewer reports GitLab failures, investigate with the same rigor as GitHub CI failures - The sync PR should ideally pass both GitHub and GitLab CI before diff --git a/skills/nightly-sync/evals/evals.json b/skills/nightly-sync/evals/evals.json new file mode 100644 index 00000000000..a030d9541b5 --- /dev/null +++ b/skills/nightly-sync/evals/evals.json @@ -0,0 +1,13 @@ +[ + { + "id": "nightly-sync-skill-specific-001", + "question": "I'm running the Megatron-LM nightly main-to-dev sync workflow. According to the nightly-sync skill: (a) which three repo paths form the 'tightly coupled triple' that must NEVER be taken from main (the skill explicitly says keep dev's versions of all three), and (b) which two GitHub PR labels must be added to the sync PR immediately after it is created? Reply with exactly these two lines and nothing else. Use plain text for every value — no backticks, no quotes, no markdown formatting:\nTriple: <path1>, <path2>, <path3>\nLabels: <label1>, <label2>", + "expected_skill": "nightly-sync", + "expected_script": null, + "ground_truth": "Triple: pyproject.toml, uv.lock, docker/Dockerfile.ci.dev\nLabels: Run functional tests, Run MBridge tests", + "expected_behavior": [ + "Names the dependency triple pyproject.toml, uv.lock, and docker/Dockerfile.ci.dev as files to keep from dev (not take from main)", + "Names both 'Run functional tests' and 'Run MBridge tests' as the labels required on the sync PR" + ] + } +] diff --git a/skills/nightly-sync/references/detailed-procedures.md b/skills/nightly-sync/references/detailed-procedures.md new file mode 100644 index 00000000000..c6615447f1d --- /dev/null +++ b/skills/nightly-sync/references/detailed-procedures.md @@ -0,0 +1,250 @@ +# Nightly Sync Detailed Procedures + +Load this reference only when executing the nightly sync workflow. Keep the +main `SKILL.md` focused on workflow decisions; this file holds copyable shell +templates and historical regression examples. + +## Merge Conflict Examples + +Squash-merge chains can go in either direction. Sometimes main is ahead because +it squash-merged dev work plus follow-up changes; sometimes dev is ahead because +it contains PR2/PR3 work that main does not have yet. Always diff both ways. + +Examples: + +- `emerging_optimizers.py`: main's version was more complete because it + squash-merged dev PRs and added more. Taking main for that section was + correct. +- `distrib_optimizer.py`: main overwrote dev's `GroupedQuantizedTensor` support. + Restore `_is_distopt_quantized_param` and the expanded quantized-shard loop + while keeping main's NVFP4 additions. +- `transformer_engine.py`: main had unrelated `TEFusedMLP` refactors while dev + had `TEFusedDenseMLP`. Restore the dev class and `gpt_layer_specs.py` + selection while preserving main's `TEFusedMLP.as_mlp_submodule` refactor. + +## Dependency Reconciliation + +Examples from prior syncs: + +- `nvidia-resiliency-ext`: main's `torch.py` imported + `get_write_results_queue`, which only existed in main's pinned git revision. + Add main's git source to dev's `pyproject.toml`. +- `nemo-run`: dev's pinned revision had a TOML parse error with uv 0.7.2. + Swap to main's revision. + +After any intentional `pyproject.toml` change, regenerate `uv.lock` inside a +CUDA container: + +```bash +docker run --rm -v $(pwd):/workspace nvcr.io/nvidia/pytorch:26.02-py3 \ + bash -c "pip install uv==0.7.2 && cd /workspace && \ + uv venv .venv --system-site-packages && uv sync --only-group build && uv lock" +docker run --rm -v $(pwd):/workspace nvcr.io/nvidia/pytorch:26.02-py3 \ + bash -c "rm -rf /workspace/.venv" +``` + +## API Mismatch Examples + +Examples from prior syncs: + +- `multi_latent_attention.py` called `off_interface.group_commit()` but dev's + interface only had `group_offload()`. +- `mamba_model.py` called `init_chunk_handler(3 params)` but dev's interface + required 6 params. +- `mamba_model.py` called `mark_not_offloadable()` but dev had + `mark_not_offload()`. +- `bulk_offload()` did `.remove()` after `bulk_offload_group()` already + `.pop()`d the same item. + +## File-Specific Lessons + +- `gated_delta_net.py`: if the merge creates code calling non-existent helpers + such as `_resolve_cu_seqlens`, take dev's version wholesale. +- `model_chunk_schedule_plan.py`: watch for missing imports such as + `CudaGraphScope` silently dropped during conflict resolution. +- `fine_grained_activation_offload.py`: critical interface file used by many + callers. If main and dev diverge on method names or signatures, prefer dev's + implementation and patch main-originated callers to match. +- `distrib_optimizer.py`: dev may have broader type abstractions covering both + FP8 and `GroupedQuantizedTensor`; restore those abstractions when main + simplifies to explicit type checks. + +## Pre-push Invariant Checks + +Run before every `git push` in the nightly sync workflow, including the initial +Phase 1 push and every Phase 3 fix push. If CODEOWNERS or the dev-feature audit +fails, stop and fix before pushing. The dependency-management triple check is a +warning because git-source reconciliation can legitimately change +`pyproject.toml`. + +```bash +MERGE_COMMIT=$(git rev-list --min-parents=2 --max-count=1 HEAD || true) +if [ -n "$MERGE_COMMIT" ]; then + DEV_REF="${MERGE_COMMIT}^1" + MAIN_REF="${MERGE_COMMIT}^2" +else + DEV_REF="origin/dev" + MAIN_REF="origin/main" +fi + +# 1. CODEOWNERS must be identical to dev's. +if ! git diff --quiet "$DEV_REF" HEAD -- .github/CODEOWNERS; then + echo "ABORT: .github/CODEOWNERS differs from dev. Restore with:" + echo " git checkout $DEV_REF -- .github/CODEOWNERS" + exit 1 +fi + +# 2. Dependency-management triple must be identical to dev's. +for f in pyproject.toml uv.lock docker/Dockerfile.ci.dev; do + if ! git diff --quiet "$DEV_REF" HEAD -- "$f"; then + echo "WARNING: $f differs from dev" + fi +done + +# 3. Dev-feature preservation audit. +INTENTIONAL_OVERRIDE_REGEX='^(megatron/training/training\.py|megatron/training/initialize\.py|megatron/training/utils\.py|megatron/training/datasets/data_samplers\.py|megatron/core/optimizer/layer_wise_optimizer\.py)$' +SKIP_REGEX='^(pyproject\.toml|uv\.lock|docker/Dockerfile\.ci\.dev|\.github/CODEOWNERS)$' + +VIOLATIONS=0 +for f in $(git diff --name-only "$DEV_REF"..HEAD \ + -- '*.py' '*.md' '*.yaml' '*.yml' '*.toml' \ + '*.sh' '*.cpp' '*.cu' '*.h' \ + | sort -u); do + [[ "$f" =~ $SKIP_REGEX ]] && continue + [[ "$f" =~ $INTENTIONAL_OVERRIDE_REGEX ]] && continue + git cat-file -e "HEAD:$f" 2>/dev/null || continue + + missing=$(comm -23 \ + <(git show "$DEV_REF:$f" 2>/dev/null | sort -u) \ + <(git show "$MAIN_REF:$f" 2>/dev/null | sort -u) \ + | comm -23 - <(git show "HEAD:$f" 2>/dev/null | sort -u) \ + | grep -E '[[:alnum:]_]' \ + || true) + + if [ -n "$missing" ]; then + echo "=== $f ===" + printf '%s\n' "$missing" + VIOLATIONS=$((VIOLATIONS + $(printf '%s\n' "$missing" | grep -c .))) + fi +done + +if [ "$VIOLATIONS" -gt 0 ]; then + echo "ABORT: $VIOLATIONS dev-only line(s) dropped by the merge. For each:" + echo " (a) MAIN INTENTIONALLY REMOVED -- find the specific commit in" + echo " 'git log origin/main -- <file>' that removed it; document the" + echo " SHA in the PR body, then the drop is acceptable." + echo " (b) MERGE ACCIDENT -- main never explicitly touched that line." + echo " RESTORE the dev line (Edit/Write to put it back)." + echo "Default to (b); only declare (a) with a specific main commit as evidence." + exit 1 +fi +``` + +Regressions this audit would have flagged: + +- `transformer_layer.py` lost `_forward_mlp_router(input_ids=None)`. +- `token_dispatcher.py` lost the `num_sms_preprocessing_api=...` kwarg on the `_HybridEPManager` call. +- `moe_layer.py` lost `self._maybe_record_overload_factor(...)`. +- `gpt_dynamic_inference_with_coordinator.py` lost `from megatron.training.arguments import parse_and_validate_args`. +- `datasets/readme.md` lost the dev-only "Packing Scheduler" section. +- PR #4882 / PR #4318 dropped `TEFusedDenseMLP` and `gpt_layer_specs.py` selection while leaving the config flag and unit test. +- `data_samplers.py`, `utils.py`, and `training.py` kept `args.hybrid_context_parallel` instead of `args.dynamic_context_parallel` after taking main's version. + +## CI Polling Template + +Use this as the single blocking Bash call inside each Phase 3 outer-loop +iteration. It waits for the aggregate gate, classifies non-exempt checks, and +prints `RESULT=FAILURE` or `RESULT=GREEN`. + +```bash +REPO='NVIDIA/Megatron-LM' +PR='<PR_NUMBER>' +# Names matched case-insensitively, anchored to the START of the name. +EXEMPT='copy-pr-bot|is-not-external-contributor|greptile|coderabbit|codeowners|.*review|.*approval|codecov|coverage|build-docs|doc-build|readthedocs|sphinx' +SENTINEL='Nemo_CICD_Test' + +while true; do + rollup=$(gh pr view "$PR" --repo "$REPO" --json statusCheckRollup --jq ' + .statusCheckRollup[] | [ + (.name // .context // "?"), + (if .__typename == "StatusContext" then + (if (.state == "PENDING" or .state == "EXPECTED") then "IN_PROGRESS" + else "COMPLETED" end) + else (.status // "UNKNOWN") end), + (if .__typename == "StatusContext" then + (if .state == "SUCCESS" then "SUCCESS" + elif (.state == "FAILURE" or .state == "ERROR") then "FAILURE" + else "NEUTRAL" end) + else (.conclusion // "UNKNOWN") end) + ] | @tsv') + + sentinel_line=$(printf '%s\n' "$rollup" | awk -F'\t' -v s="$SENTINEL" '$1 == s') + sentinel_status=$(printf '%s\n' "$sentinel_line" | awk -F'\t' 'NR==1 {print $2}') + if [ "$sentinel_status" != "COMPLETED" ]; then + echo "=== $(date -u) waiting for $SENTINEL (status: ${sentinel_status:-absent}) ===" + sleep 120 + continue + fi + + non_exempt=$(printf '%s\n' "$rollup" | awk -F'\t' -v p="^($EXEMPT)" 'tolower($1) !~ tolower(p)') + failed=$(printf '%s\n' "$non_exempt" | awk -F'\t' '$2 == "COMPLETED" && $3 !~ /^(SUCCESS|SKIPPED|NEUTRAL)$/') + pending=$(printf '%s\n' "$non_exempt" | awk -F'\t' '$2 != "COMPLETED"') + + if [ -n "$failed" ]; then + echo "=== NON-EXEMPT FAILURES ===" + printf '%s\n' "$failed" + echo "RESULT=FAILURE" + exit 0 + fi + if [ -n "$pending" ]; then + echo "=== $(date -u) sentinel done but non-exempt checks still pending ===" + printf '%s\n' "$pending" + sleep 120 + continue + fi + + echo "=== ALL NON-EXEMPT CHECKS COMPLETED GREEN ===" + printf '%s\n' "$non_exempt" + echo "RESULT=GREEN" + exit 0 +done +``` + +Do not split this into many short polls. `/ok to test <sha>` may take a few +seconds to register through `copy-pr-bot`; the polling template handles absent, +pending, queued, and in-progress checks. + +## Fix Commit Update + +When the polling template reports `RESULT=FAILURE`, diagnose logs and update the +single rolling fix commit on top of the immutable Phase 1 merge commit: + +```bash +git add -A +if git rev-parse --verify HEAD^2 >/dev/null 2>&1; then + git commit -m "fix: post-CI corrections" + git push origin "$BRANCH" +else + git commit --amend --no-edit + git push --force-with-lease origin "$BRANCH" +fi +``` + +Use `--force-with-lease`, not `--force`; if a human pushed onto the branch, +fetch and decide how to proceed instead of clobbering their work. + +## CI Anti-Patterns + +- Do not classify queued or in-progress jobs as infrastructure-blocked. Wait for + them to reach a terminal state. +- Do not mark ready while any required check is pending, queued, or in progress + on the HEAD SHA. +- Do not declare an untested job pre-existing. Pre-existing means the test ran + to completion and failed the same way on recent dev CI. +- Do not use `gh api .../actions/runs/.../jobs` alone as the gate signal. + External GitLab and bot status contexts do not appear there. +- Do not start background processes. The GitHub Actions step owns the shell, and + background processes die when the step exits. +- Do not push directly to `pull-request/<PR_NUMBER>` branches; the community bot + manages those refs. +- Do not forget `Run functional tests` and `Run MBridge tests` labels. diff --git a/skills/nightly-sync/skill-card.md b/skills/nightly-sync/skill-card.md new file mode 100644 index 00000000000..aa8da7d7fc7 --- /dev/null +++ b/skills/nightly-sync/skill-card.md @@ -0,0 +1,37 @@ +## Description: <br> +Domain knowledge for the nightly main-to-dev sync workflow, covering merge strategy, CI architecture, failure investigation, and known issues. <br> + +This skill is ready for commercial/non-commercial use. <br> + +## Owner: NVIDIA <br> + +### License/Terms of Use: <br> +Apache 2.0 <br> +## Use Case: <br> +Developers and CI automation maintaining the Megatron-LM repository use this skill to execute and troubleshoot the nightly merge of main into dev, resolving conflicts and iterating on CI failures. <br> + +### Deployment Geography for Use: <br> +Global <br> + +## Known Risks and Mitigations: <br> +Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills. <br> +Mitigation: Review and scan skill before deployment. <br> + +## Reference(s): <br> +- [Detailed Procedures](references/detailed-procedures.md) <br> + + +## Skill Output: <br> +**Output Type(s):** [Shell commands, Code, Configuration instructions] <br> +**Output Format:** [Markdown with inline bash code blocks] <br> +**Output Parameters:** [1D] <br> +**Other Properties Related to Output:** [None] <br> + +## Skill Version(s): <br> +core_v0.15.0rc7-1642-gf3431cbec (source: git tag) <br> + +## Ethical Considerations: <br> +NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse. <br> + +(For Release on NVIDIA Platforms Only) <br> +Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail). <br> diff --git a/skills/nightly-sync/skill.oms.sig b/skills/nightly-sync/skill.oms.sig new file mode 100644 index 00000000000..8fcab89ea87 --- /dev/null +++ b/skills/nightly-sync/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAibmlnaHRseS1zeW5jIiwKICAgICAgImRpZ2VzdCI6IHsKICAgICAgICAic2hhMjU2IjogIjcwMGZlMzc1OWI0NzEyNDNkZWMyODgzODExYzg0MzZmY2IxYmEzMjM0ZDVlYzYzZDc1NWNjNzJkYTM2MTdlZTkiCiAgICAgIH0KICAgIH0KICBdLAogICJwcmVkaWNhdGVUeXBlIjogImh0dHBzOi8vbW9kZWxfc2lnbmluZy9zaWduYXR1cmUvdjEuMCIsCiAgInByZWRpY2F0ZSI6IHsKICAgICJyZXNvdXJjZXMiOiBbCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICIwMWQ1NjExYmFlOGU4ZmU5NmU3YzVlODg0NGFlYzI5MDk1NmUxZTQ4YTRhMzZhZGJiOWM2MWM3NjRkOGQwYjJiIiwKICAgICAgICAibmFtZSI6ICJTS0lMTC5tZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjRjN2MzZjRiNGU1NWMwNzczOTU3ZjU2OGQwNGIxM2M2Nzc1Y2U2ZjgzNGQ2NjIwNzYxNGFmMzAwYzIwMzQ0YjEiLAogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvZGV0YWlsZWQtcHJvY2VkdXJlcy5tZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjRlYmZmYjU5MGNjMGUxOWZiYTIxZmE4YTVhZjFkMmYyZWJmMDczODUyOWVlY2ExYjk4MGQ4OThjOWRmZWZiNzUiLAogICAgICAgICJuYW1lIjogInNraWxsLWNhcmQubWQiCiAgICAgIH0KICAgIF0sCiAgICAic2VyaWFsaXphdGlvbiI6IHsKICAgICAgImhhc2hfdHlwZSI6ICJzaGEyNTYiLAogICAgICAibWV0aG9kIjogImZpbGVzIiwKICAgICAgImFsbG93X3N5bWxpbmtzIjogZmFsc2UsCiAgICAgICJpZ25vcmVfcGF0aHMiOiBbCiAgICAgICAgIi5naXRpZ25vcmUiLAogICAgICAgICIuZ2l0IiwKICAgICAgICAiLmdpdGh1YiIsCiAgICAgICAgIi5naXRhdHRyaWJ1dGVzIgogICAgICBdCiAgICB9CiAgfQp9","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGUCMAcbALmZ7B5oDTht7hte16g/V69xgdt2+6RnLiPKWHRfaj9xpcycuqYIJmJ6PVWunwIxAKmV/TuGEf5jtgQY/YD0tPLVIrE3ZT7JkRkv7qCETdOv9JjtF5eK+IRUvfYdo7B4pw==","keyid":""}]}} \ No newline at end of file