diff --git a/.changeset/autofix.md b/.changeset/autofix.md new file mode 100644 index 00000000..7f2b0478 --- /dev/null +++ b/.changeset/autofix.md @@ -0,0 +1,15 @@ +--- +"autofix": minor +--- + +Add the `autofix` workflow: opt-in, one-shot fixing of the PR reviewer's own feedback. + +Arm a PR with an `autofix: blocking` / `autofix: nits` label or an `/autofix [scope]` comment; the run fixes the reviewer's open threads in that scope, pushes one commit, replies in each thread, and removes the label. Both arming surfaces are peers resolving through one shared token vocabulary, and the trigger decides which is read, so a stale label cannot widen an explicit command. + +Everything except the code edit is deterministic. `lib/stage.ts` runs as a pre-agent step and fetches the inputs before the agent starts; `lib/plan.ts` then resolves scope, checks review currency, builds the work list, and renders the commit trailer. The plan is final: the prompt may execute it or stop, never widen it. + +Guards fail closed. Currency is checked per file so one unrelated push doesn't refuse the whole run; unparseable labels, outdated anchors, threads a human opened, an unreadable diff, and a head that moves mid-run are all excluded. Refusal is reserved for a PR with no review at all, and for a thread fetch that fails: GitHub reports GraphQL rate limits and node-access failures as HTTP 200 with an `errors` array, so staging treats any `errors` entry or an unparseable body as fatal rather than as "this PR has no threads", which would clear the arming label while the findings it was armed for stayed open. + +The reviewer's `skip-ai-review` label does not disarm autofix. It stops the reviewer's next run without withdrawing a review already posted, so a labelled PR can still carry current findings, and an explicit `autofix:` label or `/autofix` from someone with write access is the authorisation to act on them. A PR with no review is still refused, by the guard that checks for one. + +The push uses `KHAN_ACTIONS_BOT_TOKEN`, because GitHub creates no workflow runs for `GITHUB_TOKEN`-triggered events, and the re-review of the autofix commit is the intended verification a fix gets. That verification is best-effort rather than guaranteed: the chain from the push to a posted review has several links, whether a break is visible depends on how the consumer triggers its reviewer, and the human re-arming loop is the accepted backstop for v1. The run's summary comment says so on every push. Ships with a documented workaround for gh-aw's unbounded PR-branch fetch, which is otherwise fatal on large monorepos. diff --git a/.github/workflows/autofix.lock.yml b/.github/workflows/autofix.lock.yml new file mode 100644 index 00000000..c6157a16 --- /dev/null +++ b/.github/workflows/autofix.lock.yml @@ -0,0 +1,1910 @@ +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"20eb43d1e2146e94b193e6a6fdf585ee6653e72335ebb8f78ad8a881d1d9b036","body_hash":"4bfba4fa8baaf015b86831cbb1ae4502d5f60a2e032429fcc00c809bf902638d","compiler_version":"v0.83.4","strict":true,"agent_id":"claude","agent_model":"claude-opus-4-8","engine_versions":{"claude":"2.1.220"}} +# gh-aw-manifest: {"version":1,"secrets":["ANTHROPIC_API_KEY","COPILOT_GITHUB_TOKEN","GH_AW_CI_TRIGGER_TOKEN","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN","KHAN_ACTIONS_BOT_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/checkout","sha":"93cb6efe18208431cddfb8368fd83d5badbf9bfd","version":"93cb6efe18208431cddfb8368fd83d5badbf9bfd"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"e89c65e17eb281bbd5ff2ff9e9199a03e96654c7","version":"v0.83.4"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.27.42","digest":"sha256:26a8af4e5566485b02f52af59ee03803ae798271a9619d4767e94d07806deb9b","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.27.42@sha256:26a8af4e5566485b02f52af59ee03803ae798271a9619d4767e94d07806deb9b"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.42","digest":"sha256:944f2686c9ab9bec338fd14b662461662f77cd12cd0ea8a3e7cb8c0987cd1607","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.42@sha256:944f2686c9ab9bec338fd14b662461662f77cd12cd0ea8a3e7cb8c0987cd1607"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.27.42","digest":"sha256:42dfeb649c680a8558cd5423dbc530b653a69413e35ffbe5e71da5d48c94bdf0","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.27.42@sha256:42dfeb649c680a8558cd5423dbc530b653a69413e35ffbe5e71da5d48c94bdf0"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.6","digest":"sha256:fecabec51bbc41f2ad61076d6bcd9a36ef23b142e672a444e054d37fc29de93c","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.6@sha256:fecabec51bbc41f2ad61076d6bcd9a36ef23b142e672a444e054d37fc29de93c"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:a8082161d7dceda14b68f32eb39d0eaa96b825d07f5895b096afab9d9e0c7748","pinned_image":"ghcr.io/github/gh-aw-node@sha256:a8082161d7dceda14b68f32eb39d0eaa96b825d07f5895b096afab9d9e0c7748"},{"image":"ghcr.io/github/github-mcp-server:v1.7.0","digest":"sha256:c491ffdf6f4c85cb5397021bc655edb8ab825c6f5f568e7597d77a1bd7c4d308","pinned_image":"ghcr.io/github/github-mcp-server:v1.7.0@sha256:c491ffdf6f4c85cb5397021bc655edb8ab825c6f5f568e7597d77a1bd7c4d308"}],"has_pull_request":true} +# This file was automatically generated by gh-aw (v0.83.4). DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md +# +# ___ _ _ +# / _ \ | | (_) +# | |_| | __ _ ___ _ __ | |_ _ ___ +# | _ |/ _` |/ _ \ '_ \| __| |/ __| +# | | | | (_| | __/ | | | |_| | (__ +# \_| |_/\__, |\___|_| |_|\__|_|\___| +# __/ | +# _ _ |___/ +# | | | | / _| | +# | | | | ___ _ __ _ __| |_| | _____ ____ +# | |/\| |/ _ \ '__| |/ /| _| |/ _ \ \ /\ / / ___| +# \ /\ / (_) | | | | ( | | | | (_) \ V V /\__ \ +# \/ \/ \___/|_| |_|\_\|_| |_|\___/ \_/\_/ |___/ +# +# +# To update this file, edit Khan/actions/workflows/autofix/autofix.md@autofix-v0.0.0 and run: +# gh aw compile +# Not all edits will cause changes to this file. +# +# For more information: https://github.github.com/gh-aw/introduction/overview/ +# +# Addresses the PR reviewer's own feedback on demand, one run per arming. Arm it with an `/autofix [blocking|nits]` comment, or with an `autofix: blocking` / `autofix: nits` label; the two are peers. The run fixes the reviewer's open threads in that scope, pushes one commit, replies in each thread, and clears the label if one armed it. +# +# Source: Khan/actions/workflows/autofix/autofix.md@autofix-v0.0.0 +# +# Frontmatter env variables: +# - GIT_CONFIG_COUNT: (main workflow) +# - GIT_CONFIG_KEY_0: (main workflow) +# - GIT_CONFIG_KEY_1: (main workflow) +# - GIT_CONFIG_VALUE_0: (main workflow) +# - GIT_CONFIG_VALUE_1: (main workflow) +# +# Secrets used: +# - ANTHROPIC_API_KEY +# - COPILOT_GITHUB_TOKEN +# - GH_AW_CI_TRIGGER_TOKEN +# - GH_AW_GITHUB_MCP_SERVER_TOKEN +# - GH_AW_GITHUB_TOKEN +# - GITHUB_TOKEN +# - KHAN_ACTIONS_BOT_TOKEN +# +# Custom actions used: +# - actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 +# - actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 +# - actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 +# - actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # 93cb6efe18208431cddfb8368fd83d5badbf9bfd +# - actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 +# - actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 +# - actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 (source v9) +# - actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 +# - actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 +# - github/gh-aw-actions/setup@e89c65e17eb281bbd5ff2ff9e9199a03e96654c7 # v0.83.4 +# +# Container images used: +# - ghcr.io/github/gh-aw-firewall/agent:0.27.42@sha256:26a8af4e5566485b02f52af59ee03803ae798271a9619d4767e94d07806deb9b +# - ghcr.io/github/gh-aw-firewall/api-proxy:0.27.42@sha256:944f2686c9ab9bec338fd14b662461662f77cd12cd0ea8a3e7cb8c0987cd1607 +# - ghcr.io/github/gh-aw-firewall/squid:0.27.42@sha256:42dfeb649c680a8558cd5423dbc530b653a69413e35ffbe5e71da5d48c94bdf0 +# - ghcr.io/github/gh-aw-mcpg:v0.4.6@sha256:fecabec51bbc41f2ad61076d6bcd9a36ef23b142e672a444e054d37fc29de93c +# - ghcr.io/github/gh-aw-node@sha256:a8082161d7dceda14b68f32eb39d0eaa96b825d07f5895b096afab9d9e0c7748 +# - ghcr.io/github/github-mcp-server:v1.7.0@sha256:c491ffdf6f4c85cb5397021bc655edb8ab825c6f5f568e7597d77a1bd7c4d308 + +name: "PR Autofixer" +on: + issue_comment: + types: + - created + pull_request: + types: + - labeled +# roles: # Roles processed as role check in pre-activation job +# - admin # Roles processed as role check in pre-activation job +# - maintainer # Roles processed as role check in pre-activation job +# - write # Roles processed as role check in pre-activation job + +permissions: {} + +concurrency: + group: "gh-aw-${{ github.workflow }}-${{ github.event.issue.number || github.event.pull_request.number || github.run_id }}" + cancel-in-progress: true + +run-name: "PR Autofixer" + +env: + GIT_CONFIG_COUNT: "2" + GIT_CONFIG_KEY_0: remote.origin.promisor + GIT_CONFIG_KEY_1: remote.origin.partialclonefilter + GIT_CONFIG_VALUE_0: "true" + GIT_CONFIG_VALUE_1: blob:none + +jobs: + activation: + needs: pre_activation + if: "needs.pre_activation.outputs.activated == 'true' && (((github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository && startsWith(github.event.label.name, 'autofix: ')) || (github.event_name == 'issue_comment' && github.event.issue.pull_request != null && (github.event.comment.body == '/autofix' || startsWith(github.event.comment.body, '/autofix ') || startsWith(github.event.comment.body, '/autofix\n') || startsWith(github.event.comment.body, '/autofix\r') || startsWith(github.event.comment.body, '/autofix\t')))) && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.id == github.repository_id))" + runs-on: ubuntu-slim + permissions: + actions: read + contents: read + issues: write + pull-requests: write + env: + GH_AW_MAX_DAILY_AI_CREDITS: ${{ vars.GH_AW_DEFAULT_MAX_DAILY_AI_CREDITS || '5000' }} + GH_AW_RUNTIME_FEATURES: ${{ vars.GH_AW_RUNTIME_FEATURES }} + outputs: + body: ${{ steps.sanitized.outputs.body }} + comment_id: "" + comment_repo: "" + daily_ai_credits_exceeded: ${{ steps.daily-effective-workflow-guardrail.outputs.daily_ai_credits_exceeded == 'true' }} + daily_ai_credits_threshold: ${{ steps.daily-effective-workflow-guardrail.outputs.daily_ai_credits_threshold || '' }} + daily_ai_credits_total_effective_tokens: ${{ steps.daily-effective-workflow-guardrail.outputs.daily_ai_credits_total_effective_tokens || '' }} + engine_id: ${{ steps.generate_aw_info.outputs.engine_id }} + lockdown_check_failed: ${{ steps.generate_aw_info.outputs.lockdown_check_failed == 'true' }} + model: ${{ steps.generate_aw_info.outputs.model }} + oauth_token_check_failed: ${{ steps.check-oauth-tokens.outputs.oauth_token_check_failed == 'true' }} + secret_verification_result: ${{ steps.validate-secret.outputs.verification_result }} + setup-parent-span-id: ${{ steps.setup.outputs.parent-span-id || steps.setup.outputs.span-id }} + setup-span-id: ${{ steps.setup.outputs.span-id }} + setup-trace-id: ${{ steps.setup.outputs.trace-id }} + stale_lock_file_failed: ${{ steps.check-lock-file.outputs.stale_lock_file_failed == 'true' }} + text: ${{ steps.sanitized.outputs.text }} + title: ${{ steps.sanitized.outputs.title }} + steps: + - name: Setup Scripts + id: setup + uses: github/gh-aw-actions/setup@e89c65e17eb281bbd5ff2ff9e9199a03e96654c7 # v0.83.4 + with: + destination: ${{ runner.temp }}/gh-aw/actions + job-name: ${{ github.job }} + trace-id: ${{ needs.pre_activation.outputs.setup-trace-id }} + parent-span-id: ${{ needs.pre_activation.outputs.setup-parent-span-id || needs.pre_activation.outputs.setup-span-id }} + safe-output-artifact-client: ${{ env.GH_AW_MAX_DAILY_AI_CREDITS != '' }} + env: + GH_AW_SETUP_WORKFLOW_NAME: "PR Autofixer" + GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/autofix.lock.yml@${{ github.ref }} + GH_AW_INFO_VERSION: "2.1.220" + GH_AW_INFO_AWF_VERSION: "v0.27.42" + GH_AW_INFO_BODY_MODIFIED: "false" + GH_AW_INFO_ENGINE_ID: "claude" + - name: Generate agentic run info + id: generate_aw_info + env: + GH_AW_INFO_ENGINE_ID: "claude" + GH_AW_INFO_ENGINE_NAME: "Claude Code" + GH_AW_INFO_MODEL: "claude-opus-4-8" + GH_AW_INFO_VERSION: "2.1.220" + GH_AW_INFO_AGENT_VERSION: "2.1.220" + GH_AW_INFO_CLI_VERSION: "v0.83.4" + GH_AW_INFO_WORKFLOW_NAME: "PR Autofixer" + GH_AW_INFO_EXPERIMENTAL: "false" + GH_AW_INFO_SUPPORTS_TOOLS_ALLOWLIST: "true" + GH_AW_INFO_STAGED: "false" + GH_AW_INFO_ALLOWED_DOMAINS: '["defaults","github"]' + GH_AW_INFO_FIREWALL_ENABLED: "true" + GH_AW_INFO_AWF_VERSION: "v0.27.42" + GH_AW_INFO_AWMG_VERSION: "" + GH_AW_INFO_FIREWALL_TYPE: "squid" + GH_AW_INFO_FRONTMATTER_SOURCE: "Khan/actions/workflows/autofix/autofix.md@autofix-v0.0.0" + GH_AW_INFO_BODY_MODIFIED: "false" + GH_AW_COMPILED_STRICT: "true" + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/generate_aw_info.cjs'); + await main(core, context); + - name: Restore daily AIC usage cache + id: restore-daily-aic-cache + if: ${{ env.GH_AW_MAX_DAILY_AI_CREDITS != '' }} + continue-on-error: true + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + key: agentic-workflow-usage-autofix-${{ github.run_id }} + restore-keys: agentic-workflow-usage-autofix- + path: /tmp/gh-aw/agentic-workflow-usage-cache.jsonl + - name: Restore daily AIC usage cache (artifact fallback) + id: restore-daily-aic-cache-fallback + if: ${{ env.GH_AW_MAX_DAILY_AI_CREDITS != '' }} + continue-on-error: true + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_RESTORE_DAILY_AIC_CACHE_HIT: ${{ steps.restore-daily-aic-cache.outputs.cache-hit }} + GH_AW_RESTORE_DAILY_AIC_CACHE_MATCHED_KEY: ${{ steps.restore-daily-aic-cache.outputs.cache-matched-key }} + with: + github-token: ${{ secrets.GITHUB_TOKEN }} + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/restore_aic_usage_cache_fallback.cjs'); + await main(); + - name: Check daily workflow token guardrail + id: daily-effective-workflow-guardrail + if: ${{ env.GH_AW_MAX_DAILY_AI_CREDITS != '' }} + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_WORKFLOW_NAME: "PR Autofixer" + GH_AW_WORKFLOW_ID: "autofix" + GH_AW_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + GH_AW_WORKFLOW_DISPATCH_AW_CONTEXT: ${{ github.event.inputs.aw_context || '' }} + GH_AW_HAS_SLASH_COMMAND: "false" + GH_AW_HAS_LABEL_COMMAND: "false" + GH_AW_GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + GH_AW_MAX_DAILY_AI_CREDITS: ${{ vars.GH_AW_DEFAULT_MAX_DAILY_AI_CREDITS || '5000' }} + with: + github-token: ${{ secrets.GITHUB_TOKEN }} + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/check_daily_aic_workflow_guardrail.cjs'); + await main(); + - name: Add eyes reaction for immediate feedback + id: react + if: github.event_name == 'issues' || github.event_name == 'issue_comment' || github.event_name == 'pull_request_review_comment' || github.event_name == 'discussion' || github.event_name == 'discussion_comment' || github.event_name == 'pull_request' && github.event.pull_request.head.repo.id == github.repository_id + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_REACTION: "eyes" + with: + github-token: ${{ secrets.GITHUB_TOKEN }} + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/add_reaction.cjs'); + await main(); + - name: Validate ANTHROPIC_API_KEY secret + id: validate-secret + run: bash "${RUNNER_TEMP}/gh-aw/actions/validate_multi_secret.sh" ANTHROPIC_API_KEY 'Claude Code' https://github.github.com/gh-aw/reference/engines/#anthropic-claude-code + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + - name: Check for OAuth tokens + id: check-oauth-tokens + run: bash "${RUNNER_TEMP}/gh-aw/actions/check_oauth_tokens.sh" + env: + COPILOT_GITHUB_TOKEN: ${{ secrets.COPILOT_GITHUB_TOKEN }} + GH_AW_GITHUB_TOKEN: ${{ secrets.GH_AW_GITHUB_TOKEN }} + GH_AW_GITHUB_MCP_SERVER_TOKEN: ${{ secrets.GH_AW_GITHUB_MCP_SERVER_TOKEN }} + - name: Checkout .github and .agents folders + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + sparse-checkout: | + .github + .agents + .antigravity + .claude + .codex + .gemini + .opencode + .pi + sparse-checkout-cone-mode: true + fetch-depth: 1 + - name: Save agent config folders for base branch restoration + env: + GH_AW_AGENT_FOLDERS: ".agents .antigravity .claude .codex .gemini .github .opencode .pi" + GH_AW_AGENT_FILES: "AGENTS.md ANTIGRAVITY.md CLAUDE.md GEMINI.md PI.md opencode.jsonc" + # poutine:ignore untrusted_checkout_exec + run: bash "${RUNNER_TEMP}/gh-aw/actions/save_base_github_folders.sh" + - name: Check workflow lock file + id: check-lock-file + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_WORKFLOW_FILE: "autofix.lock.yml" + GH_AW_CONTEXT_WORKFLOW_REF: "${{ github.workflow_ref }}" + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/check_workflow_timestamp_api.cjs'); + await main(); + - name: Check compile-agentic version + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_COMPILED_VERSION: "v0.83.4" + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/check_version_updates.cjs'); + await main(); + - name: Compute current body text + id: sanitized + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_ALLOWED_DOMAINS: "*.githubusercontent.com,anthropic.com,api.anthropic.com,api.github.com,api.snapcraft.io,archive.ubuntu.com,azure.archive.ubuntu.com,cdn.playwright.dev,codeload.github.com,crl.geotrust.com,crl.globalsign.com,crl.identrust.com,crl.sectigo.com,crl.thawte.com,crl.usertrust.com,crl.verisign.com,crl3.digicert.com,crl4.digicert.com,crls.ssl.com,docs.github.com,files.pythonhosted.org,ghcr.io,github-cloud.githubusercontent.com,github-cloud.s3.amazonaws.com,github.blog,github.com,github.githubassets.com,host.docker.internal,json-schema.org,json.schemastore.org,keyserver.ubuntu.com,khanacademy.atlassian.net,khanacademy.dev,khanacademy.org,lfs.github.com,localhost,objects.githubusercontent.com,ocsp.digicert.com,ocsp.geotrust.com,ocsp.globalsign.com,ocsp.identrust.com,ocsp.sectigo.com,ocsp.ssl.com,ocsp.thawte.com,ocsp.usertrust.com,ocsp.verisign.com,packagecloud.io,packages.cloud.google.com,packages.microsoft.com,patch-diff.githubusercontent.com,patchdiff.githubusercontent.com,playwright.download.prss.microsoft.com,ppa.launchpad.net,pypi.org,raw.githubusercontent.com,registry.npmjs.org,s.symcb.com,s.symcd.com,security.ubuntu.com,sentry.io,statsig.anthropic.com,ts-crl.ws.symantec.com,ts-ocsp.ws.symantec.com,www.googleapis.com" + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/compute_text.cjs'); + await main(); + - name: Log runtime features + if: ${{ contains(toJSON(vars), '"GH_AW_RUNTIME_FEATURES":') }} + run: bash "${RUNNER_TEMP}/gh-aw/actions/log_runtime_features_summary.sh" + - name: Create prompt with built-in context + env: + GH_AW_PROMPT: /tmp/gh-aw/aw-prompts/prompt.txt + GH_AW_SAFE_OUTPUTS: ${{ runner.temp }}/gh-aw/safeoutputs/outputs.jsonl + GH_AW_EXPR_1A3A194A: ${{ github.event.discussion.number || (fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').item_type == 'discussion' && fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').item_number) }} + GH_AW_EXPR_463A214A: ${{ github.event.pull_request.number || (fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').item_type == 'pull_request' && fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').item_number) }} + GH_AW_EXPR_802A9F6A: ${{ github.event.issue.number || (fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').item_type == 'issue' && fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').item_number) }} + GH_AW_EXPR_AE61BB68: ${{ github.event.pull_request.number || github.event.issue.number }} + GH_AW_EXPR_FF1D34CE: ${{ github.event.comment.id || fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').comment_id }} + GH_AW_GITHUB_ACTOR: ${{ github.actor }} + GH_AW_GITHUB_REPOSITORY: ${{ github.repository }} + GH_AW_GITHUB_RUN_ID: ${{ github.run_id }} + GH_AW_GITHUB_WORKSPACE: ${{ github.workspace }} + GH_AW_IS_PR_COMMENT: ${{ github.event.issue.pull_request && 'true' || '' }} + # poutine:ignore untrusted_checkout_exec + run: | + bash "${RUNNER_TEMP}/gh-aw/actions/create_prompt_first.sh" + { + cat << 'GH_AW_PROMPT_21c666b0ea0153cc_EOF' + + GH_AW_PROMPT_21c666b0ea0153cc_EOF + cat "${RUNNER_TEMP}/gh-aw/prompts/xpia.md" + cat "${RUNNER_TEMP}/gh-aw/prompts/temp_folder_prompt.md" + cat "${RUNNER_TEMP}/gh-aw/prompts/markdown.md" + cat "${RUNNER_TEMP}/gh-aw/prompts/safe_outputs_prompt.md" + cat << 'GH_AW_PROMPT_21c666b0ea0153cc_EOF' + + Tools: add_comment, reply_to_pull_request_review_comment(max:20), remove_labels, push_to_pull_request_branch, missing_tool, missing_data, noop + GH_AW_PROMPT_21c666b0ea0153cc_EOF + cat "${RUNNER_TEMP}/gh-aw/prompts/safe_outputs_push_to_pr_branch.md" + cat << 'GH_AW_PROMPT_21c666b0ea0153cc_EOF' + + GH_AW_PROMPT_21c666b0ea0153cc_EOF + cat "${RUNNER_TEMP}/gh-aw/prompts/mcp_cli_tools_prompt.md" + cat << 'GH_AW_PROMPT_21c666b0ea0153cc_EOF' + + The following GitHub context information is available for this workflow: + {{#if github.actor}} + - **actor**: __GH_AW_GITHUB_ACTOR__ + {{/if}} + {{#if github.repository}} + - **repository**: __GH_AW_GITHUB_REPOSITORY__ + {{/if}} + {{#if github.workspace}} + - **workspace**: __GH_AW_GITHUB_WORKSPACE__ + {{/if}} + {{#if github.event.issue.number || (github.aw.context.item_type == 'issue' && github.aw.context.item_number)}} + - **issue-number**: #__GH_AW_EXPR_802A9F6A__ + {{/if}} + {{#if github.event.discussion.number || (github.aw.context.item_type == 'discussion' && github.aw.context.item_number)}} + - **discussion-number**: #__GH_AW_EXPR_1A3A194A__ + {{/if}} + {{#if github.event.pull_request.number || (github.aw.context.item_type == 'pull_request' && github.aw.context.item_number)}} + - **pull-request-number**: #__GH_AW_EXPR_463A214A__ + {{/if}} + {{#if github.event.comment.id || github.aw.context.comment_id}} + - **comment-id**: __GH_AW_EXPR_FF1D34CE__ + {{/if}} + {{#if github.run_id}} + - **workflow-run-id**: __GH_AW_GITHUB_RUN_ID__ + {{/if}} + + + GH_AW_PROMPT_21c666b0ea0153cc_EOF + cat "${RUNNER_TEMP}/gh-aw/prompts/github_mcp_tools_with_safeoutputs_prompt.md" + if [ "$GITHUB_EVENT_NAME" = "issue_comment" ] && [ -n "$GH_AW_IS_PR_COMMENT" ] || [ "$GITHUB_EVENT_NAME" = "pull_request_review_comment" ] || [ "$GITHUB_EVENT_NAME" = "pull_request_review" ]; then + cat "${RUNNER_TEMP}/gh-aw/prompts/pr_context_prompt.md" + fi + if [ "$GITHUB_EVENT_NAME" = "issue_comment" ] && [ -n "$GH_AW_IS_PR_COMMENT" ] || [ "$GITHUB_EVENT_NAME" = "pull_request_review_comment" ] || [ "$GITHUB_EVENT_NAME" = "pull_request_review" ]; then + cat "${RUNNER_TEMP}/gh-aw/prompts/pr_context_push_to_pr_branch_guidance.md" + fi + cat << 'GH_AW_PROMPT_21c666b0ea0153cc_EOF' + + {{#runtime-import .github/workflows/autofix.md}} + GH_AW_PROMPT_21c666b0ea0153cc_EOF + } > "$GH_AW_PROMPT" + - name: Interpolate variables and render templates + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_PROMPT: /tmp/gh-aw/aw-prompts/prompt.txt + GH_AW_ENGINE_ID: "claude" + GH_AW_EXPR_AE61BB68: ${{ github.event.pull_request.number || github.event.issue.number }} + GH_AW_GITHUB_REPOSITORY: ${{ github.repository }} + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/interpolate_prompt.cjs'); + await main(); + - name: Substitute placeholders + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_PROMPT: /tmp/gh-aw/aw-prompts/prompt.txt + GH_AW_EXPR_1A3A194A: ${{ github.event.discussion.number || (fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').item_type == 'discussion' && fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').item_number) }} + GH_AW_EXPR_463A214A: ${{ github.event.pull_request.number || (fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').item_type == 'pull_request' && fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').item_number) }} + GH_AW_EXPR_802A9F6A: ${{ github.event.issue.number || (fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').item_type == 'issue' && fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').item_number) }} + GH_AW_EXPR_AE61BB68: ${{ github.event.pull_request.number || github.event.issue.number }} + GH_AW_EXPR_FF1D34CE: ${{ github.event.comment.id || fromJSON(github.event.inputs.aw_context || github.event.client_payload.aw_context || '{}').comment_id }} + GH_AW_GITHUB_ACTOR: ${{ github.actor }} + GH_AW_GITHUB_REPOSITORY: ${{ github.repository }} + GH_AW_GITHUB_RUN_ID: ${{ github.run_id }} + GH_AW_GITHUB_WORKSPACE: ${{ github.workspace }} + GH_AW_IS_PR_COMMENT: ${{ github.event.issue.pull_request && 'true' || '' }} + GH_AW_MCP_CLI_SERVERS_LIST: '- `safeoutputs` — run `safeoutputs --help` to see available tools' + GH_AW_NEEDS_PRE_ACTIVATION_OUTPUTS_ACTIVATED: ${{ needs.pre_activation.outputs.activated }} + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + + const substitutePlaceholders = require('${{ runner.temp }}/gh-aw/actions/substitute_placeholders.cjs'); + + // Call the substitution function + return await substitutePlaceholders({ + file: process.env.GH_AW_PROMPT, + substitutions: { + GH_AW_EXPR_1A3A194A: process.env.GH_AW_EXPR_1A3A194A, + GH_AW_EXPR_463A214A: process.env.GH_AW_EXPR_463A214A, + GH_AW_EXPR_802A9F6A: process.env.GH_AW_EXPR_802A9F6A, + GH_AW_EXPR_AE61BB68: process.env.GH_AW_EXPR_AE61BB68, + GH_AW_EXPR_FF1D34CE: process.env.GH_AW_EXPR_FF1D34CE, + GH_AW_GITHUB_ACTOR: process.env.GH_AW_GITHUB_ACTOR, + GH_AW_GITHUB_REPOSITORY: process.env.GH_AW_GITHUB_REPOSITORY, + GH_AW_GITHUB_RUN_ID: process.env.GH_AW_GITHUB_RUN_ID, + GH_AW_GITHUB_WORKSPACE: process.env.GH_AW_GITHUB_WORKSPACE, + GH_AW_IS_PR_COMMENT: process.env.GH_AW_IS_PR_COMMENT, + GH_AW_MCP_CLI_SERVERS_LIST: process.env.GH_AW_MCP_CLI_SERVERS_LIST, + GH_AW_NEEDS_PRE_ACTIVATION_OUTPUTS_ACTIVATED: process.env.GH_AW_NEEDS_PRE_ACTIVATION_OUTPUTS_ACTIVATED + } + }); + - name: Validate prompt placeholders + env: + GH_AW_PROMPT: /tmp/gh-aw/aw-prompts/prompt.txt + # poutine:ignore untrusted_checkout_exec + run: bash "${RUNNER_TEMP}/gh-aw/actions/validate_prompt_placeholders.sh" + - name: Print prompt + env: + GH_AW_PROMPT: /tmp/gh-aw/aw-prompts/prompt.txt + # poutine:ignore untrusted_checkout_exec + run: bash "${RUNNER_TEMP}/gh-aw/actions/print_prompt_summary.sh" + - name: Upload activation artifact + if: success() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: activation + include-hidden-files: true + path: | + /tmp/gh-aw/aw_info.json + /tmp/gh-aw/models.json + /tmp/gh-aw/aw-prompts/prompt.txt + /tmp/gh-aw/aw-prompts/prompt-template.txt + /tmp/gh-aw/aw-prompts/prompt-import-tree.json + /tmp/gh-aw/github_rate_limits.jsonl + /tmp/gh-aw/base + /tmp/gh-aw/.claude/agents + /tmp/gh-aw/.claude/skills + if-no-files-found: ignore + retention-days: 1 + + agent: + needs: activation + if: needs.activation.outputs.daily_ai_credits_exceeded != 'true' + runs-on: ubuntu-latest + permissions: + contents: read + pull-requests: read + env: + DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} + GH_AW_ASSETS_ALLOWED_EXTS: "" + GH_AW_ASSETS_BRANCH: "" + GH_AW_ASSETS_MAX_SIZE_KB: 0 + GH_AW_MCP_LOG_DIR: /tmp/gh-aw/mcp-logs/safeoutputs + GH_AW_RUNTIME_FEATURES: ${{ vars.GH_AW_RUNTIME_FEATURES }} + GH_AW_WORKFLOW_ID_SANITIZED: autofix + outputs: + agentic_engine_timeout: ${{ steps.detect-agent-errors.outputs.agentic_engine_timeout || 'false' }} + ai_credits_rate_limit_error: ${{ steps.parse-mcp-gateway.outputs.ai_credits_rate_limit_error || 'false' }} + aic: ${{ steps.parse-mcp-gateway.outputs.aic }} + ambient_context: ${{ steps.parse-mcp-gateway.outputs.ambient_context }} + checkout_pr_success: ${{ steps.checkout-pr.outputs.checkout_pr_success || 'true' }} + effective_tokens: ${{ steps.parse-mcp-gateway.outputs.effective_tokens }} + has_patch: ${{ steps.collect_output.outputs.has_patch }} + http_400_response_error: ${{ steps.detect-agent-errors.outputs.http_400_response_error || 'false' }} + inference_access_error: ${{ steps.detect-agent-errors.outputs.inference_access_error || 'false' }} + invocation_cap_exceeded: ${{ steps.detect-agent-errors.outputs.invocation_cap_exceeded || 'false' }} + mcp_policy_error: ${{ steps.detect-agent-errors.outputs.mcp_policy_error || 'false' }} + model: ${{ needs.activation.outputs.model }} + model_not_supported_error: ${{ steps.detect-agent-errors.outputs.model_not_supported_error || 'false' }} + output: ${{ steps.collect_output.outputs.output }} + output_types: ${{ steps.collect_output.outputs.output_types }} + setup-parent-span-id: ${{ steps.setup.outputs.parent-span-id || steps.setup.outputs.span-id }} + setup-span-id: ${{ steps.setup.outputs.span-id }} + setup-trace-id: ${{ steps.setup.outputs.trace-id }} + unknown_model_ai_credits: ${{ steps.parse-mcp-gateway.outputs.unknown_model_ai_credits || 'false' }} + steps: + - name: Setup Scripts + id: setup + uses: github/gh-aw-actions/setup@e89c65e17eb281bbd5ff2ff9e9199a03e96654c7 # v0.83.4 + with: + destination: ${{ runner.temp }}/gh-aw/actions + job-name: ${{ github.job }} + trace-id: ${{ needs.activation.outputs.setup-trace-id }} + parent-span-id: ${{ needs.activation.outputs.setup-parent-span-id || needs.activation.outputs.setup-span-id }} + env: + GH_AW_SETUP_WORKFLOW_NAME: "PR Autofixer" + GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/autofix.lock.yml@${{ github.ref }} + GH_AW_INFO_VERSION: "2.1.220" + GH_AW_INFO_AWF_VERSION: "v0.27.42" + GH_AW_INFO_BODY_MODIFIED: "false" + GH_AW_INFO_ENGINE_ID: "claude" + - name: Set runtime paths + id: set-runtime-paths + run: | + { + echo "GH_AW_SAFE_OUTPUTS=${RUNNER_TEMP}/gh-aw/safeoutputs/outputs.jsonl" + echo "GH_AW_SAFE_OUTPUTS_CONFIG_PATH=${RUNNER_TEMP}/gh-aw/safeoutputs/config.json" + echo "GH_AW_SAFE_OUTPUTS_TOOLS_PATH=${RUNNER_TEMP}/gh-aw/safeoutputs/tools.json" + } >> "$GITHUB_OUTPUT" + - name: Checkout repository + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - name: Create gh-aw temp directory + run: bash "${RUNNER_TEMP}/gh-aw/actions/create_gh_aw_tmp_dir.sh" + - name: Configure gh CLI for GitHub Enterprise + run: bash "${RUNNER_TEMP}/gh-aw/actions/configure_gh_for_ghe.sh" + env: + GH_TOKEN: ${{ github.token }} + - name: Download activation artifact + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: activation + path: /tmp/gh-aw + - name: Configure Git credentials + env: + GITHUB_REPOSITORY: ${{ github.repository }} + GITHUB_SERVER_URL: ${{ github.server_url }} + GITHUB_TOKEN: ${{ github.token }} + run: bash "${RUNNER_TEMP}/gh-aw/actions/configure_git_credentials.sh" + - name: Checkout PR branch + id: checkout-pr + if: | + github.event.pull_request || github.event.issue.pull_request || github.event_name == 'workflow_dispatch' && fromJSON(github.event.inputs.aw_context || '{}').item_type == 'pull_request' + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_TOKEN: ${{ secrets.GH_AW_GITHUB_MCP_SERVER_TOKEN || secrets.GH_AW_GITHUB_TOKEN || secrets.GITHUB_TOKEN }} + with: + github-token: ${{ secrets.GH_AW_GITHUB_MCP_SERVER_TOKEN || secrets.GH_AW_GITHUB_TOKEN || secrets.GITHUB_TOKEN }} + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/checkout_pr_branch.cjs'); + await main(); + - name: Setup Node.js + uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: '24' + package-manager-cache: false + - name: Install AWF binary + run: bash "${RUNNER_TEMP}/gh-aw/actions/install_awf_binary.sh" v0.27.42 --rootless + - name: Install Claude Code CLI + run: npm install -g @anthropic-ai/claude-code@2.1.220 + - name: Determine automatic lockdown mode for GitHub MCP Server + id: determine-automatic-lockdown + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 (source v9) + env: + GH_AW_GITHUB_TOKEN: ${{ secrets.GH_AW_GITHUB_TOKEN }} + GH_AW_GITHUB_MCP_SERVER_TOKEN: ${{ secrets.GH_AW_GITHUB_MCP_SERVER_TOKEN }} + GH_AW_GITHUB_MIN_INTEGRITY: 'none' + with: + script: | + const determineAutomaticLockdown = require('${{ runner.temp }}/gh-aw/actions/determine_automatic_lockdown.cjs'); + await determineAutomaticLockdown(github, context, core); + - name: Parse integrity filter lists + id: parse-guard-vars + env: + GH_AW_BLOCKED_USERS_VAR: ${{ vars.GH_AW_GITHUB_BLOCKED_USERS || '' }} + GH_AW_TRUSTED_USERS_VAR: ${{ vars.GH_AW_GITHUB_TRUSTED_USERS || '' }} + GH_AW_APPROVAL_LABELS_VAR: ${{ vars.GH_AW_GITHUB_APPROVAL_LABELS || '' }} + run: bash "${RUNNER_TEMP}/gh-aw/actions/parse_guard_list.sh" + - name: Restore agent config folders from base branch + if: steps.checkout-pr.outcome == 'success' + env: + GH_AW_AGENT_FOLDERS: ".agents .antigravity .claude .codex .gemini .github .opencode .pi" + GH_AW_AGENT_FILES: "AGENTS.md ANTIGRAVITY.md CLAUDE.md GEMINI.md PI.md opencode.jsonc" + run: bash "${RUNNER_TEMP}/gh-aw/actions/restore_base_github_folders.sh" + - name: Restore inline sub-agents from activation artifact + env: + GH_AW_SUB_AGENT_DIR: ".claude/agents" + GH_AW_SUB_AGENT_EXT: ".md" + run: bash "${RUNNER_TEMP}/gh-aw/actions/restore_inline_sub_agents.sh" + - name: Restore inline skills from activation artifact + env: + GH_AW_SKILL_DIR: ".claude/skills" + run: bash "${RUNNER_TEMP}/gh-aw/actions/restore_inline_skills.sh" + - name: Check out shared workflow lib (Khan/actions) + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # 93cb6efe18208431cddfb8368fd83d5badbf9bfd + with: + path: gh-aw-autofix-lib + persist-credentials: false + ref: autofix-v0.0.0 + repository: Khan/actions + - env: + AUTOFIX_COMMAND_BODY: ${{ github.event_name == 'issue_comment' && github.event.comment.body || '' }} + AUTOFIX_PR_NUMBER: ${{ github.event.pull_request.number || github.event.issue.number }} + GITHUB_REPOSITORY: ${{ github.repository }} + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + name: Stage the plan's inputs (deterministic) + run: npx -y tsx workflows/autofix/lib/stage.ts + working-directory: gh-aw-autofix-lib + + - name: Download container images + run: bash "${RUNNER_TEMP}/gh-aw/actions/download_docker_images.sh" ghcr.io/github/gh-aw-firewall/agent:0.27.42@sha256:26a8af4e5566485b02f52af59ee03803ae798271a9619d4767e94d07806deb9b ghcr.io/github/gh-aw-firewall/api-proxy:0.27.42@sha256:944f2686c9ab9bec338fd14b662461662f77cd12cd0ea8a3e7cb8c0987cd1607 ghcr.io/github/gh-aw-firewall/squid:0.27.42@sha256:42dfeb649c680a8558cd5423dbc530b653a69413e35ffbe5e71da5d48c94bdf0 ghcr.io/github/gh-aw-mcpg:v0.4.6@sha256:fecabec51bbc41f2ad61076d6bcd9a36ef23b142e672a444e054d37fc29de93c ghcr.io/github/gh-aw-node@sha256:a8082161d7dceda14b68f32eb39d0eaa96b825d07f5895b096afab9d9e0c7748 ghcr.io/github/github-mcp-server:v1.7.0@sha256:c491ffdf6f4c85cb5397021bc655edb8ab825c6f5f568e7597d77a1bd7c4d308 + - name: Generate Safe Outputs Config + env: + GH_AW_SECRET_KHAN_ACTIONS_BOT_TOKEN: ${{ secrets.KHAN_ACTIONS_BOT_TOKEN }} + run: | + mkdir -p "${RUNNER_TEMP}/gh-aw/safeoutputs" + mkdir -p /tmp/gh-aw/safeoutputs + mkdir -p /tmp/gh-aw/mcp-logs/safeoutputs + mkdir -p "${RUNNER_TEMP}/gh-aw/safeoutputs/upload-artifacts" + cat > "${RUNNER_TEMP}/gh-aw/safeoutputs/config.json" << 'GH_AW_SAFE_OUTPUTS_CONFIG_5c591a0b3081d14f_EOF' + {"add_comment":{"discussions":false,"footer":false,"hide_older_comments":true,"max":1,"target":"triggering"},"create_report_incomplete_issue":{},"missing_data":{},"missing_tool":{},"noop":{"max":1,"report-as-issue":"true"},"push_to_pull_request_branch":{"github-token":"${GH_AW_SECRET_KHAN_ACTIONS_BOT_TOKEN}","if_no_changes":"ignore","max":1,"max_patch_size":4096,"protect_top_level_dot_folders":true,"protected_files":["package.json","bun.lockb","bunfig.toml","deno.json","deno.jsonc","deno.lock","global.json","NuGet.Config","Directory.Packages.props","mix.exs","mix.lock","go.mod","go.sum","stack.yaml","stack.yaml.lock","pom.xml","build.gradle","build.gradle.kts","settings.gradle","settings.gradle.kts","gradle.properties","package-lock.json","yarn.lock","pnpm-lock.yaml","npm-shrinkwrap.json","requirements.txt","Pipfile","Pipfile.lock","pyproject.toml","setup.py","setup.cfg","Gemfile","Gemfile.lock","uv.lock","CODEOWNERS","DESIGN.md","README.md","CONTRIBUTING.md","CHANGELOG.md","SECURITY.md","CODE_OF_CONDUCT.md","CLAUDE.md","AGENTS.md"],"target":"triggering"},"remove_labels":{"allowed":["autofix: blocking","autofix: nits","autofix: loop","autofix: human","autofix: author"]},"reply_to_pull_request_review_comment":{"footer":false,"max":20,"target":"triggering"},"report_incomplete":{},"upload_artifact":{"allowed-paths":["out/**","/tmp/gh-aw/autofix/out/**"],"max-size-bytes":104857600,"max-uploads":1,"retention-days":30}} + GH_AW_SAFE_OUTPUTS_CONFIG_5c591a0b3081d14f_EOF + - name: Generate Safe Outputs Tools + env: + GH_AW_TOOLS_META_JSON: | + { + "description_suffixes": { + "add_comment": " CONSTRAINTS: Maximum 1 comment(s) can be added. Target: triggering. Supports reply_to_id for discussion threading.", + "push_to_pull_request_branch": " CONSTRAINTS: Maximum 1 push(es) can be made.", + "remove_labels": " CONSTRAINTS: Only these labels can be removed: [autofix: blocking autofix: nits autofix: loop autofix: human autofix: author].", + "reply_to_pull_request_review_comment": " CONSTRAINTS: Maximum 20 reply/replies can be created." + }, + "repo_params": {}, + "dynamic_tools": [] + } + GH_AW_VALIDATION_JSON: | + { + "add_comment": { + "defaultMax": 1, + "fields": { + "body": { + "required": true, + "type": "string", + "sanitize": true, + "maxLength": 65000 + }, + "item_number": { + "issueOrPRNumber": true + }, + "reply_to_id": { + "type": "string", + "maxLength": 256 + }, + "repo": { + "type": "string", + "maxLength": 256 + } + } + }, + "missing_data": { + "defaultMax": 20, + "fields": { + "alternatives": { + "type": "string", + "sanitize": true, + "maxLength": 256 + }, + "context": { + "type": "string", + "sanitize": true, + "maxLength": 256 + }, + "data_type": { + "type": "string", + "sanitize": true, + "maxLength": 128 + }, + "reason": { + "type": "string", + "sanitize": true, + "maxLength": 256 + } + } + }, + "missing_tool": { + "defaultMax": 20, + "fields": { + "alternatives": { + "type": "string", + "sanitize": true, + "maxLength": 512 + }, + "reason": { + "required": true, + "type": "string", + "sanitize": true, + "maxLength": 256 + }, + "tool": { + "type": "string", + "sanitize": true, + "maxLength": 128 + } + } + }, + "noop": { + "defaultMax": 1, + "fields": { + "message": { + "required": true, + "type": "string", + "sanitize": true, + "maxLength": 65000 + } + } + }, + "push_to_pull_request_branch": { + "defaultMax": 1, + "fields": { + "branch": { + "type": "string", + "sanitize": true, + "maxLength": 256 + }, + "message": { + "required": true, + "type": "string", + "sanitize": true, + "maxLength": 65000 + }, + "pull_request_number": { + "issueOrPRNumber": true + } + } + }, + "remove_labels": { + "defaultMax": 5, + "fields": { + "item_number": { + "issueNumberOrTemporaryId": true + }, + "labels": { + "required": true, + "type": "array" + }, + "repo": { + "type": "string", + "maxLength": 256 + } + } + }, + "reply_to_pull_request_review_comment": { + "defaultMax": 10, + "fields": { + "body": { + "required": true, + "type": "string", + "sanitize": true, + "maxLength": 65000 + }, + "comment_id": { + "required": true, + "positiveInteger": true + }, + "pull_request_number": { + "optionalPositiveInteger": true + }, + "repo": { + "type": "string", + "maxLength": 256 + } + } + }, + "report_incomplete": { + "defaultMax": 5, + "fields": { + "details": { + "type": "string", + "sanitize": true, + "maxLength": 65000 + }, + "reason": { + "required": true, + "type": "string", + "sanitize": true, + "maxLength": 1024 + } + } + } + } + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/generate_safe_outputs_tools.cjs'); + await main(); + - name: Start MCP Gateway + id: start-mcp-gateway + env: + GH_AW_POLICY_ALLOW_CREATE_PULL_REQUEST: ${{ vars.GH_AW_POLICY_ALLOW_CREATE_PULL_REQUEST || 'true' }} + GH_AW_SAFE_OUTPUTS: ${{ steps.set-runtime-paths.outputs.GH_AW_SAFE_OUTPUTS }} + GH_AW_SAFE_OUTPUTS_CONFIG_PATH: ${{ steps.set-runtime-paths.outputs.GH_AW_SAFE_OUTPUTS_CONFIG_PATH }} + GH_AW_SAFE_OUTPUTS_TOOLS_PATH: ${{ steps.set-runtime-paths.outputs.GH_AW_SAFE_OUTPUTS_TOOLS_PATH }} + GITHUB_MCP_SERVER_TOKEN: ${{ secrets.GH_AW_GITHUB_MCP_SERVER_TOKEN || secrets.GH_AW_GITHUB_TOKEN || secrets.GITHUB_TOKEN }} + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + set -eo pipefail + mkdir -p "${RUNNER_TEMP}/gh-aw/mcp-config" + + # Export gateway environment variables for MCP config and gateway script + export MCP_GATEWAY_PORT="8080" + export MCP_GATEWAY_DOMAIN="awmg-mcpg" + export MCP_GATEWAY_HOST_DOMAIN="localhost" + MCP_GATEWAY_API_KEY=$(openssl rand -base64 45 | tr -d '/+=') + echo "::add-mask::${MCP_GATEWAY_API_KEY}" + export MCP_GATEWAY_API_KEY + export MCP_GATEWAY_PAYLOAD_DIR="/tmp/gh-aw/mcp-payloads" + mkdir -p "${MCP_GATEWAY_PAYLOAD_DIR}" + export MCP_GATEWAY_PAYLOAD_SIZE_THRESHOLD="524288" + export DEBUG="*" + + export GH_AW_ENGINE="claude" + MCP_GATEWAY_UID=$(id -u 2>/dev/null || echo '0') + MCP_GATEWAY_GID=$(id -g 2>/dev/null || echo '0') + source "${RUNNER_TEMP}/gh-aw/actions/resolve_docker_socket_gid.sh" + export MCP_GATEWAY_DOCKER_COMMAND='docker run -i --rm --network bridge -p 127.0.0.1:'"${MCP_GATEWAY_PORT}"':'"${MCP_GATEWAY_PORT}"' --name awmg-mcpg --add-host host.docker.internal:host-gateway --user '"${MCP_GATEWAY_UID}"':'"${MCP_GATEWAY_GID}"' --group-add '"${DOCKER_SOCK_GID}"' -v '"${DOCKER_SOCK_PATH}"':/var/run/docker.sock -e MCP_GATEWAY_PORT -e MCP_GATEWAY_DOMAIN -e MCP_GATEWAY_API_KEY -e MCP_GATEWAY_PAYLOAD_DIR -e MCP_GATEWAY_PAYLOAD_SIZE_THRESHOLD -e DOCKER_HOST=unix:///var/run/docker.sock -e DEBUG -e MCP_GATEWAY_LOG_DIR -e GH_AW_MCP_LOG_DIR -e GH_AW_SAFE_OUTPUTS -e GH_AW_SAFE_OUTPUTS_CONFIG_PATH -e GH_AW_SAFE_OUTPUTS_TOOLS_PATH -e GH_AW_POLICY_ALLOW_CREATE_PULL_REQUEST -e GH_AW_ASSETS_BRANCH -e GH_AW_ASSETS_MAX_SIZE_KB -e GH_AW_ASSETS_ALLOWED_EXTS -e DEFAULT_BRANCH -e GITHUB_MCP_SERVER_TOKEN -e GITHUB_MCP_GUARD_MIN_INTEGRITY -e GITHUB_MCP_GUARD_REPOS -e GITHUB_REPOSITORY -e GITHUB_SERVER_URL -e GITHUB_SHA -e GITHUB_WORKSPACE -e GITHUB_TOKEN -e GITHUB_RUN_ID -e GITHUB_RUN_NUMBER -e GITHUB_RUN_ATTEMPT -e GITHUB_JOB -e GITHUB_ACTION -e GITHUB_EVENT_NAME -e GITHUB_EVENT_PATH -e GITHUB_ACTOR -e GITHUB_ACTOR_ID -e GITHUB_TRIGGERING_ACTOR -e GITHUB_WORKFLOW -e GITHUB_WORKFLOW_REF -e GITHUB_WORKFLOW_SHA -e GITHUB_REF -e GITHUB_REF_NAME -e GITHUB_REF_TYPE -e GITHUB_HEAD_REF -e GITHUB_BASE_REF -e RUNNER_TEMP -v /tmp/gh-aw/mcp-payloads:/tmp/gh-aw/mcp-payloads:rw -v /opt:/opt:ro -v /tmp:/tmp:rw -v '"${GITHUB_WORKSPACE}"':'"${GITHUB_WORKSPACE}"':rw -v '"${RUNNER_TEMP}"'/gh-aw/safeoutputs:'"${RUNNER_TEMP}"'/gh-aw/safeoutputs:rw ghcr.io/github/gh-aw-mcpg:v0.4.6' + + GH_AW_NODE=$(which node 2>/dev/null || command -v node 2>/dev/null || echo node) + cat << GH_AW_MCP_CONFIG_2e313451a36ccf23_EOF | "$GH_AW_NODE" "${RUNNER_TEMP}/gh-aw/actions/start_mcp_gateway.cjs" + { + "mcpServers": { + "github": { + "container": "ghcr.io/github/github-mcp-server:v1.7.0", + "env": { + "GITHUB_FEATURES": "fields_param", + "GITHUB_HOST": "$GITHUB_SERVER_URL", + "GITHUB_PERSONAL_ACCESS_TOKEN": "$GITHUB_MCP_SERVER_TOKEN", + "GITHUB_READ_ONLY": "1", + "GITHUB_TOOLSETS": "pull_requests,repos" + }, + "guard-policies": { + "allow-only": { + "approval-labels": ${{ steps.parse-guard-vars.outputs.approval_labels }}, + "blocked-users": ${{ steps.parse-guard-vars.outputs.blocked_users }}, + "min-integrity": "none", + "repos": "all", + "trusted-users": ${{ steps.parse-guard-vars.outputs.trusted_users }} + } + } + }, + "safeoutputs": { + "container": "ghcr.io/github/gh-aw-node", + "mounts": ["\${GITHUB_WORKSPACE}:\${GITHUB_WORKSPACE}:rw", "${RUNNER_TEMP}/gh-aw/safeoutputs:${RUNNER_TEMP}/gh-aw/safeoutputs:rw", "/tmp/gh-aw:/tmp/gh-aw:rw"], + "args": ["-w", "\${GITHUB_WORKSPACE}"], + "entrypoint": "sh", + "entrypointArgs": ["-c", "sh ${RUNNER_TEMP}/gh-aw/safeoutputs/start_safe_outputs_mcp.sh"], + "env": { + "DEBUG": "*", + "DEFAULT_BRANCH": "\${DEFAULT_BRANCH}", + "GH_AW_ASSETS_ALLOWED_EXTS": "\${GH_AW_ASSETS_ALLOWED_EXTS}", + "GH_AW_ASSETS_BRANCH": "\${GH_AW_ASSETS_BRANCH}", + "GH_AW_ASSETS_MAX_SIZE_KB": "\${GH_AW_ASSETS_MAX_SIZE_KB}", + "GH_AW_MCP_LOG_DIR": "\${GH_AW_MCP_LOG_DIR}", + "GH_AW_SAFE_OUTPUTS": "\${GH_AW_SAFE_OUTPUTS}", + "GH_AW_SAFE_OUTPUTS_CONFIG_PATH": "\${GH_AW_SAFE_OUTPUTS_CONFIG_PATH}", + "GH_AW_SAFE_OUTPUTS_TOOLS_PATH": "\${GH_AW_SAFE_OUTPUTS_TOOLS_PATH}", + "GH_AW_POLICY_ALLOW_CREATE_PULL_REQUEST": "\${GH_AW_POLICY_ALLOW_CREATE_PULL_REQUEST}", + "GITHUB_REPOSITORY": "\${GITHUB_REPOSITORY}", + "GITHUB_SHA": "\${GITHUB_SHA}", + "GITHUB_TOKEN": "\${GITHUB_TOKEN}", + "GITHUB_WORKSPACE": "\${GITHUB_WORKSPACE}", + "RUNNER_TEMP": "\${RUNNER_TEMP}" + }, + "guard-policies": { + "write-sink": { + "accept": [ + "*" + ], + "sink-visibility": ${{ toJSON(steps.determine-automatic-lockdown.outputs.visibility) }} + } + } + } + }, + "gateway": { + "port": $MCP_GATEWAY_PORT, + "domain": "${MCP_GATEWAY_DOMAIN}", + "apiKey": "${MCP_GATEWAY_API_KEY}", + "payloadDir": "${MCP_GATEWAY_PAYLOAD_DIR}", + "startupTimeout": 120 + } + } + GH_AW_MCP_CONFIG_2e313451a36ccf23_EOF + - name: Mount MCP servers as CLIs + id: mount-mcp-clis + continue-on-error: true + env: + MCP_GATEWAY_API_KEY: ${{ steps.start-mcp-gateway.outputs.gateway-api-key }} + MCP_GATEWAY_DOMAIN: ${{ steps.start-mcp-gateway.outputs.gateway-domain }} + MCP_GATEWAY_PORT: ${{ steps.start-mcp-gateway.outputs.gateway-port }} + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io); + const { main } = require('${{ runner.temp }}/gh-aw/actions/mount_mcp_as_cli.cjs'); + await main(); + - name: Clean credentials + continue-on-error: true + run: bash "${RUNNER_TEMP}/gh-aw/actions/clean_git_credentials.sh" + - name: Audit pre-agent workspace + id: pre_agent_audit + continue-on-error: true + run: bash "${RUNNER_TEMP}/gh-aw/actions/audit_pre_agent_workspace.sh" + - name: Execute Claude Code CLI + id: agentic_execution + # Allowed tools (sorted): + # - Bash(cat) + # - Bash(cat:*) + # - Bash(date) + # - Bash(date:*) + # - Bash(echo) + # - Bash(git add:*) + # - Bash(git branch:*) + # - Bash(git checkout:*) + # - Bash(git commit:*) + # - Bash(git merge:*) + # - Bash(git rm:*) + # - Bash(git status) + # - Bash(git switch:*) + # - Bash(git:*) + # - Bash(grep) + # - Bash(head) + # - Bash(ls) + # - Bash(ls:*) + # - Bash(mkdir:*) + # - Bash(node:*) + # - Bash(npx:*) + # - Bash(printf) + # - Bash(pwd) + # - Bash(safeoutputs:*) + # - Bash(sort) + # - Bash(tail) + # - Bash(uniq) + # - Bash(wc) + # - Bash(yq) + # - BashOutput + # - Edit + # - Edit(/tmp/*) + # - Edit(/tmp/gh-aw/agent/*) + # - ExitPlanMode + # - Glob + # - Grep + # - KillBash + # - LS + # - MultiEdit + # - MultiEdit(/tmp/*) + # - MultiEdit(/tmp/gh-aw/agent/*) + # - NotebookEdit + # - NotebookRead + # - Read + # - Read(/tmp/*) + # - Read(/tmp/gh-aw/agent/*) + # - Task + # - TodoWrite + # - Write + # - Write(/tmp/*) + # - Write(/tmp/gh-aw/agent/*) + # - mcp__github__actions_get + # - mcp__github__actions_list + # - mcp__github__get_code_scanning_alert + # - mcp__github__get_commit + # - mcp__github__get_dependabot_alert + # - mcp__github__get_discussion + # - mcp__github__get_discussion_comments + # - mcp__github__get_file_contents + # - mcp__github__get_job_logs + # - mcp__github__get_label + # - mcp__github__get_latest_release + # - mcp__github__get_me + # - mcp__github__get_notification_details + # - mcp__github__get_pull_request + # - mcp__github__get_pull_request_comments + # - mcp__github__get_pull_request_diff + # - mcp__github__get_pull_request_files + # - mcp__github__get_pull_request_review_comments + # - mcp__github__get_pull_request_reviews + # - mcp__github__get_pull_request_status + # - mcp__github__get_release_by_tag + # - mcp__github__get_secret_scanning_alert + # - mcp__github__get_tag + # - mcp__github__issue_read + # - mcp__github__list_branches + # - mcp__github__list_code_scanning_alerts + # - mcp__github__list_commits + # - mcp__github__list_dependabot_alerts + # - mcp__github__list_discussion_categories + # - mcp__github__list_discussions + # - mcp__github__list_issue_types + # - mcp__github__list_issues + # - mcp__github__list_label + # - mcp__github__list_notifications + # - mcp__github__list_pull_requests + # - mcp__github__list_releases + # - mcp__github__list_secret_scanning_alerts + # - mcp__github__list_starred_repositories + # - mcp__github__list_tags + # - mcp__github__pull_request_read + # - mcp__github__search_code + # - mcp__github__search_issues + # - mcp__github__search_orgs + # - mcp__github__search_pull_requests + # - mcp__github__search_repositories + # - mcp__github__search_users + # - mcp__safeoutputs + timeout-minutes: 20 + run: | + set -o pipefail + printf '%s' "$(date +%s%3N)" > /tmp/gh-aw/agent_cli_start_ms.txt + touch /tmp/gh-aw/agent-step-summary.md + (umask 177 && touch /tmp/gh-aw/agent-stdio.log) + # shellcheck disable=SC2016 + printf '%s\n' '{"$schema":"https://github.com/github/gh-aw-firewall/releases/download/v0.27.42/awf-config.schema.json","network":{"allowDomains":["*.githubusercontent.com","anthropic.com","api.anthropic.com","api.github.com","api.snapcraft.io","archive.ubuntu.com","azure.archive.ubuntu.com","cdn.playwright.dev","codeload.github.com","crl.geotrust.com","crl.globalsign.com","crl.identrust.com","crl.sectigo.com","crl.thawte.com","crl.usertrust.com","crl.verisign.com","crl3.digicert.com","crl4.digicert.com","crls.ssl.com","docs.github.com","files.pythonhosted.org","ghcr.io","github-cloud.githubusercontent.com","github-cloud.s3.amazonaws.com","github.blog","github.com","github.githubassets.com","host.docker.internal","json-schema.org","json.schemastore.org","keyserver.ubuntu.com","lfs.github.com","objects.githubusercontent.com","ocsp.digicert.com","ocsp.geotrust.com","ocsp.globalsign.com","ocsp.identrust.com","ocsp.sectigo.com","ocsp.ssl.com","ocsp.thawte.com","ocsp.usertrust.com","ocsp.verisign.com","packagecloud.io","packages.cloud.google.com","packages.microsoft.com","patch-diff.githubusercontent.com","patchdiff.githubusercontent.com","playwright.download.prss.microsoft.com","ppa.launchpad.net","pypi.org","raw.githubusercontent.com","registry.npmjs.org","s.symcb.com","s.symcd.com","security.ubuntu.com","sentry.io","statsig.anthropic.com","ts-crl.ws.symantec.com","ts-ocsp.ws.symantec.com","www.googleapis.com"],"isolation":true,"topologyAttach":["awmg-mcpg"]},"apiProxy":{"enabled":true,"enableTokenSteering":true,"maxRuns":500,"maxCacheMisses":5,"maxAiCredits":1000,"models":{"agent":["sonnet-6x","gpt-5.4","gpt-5.5","gpt-5.6","gpt-5.3","gemini-pro","any"],"antigravity":["copilot/antigravity*","google/antigravity*","gemini/antigravity*"],"any":["copilot/*","anthropic/*","openai/*","google/*","gemini/*"],"claude":["agent"],"codex":["agent"],"coding":["copilot/gpt-5*codex*","openai/gpt-5*codex*","gpt-5-codex","kimi"],"computer-use":["copilot/*computer-use*","google/*computer-use*","gemini/*computer-use*","openai/*computer-use*"],"copilot":["agent"],"deep-research":["copilot/deep-research*","copilot/o3-deep-research*","copilot/o4-mini-deep-research*","google/deep-research*","gemini/deep-research*","openai/o3-deep-research*","openai/o4-mini-deep-research*"],"fable":["copilot/*fable*","anthropic/*fable*"],"gemini":["agent"],"gemini-3-flash":["copilot/gemini-3*flash*","google/gemini-3*flash*","gemini/gemini-3*flash*"],"gemini-3-pro":["copilot/gemini-3*pro*","google/gemini-3*pro*","google/nano-banana*","gemini/gemini-3*pro*"],"gemini-3.1-flash":["copilot/gemini-3.1*flash*","google/gemini-3.1*flash*","gemini/gemini-3.1*flash*"],"gemini-3.1-pro":["copilot/gemini-3.1*pro*","google/gemini-3.1*pro*","gemini/gemini-3.1*pro*"],"gemini-3.5-flash":["copilot/gemini-3.5*flash*","google/gemini-3.5*flash*","gemini/gemini-3.5*flash*"],"gemini-3.6-flash":["copilot/gemini-3.6*flash*","google/gemini-3.6*flash*","gemini/gemini-3.6*flash*"],"gemini-flash":["copilot/gemini-*flash*","google/gemini-*flash*","gemini/gemini-*flash*"],"gemini-flash-lite":["copilot/gemini-*flash*lite*","google/gemini-*flash*lite*","gemini/gemini-*flash*lite*"],"gemini-omni":["copilot/gemini-omni*","google/gemini-omni*","gemini/gemini-omni*"],"gemini-pro":["copilot/gemini-*pro*","google/gemini-*pro*","gemini/gemini-*pro*"],"gemma":["copilot/gemma*","google/gemma*","gemini/gemma*"],"gpt-5":["copilot/gpt-5*","openai/gpt-5*"],"gpt-5-codex":["copilot/gpt-5*codex*","openai/gpt-5*codex*"],"gpt-5-mini":["copilot/gpt-5*mini*","openai/gpt-5*mini*"],"gpt-5-nano":["copilot/gpt-5*nano*","openai/gpt-5*nano*"],"gpt-5-pro":["copilot/gpt-5*pro*","openai/gpt-5*pro*"],"gpt-5.1":["copilot/gpt-5.1*","openai/gpt-5.1*"],"gpt-5.2":["copilot/gpt-5.2*","openai/gpt-5.2*"],"gpt-5.3":["copilot/gpt-5.3*","openai/gpt-5.3*"],"gpt-5.4":["copilot/gpt-5.4*","openai/gpt-5.4*"],"gpt-5.5":["copilot/gpt-5.5*","openai/gpt-5.5*"],"gpt-5.6":["copilot/gpt-5.6*","openai/gpt-5.6*"],"haiku":["copilot/*haiku*","anthropic/*haiku*"],"image-generation":["copilot/gpt-image*","openai/gpt-image*","openai/chatgpt-image*","copilot/gemini-*image*","google/gemini-*image*","gemini/gemini-*image*","google/imagen*"],"kimi":["copilot/kimi*","openai/kimi*"],"kiwi":["copilot/kiwi*","openai/kiwi*"],"large":["fable","sonnet","gpt-5-pro","gpt-5","gemini-pro"],"lyria":["google/lyria*","gemini/lyria*","copilot/lyria*"],"mai-code":["copilot/MAI-Code*","copilot/mai-code*","openai/MAI-Code*"],"mai-code-1-flash-picker":["copilot/MAI-Code-1-Flash-picker*","copilot/mai-code-1-flash-picker*","openai/MAI-Code-1-Flash-picker*"],"mini":["haiku","gpt-5-mini","gpt-5-nano","gemini-flash-lite"],"nano-banana":["copilot/nano-banana*","google/nano-banana*","gemini/nano-banana*"],"opus":["copilot/*opus*","anthropic/*opus*"],"opusplan":["opus?effort=high"],"raptor-mini":["copilot/raptor*","openai/raptor*"],"reasoning":["copilot/o1*","copilot/o3*","copilot/o4*","openai/o1*","openai/o3*","openai/o4*"],"robotics":["copilot/*robotics*","google/*robotics*","gemini/*robotics*"],"small":["mini"],"small-agent":["haiku","gpt-5-mini","gemini-flash"],"sonnet":["copilot/*sonnet*","anthropic/*sonnet*"],"sonnet-6x":["copilot/*sonnet-4.5*","copilot/*sonnet-4.6*","copilot/*sonnet-5*","copilot/*sonnet-4-5-*","anthropic/*sonnet-4-5-*","copilot/*sonnet-4-6*","anthropic/*sonnet-4-6*","anthropic/*sonnet-5*"],"summarization":["haiku","gpt-5-mini","gemini-flash-lite","mini"],"veo":["google/veo*","gemini/veo*"],"vision":["copilot/gemini-*image*","google/gemini-*image*","gemini/gemini-*image*","copilot/gemini-*flash*","google/gemini-*flash*","gemini/gemini-*flash*"]}},"container":{"imageTag":"0.27.42,squid=sha256:42dfeb649c680a8558cd5423dbc530b653a69413e35ffbe5e71da5d48c94bdf0,agent=sha256:26a8af4e5566485b02f52af59ee03803ae798271a9619d4767e94d07806deb9b,agent-act=sha256:a14ad974484aa518aab83d40f3f141175dfd171d3745e01c092375b970f73a20,api-proxy=sha256:944f2686c9ab9bec338fd14b662461662f77cd12cd0ea8a3e7cb8c0987cd1607,cli-proxy=sha256:da006bf96d2d246dd269d57b233c1798d2ad63d6cd64ca02f7bf71045028781f"},"logging":{"proxyLogsDir":"/tmp/gh-aw/sandbox/firewall/logs","auditDir":"/tmp/gh-aw/sandbox/firewall/audit"}}' > "${RUNNER_TEMP}/gh-aw/awf-config.json" + cp "${RUNNER_TEMP}/gh-aw/awf-config.json" /tmp/gh-aw/awf-config.json + export GH_AW_MODELS_JSON_PATH="/tmp/gh-aw/models.json" + GH_AW_DOCKER_HOST="" + if [[ "${DOCKER_HOST:-}" =~ ^tcp:// ]]; then + GH_AW_DOCKER_HOST="${DOCKER_HOST}" + fi + if [[ "${DOCKER_HOST:-}" =~ ^tcp:// ]]; then + GH_AW_CHROOT_BINARIES_SOURCE_PATH="${RUNNER_TEMP}/gh-aw" GH_AW_CHROOT_IDENTITY_HOME="${RUNNER_TEMP}/gh-aw/home" node "${RUNNER_TEMP}/gh-aw/actions/patch_awf_chroot_config.cjs" + fi + GH_AW_TOOL_CACHE_MOUNT="" + GH_AW_TOOL_CACHE="${RUNNER_TOOL_CACHE:?RUNNER_TOOL_CACHE must be set}" + if [ -d "$GH_AW_TOOL_CACHE" ]; then + if [[ "$GH_AW_TOOL_CACHE" != /opt/* ]]; then + GH_AW_TOOL_CACHE_MOUNT="$GH_AW_TOOL_CACHE:$GH_AW_TOOL_CACHE:ro" + fi + fi + # shellcheck disable=SC1003,SC2016,SC2086 + awf --config "${RUNNER_TEMP}/gh-aw/awf-config.json" --container-workdir "${GITHUB_WORKSPACE}" --mount "${RUNNER_TEMP}/gh-aw:${RUNNER_TEMP}/gh-aw:ro" --mount "${RUNNER_TEMP}/gh-aw:/host${RUNNER_TEMP}/gh-aw:ro" --mount "${RUNNER_TEMP}/gh-aw/safeoutputs/upload-artifacts:${RUNNER_TEMP}/gh-aw/safeoutputs/upload-artifacts:rw" ${GH_AW_TOOL_CACHE_MOUNT:+--mount "$GH_AW_TOOL_CACHE_MOUNT"} ${GH_AW_DOCKER_HOST:+--docker-host "$GH_AW_DOCKER_HOST"} --tty --env-all --exclude-env ANTHROPIC_API_KEY --exclude-env GITHUB_MCP_SERVER_TOKEN --exclude-env MCP_GATEWAY_API_KEY --log-level info --skip-pull \ + -- /bin/bash -c 'set +o histexpand; export PATH="${RUNNER_TEMP}/gh-aw/mcp-cli/bin:$PATH" && : "${RUNNER_TOOL_CACHE:?RUNNER_TOOL_CACHE must be set}"; GH_AW_TOOL_CACHE="$RUNNER_TOOL_CACHE"; export PATH="$(find "$GH_AW_TOOL_CACHE" -maxdepth 5 -type d -name bin 2>/dev/null | tr '\''\n'\'' '\'':'\'')$PATH"; [ -n "$GOROOT" ] && export PATH="$GOROOT/bin:$PATH" || true; [ -n "$ERLANG_HOME" ] && export PATH="$ERLANG_HOME/bin:$PATH" || true && GH_AW_NODE_EXEC="${GH_AW_NODE_BIN:-}"; if [ -z "$GH_AW_NODE_EXEC" ] || [ ! -x "$GH_AW_NODE_EXEC" ]; then GH_AW_NODE_EXEC="$(command -v node 2>/dev/null || true)"; fi; if [ -z "$GH_AW_NODE_EXEC" ]; then echo "node runtime missing on this runner — check runtimes.node in workflow YAML" >&2; exit 127; fi; GH_AW_NPM_GLOBAL_ROOT="$(npm root -g 2>/dev/null || true)"; if [ -n "$GH_AW_NPM_GLOBAL_ROOT" ]; then export NODE_PATH="${GH_AW_NPM_GLOBAL_ROOT}${NODE_PATH:+:${NODE_PATH}}"; fi; "$GH_AW_NODE_EXEC" ${RUNNER_TEMP}/gh-aw/actions/claude_harness.cjs claude --print --no-chrome --allowed-tools '\''Bash(cat),Bash(cat:*),Bash(date),Bash(date:*),Bash(echo),Bash(git add:*),Bash(git branch:*),Bash(git checkout:*),Bash(git commit:*),Bash(git merge:*),Bash(git rm:*),Bash(git status),Bash(git switch:*),Bash(git:*),Bash(grep),Bash(head),Bash(ls),Bash(ls:*),Bash(mkdir:*),Bash(node:*),Bash(npx:*),Bash(printf),Bash(pwd),Bash(safeoutputs:*),Bash(sort),Bash(tail),Bash(uniq),Bash(wc),Bash(yq),BashOutput,Edit,Edit(/tmp/*),Edit(/tmp/gh-aw/agent/*),ExitPlanMode,Glob,Grep,KillBash,LS,MultiEdit,MultiEdit(/tmp/*),MultiEdit(/tmp/gh-aw/agent/*),NotebookEdit,NotebookRead,Read,Read(/tmp/*),Read(/tmp/gh-aw/agent/*),Task,TodoWrite,Write,Write(/tmp/*),Write(/tmp/gh-aw/agent/*),mcp__github__actions_get,mcp__github__actions_list,mcp__github__get_code_scanning_alert,mcp__github__get_commit,mcp__github__get_dependabot_alert,mcp__github__get_discussion,mcp__github__get_discussion_comments,mcp__github__get_file_contents,mcp__github__get_job_logs,mcp__github__get_label,mcp__github__get_latest_release,mcp__github__get_me,mcp__github__get_notification_details,mcp__github__get_pull_request,mcp__github__get_pull_request_comments,mcp__github__get_pull_request_diff,mcp__github__get_pull_request_files,mcp__github__get_pull_request_review_comments,mcp__github__get_pull_request_reviews,mcp__github__get_pull_request_status,mcp__github__get_release_by_tag,mcp__github__get_secret_scanning_alert,mcp__github__get_tag,mcp__github__issue_read,mcp__github__list_branches,mcp__github__list_code_scanning_alerts,mcp__github__list_commits,mcp__github__list_dependabot_alerts,mcp__github__list_discussion_categories,mcp__github__list_discussions,mcp__github__list_issue_types,mcp__github__list_issues,mcp__github__list_label,mcp__github__list_notifications,mcp__github__list_pull_requests,mcp__github__list_releases,mcp__github__list_secret_scanning_alerts,mcp__github__list_starred_repositories,mcp__github__list_tags,mcp__github__pull_request_read,mcp__github__search_code,mcp__github__search_issues,mcp__github__search_orgs,mcp__github__search_pull_requests,mcp__github__search_repositories,mcp__github__search_users,mcp__safeoutputs'\'' --debug-file /tmp/gh-aw/agent-stdio.log --verbose --permission-mode acceptEdits --output-format stream-json --mcp-config "${RUNNER_TEMP}/gh-aw/mcp-config/mcp-servers.json" --prompt-file /tmp/gh-aw/aw-prompts/prompt.txt' 2>&1 | tee -a /tmp/gh-aw/agent-stdio.log + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + ANTHROPIC_MAX_RETRIES: 0 + ANTHROPIC_MODEL: claude-opus-4-8 + BASH_DEFAULT_TIMEOUT_MS: 60000 + BASH_MAX_TIMEOUT_MS: 60000 + CLAUDE_CODE_DISABLE_FAST_MODE: 1 + DISABLE_BUG_COMMAND: 1 + DISABLE_ERROR_REPORTING: 1 + DISABLE_TELEMETRY: 1 + GH_AW_LLM_PROVIDER: anthropic + GH_AW_MAX_TURNS: ${{ vars.GH_AW_DEFAULT_MAX_TURNS || '' }} + GH_AW_MCP_CONFIG: ${{ runner.temp }}/gh-aw/mcp-config/mcp-servers.json + GH_AW_PHASE: agent + GH_AW_PROMPT: /tmp/gh-aw/aw-prompts/prompt.txt + GH_AW_SAFE_OUTPUTS: ${{ steps.set-runtime-paths.outputs.GH_AW_SAFE_OUTPUTS }} + GH_AW_VERSION: v0.83.4 + GITHUB_AW: true + GITHUB_STEP_SUMMARY: /tmp/gh-aw/agent-step-summary.md + GITHUB_WORKSPACE: ${{ github.workspace }} + GIT_AUTHOR_EMAIL: github-actions[bot]@users.noreply.github.com + GIT_AUTHOR_NAME: github-actions[bot] + GIT_COMMITTER_EMAIL: github-actions[bot]@users.noreply.github.com + GIT_COMMITTER_NAME: github-actions[bot] + MCP_TIMEOUT: 120000 + MCP_TOOL_TIMEOUT: 60000 + RUNNER_TEMP: ${{ runner.temp }} + TRACEPARENT: ${{ env.GITHUB_AW_OTEL_TRACE_ID != '' && env.GITHUB_AW_OTEL_PARENT_SPAN_ID != '' && format('00-{0}-{1}-01', env.GITHUB_AW_OTEL_TRACE_ID, env.GITHUB_AW_OTEL_PARENT_SPAN_ID) || '' }} + - name: Detect agent errors + if: always() + id: detect-agent-errors + continue-on-error: true + run: node "${RUNNER_TEMP}/gh-aw/actions/detect_agent_errors.cjs" + - name: Configure Git credentials + env: + GITHUB_REPOSITORY: ${{ github.repository }} + GITHUB_SERVER_URL: ${{ github.server_url }} + GITHUB_TOKEN: ${{ github.token }} + run: bash "${RUNNER_TEMP}/gh-aw/actions/configure_git_credentials.sh" + - name: Stop MCP Gateway + if: always() + continue-on-error: true + env: + MCP_GATEWAY_PORT: ${{ steps.start-mcp-gateway.outputs.gateway-port }} + MCP_GATEWAY_API_KEY: ${{ steps.start-mcp-gateway.outputs.gateway-api-key }} + GATEWAY_PID: ${{ steps.start-mcp-gateway.outputs.gateway-pid }} + run: | + bash "${RUNNER_TEMP}/gh-aw/actions/stop_mcp_gateway.sh" "$GATEWAY_PID" + - name: Redact secrets in logs + if: always() + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/redact_secrets.cjs'); + await main(); + env: + GH_AW_SECRET_NAMES: 'ANTHROPIC_API_KEY,GH_AW_GITHUB_MCP_SERVER_TOKEN,GH_AW_GITHUB_TOKEN,GITHUB_TOKEN,KHAN_ACTIONS_BOT_TOKEN' + SECRET_ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + SECRET_GH_AW_GITHUB_MCP_SERVER_TOKEN: ${{ secrets.GH_AW_GITHUB_MCP_SERVER_TOKEN }} + SECRET_GH_AW_GITHUB_TOKEN: ${{ secrets.GH_AW_GITHUB_TOKEN }} + SECRET_GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + SECRET_KHAN_ACTIONS_BOT_TOKEN: ${{ secrets.KHAN_ACTIONS_BOT_TOKEN }} + - name: Append agent step summary + if: always() + run: bash "${RUNNER_TEMP}/gh-aw/actions/append_agent_step_summary.sh" + - name: Copy Safe Outputs + if: always() + env: + GH_AW_SAFE_OUTPUTS: ${{ steps.set-runtime-paths.outputs.GH_AW_SAFE_OUTPUTS }} + run: | + mkdir -p /tmp/gh-aw + cp "$GH_AW_SAFE_OUTPUTS" /tmp/gh-aw/safeoutputs.jsonl 2>/dev/null || true + - name: Ingest agent output + id: collect_output + if: always() + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_SAFE_OUTPUTS: ${{ steps.set-runtime-paths.outputs.GH_AW_SAFE_OUTPUTS }} + GH_AW_ALLOWED_DOMAINS: "*.githubusercontent.com,anthropic.com,api.anthropic.com,api.github.com,api.snapcraft.io,archive.ubuntu.com,azure.archive.ubuntu.com,cdn.playwright.dev,codeload.github.com,crl.geotrust.com,crl.globalsign.com,crl.identrust.com,crl.sectigo.com,crl.thawte.com,crl.usertrust.com,crl.verisign.com,crl3.digicert.com,crl4.digicert.com,crls.ssl.com,docs.github.com,files.pythonhosted.org,ghcr.io,github-cloud.githubusercontent.com,github-cloud.s3.amazonaws.com,github.blog,github.com,github.githubassets.com,host.docker.internal,json-schema.org,json.schemastore.org,keyserver.ubuntu.com,khanacademy.atlassian.net,khanacademy.dev,khanacademy.org,lfs.github.com,localhost,objects.githubusercontent.com,ocsp.digicert.com,ocsp.geotrust.com,ocsp.globalsign.com,ocsp.identrust.com,ocsp.sectigo.com,ocsp.ssl.com,ocsp.thawte.com,ocsp.usertrust.com,ocsp.verisign.com,packagecloud.io,packages.cloud.google.com,packages.microsoft.com,patch-diff.githubusercontent.com,patchdiff.githubusercontent.com,playwright.download.prss.microsoft.com,ppa.launchpad.net,pypi.org,raw.githubusercontent.com,registry.npmjs.org,s.symcb.com,s.symcd.com,security.ubuntu.com,sentry.io,statsig.anthropic.com,ts-crl.ws.symantec.com,ts-ocsp.ws.symantec.com,www.googleapis.com" + GITHUB_SERVER_URL: ${{ github.server_url }} + GITHUB_API_URL: ${{ github.api_url }} + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/collect_ndjson_output.cjs'); + await main(); + - name: Parse agent logs for step summary + if: always() + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_AGENT_OUTPUT: /tmp/gh-aw/agent-stdio.log + GH_AW_SAFE_OUTPUTS: ${{ steps.set-runtime-paths.outputs.GH_AW_SAFE_OUTPUTS }} + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/parse_claude_log.cjs'); + await main(); + - name: Parse MCP Gateway logs for step summary + if: always() + id: parse-mcp-gateway + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/parse_mcp_gateway_log.cjs'); + await main(); + - name: Print firewall logs + if: always() + continue-on-error: true + env: + AWF_LOGS_DIR: /tmp/gh-aw/sandbox/firewall/logs + run: bash "${RUNNER_TEMP}/gh-aw/actions/print_firewall_logs.sh" --rootless + - name: Parse token usage for step summary + if: always() + continue-on-error: true + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/parse_token_usage.cjs'); + await main(); + - name: Print AWF reflect summary + if: always() + continue-on-error: true + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/awf_reflect_summary.cjs'); + await main(); + - name: Write agent output placeholder if missing + if: always() + run: | + if [ ! -f /tmp/gh-aw/agent_output.json ]; then + echo '{"items":[]}' > /tmp/gh-aw/agent_output.json + fi + # Upload safe-outputs upload-artifact staging for the upload_artifact job + - name: Upload upload-artifact staging + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: safe-outputs-upload-artifacts + path: ${{ runner.temp }}/gh-aw/safeoutputs/upload-artifacts/ + retention-days: 1 + if-no-files-found: ignore + - name: Upload agent artifacts + if: always() + continue-on-error: true + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: agent + path: | + /tmp/gh-aw/aw-prompts/prompt.txt + /tmp/gh-aw/mcp-logs/ + /tmp/gh-aw/proxy-logs/ + !/tmp/gh-aw/proxy-logs/proxy-tls/ + /tmp/gh-aw/agent_usage.json + /tmp/gh-aw/agent-stdio.log + /tmp/gh-aw/pre-agent-audit.txt + /tmp/gh-aw/agent/ + /tmp/gh-aw/github_rate_limits.jsonl + /tmp/gh-aw/safeoutputs.jsonl + /tmp/gh-aw/agent_output.json + /tmp/gh-aw/aw-*.patch + /tmp/gh-aw/aw-*.bundle + /tmp/gh-aw/awf-config.json + /tmp/gh-aw/sandbox/firewall/logs/ + /tmp/gh-aw/sandbox/firewall/audit/ + /tmp/gh-aw/sandbox/firewall/awf-reflect.json + if-no-files-found: ignore + + conclusion: + needs: + - activation + - agent + - detection + - safe_outputs + if: > + always() && (needs.agent.result != 'skipped' || needs.activation.outputs.lockdown_check_failed == 'true' || + needs.activation.outputs.oauth_token_check_failed == 'true' || needs.activation.outputs.stale_lock_file_failed == 'true' || + needs.activation.outputs.secret_verification_result == 'failed' || needs.activation.outputs.daily_ai_credits_exceeded == 'true') + runs-on: ubuntu-slim + permissions: + contents: write + issues: write + pull-requests: write + concurrency: + group: "gh-aw-conclusion-autofix" + cancel-in-progress: false + queue: max + env: + GH_AW_RUNTIME_FEATURES: ${{ vars.GH_AW_RUNTIME_FEATURES }} + outputs: + incomplete_count: ${{ steps.report_incomplete.outputs.incomplete_count }} + noop_message: ${{ steps.noop.outputs.noop_message }} + tools_reported: ${{ steps.missing_tool.outputs.tools_reported }} + total_count: ${{ steps.missing_tool.outputs.total_count }} + steps: + - name: Setup Scripts + id: setup + uses: github/gh-aw-actions/setup@e89c65e17eb281bbd5ff2ff9e9199a03e96654c7 # v0.83.4 + with: + destination: ${{ runner.temp }}/gh-aw/actions + job-name: ${{ github.job }} + trace-id: ${{ needs.activation.outputs.setup-trace-id }} + parent-span-id: ${{ needs.activation.outputs.setup-parent-span-id || needs.activation.outputs.setup-span-id }} + env: + GH_AW_SETUP_WORKFLOW_NAME: "PR Autofixer" + GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/autofix.lock.yml@${{ github.ref }} + GH_AW_INFO_VERSION: "2.1.220" + GH_AW_INFO_AWF_VERSION: "v0.27.42" + GH_AW_INFO_BODY_MODIFIED: "false" + GH_AW_INFO_ENGINE_ID: "claude" + - name: Download agent output artifact + id: download-agent-output + continue-on-error: true + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: agent + path: /tmp/gh-aw/ + - name: Setup agent output environment variable + id: setup-agent-output-env + if: steps.download-agent-output.outcome == 'success' + run: | + mkdir -p /tmp/gh-aw/ + find "/tmp/gh-aw/" -type f -print + echo "GH_AW_AGENT_OUTPUT=/tmp/gh-aw/agent_output.json" >> "$GITHUB_OUTPUT" + - name: Download safe outputs items manifest + id: download-safe-outputs-manifest + if: always() + continue-on-error: true + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: safe-outputs-items + path: /tmp/gh-aw/ + - name: Collect usage artifact files + if: always() + continue-on-error: true + run: | + mkdir -p /tmp/gh-aw/usage/agent /tmp/gh-aw/usage/detection + echo "Usage artifact source file status:" + for file in /tmp/gh-aw/aw_info.json /tmp/gh-aw/aw-info.jsonl /tmp/gh-aw/agent_usage.json /tmp/gh-aw/agent_usage.jsonl /tmp/gh-aw/detection_usage.jsonl /tmp/gh-aw/evals/evals.jsonl /tmp/gh-aw/github_rate_limits.jsonl /tmp/gh-aw/sandbox/firewall-audit-logs/api-proxy-logs/token-usage.jsonl /tmp/gh-aw/sandbox/firewall/logs/api-proxy-logs/token-usage.jsonl /tmp/gh-aw/sandbox/firewall/audit/api-proxy-logs/token-usage.jsonl /tmp/gh-aw/threat-detection/sandbox/firewall-audit-logs/api-proxy-logs/token-usage.jsonl /tmp/gh-aw/threat-detection/sandbox/firewall/logs/api-proxy-logs/token-usage.jsonl /tmp/gh-aw/threat-detection/sandbox/firewall/audit/api-proxy-logs/token-usage.jsonl; do + [ -f "$file" ] && echo "FOUND: $file" || echo "MISSING: $file" + done + [ -f /tmp/gh-aw/aw_info.json ] && cp /tmp/gh-aw/aw_info.json /tmp/gh-aw/usage/aw_info.json || true + [ -f /tmp/gh-aw/aw-info.jsonl ] && cp /tmp/gh-aw/aw-info.jsonl /tmp/gh-aw/usage/aw-info.jsonl || true + [ -f /tmp/gh-aw/agent_usage.json ] && cp /tmp/gh-aw/agent_usage.json /tmp/gh-aw/usage/agent_usage.json || true + [ -f /tmp/gh-aw/agent_usage.jsonl ] && cp /tmp/gh-aw/agent_usage.jsonl /tmp/gh-aw/usage/agent_usage.jsonl || true + [ -f /tmp/gh-aw/detection_usage.jsonl ] && cp /tmp/gh-aw/detection_usage.jsonl /tmp/gh-aw/usage/detection_usage.jsonl || true + [ -f /tmp/gh-aw/evals/evals.jsonl ] && cp /tmp/gh-aw/evals/evals.jsonl /tmp/gh-aw/usage/evals.jsonl || true + [ -f /tmp/gh-aw/github_rate_limits.jsonl ] && cp /tmp/gh-aw/github_rate_limits.jsonl /tmp/gh-aw/usage/github_rate_limits.jsonl || true + [ -s /tmp/gh-aw/sandbox/firewall-audit-logs/api-proxy-logs/token-usage.jsonl ] && cp /tmp/gh-aw/sandbox/firewall-audit-logs/api-proxy-logs/token-usage.jsonl /tmp/gh-aw/usage/agent/token_usage.jsonl || true + [ -s /tmp/gh-aw/sandbox/firewall/audit/api-proxy-logs/token-usage.jsonl ] && cp /tmp/gh-aw/sandbox/firewall/audit/api-proxy-logs/token-usage.jsonl /tmp/gh-aw/usage/agent/token_usage.jsonl || true + [ -s /tmp/gh-aw/sandbox/firewall/logs/api-proxy-logs/token-usage.jsonl ] && cp /tmp/gh-aw/sandbox/firewall/logs/api-proxy-logs/token-usage.jsonl /tmp/gh-aw/usage/agent/token_usage.jsonl || true + [ -s /tmp/gh-aw/threat-detection/sandbox/firewall-audit-logs/api-proxy-logs/token-usage.jsonl ] && cp /tmp/gh-aw/threat-detection/sandbox/firewall-audit-logs/api-proxy-logs/token-usage.jsonl /tmp/gh-aw/usage/detection/token_usage.jsonl || true + [ -s /tmp/gh-aw/threat-detection/sandbox/firewall/audit/api-proxy-logs/token-usage.jsonl ] && cp /tmp/gh-aw/threat-detection/sandbox/firewall/audit/api-proxy-logs/token-usage.jsonl /tmp/gh-aw/usage/detection/token_usage.jsonl || true + [ -s /tmp/gh-aw/threat-detection/sandbox/firewall/logs/api-proxy-logs/token-usage.jsonl ] && cp /tmp/gh-aw/threat-detection/sandbox/firewall/logs/api-proxy-logs/token-usage.jsonl /tmp/gh-aw/usage/detection/token_usage.jsonl || true + [ -f /tmp/gh-aw/usage/agent/token_usage.jsonl ] || : > /tmp/gh-aw/usage/agent/token_usage.jsonl + [ -f /tmp/gh-aw/usage/detection/token_usage.jsonl ] || : > /tmp/gh-aw/usage/detection/token_usage.jsonl + mkdir -p /tmp/gh-aw/usage/activity + node "${RUNNER_TEMP}/gh-aw/actions/generate_usage_activity_summary.cjs" + find /tmp/gh-aw/usage -type f -print | sort + - name: Upload usage artifact + if: always() + continue-on-error: true + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: usage + path: | + /tmp/gh-aw/usage/aw_info.json + /tmp/gh-aw/usage/aw-info.jsonl + /tmp/gh-aw/usage/agent_usage.json + /tmp/gh-aw/usage/agent_usage.jsonl + /tmp/gh-aw/usage/detection_usage.jsonl + /tmp/gh-aw/usage/evals.jsonl + /tmp/gh-aw/usage/github_rate_limits.jsonl + /tmp/gh-aw/usage/agent/token_usage.jsonl + /tmp/gh-aw/usage/detection/token_usage.jsonl + /tmp/gh-aw/usage/activity/summary.json + if-no-files-found: ignore + - name: Restore daily AIC usage cache + id: restore-daily-aic-cache-conclusion + if: always() + continue-on-error: true + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + key: agentic-workflow-usage-autofix-${{ github.run_id }} + restore-keys: agentic-workflow-usage-autofix- + path: /tmp/gh-aw/agentic-workflow-usage-cache.jsonl + - name: Write daily AIC usage cache entry + id: write-daily-aic-cache + if: always() + continue-on-error: true + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + with: + github-token: ${{ github.token }} + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context); + const { main } = require('${{ runner.temp }}/gh-aw/actions/write_daily_aic_usage_cache.cjs'); + await main(); + - name: Save daily AIC usage cache + id: save-daily-aic-cache + if: always() + continue-on-error: true + uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + key: agentic-workflow-usage-autofix-${{ github.run_id }} + path: /tmp/gh-aw/agentic-workflow-usage-cache.jsonl + - name: Upload daily AIC usage cache artifact + id: upload-daily-aic-cache + if: always() + continue-on-error: true + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: aic-usage-cache + path: /tmp/gh-aw/agentic-workflow-usage-cache.jsonl + if-no-files-found: ignore + retention-days: 7 + - name: Process no-op messages + id: noop + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_AGENT_OUTPUT: ${{ steps.setup-agent-output-env.outputs.GH_AW_AGENT_OUTPUT }} + GH_AW_NOOP_MAX: "1" + GH_AW_WORKFLOW_NAME: "PR Autofixer" + GH_AW_WORKFLOW_SOURCE: "Khan/actions/workflows/autofix/autofix.md@autofix-v0.0.0" + GH_AW_WORKFLOW_SOURCE_URL: "${{ github.server_url }}/Khan/actions/blob/autofix-v0.0.0/workflows/autofix/autofix.md" + GH_AW_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + GH_AW_AGENT_CONCLUSION: ${{ needs.agent.result }} + GH_AW_NOOP_REPORT_AS_ISSUE: "true" + GH_AW_AIC: ${{ needs.agent.outputs.aic }} + GH_AW_THREAT_DETECTION_AIC: ${{ needs.detection.outputs.aic }} + GH_AW_AMBIENT_CONTEXT: ${{ needs.agent.outputs.ambient_context }} + GH_AW_WORKFLOW_ID: "autofix" + with: + github-token: ${{ secrets.GH_AW_GITHUB_TOKEN || secrets.GITHUB_TOKEN }} + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/handle_noop_message.cjs'); + await main(); + - name: Log detection run + id: detection_runs + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_AGENT_OUTPUT: ${{ steps.setup-agent-output-env.outputs.GH_AW_AGENT_OUTPUT }} + GH_AW_WORKFLOW_NAME: "PR Autofixer" + GH_AW_WORKFLOW_SOURCE: "Khan/actions/workflows/autofix/autofix.md@autofix-v0.0.0" + GH_AW_WORKFLOW_SOURCE_URL: "${{ github.server_url }}/Khan/actions/blob/autofix-v0.0.0/workflows/autofix/autofix.md" + GH_AW_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + GH_AW_DETECTION_CONCLUSION: ${{ needs.detection.outputs.detection_conclusion }} + GH_AW_DETECTION_REASON: ${{ needs.detection.outputs.detection_reason }} + with: + github-token: ${{ secrets.GH_AW_GITHUB_TOKEN || secrets.GITHUB_TOKEN }} + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/handle_detection_runs.cjs'); + await main(); + - name: Record missing tool + id: missing_tool + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_AGENT_OUTPUT: ${{ steps.setup-agent-output-env.outputs.GH_AW_AGENT_OUTPUT }} + GH_AW_MISSING_TOOL_CREATE_ISSUE: "true" + GH_AW_WORKFLOW_NAME: "PR Autofixer" + GH_AW_WORKFLOW_SOURCE: "Khan/actions/workflows/autofix/autofix.md@autofix-v0.0.0" + GH_AW_WORKFLOW_SOURCE_URL: "${{ github.server_url }}/Khan/actions/blob/autofix-v0.0.0/workflows/autofix/autofix.md" + with: + github-token: ${{ secrets.GH_AW_GITHUB_TOKEN || secrets.GITHUB_TOKEN }} + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/missing_tool.cjs'); + await main(); + - name: Record incomplete + id: report_incomplete + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_AGENT_OUTPUT: ${{ steps.setup-agent-output-env.outputs.GH_AW_AGENT_OUTPUT }} + GH_AW_REPORT_INCOMPLETE_CREATE_ISSUE: "true" + GH_AW_WORKFLOW_NAME: "PR Autofixer" + GH_AW_WORKFLOW_SOURCE: "Khan/actions/workflows/autofix/autofix.md@autofix-v0.0.0" + GH_AW_WORKFLOW_SOURCE_URL: "${{ github.server_url }}/Khan/actions/blob/autofix-v0.0.0/workflows/autofix/autofix.md" + with: + github-token: ${{ secrets.GH_AW_GITHUB_TOKEN || secrets.GITHUB_TOKEN }} + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/report_incomplete_handler.cjs'); + await main(); + - name: Handle agent failure + id: handle_agent_failure + if: always() + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_AGENT_OUTPUT: ${{ steps.setup-agent-output-env.outputs.GH_AW_AGENT_OUTPUT }} + GH_AW_WORKFLOW_NAME: "PR Autofixer" + GH_AW_WORKFLOW_SOURCE: "Khan/actions/workflows/autofix/autofix.md@autofix-v0.0.0" + GH_AW_WORKFLOW_SOURCE_URL: "${{ github.server_url }}/Khan/actions/blob/autofix-v0.0.0/workflows/autofix/autofix.md" + GH_AW_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + GH_AW_AGENT_CONCLUSION: ${{ needs.agent.result }} + GH_AW_WORKFLOW_ID: "autofix" + GH_AW_ACTION_FAILURE_ISSUE_EXPIRES_HOURS: "168" + GH_AW_ENGINE_ID: "claude" + GH_AW_SECRET_VERIFICATION_RESULT: ${{ needs.activation.outputs.secret_verification_result }} + GH_AW_CHECKOUT_PR_SUCCESS: ${{ needs.agent.outputs.checkout_pr_success }} + GH_AW_EFFECTIVE_TOKENS: ${{ needs.agent.outputs.effective_tokens || '' }} + GH_AW_AI_CREDITS_RATE_LIMIT_ERROR: ${{ needs.agent.outputs.ai_credits_rate_limit_error || 'false' }} + GH_AW_UNKNOWN_MODEL_AI_CREDITS: ${{ needs.agent.outputs.unknown_model_ai_credits || 'false' }} + GH_AW_AIC: ${{ needs.agent.outputs.aic }} + GH_AW_THREAT_DETECTION_AIC: ${{ needs.detection.outputs.aic }} + GH_AW_MAX_AI_CREDITS: "1000" + GH_AW_INFERENCE_ACCESS_ERROR: ${{ needs.agent.outputs.inference_access_error }} + GH_AW_MCP_POLICY_ERROR: ${{ needs.agent.outputs.mcp_policy_error }} + GH_AW_AGENTIC_ENGINE_TIMEOUT: ${{ needs.agent.outputs.agentic_engine_timeout }} + GH_AW_MODEL_NOT_SUPPORTED_ERROR: ${{ needs.agent.outputs.model_not_supported_error }} + GH_AW_HTTP_400_RESPONSE_ERROR: ${{ needs.agent.outputs.http_400_response_error }} + GH_AW_ENGINE_API_HOSTS: "api.anthropic.com" + GH_AW_CODE_PUSH_FAILURE_ERRORS: ${{ needs.safe_outputs.outputs.code_push_failure_errors }} + GH_AW_CODE_PUSH_FAILURE_COUNT: ${{ needs.safe_outputs.outputs.code_push_failure_count }} + GH_AW_LOCKDOWN_CHECK_FAILED: ${{ needs.activation.outputs.lockdown_check_failed }} + GH_AW_OAUTH_TOKEN_CHECK_FAILED: ${{ needs.activation.outputs.oauth_token_check_failed }} + GH_AW_STALE_LOCK_FILE_FAILED: ${{ needs.activation.outputs.stale_lock_file_failed }} + GH_AW_DAILY_AI_CREDITS_EXCEEDED: ${{ needs.activation.outputs.daily_ai_credits_exceeded }} + GH_AW_DAILY_AI_CREDITS_TOTAL_EFFECTIVE_TOKENS: ${{ needs.activation.outputs.daily_ai_credits_total_effective_tokens }} + GH_AW_DAILY_AI_CREDITS_THRESHOLD: ${{ needs.activation.outputs.daily_ai_credits_threshold }} + GH_AW_GROUP_REPORTS: "false" + GH_AW_FAILURE_REPORT_AS_ISSUE: "true" + GH_AW_MISSING_TOOL_REPORT_AS_FAILURE: "true" + GH_AW_MISSING_DATA_REPORT_AS_FAILURE: "true" + GH_AW_TIMEOUT_MINUTES: "20" + with: + github-token: ${{ secrets.GH_AW_GITHUB_TOKEN || secrets.GITHUB_TOKEN }} + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/handle_agent_failure.cjs'); + await main(); + + detection: + needs: + - activation + - agent + if: always() && needs.agent.result != 'skipped' + runs-on: ubuntu-latest + permissions: + contents: read + env: + GH_AW_RUNTIME_FEATURES: ${{ vars.GH_AW_RUNTIME_FEATURES }} + outputs: + aic: ${{ steps.parse_detection_token_usage.outputs.aic }} + detection_conclusion: ${{ steps.detection_conclusion.outputs.conclusion }} + detection_reason: ${{ steps.detection_conclusion.outputs.reason }} + detection_success: ${{ steps.detection_conclusion.outputs.success }} + steps: + - name: Setup Scripts + id: setup + uses: github/gh-aw-actions/setup@e89c65e17eb281bbd5ff2ff9e9199a03e96654c7 # v0.83.4 + with: + destination: ${{ runner.temp }}/gh-aw/actions + job-name: ${{ github.job }} + trace-id: ${{ needs.activation.outputs.setup-trace-id }} + parent-span-id: ${{ needs.activation.outputs.setup-parent-span-id || needs.activation.outputs.setup-span-id }} + env: + GH_AW_SETUP_WORKFLOW_NAME: "PR Autofixer" + GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/autofix.lock.yml@${{ github.ref }} + GH_AW_INFO_VERSION: "2.1.220" + GH_AW_INFO_AWF_VERSION: "v0.27.42" + GH_AW_INFO_BODY_MODIFIED: "false" + GH_AW_INFO_ENGINE_ID: "claude" + - name: Download agent output artifact + id: download-agent-output + continue-on-error: true + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: agent + path: /tmp/gh-aw/ + - name: Setup agent output environment variable + id: setup-agent-output-env + if: steps.download-agent-output.outcome == 'success' + run: | + mkdir -p /tmp/gh-aw/ + find "/tmp/gh-aw/" -type f -print + echo "GH_AW_AGENT_OUTPUT=/tmp/gh-aw/agent_output.json" >> "$GITHUB_OUTPUT" + - name: Checkout repository for patch context + if: needs.agent.outputs.has_patch == 'true' + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + # --- Threat Detection --- + - name: Clean stale firewall files from agent artifact + run: | + rm -rf /tmp/gh-aw/sandbox/firewall/logs + rm -rf /tmp/gh-aw/sandbox/firewall/audit + - name: Download container images + run: bash "${RUNNER_TEMP}/gh-aw/actions/download_docker_images.sh" ghcr.io/github/gh-aw-firewall/agent:0.27.42@sha256:26a8af4e5566485b02f52af59ee03803ae798271a9619d4767e94d07806deb9b ghcr.io/github/gh-aw-firewall/api-proxy:0.27.42@sha256:944f2686c9ab9bec338fd14b662461662f77cd12cd0ea8a3e7cb8c0987cd1607 ghcr.io/github/gh-aw-firewall/squid:0.27.42@sha256:42dfeb649c680a8558cd5423dbc530b653a69413e35ffbe5e71da5d48c94bdf0 + - name: Check if detection needed + id: detection_guard + if: always() + env: + OUTPUT_TYPES: ${{ needs.agent.outputs.output_types }} + HAS_PATCH: ${{ needs.agent.outputs.has_patch }} + run: | + if [[ -n "$OUTPUT_TYPES" || "$HAS_PATCH" == "true" ]]; then + echo "run_detection=true" >> "$GITHUB_OUTPUT" + echo "Detection will run: output_types=$OUTPUT_TYPES, has_patch=$HAS_PATCH" + else + echo "run_detection=false" >> "$GITHUB_OUTPUT" + echo "Detection skipped: no agent outputs or patches to analyze" + fi + - name: Clear MCP Config for detection + if: always() && steps.detection_guard.outputs.run_detection == 'true' + run: | + rm -f "${RUNNER_TEMP}/gh-aw/mcp-config/mcp-servers.json" + rm -f "$HOME/.copilot/mcp-config.json" + rm -f "$GITHUB_WORKSPACE/.gemini/settings.json" + - name: Prepare threat detection files + if: always() && steps.detection_guard.outputs.run_detection == 'true' + run: | + mkdir -p /tmp/gh-aw/threat-detection/aw-prompts + rm -f /tmp/gh-aw/agent_usage.json + cp /tmp/gh-aw/aw-prompts/prompt.txt /tmp/gh-aw/threat-detection/aw-prompts/prompt.txt 2>/dev/null || true + if [ ! -s /tmp/gh-aw/threat-detection/aw-prompts/prompt.txt ]; then + echo "::warning::ERR_VALIDATION: Missing or empty detection context prompt at /tmp/gh-aw/threat-detection/aw-prompts/prompt.txt. Ensure the agent artifact includes /tmp/gh-aw/aw-prompts/prompt.txt. Detection will continue with fallback workflow context." + fi + cp /tmp/gh-aw/agent_output.json /tmp/gh-aw/threat-detection/agent_output.json 2>/dev/null || true + for f in /tmp/gh-aw/aw-*.patch; do + [ -f "$f" ] && cp "$f" /tmp/gh-aw/threat-detection/ 2>/dev/null || true + done + for f in /tmp/gh-aw/aw-*.bundle; do + [ -f "$f" ] && cp "$f" /tmp/gh-aw/threat-detection/ 2>/dev/null || true + done + echo "Prepared threat detection files:" + ls -la /tmp/gh-aw/threat-detection/ 2>/dev/null || true + - name: Setup threat detection + if: always() && steps.detection_guard.outputs.run_detection == 'true' + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + WORKFLOW_NAME: "PR Autofixer" + WORKFLOW_DESCRIPTION: "Addresses the PR reviewer's own feedback on demand, one run per arming. Arm it with an `/autofix [blocking|nits]` comment, or with an `autofix: blocking` / `autofix: nits` label; the two are peers. The run fixes the reviewer's open threads in that scope, pushes one commit, replies in each thread, and clears the label if one armed it." + HAS_PATCH: ${{ needs.agent.outputs.has_patch }} + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/setup_threat_detection.cjs'); + await main(); + - name: Ensure threat-detection directory and log + if: always() && steps.detection_guard.outputs.run_detection == 'true' + run: | + mkdir -p /tmp/gh-aw/threat-detection + touch /tmp/gh-aw/threat-detection/detection.log + - name: Setup Node.js + uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: '24' + package-manager-cache: false + - name: Install AWF binary + run: bash "${RUNNER_TEMP}/gh-aw/actions/install_awf_binary.sh" v0.27.42 + - name: Install Claude Code CLI + run: npm install -g @anthropic-ai/claude-code@2.1.220 + - name: Execute Claude Code CLI + if: always() && steps.detection_guard.outputs.run_detection == 'true' + continue-on-error: true + id: detection_agentic_execution + # Allowed tools (sorted): + # - Bash + # - BashOutput + # - Edit(/tmp/*) + # - ExitPlanMode + # - Glob + # - Grep + # - KillBash + # - LS + # - MultiEdit(/tmp/*) + # - NotebookRead + # - Read + # - Read(/tmp/*) + # - Task + # - TodoWrite + # - Write(/tmp/*) + timeout-minutes: 20 + run: | + set -o pipefail + printf '%s' "$(date +%s%3N)" > /tmp/gh-aw/agent_cli_start_ms.txt + touch /tmp/gh-aw/agent-step-summary.md + (umask 177 && touch /tmp/gh-aw/threat-detection/detection.log) + GH_AW_MAX_AI_CREDITS="${{ vars.GH_AW_DEFAULT_DETECTION_MAX_AI_CREDITS || '400' }}" + printf '%s\n' "{\"\$schema\":\"https://github.com/github/gh-aw-firewall/releases/download/v0.27.42/awf-config.schema.json\",\"network\":{\"allowDomains\":[\"*.githubusercontent.com\",\"anthropic.com\",\"api.anthropic.com\",\"api.github.com\",\"api.snapcraft.io\",\"archive.ubuntu.com\",\"azure.archive.ubuntu.com\",\"cdn.playwright.dev\",\"codeload.github.com\",\"crl.geotrust.com\",\"crl.globalsign.com\",\"crl.identrust.com\",\"crl.sectigo.com\",\"crl.thawte.com\",\"crl.usertrust.com\",\"crl.verisign.com\",\"crl3.digicert.com\",\"crl4.digicert.com\",\"crls.ssl.com\",\"files.pythonhosted.org\",\"ghcr.io\",\"github-cloud.githubusercontent.com\",\"github-cloud.s3.amazonaws.com\",\"github.com\",\"host.docker.internal\",\"json-schema.org\",\"json.schemastore.org\",\"keyserver.ubuntu.com\",\"lfs.github.com\",\"objects.githubusercontent.com\",\"ocsp.digicert.com\",\"ocsp.geotrust.com\",\"ocsp.globalsign.com\",\"ocsp.identrust.com\",\"ocsp.sectigo.com\",\"ocsp.ssl.com\",\"ocsp.thawte.com\",\"ocsp.usertrust.com\",\"ocsp.verisign.com\",\"packagecloud.io\",\"packages.cloud.google.com\",\"packages.microsoft.com\",\"playwright.download.prss.microsoft.com\",\"ppa.launchpad.net\",\"pypi.org\",\"raw.githubusercontent.com\",\"registry.npmjs.org\",\"s.symcb.com\",\"s.symcd.com\",\"security.ubuntu.com\",\"sentry.io\",\"statsig.anthropic.com\",\"ts-crl.ws.symantec.com\",\"ts-ocsp.ws.symantec.com\"]},\"apiProxy\":{\"enabled\":true,\"enableTokenSteering\":true,\"maxRuns\":500,\"maxAiCredits\":${GH_AW_MAX_AI_CREDITS},\"maxCacheMisses\":5,\"models\":{\"agent\":[\"sonnet-6x\",\"gpt-5.4\",\"gpt-5.5\",\"gpt-5.6\",\"gpt-5.3\",\"gemini-pro\",\"any\"],\"antigravity\":[\"copilot/antigravity*\",\"google/antigravity*\",\"gemini/antigravity*\"],\"any\":[\"copilot/*\",\"anthropic/*\",\"openai/*\",\"google/*\",\"gemini/*\"],\"claude\":[\"agent\"],\"codex\":[\"agent\"],\"coding\":[\"copilot/gpt-5*codex*\",\"openai/gpt-5*codex*\",\"gpt-5-codex\",\"kimi\"],\"computer-use\":[\"copilot/*computer-use*\",\"google/*computer-use*\",\"gemini/*computer-use*\",\"openai/*computer-use*\"],\"copilot\":[\"agent\"],\"deep-research\":[\"copilot/deep-research*\",\"copilot/o3-deep-research*\",\"copilot/o4-mini-deep-research*\",\"google/deep-research*\",\"gemini/deep-research*\",\"openai/o3-deep-research*\",\"openai/o4-mini-deep-research*\"],\"fable\":[\"copilot/*fable*\",\"anthropic/*fable*\"],\"gemini\":[\"agent\"],\"gemini-3-flash\":[\"copilot/gemini-3*flash*\",\"google/gemini-3*flash*\",\"gemini/gemini-3*flash*\"],\"gemini-3-pro\":[\"copilot/gemini-3*pro*\",\"google/gemini-3*pro*\",\"google/nano-banana*\",\"gemini/gemini-3*pro*\"],\"gemini-3.1-flash\":[\"copilot/gemini-3.1*flash*\",\"google/gemini-3.1*flash*\",\"gemini/gemini-3.1*flash*\"],\"gemini-3.1-pro\":[\"copilot/gemini-3.1*pro*\",\"google/gemini-3.1*pro*\",\"gemini/gemini-3.1*pro*\"],\"gemini-3.5-flash\":[\"copilot/gemini-3.5*flash*\",\"google/gemini-3.5*flash*\",\"gemini/gemini-3.5*flash*\"],\"gemini-3.6-flash\":[\"copilot/gemini-3.6*flash*\",\"google/gemini-3.6*flash*\",\"gemini/gemini-3.6*flash*\"],\"gemini-flash\":[\"copilot/gemini-*flash*\",\"google/gemini-*flash*\",\"gemini/gemini-*flash*\"],\"gemini-flash-lite\":[\"copilot/gemini-*flash*lite*\",\"google/gemini-*flash*lite*\",\"gemini/gemini-*flash*lite*\"],\"gemini-omni\":[\"copilot/gemini-omni*\",\"google/gemini-omni*\",\"gemini/gemini-omni*\"],\"gemini-pro\":[\"copilot/gemini-*pro*\",\"google/gemini-*pro*\",\"gemini/gemini-*pro*\"],\"gemma\":[\"copilot/gemma*\",\"google/gemma*\",\"gemini/gemma*\"],\"gpt-5\":[\"copilot/gpt-5*\",\"openai/gpt-5*\"],\"gpt-5-codex\":[\"copilot/gpt-5*codex*\",\"openai/gpt-5*codex*\"],\"gpt-5-mini\":[\"copilot/gpt-5*mini*\",\"openai/gpt-5*mini*\"],\"gpt-5-nano\":[\"copilot/gpt-5*nano*\",\"openai/gpt-5*nano*\"],\"gpt-5-pro\":[\"copilot/gpt-5*pro*\",\"openai/gpt-5*pro*\"],\"gpt-5.1\":[\"copilot/gpt-5.1*\",\"openai/gpt-5.1*\"],\"gpt-5.2\":[\"copilot/gpt-5.2*\",\"openai/gpt-5.2*\"],\"gpt-5.3\":[\"copilot/gpt-5.3*\",\"openai/gpt-5.3*\"],\"gpt-5.4\":[\"copilot/gpt-5.4*\",\"openai/gpt-5.4*\"],\"gpt-5.5\":[\"copilot/gpt-5.5*\",\"openai/gpt-5.5*\"],\"gpt-5.6\":[\"copilot/gpt-5.6*\",\"openai/gpt-5.6*\"],\"haiku\":[\"copilot/*haiku*\",\"anthropic/*haiku*\"],\"image-generation\":[\"copilot/gpt-image*\",\"openai/gpt-image*\",\"openai/chatgpt-image*\",\"copilot/gemini-*image*\",\"google/gemini-*image*\",\"gemini/gemini-*image*\",\"google/imagen*\"],\"kimi\":[\"copilot/kimi*\",\"openai/kimi*\"],\"kiwi\":[\"copilot/kiwi*\",\"openai/kiwi*\"],\"large\":[\"fable\",\"sonnet\",\"gpt-5-pro\",\"gpt-5\",\"gemini-pro\"],\"lyria\":[\"google/lyria*\",\"gemini/lyria*\",\"copilot/lyria*\"],\"mai-code\":[\"copilot/MAI-Code*\",\"copilot/mai-code*\",\"openai/MAI-Code*\"],\"mai-code-1-flash-picker\":[\"copilot/MAI-Code-1-Flash-picker*\",\"copilot/mai-code-1-flash-picker*\",\"openai/MAI-Code-1-Flash-picker*\"],\"mini\":[\"haiku\",\"gpt-5-mini\",\"gpt-5-nano\",\"gemini-flash-lite\"],\"nano-banana\":[\"copilot/nano-banana*\",\"google/nano-banana*\",\"gemini/nano-banana*\"],\"opus\":[\"copilot/*opus*\",\"anthropic/*opus*\"],\"opusplan\":[\"opus?effort=high\"],\"raptor-mini\":[\"copilot/raptor*\",\"openai/raptor*\"],\"reasoning\":[\"copilot/o1*\",\"copilot/o3*\",\"copilot/o4*\",\"openai/o1*\",\"openai/o3*\",\"openai/o4*\"],\"robotics\":[\"copilot/*robotics*\",\"google/*robotics*\",\"gemini/*robotics*\"],\"small\":[\"mini\"],\"small-agent\":[\"haiku\",\"gpt-5-mini\",\"gemini-flash\"],\"sonnet\":[\"copilot/*sonnet*\",\"anthropic/*sonnet*\"],\"sonnet-6x\":[\"copilot/*sonnet-4.5*\",\"copilot/*sonnet-4.6*\",\"copilot/*sonnet-5*\",\"copilot/*sonnet-4-5-*\",\"anthropic/*sonnet-4-5-*\",\"copilot/*sonnet-4-6*\",\"anthropic/*sonnet-4-6*\",\"anthropic/*sonnet-5*\"],\"summarization\":[\"haiku\",\"gpt-5-mini\",\"gemini-flash-lite\",\"mini\"],\"veo\":[\"google/veo*\",\"gemini/veo*\"],\"vision\":[\"copilot/gemini-*image*\",\"google/gemini-*image*\",\"gemini/gemini-*image*\",\"copilot/gemini-*flash*\",\"google/gemini-*flash*\",\"gemini/gemini-*flash*\"]}},\"container\":{\"imageTag\":\"0.27.42,squid=sha256:42dfeb649c680a8558cd5423dbc530b653a69413e35ffbe5e71da5d48c94bdf0,agent=sha256:26a8af4e5566485b02f52af59ee03803ae798271a9619d4767e94d07806deb9b,agent-act=sha256:a14ad974484aa518aab83d40f3f141175dfd171d3745e01c092375b970f73a20,api-proxy=sha256:944f2686c9ab9bec338fd14b662461662f77cd12cd0ea8a3e7cb8c0987cd1607,cli-proxy=sha256:da006bf96d2d246dd269d57b233c1798d2ad63d6cd64ca02f7bf71045028781f\"},\"logging\":{\"proxyLogsDir\":\"/tmp/gh-aw/sandbox/firewall/logs\",\"auditDir\":\"/tmp/gh-aw/sandbox/firewall/audit\"}}" > "${RUNNER_TEMP}/gh-aw/awf-config.json" + cp "${RUNNER_TEMP}/gh-aw/awf-config.json" /tmp/gh-aw/awf-config.json + export GH_AW_MODELS_JSON_PATH="/tmp/gh-aw/models.json" + GH_AW_DOCKER_HOST="" + if [[ "${DOCKER_HOST:-}" =~ ^tcp:// ]]; then + GH_AW_DOCKER_HOST="${DOCKER_HOST}" + fi + if [[ "${DOCKER_HOST:-}" =~ ^tcp:// ]]; then + _GH_AW_CHROOT_JSON=$(jq -c --arg src "${RUNNER_TEMP}/gh-aw" --arg user "$(id -un)" --argjson uid "$(id -u)" --argjson gid "$(id -g)" --arg home "${RUNNER_TEMP}/gh-aw/home" '.chroot={"binariesSourcePath":$src,"identity":{"user":$user,"uid":$uid,"gid":$gid,"home":$home}}' "${RUNNER_TEMP}/gh-aw/awf-config.json") || { echo "chroot config patch failed" >&2; exit 1; } + printf '%s\n' "$_GH_AW_CHROOT_JSON" > "${RUNNER_TEMP}/gh-aw/awf-config.json" + printf '%s\n' "$_GH_AW_CHROOT_JSON" > "${RUNNER_TEMP}/gh-aw/awf-config.json" + fi + GH_AW_TOOL_CACHE_MOUNT="" + GH_AW_TOOL_CACHE="${RUNNER_TOOL_CACHE:?RUNNER_TOOL_CACHE must be set}" + if [ -d "$GH_AW_TOOL_CACHE" ]; then + if [[ "$GH_AW_TOOL_CACHE" != /opt/* ]]; then + GH_AW_TOOL_CACHE_MOUNT="$GH_AW_TOOL_CACHE:$GH_AW_TOOL_CACHE:ro" + fi + fi + # shellcheck disable=SC1003,SC2016,SC2086 + awf --config "${RUNNER_TEMP}/gh-aw/awf-config.json" --container-workdir "${GITHUB_WORKSPACE}" --mount "${RUNNER_TEMP}/gh-aw:${RUNNER_TEMP}/gh-aw:ro" --mount "${RUNNER_TEMP}/gh-aw:/host${RUNNER_TEMP}/gh-aw:ro" ${GH_AW_TOOL_CACHE_MOUNT:+--mount "$GH_AW_TOOL_CACHE_MOUNT"} ${GH_AW_DOCKER_HOST:+--docker-host "$GH_AW_DOCKER_HOST"} --tty --env-all --exclude-env ANTHROPIC_API_KEY --log-level info --skip-pull \ + -- /bin/bash -c 'set +o histexpand; : "${RUNNER_TOOL_CACHE:?RUNNER_TOOL_CACHE must be set}"; GH_AW_TOOL_CACHE="$RUNNER_TOOL_CACHE"; export PATH="$(find "$GH_AW_TOOL_CACHE" -maxdepth 5 -type d -name bin 2>/dev/null | tr '\''\n'\'' '\'':'\'')$PATH"; [ -n "$GOROOT" ] && export PATH="$GOROOT/bin:$PATH" || true; [ -n "$ERLANG_HOME" ] && export PATH="$ERLANG_HOME/bin:$PATH" || true && GH_AW_NODE_EXEC="${GH_AW_NODE_BIN:-}"; if [ -z "$GH_AW_NODE_EXEC" ] || [ ! -x "$GH_AW_NODE_EXEC" ]; then GH_AW_NODE_EXEC="$(command -v node 2>/dev/null || true)"; fi; if [ -z "$GH_AW_NODE_EXEC" ]; then echo "node runtime missing on this runner — check runtimes.node in workflow YAML" >&2; exit 127; fi; GH_AW_NPM_GLOBAL_ROOT="$(npm root -g 2>/dev/null || true)"; if [ -n "$GH_AW_NPM_GLOBAL_ROOT" ]; then export NODE_PATH="${GH_AW_NPM_GLOBAL_ROOT}${NODE_PATH:+:${NODE_PATH}}"; fi; "$GH_AW_NODE_EXEC" ${RUNNER_TEMP}/gh-aw/actions/claude_harness.cjs claude --print --no-chrome --allowed-tools '\''Bash,BashOutput,Edit(/tmp/*),ExitPlanMode,Glob,Grep,KillBash,LS,MultiEdit(/tmp/*),NotebookRead,Read,Read(/tmp/*),Task,TodoWrite,Write(/tmp/*)'\'' --debug-file /tmp/gh-aw/threat-detection/detection.log --verbose --permission-mode acceptEdits --output-format stream-json --prompt-file /tmp/gh-aw/aw-prompts/prompt.txt' 2>&1 | tee -a /tmp/gh-aw/threat-detection/detection.log + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + ANTHROPIC_MODEL: claude-opus-4-8 + BASH_DEFAULT_TIMEOUT_MS: 60000 + BASH_MAX_TIMEOUT_MS: 60000 + CLAUDE_CODE_DISABLE_FAST_MODE: 1 + DISABLE_BUG_COMMAND: 1 + DISABLE_ERROR_REPORTING: 1 + DISABLE_TELEMETRY: 1 + GH_AW_LLM_PROVIDER: anthropic + GH_AW_MAX_TURNS: ${{ vars.GH_AW_DEFAULT_MAX_TURNS || '' }} + GH_AW_PHASE: detection + GH_AW_PROMPT: /tmp/gh-aw/aw-prompts/prompt.txt + GH_AW_VERSION: v0.83.4 + GITHUB_AW: true + GITHUB_STEP_SUMMARY: /tmp/gh-aw/agent-step-summary.md + GITHUB_WORKSPACE: ${{ github.workspace }} + GIT_AUTHOR_EMAIL: github-actions[bot]@users.noreply.github.com + GIT_AUTHOR_NAME: github-actions[bot] + GIT_COMMITTER_EMAIL: github-actions[bot]@users.noreply.github.com + GIT_COMMITTER_NAME: github-actions[bot] + MCP_TIMEOUT: 120000 + MCP_TOOL_TIMEOUT: 60000 + RUNNER_TEMP: ${{ runner.temp }} + TRACEPARENT: ${{ env.GITHUB_AW_OTEL_TRACE_ID != '' && env.GITHUB_AW_OTEL_PARENT_SPAN_ID != '' && format('00-{0}-{1}-01', env.GITHUB_AW_OTEL_TRACE_ID, env.GITHUB_AW_OTEL_PARENT_SPAN_ID) || '' }} + - name: Parse threat detection token usage for step summary + id: parse_detection_token_usage + if: always() + continue-on-error: true + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_TOKEN_USAGE_SUMMARY_TITLE: Threat Detection Token Usage + with: + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/parse_token_usage.cjs'); + await main(); + - name: Upload threat detection log + if: always() && steps.detection_guard.outputs.run_detection == 'true' + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: detection + path: /tmp/gh-aw/threat-detection/detection.log + if-no-files-found: ignore + - name: Parse and conclude threat detection + id: detection_conclusion + if: always() + continue-on-error: true + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + RUN_DETECTION: ${{ steps.detection_guard.outputs.run_detection }} + DETECTION_AGENTIC_EXECUTION_OUTCOME: ${{ steps.detection_agentic_execution.outcome }} + GH_AW_DETECTION_CONTINUE_ON_ERROR: "true" + with: + script: | + try { + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/parse_threat_detection_results.cjs'); + await main(); + } catch (loadErr) { + const continueOnError = process.env.GH_AW_DETECTION_CONTINUE_ON_ERROR !== 'false'; + const detectionExecutionFailed = process.env.DETECTION_AGENTIC_EXECUTION_OUTCOME === 'failure'; + const msg = 'ERR_SYSTEM: \u274C Unexpected error loading threat detection module: ' + (loadErr && loadErr.message ? loadErr.message : String(loadErr)); + core.error(msg); + core.setOutput('reason', 'parse_error'); + if (continueOnError && !detectionExecutionFailed) { + core.warning('\u26A0\uFE0F ' + msg); + core.setOutput('conclusion', 'warning'); + core.setOutput('success', 'false'); + } else { + core.setOutput('conclusion', 'failure'); + core.setOutput('success', 'false'); + core.setFailed(msg); + } + } + + pre_activation: + if: "(github.event_name != 'issue_comment' || contains(fromJSON('[\"OWNER\",\"MEMBER\",\"COLLABORATOR\"]'), github.event.comment.author_association)) && (((github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository && startsWith(github.event.label.name, 'autofix: ')) || (github.event_name == 'issue_comment' && github.event.issue.pull_request != null && (github.event.comment.body == '/autofix' || startsWith(github.event.comment.body, '/autofix ') || startsWith(github.event.comment.body, '/autofix\n') || startsWith(github.event.comment.body, '/autofix\r') || startsWith(github.event.comment.body, '/autofix\t')))) && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.id == github.repository_id))" + runs-on: ubuntu-slim + env: + GH_AW_RUNTIME_FEATURES: ${{ vars.GH_AW_RUNTIME_FEATURES }} + outputs: + activated: ${{ steps.check_membership.outputs.is_team_member == 'true' }} + matched_command: '' + setup-parent-span-id: ${{ steps.setup.outputs.parent-span-id || steps.setup.outputs.span-id }} + setup-span-id: ${{ steps.setup.outputs.span-id }} + setup-trace-id: ${{ steps.setup.outputs.trace-id }} + steps: + - name: Setup Scripts + id: setup + uses: github/gh-aw-actions/setup@e89c65e17eb281bbd5ff2ff9e9199a03e96654c7 # v0.83.4 + with: + destination: ${{ runner.temp }}/gh-aw/actions + job-name: ${{ github.job }} + env: + GH_AW_SETUP_WORKFLOW_NAME: "PR Autofixer" + GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/autofix.lock.yml@${{ github.ref }} + GH_AW_INFO_VERSION: "2.1.220" + GH_AW_INFO_AWF_VERSION: "v0.27.42" + GH_AW_INFO_BODY_MODIFIED: "false" + GH_AW_INFO_ENGINE_ID: "claude" + - name: Check team membership for workflow + id: check_membership + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_REQUIRED_ROLES: "admin,maintainer,write" + with: + github-token: ${{ secrets.GITHUB_TOKEN }} + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/check_membership.cjs'); + await main(); + + safe_outputs: + needs: + - activation + - agent + - detection + if: (!cancelled()) && needs.agent.result != 'skipped' && needs.detection.result == 'success' + runs-on: ubuntu-slim + permissions: + contents: write + issues: write + pull-requests: write + timeout-minutes: 45 + env: + GH_AW_AGENT_AIC: ${{ needs.agent.outputs.aic }} + GH_AW_AIC: ${{ needs.agent.outputs.aic }} + GH_AW_AMBIENT_CONTEXT: ${{ needs.agent.outputs.ambient_context }} + GH_AW_CALLER_WORKFLOW_ID: "${{ github.repository }}/autofix" + GH_AW_DETECTION_CONCLUSION: ${{ needs.detection.outputs.detection_conclusion }} + GH_AW_DETECTION_REASON: ${{ needs.detection.outputs.detection_reason }} + GH_AW_EFFECTIVE_TOKENS: ${{ needs.agent.outputs.effective_tokens }} + GH_AW_ENGINE_ID: "claude" + GH_AW_ENGINE_MODEL: "claude-opus-4-8" + GH_AW_RUNTIME_FEATURES: ${{ vars.GH_AW_RUNTIME_FEATURES }} + GH_AW_THREAT_DETECTION_AIC: ${{ needs.detection.outputs.aic }} + GH_AW_WORKFLOW_ID: "autofix" + GH_AW_WORKFLOW_NAME: "PR Autofixer" + GH_AW_WORKFLOW_SOURCE: "Khan/actions/workflows/autofix/autofix.md@autofix-v0.0.0" + GH_AW_WORKFLOW_SOURCE_URL: "${{ github.server_url }}/Khan/actions/blob/autofix-v0.0.0/workflows/autofix/autofix.md" + outputs: + code_push_failure_count: ${{ steps.process_safe_outputs.outputs.code_push_failure_count }} + code_push_failure_errors: ${{ steps.process_safe_outputs.outputs.code_push_failure_errors }} + comment_id: ${{ steps.process_safe_outputs.outputs.comment_id }} + comment_url: ${{ steps.process_safe_outputs.outputs.comment_url }} + create_discussion_error_count: ${{ steps.process_safe_outputs.outputs.create_discussion_error_count }} + create_discussion_errors: ${{ steps.process_safe_outputs.outputs.create_discussion_errors }} + process_safe_outputs_processed_count: ${{ steps.process_safe_outputs.outputs.processed_count }} + process_safe_outputs_temporary_id_map: ${{ steps.process_safe_outputs.outputs.temporary_id_map }} + push_commit_sha: ${{ steps.process_safe_outputs.outputs.push_commit_sha }} + push_commit_url: ${{ steps.process_safe_outputs.outputs.push_commit_url }} + upload_artifact_count: ${{ steps.process_safe_outputs.outputs.upload_artifact_count }} + upload_artifact_slot_0_tmp_id: ${{ steps.process_safe_outputs.outputs.slot_0_tmp_id }} + steps: + - name: Setup Scripts + id: setup + uses: github/gh-aw-actions/setup@e89c65e17eb281bbd5ff2ff9e9199a03e96654c7 # v0.83.4 + with: + destination: ${{ runner.temp }}/gh-aw/actions + job-name: ${{ github.job }} + trace-id: ${{ needs.activation.outputs.setup-trace-id }} + parent-span-id: ${{ needs.activation.outputs.setup-parent-span-id || needs.activation.outputs.setup-span-id }} + safe-output-artifact-client: 'true' + env: + GH_AW_SETUP_WORKFLOW_NAME: "PR Autofixer" + GH_AW_CURRENT_WORKFLOW_REF: ${{ github.repository }}/.github/workflows/autofix.lock.yml@${{ github.ref }} + GH_AW_INFO_VERSION: "2.1.220" + GH_AW_INFO_AWF_VERSION: "v0.27.42" + GH_AW_INFO_BODY_MODIFIED: "false" + GH_AW_INFO_ENGINE_ID: "claude" + - name: Download agent output artifact + id: download-agent-output + continue-on-error: true + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: agent + path: /tmp/gh-aw/ + - name: Setup agent output environment variable + id: setup-agent-output-env + if: steps.download-agent-output.outcome == 'success' + run: | + mkdir -p /tmp/gh-aw/ + find "/tmp/gh-aw/" -type f -print + echo "GH_AW_AGENT_OUTPUT=/tmp/gh-aw/agent_output.json" >> "$GITHUB_OUTPUT" + - name: Download patch artifact + continue-on-error: true + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: agent + path: /tmp/gh-aw/ + - name: Checkout repository + if: (!cancelled()) && needs.agent.result != 'skipped' && contains(needs.agent.outputs.output_types, 'push_to_pull_request_branch') + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: true + token: ${{ secrets.KHAN_ACTIONS_BOT_TOKEN }} + - name: Configure Git credentials + if: (!cancelled()) && needs.agent.result != 'skipped' && contains(needs.agent.outputs.output_types, 'push_to_pull_request_branch') + env: + GITHUB_REPOSITORY: ${{ github.repository }} + GITHUB_SERVER_URL: ${{ github.server_url }} + GIT_TOKEN: ${{ secrets.KHAN_ACTIONS_BOT_TOKEN }} + run: bash "${RUNNER_TEMP}/gh-aw/actions/configure_git_credentials.sh" + - name: Configure GH_HOST for enterprise compatibility + id: ghes-host-config + shell: bash + run: | # zizmor: ignore[github-env] - GITHUB_SERVER_URL is set by GitHub Actions, not user input. + # Derive GH_HOST from GITHUB_SERVER_URL so the gh CLI targets the correct + # GitHub instance (GHES/GHEC). On github.com this is a harmless no-op. + GH_HOST="${GITHUB_SERVER_URL#https://}" + GH_HOST="${GH_HOST#http://}" + echo "GH_HOST=${GH_HOST}" >> "$GITHUB_ENV" + - name: Download upload-artifact staging + continue-on-error: true + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: safe-outputs-upload-artifacts + path: ${{ runner.temp }}/gh-aw/safeoutputs/upload-artifacts/ + - name: Process Safe Outputs + id: process_safe_outputs + uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 + env: + GH_AW_AGENT_OUTPUT: ${{ steps.setup-agent-output-env.outputs.GH_AW_AGENT_OUTPUT }} + GH_AW_COMMENT_ID: ${{ needs.activation.outputs.comment_id }} + GH_AW_ALLOWED_DOMAINS: "*.githubusercontent.com,anthropic.com,api.anthropic.com,api.github.com,api.snapcraft.io,archive.ubuntu.com,azure.archive.ubuntu.com,cdn.playwright.dev,codeload.github.com,crl.geotrust.com,crl.globalsign.com,crl.identrust.com,crl.sectigo.com,crl.thawte.com,crl.usertrust.com,crl.verisign.com,crl3.digicert.com,crl4.digicert.com,crls.ssl.com,docs.github.com,files.pythonhosted.org,ghcr.io,github-cloud.githubusercontent.com,github-cloud.s3.amazonaws.com,github.blog,github.com,github.githubassets.com,host.docker.internal,json-schema.org,json.schemastore.org,keyserver.ubuntu.com,khanacademy.atlassian.net,khanacademy.dev,khanacademy.org,lfs.github.com,localhost,objects.githubusercontent.com,ocsp.digicert.com,ocsp.geotrust.com,ocsp.globalsign.com,ocsp.identrust.com,ocsp.sectigo.com,ocsp.ssl.com,ocsp.thawte.com,ocsp.usertrust.com,ocsp.verisign.com,packagecloud.io,packages.cloud.google.com,packages.microsoft.com,patch-diff.githubusercontent.com,patchdiff.githubusercontent.com,playwright.download.prss.microsoft.com,ppa.launchpad.net,pypi.org,raw.githubusercontent.com,registry.npmjs.org,s.symcb.com,s.symcd.com,security.ubuntu.com,sentry.io,statsig.anthropic.com,ts-crl.ws.symantec.com,ts-ocsp.ws.symantec.com,www.googleapis.com" + GITHUB_SERVER_URL: ${{ github.server_url }} + GITHUB_API_URL: ${{ github.api_url }} + GH_AW_SAFE_OUTPUTS_HANDLER_CONFIG: "{\"add_comment\":{\"discussions\":false,\"footer\":false,\"hide_older_comments\":true,\"max\":1,\"target\":\"triggering\"},\"create_report_incomplete_issue\":{},\"missing_data\":{},\"missing_tool\":{},\"noop\":{\"max\":1,\"report-as-issue\":\"true\"},\"push_to_pull_request_branch\":{\"github-token\":\"${{ secrets.KHAN_ACTIONS_BOT_TOKEN }}\",\"if_no_changes\":\"ignore\",\"max\":1,\"max_patch_size\":4096,\"protect_top_level_dot_folders\":true,\"protected_files\":[\"package.json\",\"bun.lockb\",\"bunfig.toml\",\"deno.json\",\"deno.jsonc\",\"deno.lock\",\"global.json\",\"NuGet.Config\",\"Directory.Packages.props\",\"mix.exs\",\"mix.lock\",\"go.mod\",\"go.sum\",\"stack.yaml\",\"stack.yaml.lock\",\"pom.xml\",\"build.gradle\",\"build.gradle.kts\",\"settings.gradle\",\"settings.gradle.kts\",\"gradle.properties\",\"package-lock.json\",\"yarn.lock\",\"pnpm-lock.yaml\",\"npm-shrinkwrap.json\",\"requirements.txt\",\"Pipfile\",\"Pipfile.lock\",\"pyproject.toml\",\"setup.py\",\"setup.cfg\",\"Gemfile\",\"Gemfile.lock\",\"uv.lock\",\"CODEOWNERS\",\"DESIGN.md\",\"README.md\",\"CONTRIBUTING.md\",\"CHANGELOG.md\",\"SECURITY.md\",\"CODE_OF_CONDUCT.md\",\"CLAUDE.md\",\"AGENTS.md\"],\"target\":\"triggering\"},\"remove_labels\":{\"allowed\":[\"autofix: blocking\",\"autofix: nits\",\"autofix: loop\",\"autofix: human\",\"autofix: author\"]},\"reply_to_pull_request_review_comment\":{\"footer\":false,\"max\":20,\"target\":\"triggering\"},\"report_incomplete\":{},\"upload_artifact\":{\"allowed-paths\":[\"out/**\",\"/tmp/gh-aw/autofix/out/**\"],\"max-size-bytes\":104857600,\"max-uploads\":1,\"retention-days\":30}}" + GH_AW_CI_TRIGGER_TOKEN: ${{ secrets.GH_AW_CI_TRIGGER_TOKEN }} + GITHUB_TOKEN: ${{ secrets.KHAN_ACTIONS_BOT_TOKEN }} + with: + github-token: ${{ secrets.GH_AW_GITHUB_TOKEN || secrets.GITHUB_TOKEN }} + script: | + const { setupGlobals } = require('${{ runner.temp }}/gh-aw/actions/setup_globals.cjs'); + setupGlobals(core, github, context, exec, io, getOctokit); + const { main } = require('${{ runner.temp }}/gh-aw/actions/process_safe_outputs.cjs'); + await main(); + - name: Upload Safe Outputs Items + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: safe-outputs-items + path: | + /tmp/gh-aw/safe-output-items.jsonl + /tmp/gh-aw/temporary-id-map.json + /tmp/gh-aw/process-safe-outputs.stdout.log + /tmp/gh-aw/process-safe-outputs.stderr.log + if-no-files-found: ignore diff --git a/.github/workflows/autofix.md b/.github/workflows/autofix.md new file mode 100644 index 00000000..68f10fe3 --- /dev/null +++ b/.github/workflows/autofix.md @@ -0,0 +1,660 @@ +--- +description: > + Addresses the PR reviewer's own feedback on demand, one run per arming. Arm it + with an `/autofix [blocking|nits]` comment, or with an `autofix: blocking` / + `autofix: nits` label; the two are peers. The run fixes the reviewer's open + threads in that scope, pushes one commit, replies in each thread, and clears + the label if one armed it. + +on: + # Two arming surfaces, and they are PEERS — neither is a shorthand for the + # other. A label is state you click; a command is an event you type and can + # pass arguments to. Both resolve through one shared resolver (`scope.ts`), so + # a value can never mean one thing as a label and another as a command. + pull_request: + types: [labeled] + issue_comment: + types: [created] + # Acknowledge an `/autofix` comment immediately, the same way the reviewer + # acknowledges `/review`. Without this the author has no signal between typing + # the command and the summary comment several minutes later. + reaction: eyes + # No status comment: the run posts exactly one summary comment of its own + # (Step 7), and a gh-aw "started/completed" comment on top of that would + # double the noise on a PR that is already carrying a full review. + status-comment: false + # Autofix writes code to someone's branch, so the actor who armed it must be + # able to write to the repo themselves. This is deliberately NOT the + # reviewer's `roles: all` override: the reviewer only reads and comments, and + # its gate is relaxed so a collaborator's push still triggers a review. On the + # comment path this role check is the PRIMARY gate — see below. + roles: [admin, maintainer, write] + +# One gate per surface. Double-quoted YAML so the `\n`/`\r`/`\t` escapes below +# become real characters in the expression rather than literal backslashes. +# +# LABEL PATH. Two cheap gates before the agent starts: +# 1. Same-repo branches only. A fork PR gets no secrets, so the push would +# fail anyway. +# 2. The label that fired this event is an autofix label. Every other label +# addition on the PR is a run we never pay for. +# +# `skip-ai-review` is deliberately NOT a gate here, and that is a decision, not +# an omission. The label stops the reviewer from running again; it does not +# dismiss a review already posted (the reviewer's own `if:` says so: "adding it +# prevents the *next* run"). The reviewer even suggests the label from inside a +# review body it just posted, so "labelled" and "has current findings" is a +# state the workflow steers users into, not a corner case. Reading the label as +# "no AI may act on this PR" would silently swallow an explicit `autofix:` label +# from someone with write access; that opt-in IS the authorisation, while +# autofix only ever runs when a human arms it. Revisit when autofix runs +# automatically: that is when a push nobody asked for becomes possible, and when +# autofix should get its own opt-out rather than borrowing the reviewer's. +# +# COMMAND PATH — deliberately weaker, and worth understanding before you touch +# it. `issue_comment` carries no `github.event.pull_request`, so the fork guard +# CANNOT be evaluated here at all. It is instead enforced in `plan.ts`, which +# refuses a fork from the staged `context.json`. That is real code, not an +# aspiration: an earlier version of this comment claimed the check "moves into +# the plan" while the plan did not implement it, and Khan/actions#298's review +# caught it. The cost is that an `/autofix` on a fork PR the label path would +# have rejected for free still burns a job before refusing. +# +# The gate that actually matters is unaffected: gh-aw's `roles` check above +# still runs, so a comment from someone without write access never reaches the +# agent. +# +# ONE STRUCTURAL LIMIT. `issue_comment` is a repository-level event, so GitHub +# reads the workflow file from the DEFAULT BRANCH, never from the PR's head. +# `/autofix` therefore cannot fire for an install that only exists on a branch, +# which is why every run of the Khan/webapp#41140 trial was `pull_request` and a +# reviewer's `/autofix` comment there did nothing. The command surface can only +# be exercised once this workflow is on the consuming repo's default branch. +# +# The command match is written out longhand rather than using gh-aw's +# `slash_command` trigger. gh-aw's compiled gate only matches the command +# followed by a space, a bare `\n`, or end-of-body, so a comment saved with a +# trailing CRLF — which the GitHub web UI produces when you press Enter after +# the command — never activates the workflow. That silently killed `/review` in +# Khan/webapp#40943. `scope.ts`'s parser tolerates the same shapes; keep the two +# in step. +if: "(github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository && startsWith(github.event.label.name, 'autofix: ')) || (github.event_name == 'issue_comment' && github.event.issue.pull_request != null && (github.event.comment.body == '/autofix' || startsWith(github.event.comment.body, '/autofix ') || startsWith(github.event.comment.body, '/autofix\n') || startsWith(github.event.comment.body, '/autofix\r') || startsWith(github.event.comment.body, '/autofix\t')))" + +permissions: + contents: read + pull-requests: read + +tools: + github: + lockdown: false + min-integrity: none + toolsets: [pull_requests, repos] + edit: + # NOTE THE `:*` SUFFIX. gh-aw's schema documents `"npx *"` (space-star) as + # "command with any args", but it compiles that form to the Claude Code + # permission `Bash(npx)`, which matches ONLY a bare `npx` with no arguments — + # so `npx -y tsx …` is denied and the plan CLI never runs. `"npx:*"` compiles + # to `Bash(npx:*)`, which is the form gh-aw's own defaults use + # (`Bash(git add:*)`). Observed on gh-aw v0.83.4; verify the compiled + # `--allowed-tools` list still carries the `:*` suffix after any gh-aw bump. + # + # Declaring this list at all NARROWS the agent: a workflow with no `bash:` key + # (the reviewer, for one) compiles to unrestricted `Bash`. That is the + # trade being made here deliberately, which is why the list must be right. + bash: + - "git:*" + - "npx:*" + - "node:*" + - "cat:*" + - "ls:*" + - "date:*" + - "mkdir:*" + +safe-outputs: + allowed-domains: + - github.com + - khanacademy.org + - khanacademy.dev + - khanacademy.atlassian.net + + # The commit. `KHAN_ACTIONS_BOT_TOKEN` rather than the default GITHUB_TOKEN is + # load-bearing, not incidental: GitHub does not create workflow runs for + # events triggered by GITHUB_TOKEN, so a push made with it would emit no + # `synchronize` and the reviewer would never re-review the fix. + # + # That re-review is the INTENDED verification for an autofix commit, and it is + # best-effort rather than guaranteed: the chain from this push to a posted + # review has several links and any of them can fail. Khan/webapp#41194 lost + # one to a gh-aw setup failure that was invisible on the PR. So the reason for + # the token is the stronger one: GITHUB_TOKEN guarantees zero re-review, the + # bot token buys a best-effort one. Step 7 states the pending status on the PR + # so the human re-arm can act as the backstop, and the README's "Verification + # is best-effort" carries the measured numbers. + # + # `if-no-changes: ignore` because "the agent decided nothing needed changing" + # is a legitimate outcome that Step 7 already reports in prose; failing the + # job on it would turn a correct no-op into a red X on the PR. + push-to-pull-request-branch: + target: "triggering" + max: 1 + if-no-changes: "ignore" + github-token: ${{ secrets.KHAN_ACTIONS_BOT_TOKEN }} + + # One reply per fixed thread (Step 6). Replying rather than resolving is + # deliberate: `thread-reconciler` already weighs author replies when the next + # review decides whether a thread is settled, and the re-review accountability + # section links threads that are still open. Resolving here would bypass both + # and destroy the record of whether the fix actually worked. + reply-to-pull-request-review-comment: + max: 20 + target: "triggering" + footer: false + + # The run summary (Step 7): what was fixed, what was skipped and why. Exactly + # one per run, and older ones collapse, so a PR armed several times keeps only + # the latest visible. + add-comment: + target: "triggering" + max: 1 + discussions: false + hide-older-comments: true + footer: false + + # The label is a button, not a mode: it is removed on EVERY outcome, including + # a refusal. A label left on after the run reads as "still queued" when + # nothing is, and re-arming is one click. + remove-labels: + allowed: + - "autofix: blocking" + - "autofix: nits" + - "autofix: loop" + - "autofix: human" + - "autofix: author" + + # The plan artifact. `plan.json` is the only record of what the run was asked + # to do versus what it did, which is what the trial reads to score fix quality + # against the next review's findings. + upload-artifact: + max-uploads: 1 + retention-days: 30 + allowed-paths: + - "out/**" + - "/tmp/gh-aw/autofix/out/**" + +network: + allowed: + - defaults + - github + +# HELD AT OPUS 4.8. This should be `claude-opus-5`, matching the roster +# Khan/actions#294 moves the reviewer to. Three live runs on Khan/webapp#41140 +# failed to get there, and the cause is not yet established, so the model stays +# where it demonstrably works rather than where we want it. +# +# WHAT WAS OBSERVED. The api-proxy's AI-credits guard rejects an un-priced model +# with a 400 before the request reaches the model, and `claude-opus-5` is in no +# firewall release's curated pricing table. +# - 30416237794: no fallback configured. 400, as expected. +# - 30421726630: `models.default-ai-credits-pricing` configured, firewall +# v0.27.42 (the compiler default). Staged awf-config.json confirmed to carry +# `apiProxy.defaultAiCreditsPricing: {input: 5, output: 25}`. Still 400. +# - 30422315631: same, firewall pinned to v0.27.27. Config confirmed present, +# image confirmed pulled. Still 400. +# +# WHAT IS ESTABLISHED. The mechanism exists and is documented: awf-config-spec +# 10.7.3 says a configured fallback makes an unresolvable model "proceed +# normally", and `config-mapper.ts` maps the field to +# `AWF_DEFAULT_AI_CREDITS_PRICING` in BOTH v0.27.27 and v0.27.42. So the pin to +# v0.27.27 above was pointless and has been removed; version is not the +# variable. Note also that #294's own `review.lock.yml` contains zero +# occurrences of `claude-opus-5` (only the shared review.md was edited, never +# recompiled), so its "verified" claim is a reading of the spec rather than a +# run, which is consistent with these three failures. +# +# WHAT IS UNTESTED, and the next thing to try. Spec 10.7.1 applies the fallback +# only when the model "cannot be resolved from the curated table or the bundled +# models.dev catalog". These runs supplied BOTH the fallback AND a +# `models.providers.anthropic.models.claude-opus-5.cost` entry (copied from +# #294). If that providers entry makes the model *resolve* to something carrying +# no AI-credits pricing, it would short-circuit the fallback and reject, which +# is exactly the symptom. The untried combination is: keep +# `default-ai-credits-pricing`, DROP the `models.providers` block, leave the +# firewall at the default. One run settles it. +# +# Until then this is a one-line change plus its `models:` block. Do not restore +# them without a run that reaches the model. +engine: + id: claude +model: claude-opus-4-8 + +timeout-minutes: 20 + +# Autofix reads the reviewer's staged artifacts and the reviewer's own label +# taxonomy, so it checks out Khan/actions for both libs at once: one tag, one +# tree, `workflows/autofix/lib` and `workflows/review/lib` guaranteed to be the +# versions that were released together. The ref is rewritten by +# utils/sync-workflow-versions.ts during the release, and +# workflows/autofix/version-sync.test.ts fails CI if it ever drifts from the +# `autofix` package version. +pre-agent-steps: + - name: Check out shared workflow lib (Khan/actions) + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5 + with: + repository: Khan/actions + ref: autofix-v0.0.0 + path: gh-aw-autofix-lib + persist-credentials: false + + # Staging is deterministic and runs BEFORE the agent, so it costs zero + # assistant turns. This follows the reviewer's orchestrator slice 1 (#280): + # anything that never needed model output belongs in a pre-agent step. + # + # The first live run measured why. Staging by prose cost roughly fifteen of + # that run's 131 turns (seven creating a directory, five hand-assembling JSON + # through repeated `node -e` scripts, three reading this workflow's own lib + # source), and turns are what autofix costs: each re-reads the whole context, + # so 61% of the bill was cache reads. Caching was already near-optimal at a + # 40:1 read-to-write ratio; there were simply too many turns. + # + # A staging failure fails this step before any AI credits are spent. + # + # The comment body is passed through `env:` rather than interpolated into + # `run:`. It is attacker-controlled text, and `env:` keeps it out of the + # shell's parse. + - name: Stage the plan's inputs (deterministic) + working-directory: gh-aw-autofix-lib + env: + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + GITHUB_REPOSITORY: ${{ github.repository }} + AUTOFIX_PR_NUMBER: ${{ github.event.pull_request.number || github.event.issue.number }} + AUTOFIX_COMMAND_BODY: ${{ github.event_name == 'issue_comment' && github.event.comment.body || '' }} + run: npx -y tsx workflows/autofix/lib/stage.ts + +# A fix run is a fraction of a review run: no reviewer roster, no lenses, one +# agent editing a bounded set of files. 1000 credits ($10) is the gh-aw default +# and is generous for that shape; the daily ceiling stays on, because unlike +# reviews (which must never be skipped) a deferred autofix costs nothing but a +# re-click. +max-ai-credits: 1000 + +# ───────────────────────────────────────────────────────────────────────────── +# WORKAROUND for a gh-aw bug. Remove once it is fixed upstream. +# +# THE BUG. The safe-outputs job checks out with `actions/checkout` and no +# `fetch-depth`, so it gets a depth-1 shallow clone of `refs/pull/N/merge` — the +# merge commit alone, without its parents. `push_to_pull_request_branch.cjs:935` +# then fetches the PR branch with NO `--depth`: +# +# git fetch origin :refs/remotes/origin/ +# +# The branch tip is a parent of the merge commit and therefore absent, and its +# history never reaches the existing shallow boundary, so git walks the branch +# all the way back: a full-history fetch. On a small repo this is invisible +# (the branch's parent usually IS the boundary). On Khan/webapp it is fatal. +# +# MEASURED, reproducing the exact command against Khan/webapp from a faithful +# depth-1 checkout of refs/pull/41130/merge: +# - as gh-aw runs it: >14 min, 5.6 GB and still climbing, never finished +# (this is what cancelled the safe_outputs job on Khan/webapp#41130) +# - with the filter below: 91 s, ~557 MB, exit 0 +# +# THE FIX WE CANNOT APPLY. `--depth=1` on that fetch. git has no `fetch.depth` +# config, so it cannot be injected; there is no gh-aw option for it (none of the +# 20 `push-to-pull-request-branch` keys touch fetch or checkout); `checkout:` in +# frontmatter configures the AGENT job only; and `pre-agent-steps`/`post-steps` +# cannot add steps to the safe-outputs job. gh-aw's own +# `checkout_pr_branch.cjs:229` does pass `--depth`, and this same file passes +# `--depth=1` at :1001 and `--filter=blob:none` at :1065, so the omission at +# :935 is an oversight rather than a design choice. +# +# WHAT THIS DOES. git honours `GIT_CONFIG_COUNT` / `GIT_CONFIG_KEY_` / +# `GIT_CONFIG_VALUE_` as if passed via `-c`. The handler runs the fetch with +# `env: {...process.env, ...gitAuthEnv}` (`:934`) and `gitAuthEnv` is empty +# (`:343`), so these reach it. We cannot bound the DEPTH, but we can make the +# fetch a partial clone that skips blobs, which is where the bulk sits. +# +# SAFE AGAINST gh-aw's OWN USE. `ensureSafeDirectoryTrust` +# (`git_helpers.cjs:64-76`) reads the existing `GIT_CONFIG_COUNT` and APPENDS +# its `safe.directory` entry at the next free index, so indices 0 and 1 here +# compose with it rather than clobbering it. +# +# KNOWN TRADE-OFFS. +# - Workflow-level `env:` reaches every job, so the agent job's checkout also +# becomes a partial clone and file reads lazily fetch blobs. For an agent +# that reads a handful of files that is fine and probably faster; gh-aw +# itself notes the lazy-fetch cost at `push_to_pull_request_branch.cjs:1062`. +# gh-aw exposes no per-job env, so this cannot be scoped more tightly. +# - The fetch still pulls all tags. `remote.origin.tagOpt=--no-tags` would +# trim more, but it is NOT set here because it has not been measured; add it +# only with a number behind it. +# ───────────────────────────────────────────────────────────────────────────── +env: + GIT_CONFIG_COUNT: "2" + GIT_CONFIG_KEY_0: remote.origin.promisor + GIT_CONFIG_VALUE_0: "true" + GIT_CONFIG_KEY_1: remote.origin.partialclonefilter + GIT_CONFIG_VALUE_1: "blob:none" + +source: Khan/actions/workflows/autofix/autofix.md@autofix-v0.0.0 +--- + +# PR Autofixer + +You address feedback the PR reviewer left on this pull request. You are not a +reviewer: you do not judge whether a finding is worth fixing, and you do not +look for problems nobody raised. Your scope is exactly the threads the plan +hands you in Step 2. + +## Current Context + +- **Repository**: ${{ github.repository }} +- **Pull Request**: #${{ github.event.pull_request.number || github.event.issue.number }} + +This workflow fires on two events, so the PR number is read from whichever one +carries it. What armed the run is not interpolated here on purpose: gh-aw's +expression allowlist excludes `github.event.label.name` from prompt bodies, and +the plan resolves the arming itself in Step 2, which is the authoritative answer +either way. + +## Cost: turns are the expensive thing + +This workflow's cost is dominated by the number of assistant turns, not by how +much data any one of them moves. Every turn re-reads the whole accumulated +context. The first live run took 131 turns and 460 AI credits to land a +six-line fix, with only ~93 KB of tool output in the entire run: the payload was +trivial and the turn count was not. + +So, concretely: + +- **Batch shell work.** One command with `&&` beats three round trips. Create + every directory you need in a single `mkdir -p` and do not verify it + afterwards; `mkdir -p` does not fail on an existing directory, and `ls` to + confirm is a wasted turn. +- **Do not read the lib source.** `gh-aw-autofix-lib` holds released, + tested code. Reading `plan.ts` or `staleness.ts` to work out what the plan + will say costs turns and context, and tempts you to re-derive a decision the + plan has already made. Run the CLI and read its output. +- **Do not explore the `safeoutputs` CLI.** Its invocations are written out at + each step below. Calling `--help` first is a wasted turn. +- **Do not re-verify your own writes.** If a command exits zero, it worked. +- **Prefer a heredoc to a chain of `node -e` scripts.** If you find yourself + writing the same inline script twice with small edits, write it to a file once + and run it. + +## Step 1: Read the staged inputs + +**There is nothing to stage.** A deterministic pre-agent step +(`workflows/autofix/lib/stage.ts`) has already fetched everything and written it +to `/tmp/gh-aw/autofix/` before you started. Do not fetch any of it again, and +do not rewrite any of these files: + +| file | what it holds | +| --- | --- | +| `labels.json` | the PR's current label names | +| `threads.json` | the reviewer's unresolved threads, each with its full reply chain, verbatim | +| `prior-reviews.json` | every review by the reviewer bot, whatever its state | +| `pr.diff` | the PR's unified diff | +| `commits.json` | commit messages on the head, the autofix cycle ledger | +| `head-sha.txt` | the head SHA when the run started; Step 5 compares against it | +| `command.txt` | the `/autofix` comment body, **only** on a command-armed run | + +You do not need to read most of these. Step 2's CLI parses them; the only ones +you will open yourself are `plan.json` (which Step 2 produces) and +`head-sha.txt` (Step 5). Reading `pr.diff` or `threads.json` in full is a waste +of context: the plan already extracted what matters into `plan.items`. + +If a file is missing, the staging step failed and the run should not have +reached you. Say so in Step 7 and stop; do not reconstruct it by hand. + +## Step 2: Build the plan (deterministic code) + +Run the plan CLI once, from the shared lib checkout: + +``` +cd gh-aw-autofix-lib && npx -y tsx workflows/autofix/lib/plan.ts +``` + +It writes `/tmp/gh-aw/autofix/plan.json` and prints a summary. Copy +`plan.json` to `/tmp/gh-aw/autofix/out/plan.json` now, so the run artifact +records what was planned even if a later step fails. + +**The plan is final.** It decided the scope, which threads are in it, which were +skipped and why, and what the commit trailer says. Do not widen it, narrow it, +re-classify a skipped thread, or act on a finding it did not hand you. If you +disagree with the plan, say so in the Step 7 comment; do not act on the +disagreement. + +**If the CLI does not run, the run is over.** If `npx` is unavailable, the +command errors, or the tool call is denied, do **not** reconstruct the plan by +reading the library source and reasoning about it. A hand-simulated plan is +exactly the thing this workflow's determinism boundary exists to prevent, and +it produces a confident-looking result nobody can audit. Instead: change +nothing, post a Step 7 comment saying the plan CLI could not be executed and +quoting the error, remove the labels (Step 8), and stop. + +The plan resolves the arming itself, from whichever surface triggered the run: +`command.txt` when it exists, the PR's labels otherwise. **The trigger decides, +and the two never union** — a stale `autofix: nits` label must not silently +widen someone's `/autofix blocking`. `plan.surface` records which one won. + +`plan.json` has a `status`: + +- **`refused`** — the run cannot proceed safely (an autofix label on an axis + this version does not implement, no reviewer feedback to act on, or a review + that cannot be matched to the current diff). Skip to Step 7, post the comment + with the plan's `reason` verbatim, remove the labels, and stop. Change + nothing. +- **`no-op`** — the labels were understood and there is nothing to fix. Skip to + Step 7 the same way. This is a success, not a failure; say so plainly. +- **`armed`** — continue to Step 3. + +## Step 3: Understand each finding before changing anything + +For every item in `plan.items`, read the file at `path` around `line` from the +Actions workspace (the PR head is checked out; read from disk, not through the +API). The item's `body` is the reviewer's statement of the problem, verbatim. + +Work out what the reviewer meant. If a finding is ambiguous, or you cannot +determine what a correct fix would be, **leave it alone** and record it as +not-fixed for Step 7. A wrong fix on a blocking finding is worse than no fix: +the author now has to review a change they did not write, on a problem they had +not yet looked at. + +## Step 4: Make the fixes + +Edit the files directly in the workspace. Rules, all hard: + +- **Fix the finding, nothing else.** No drive-by refactors, no reformatting, no + fixing things you noticed on the way. Every hunk you write must be traceable + to an item in `plan.items`. +- **Never weaken a test to satisfy a finding.** If the honest fix makes a test + fail, fix the source. If you believe the test itself encodes the bug, leave + the finding unfixed and explain why in Step 7. Deleting an assertion, loosening + a matcher, adding a skip, or widening an expected range to make something pass + is never an acceptable outcome of this workflow. +- **Do not touch files no item points at.** The one exception is a change that + is mechanically forced by a fix (a caller that must be updated for a changed + signature); note any such file in Step 7. +- **Do not amend, rebase, or force-push.** You produce working-tree changes; + the push is a safe output. +- If a fix would require a design decision the reviewer did not make for you, + leave it unfixed and say so. + +**Verify what you can, and be honest about what you cannot.** Nothing between +your edit and the push checks that the code still compiles or that its tests +still pass; the re-review and the repo's CI both run only after the commit is on +the author's branch. If this repository has a cheap check you can run from the +allowlisted commands, run it. If you cannot verify (no allowlisted runner, or +the check needs a toolchain that is not installed), that is expected, not a +failure; say so in the Step 7 summary rather than implying the fix was +validated. An unverified fix presented as a verified one is the thing to +avoid. + +## Step 5: Push one commit + +Compare the live head SHA against `/tmp/gh-aw/autofix/head-sha.txt`, which +staging captured from the job's checkout (`git rev-parse HEAD`) rather than from +the API, because the checkout is what your edits are actually against. +**If it changed, do not push.** The author pushed while you were working, and +your edits are against a base that no longer exists. Skip to Step 7, report that +the run was abandoned for that reason, and remove the labels; the author can +re-label once their push settles. + +Otherwise commit your working-tree changes locally, then emit a single +`push-to-pull-request-branch`. The engine builds its patch and bundle from that +commit, so both steps are needed: + +``` +git -C "$GITHUB_WORKSPACE" add -- +git -C "$GITHUB_WORKSPACE" commit -F /tmp/gh-aw/autofix/commitmsg.txt +printf '{"message":%s}' "$(...)" | safeoutputs push_to_pull_request_branch . +``` + +Write the message to `commitmsg.txt` first rather than passing it inline; it is +multi-paragraph and shell-quoting it is a reliable way to lose the trailer. + +**The commit message must stand on its own.** Someone reading `git log` a year +from now, with no PR open and no reviewer thread to click through to, should be +able to tell what changed and why. Write it as you would any commit; the fact +that a bot wrote it is not the interesting part. + +``` +autofix: + + + +:: `, only when +there is more than one finding; with a single finding the paragraph above has +already said it.> + + +``` + +Rules for the subject line, which is the part that ages worst: + +- **Never** use a generic subject. `autofix: address reviewer feedback` is + banned: it is identical on every run, so a branch with several autofix + commits becomes a wall of indistinguishable `git log --oneline` entries. +- Name the change, not the process. `autofix: clamp Page start into range`, + not `autofix: fix blocking finding` or `autofix: apply review comments`. +- Under 60 characters, imperative mood, no trailing period. + +Do not reference thread ids, run urls, or the reviewer by name in the prose; +that is what the trailer is for. + +The trailer block must be the last paragraph and must be copied exactly as +`plan.json` renders it. It is what a later run reads to know this one happened. +It is machine metadata: never describe it, expand it, or move it into the body. + +## Step 6: Reply in each thread + +For every item you fixed, emit one `reply-to-pull-request-review-comment` on +that item's thread, stating what you changed in one or two sentences. Be +specific: "Renamed to `parsedConfig` and updated the three call sites" beats +"Fixed". + +Do **not** resolve any thread. The next review decides whether the fix settled +the finding; that is the whole verification story for this workflow, it is +best-effort (see Step 7), and resolving here would erase it. + +For an item you deliberately left unfixed (Step 3 or Step 4), reply saying so +and why, in one sentence. A finding that was handed to you and silently skipped +is the one outcome an author cannot debug. + +## Step 7: Post the run summary + +**Post exactly one `add-comment`, on every path through this workflow.** A +refusal, a no-op, a finding left unfixed, an abandoned push, a run that fixed +everything cleanly: all of them get one comment, and never more than one. + +An earlier version of this step stayed quiet when a run was unremarkable, on the +reasoning that a clean run already tells its own story in three other places: +the thread reply on each fixed finding, the commit in the PR timeline, and the +engine's own "Commit pushed" comment, so a fourth notification repeating them is +noise. That reasoning was right about noise and wrong about what still needed +saying. All three of those places record that something *changed*; none of them +records that nothing has *checked* it. Item 8 below is the only place a reader +learns that, so the comment carrying it cannot be optional. + +The quiet branch was therefore retracted deliberately, not lost. It was also +unreachable in practice: it required `plan.degradedNote` to be empty, and the +reviewer's hidden fingerprint stamp is stripped from every posted review (the +README's "Degrading when there is no fingerprint" documents it), so that note is +essentially always set. Do not reintroduce the branch without first answering +where the pending-verification statement goes instead. + +Do **not** try to add a hidden HTML-comment marker of your own. gh-aw's +safe-output ingest strips every XML/HTML comment before posting +(`removeXmlComments` in `sanitize_content_core.cjs`, a depth-tracking scan with +no allowlist), so such a marker is silently deleted. An earlier version of this +step asked for ``; the posted comments never +carried it. Collapsing older comments still works, because the engine adds its +own `gh-aw-workflow-call-id` marker after sanitisation. + +Write the body directly, in this order, including only the parts that apply: + +1. One sentence of plain past-tense prose saying what happened. Take the + substance from the plan's `reason` but write it as a sentence to a person: + `Fixed 1 blocking finding.`, not `fixing 1 blocking finding(s).` Get the + tense and the plural right; the work is already done by the time anyone + reads this. +2. If anything was fixed: a list, one line per finding, `path:line` plus what + changed. Link each to its thread `url` when the item has one. +3. If anything was left unfixed: a list, one line each, with the reason. This + is the most important section in the comment; never omit or soften it. +4. If `plan.skipped` contains entries whose reason is **not** `out-of-scope`: + one line each with the reason (`outdated-anchor`, `unparseable-label`, + `stale-path`). Put any `out-of-scope` entries in a collapsed + `
N thread(s) outside this run's scope` block, or + omit them entirely when the comment already has more urgent content: they + are the expected consequence of the scope the author picked. +5. If `plan.stalePaths` is non-empty, one line: `Files changed since the last + review, so findings in them were not acted on: .` +6. If you could not run any check against your own edit, one line saying so, in + plain terms: `Not verified locally: no test or build command is available to + this workflow.` Never imply a fix was validated when it was not. +7. If `plan.degradedNote` is non-empty, that note **verbatim** on its own line. + Never omit it and never soften it: a weaker check that goes unmentioned is + indistinguishable from the full one. +8. When anything was pushed, last two lines, exactly: + + `Not verified: nothing has checked this commit. Autofix does not resolve its + own threads; whether these fixes settled the findings is decided by the + reviewer's next review of this branch, not by this run.` + + `That re-review can fail, or never trigger at all; a /review comment asks for + one at any time.` + +Items 6 and 8 are different claims and both can appear in the same comment: item +6 is about what this run could check *before* pushing, item 8 about what checks +the commit *after*. Do not merge them or drop one as a duplicate. + +Both of item 8's lines have to still be true a week later, once the re-review +has landed and approved. The first is tensed to the moment of writing, which is +what makes it safe. The second states a standing fact and a standing capability, +not a conditional instruction, because a one-shot run can never come back and +retract one: `If no review appears, comment /review` would leave every verified +PR permanently carrying an instruction to go and trigger a review. The same +objection rules out putting verification state in the commit trailer, where +nothing could ever update it either. + +Write nothing else. No preamble, no summary of the PR, no opinion on the code. +Do not use em dashes; a semicolon, colon, or full stop reads better and matches +the rest of this repo's bot output. + +## Step 8: Remove the labels + +Emit `remove-labels` for every label in the plan's `labelsToRemove`. Do this on +every path through this workflow, including refusals and no-ops. The label is a +button: once the run is over it must be off, so that its presence always means +"queued" and never "already done". + +On a command-armed run `labelsToRemove` is empty, and that is correct, not an +oversight: a comment is self-clearing, and any autofix label sitting on the PR +was not what armed this run. Removing it would clear an intent nobody acted on. +Emit nothing in that case. + +## Step 9: Upload the artifact + +Upload `/tmp/gh-aw/autofix/out/` with `upload-artifact` in one call. diff --git a/workflows/autofix/README.md b/workflows/autofix/README.md new file mode 100644 index 00000000..0f3a0d29 --- /dev/null +++ b/workflows/autofix/README.md @@ -0,0 +1,372 @@ +# `autofix` — opt-in reviewer-feedback autofixer + +Addresses the [`review`](../review) workflow's own feedback on a PR, on demand, +one run per arming. Add a label or comment `/autofix`, get a commit. + +It is deliberately narrow. It fixes findings the reviewer already raised; it +does not review, does not look for problems nobody flagged, and does not resolve +its own threads. + +## Using it + +Two ways to arm it, and they are peers. Neither is a shorthand for the other. + +**Label the PR:** + +| Label | Fixes | +| ------------------ | ---------------------------------------------------------- | +| `autofix: blocking` | The reviewer's open blocking threads (`issue (blocking)`, `issue (blocking, best-practice)`, `todo (blocking)`) | +| `autofix: nits` | The reviewer's open non-blocking threads (suggestions, nitpicks, questions, thoughts, notes) | + +**Or comment on the PR:** + +``` +/autofix # same as /autofix blocking +/autofix nits +/autofix blocking nits +``` + +Prose after the command line is ignored by the parser, so you can leave context +for whoever reads the thread later: + +``` +/autofix blocking + +but keep the existing naming, it matches the RFC +``` + +Both labels may be on at once and both arguments may be given at once; the +scopes union. **The trigger decides, and the two surfaces never union with each +other**: a stale `autofix: nits` label will not widen an explicit +`/autofix blocking`. Whichever one fired the run is the one that is read. + +The run then: + +1. checks that the reviewer's feedback is current for this head, +2. fixes what it can, in one commit pushed to the PR branch, +3. replies in each thread saying what it did (or why it did not), +4. posts one summary comment, but only when it has something non-obvious to + say (see below); a clean run stays quiet, +5. **removes the label** (label-armed runs only). + +The label is a button, not a mode. It comes off on every outcome, including +refusals, so its presence always means "queued" and never "already done". +Re-arming is one click. + +A command-armed run removes nothing, and that is deliberate: a comment is +already self-clearing, and any autofix label sitting on the PR was not what +armed the run. Clearing it would discard an intent nobody acted on. + +The push is made with `KHAN_ACTIONS_BOT_TOKEN`, so it can trigger a re-review. +That re-review is the **intended** verification: autofix never resolves a +thread, and whether a fix actually settled a finding is decided by the next +review, not by the run that wrote it. + +### Verification is best-effort + +Nothing between the fix and the merge gate is guaranteed to check an autofix +commit, and the summary comment says so on every push. + +The chain from the push to a posted review has links, and how many depends on +the consumer. With the shared push-triggered reviewer it is two (push → +`synchronize` → reviewer). In Khan/webapp, where the reviewer is an +`issue_comment` local override, it is four (push → `synchronize` → +`review-kore-prs.yml` posts `/review` → reviewer). Any link can fail +independently of gh-aw. + +Whether a break is *visible* also depends on the consumer, and this is the part +worth knowing before trialling autofix in a new repo: + +- **Push-triggered reviewer:** the reviewer's jobs join the PR's check suite, so + a failed re-review is a red X on the commit autofix pushed. +- **`issue_comment`-triggered reviewer:** the run's head SHA is a default-branch + merge commit, so it never joins the PR's check suite at all. With + `status-comment: false` it posts nothing either. gh-aw's own fallback + (`failure-report-as-issue`) is then the last line of defence, and it is + unavailable in a repo with issues disabled. + +In Khan/webapp all three of those conditions hold at once: the reviewer is an +`issue_comment` override, `status-comment` is false, and issues are disabled. On +#41194 the re-review of the autofix commit `ad8da8d4` failed in `Install AWF +binary`, before the model ran, so it cost zero AI credits and produced no +output; gh-aw tried to file its failure issue and got `410 Issues has been +disabled in this repository`. The only trace +on the PR was the 👀 the activation job had already put on the `/review` +comment, which is indistinguishable from "still running". The commit sat +unverified for 36 minutes, and the human who eventually re-triggered it found it +by querying the Actions runs list, not from anything on the PR. This is not a +rare shape: of that repo's last 100 reviewer runs, 15 of the 53 that started +ended in `failure`. + +**The human re-arming loop is the backstop, and that is accepted for v1.** It is +the same loop that arms autofix in the first place. What v1 owes it is the +pending statement, not machinery. + +A detector is deferred, not blocked, and needs nothing added here. It does not +require reading the trailer back: the question is "does a review by the reviewer +bot exist whose `commit_id` is this autofix commit or later", answerable from +the SHA alone. What it needs is a home that runs *later* than the autofix run, +which a one-shot workflow does not have. Note also that verification state must +never go **in** the trailer: nothing in v1 could ever flip an `Autofix-Verified: +pending` field, so it would sit permanently wrong on every commit that was in +fact verified. + +## The axis model + +Both surfaces share one currency: the **token**, the value after the namespace. +`autofix: blocking` and `/autofix blocking` carry the same token, resolve +through the same function (`scope.ts` `resolveTokens`), and cannot drift. + +The token space is flat while the semantics are not. Three axes exist; a token +names a value on exactly one of them: + +| Axis | Tokens | Combination rule | Implemented | +| ----------- | ----------------------------- | ---------------------- | ----------- | +| **scope** | `blocking`, `nits` | union | yes | +| **cadence** | `loop` (absent = once) | flag | no | +| **source** | `human`, `author` (absent = the reviewer bot) | union | no | + +Read this before adding a token to the vocabulary. `nits` and `loop` look like +peers and are not, and the day both are requested the rule that resolves them +has to already exist. + +Because unioning happens *within* an axis, the vocabulary stays bounded by the +axes (five tokens across all three) rather than growing as their product. There +is no `blocking-loop` token and there must never be one. + +A token on an unimplemented axis is **rejected, not ignored** (`scope.ts` +`UNIMPLEMENTED_TOKENS`). Honouring the blocking half of `blocking + loop` would +present as a loop that mysteriously stopped after one cycle, which is worse than +a clear refusal. + +A bare `/autofix` means `blocking`, the scope that terminates at the merge gate; +a bare command must not silently do the open-ended thing. There is deliberately +no bare `autofix` label equivalent, since a label carries no arguments and the +two forms would be indistinguishable at a glance. + +### One constraint that outlives v1: nits never loop + +`isLoopEligible` is enforced in code rather than left to convention. +Non-blocking findings have no fixed point — the reviewer will always find +something cosmetic in the autofixer's own output — so a nits-scoped loop cannot +converge. Blocking scope terminates naturally at the merge gate, which is why it +is the scope a cadence axis would be built on. + +## What it refuses to do + +Every refusal fails closed: when the run cannot establish that acting is safe, +it does nothing, clears any label that armed it, and says why. + +- **No reviewer feedback at all.** Nothing to fix. This is the *only* currency + state that refuses. +- **The review does not match this head.** Currency is checked against the + reviewer's own hidden fingerprint stamp (`review.md` Step 6), which survives + force-pushes and rebases because it hashes added-line content rather than + SHAs. The check is **per file**: if the author pushed one unrelated fix after + the review, findings in the files that did not change are still fixed, and + only the affected ones are dropped. An all-or-nothing gate would refuse + routine PRs constantly. +- **The thread's label will not parse.** Note this fails *closed in the opposite + direction* from `rereview.ts`, where an unparseable label is treated as + blocking so the thread is kept. Here an unclassifiable finding is excluded, + because the risk being managed is an agent editing code on the strength of a + finding it could not classify. +- **The thread is outdated** (GitHub reports no anchor line). The code the + finding was written about is gone. +- **The head moved while the run was working.** The edits are against a base + that no longer exists, so the push is abandoned. + +### Degrading when there is no fingerprint (the normal case) + +If the reviewer's review carries no diff fingerprint, the file-level check +cannot run. Autofix **degrades rather than refusing**, and says so in the +summary. + +This is not an edge case. gh-aw's safe-output ingest strips every XML/HTML +comment before a review posts (`removeXmlComments` in +`sanitize_content_core.cjs`), so the reviewer's hidden stamp is deleted on the +way out and has never reached a posted review; Khan/actions#287 documents it end +to end and gives the reviewer a second carrier in its cache-memory record. That +carrier is not reachable from here, because cache memory is scoped per workflow +and autofix is a different workflow. So the per-thread anchor check is what +autofix actually runs on, and the fingerprint branch is the optimisation. + +An earlier version refused outright, which made autofix unusable against the +reviewer as actually deployed: on Khan/webapp#41130 the reviewer posted a correct +blocking finding under a body of exactly `Changes requested — see inline +comments.` with no stamp, and autofix refused every time while reporting "no +reviewer feedback has been posted on this PR". + +The fingerprint is not the only currency signal and not even the primary one. +GitHub marks a review comment outdated when the diff hunk it anchors to changes, +which is the per-thread check above, and it covers the case that actually +matters: the author edited the flagged code. The fingerprint adds coarser +file-level detection whose failure mode is a redundant fix the next re-review +catches. So: use the fingerprint when it is there, fall back to anchors when it +is not, and never let the weaker check be silent. + +## What it will not do to your code + +Enforced in the prompt, not in code — treat these as the contract the trial is +measuring, not as a guarantee: + +- No drive-by refactors. Every hunk traces to a finding it was handed. +- **Never weakens a test to satisfy a finding.** Deleting an assertion, + loosening a matcher, adding a skip, or widening an expected range is never an + acceptable outcome; the finding is left unfixed and reported instead. +- No amend, rebase, or force-push. +- Ambiguous findings are left alone and reported, not guessed at. + +One further limit is enforced by gh-aw itself rather than by us: its +`push-to-pull-request-branch` handler refuses to commit changes to a +protected-file list that includes every dependency manifest and lockfile +(`package.json`, `pnpm-lock.yaml`, `go.mod`, …), `CODEOWNERS`, and the +repo-root markdown (`README.md`, `CHANGELOG.md`, `CLAUDE.md`, `AGENTS.md`), and +it blocks writes to top-level dot-folders. A finding in one of those files +cannot be autofixed; the run reports it unfixed. + +## The commit trailer + +Every autofix commit ends with a machine-readable trailer: + +``` +Autofix-Version: 1 +Autofix-Scope: blocking +Autofix-Cycle: 1 +Autofix-Threads: PRRT_kwDO…,PRRT_kwDO… +``` + +v1 runs once per arming and never reads this back. It is written anyway because +the branch is the only cycle store that survives cache eviction, needs no +external state, and is legible to a human reading the PR — the same reasoning +that put the reviewer's authoritative fingerprint in the review body rather than +in cache memory. `Autofix-Threads` is the attempted-finding ledger: diffing it +against what the next review still reports open is how the trial answers whether +a fix actually cleared the finding. + +## Install + +```sh +gh aw add Khan/actions/workflows/autofix +gh aw compile +``` + +Requires the `review` workflow to be installed and running in the same repo: +autofix reads its threads, its label taxonomy, and its fingerprint stamp. + +### Required secrets + +- `ANTHROPIC_API_KEY` — the `claude` engine. +- `KHAN_ACTIONS_BOT_TOKEN` — the push. **Not optional and not substitutable + with `GITHUB_TOKEN`**: GitHub creates no workflow runs for events triggered by + `GITHUB_TOKEN`, so a push made with it emits no `synchronize` and the reviewer + is never even asked to re-review. With the bot token it is asked; see + [Verification is best-effort](#verification-is-best-effort) for what that does + and does not guarantee. + +### Repository setup + +Create the two labels (`autofix: blocking`, `autofix: nits`) if you want the +label surface; the `/autofix` command needs no setup. Nothing else is +configured per repo: scope is chosen per PR by label or argument. + +## Design notes + +### Why the command is written out longhand + +The `/autofix` gate is spelled out in the workflow's `if:` rather than using +gh-aw's `slash_command` trigger. gh-aw's compiled gate only matches the command +followed by a space, a bare `\n`, or end-of-body, so a comment saved with a +trailing CRLF — which the GitHub web UI produces when you press Enter after the +command — never activates the workflow. That silently killed `/review` in +Khan/webapp#40943. The parser in `scope.ts` tolerates the same shapes; the two +must stay in step, and there is a test pinning the CRLF case specifically. + +### The command path's gates are weaker + +Worth knowing before relying on it. `issue_comment` carries no +`github.event.pull_request`, so the fork guard cannot be evaluated in the `if:` +at all; it moves into the plan, after the agent job has started. An `/autofix` +on a fork PR the label path would have rejected for free still costs a job. + +It is instead enforced in `plan.ts`, from the staged `context.json`, on every +path rather than only the command one. A duplicated guard is cheap; a missing +one authorises a code push. + +### `skip-ai-review` does not disarm autofix + +Deliberately. That label stops the reviewer's *next* run; it does not withdraw a +review already posted, so a labelled PR can still be carrying current findings. +The reviewer even suggests the label from inside a review body it just posted, so +that state is one the workflow steers people into. An explicit `autofix:` label +or `/autofix` from someone with write access is the authorisation to act on those +findings, and the earlier gate swallowed it silently. + +The case that gate was justified by ("no review, so nothing to fix") is real and +still covered, by the review-currency guard that actually checks for a review. + +Revisit when autofix runs automatically rather than only when a human arms it: +that is when a push nobody asked for becomes possible, and autofix should then +get its own opt-out rather than borrowing the reviewer's. + +The gate that actually matters is unaffected: gh-aw's `roles` check still runs, +compiling to an `author_association` test against `OWNER`/`MEMBER`/ +`COLLABORATOR`, so a comment from someone without write access never reaches the +agent. That is the gate standing between a drive-by comment and a code push. + +**`/autofix` only works from the default branch.** `issue_comment` is a +repository-level event, so GitHub reads the workflow from the default branch and +never from a PR head. An install that exists only on a branch cannot be driven +by the command at all; the label is the only surface available to it. + +### Why the label, and not a 🚀 on a comment + +Per-comment triggering was considered and dropped for v1. GitHub emits **no +webhook for reactions** — the feature request has been open since 2022 — which +is why the review workflow's own thumbs sweep is a two-hourly cron. A +reaction-triggered autofix would inherit that latency, or need a second poll to +shave a delay it still could not bound. `pull_request: labeled` and +`issue_comment: created` both fire immediately. + +Note also that 🚀 is already live signal: `thumbs-sweep.ts` counts it as a +positive reaction feeding the reviewer's tuning loop, so overloading it would +corrupt that channel. + +The command surface makes per-comment autofix nearly free when it lands: an +`/autofix` posted as a **reply inside a review thread** fires +`pull_request_review_comment: created` and carries `in_reply_to_id`, naming the +exact finding with no matching heuristics. The parser already handles the +command; only the trigger and the thread-scoping would be new. + +### Why not suggestion blocks + +The reviewer already emits ```suggestion blocks for single-line mechanical +fixes, which GitHub lets an author batch-commit with one click at zero CI and +zero credit cost. Autofix earns its keep on what a suggestion block cannot +express: multi-line, cross-file, needs-a-test changes. + +### Division of labour + +Code decides; the model edits. Two deterministic stages run before the agent is +asked to change anything: + +- **`lib/stage.ts` runs as a `pre-agent-steps:` step**, before the agent starts, + and fetches everything the plan needs (labels, the reviewer's unresolved + threads with their full reply chains, prior reviews, the diff, commit + messages, the head SHA). It costs zero assistant turns, and a staging failure + fails the step before any AI credits are spent. This follows the reviewer's + own orchestrator slice 1 (Khan/actions#280): anything that never needed model + output belongs in a pre-agent step. +- **`lib/plan.ts`** then resolves the scope, checks currency, builds the work + list, and renders the trailer. The plan is final — +the prompt's contract is to execute it or stop, never to widen it, narrow it, or +re-classify a skipped thread. Nothing in `lib/` composes a sentence about the +code under review. + +## Versioning + +`autofix.md` pins `Khan/actions` at `autofix-v` in both its +`pre-agent-steps` checkout and its `source:`, so prompt and code always come +from one release. `utils/sync-workflow-versions.ts` rewrites those literals +during the release; `version-sync.test.ts` fails CI if they ever drift from the +package version. diff --git a/workflows/autofix/autofix.md b/workflows/autofix/autofix.md new file mode 100644 index 00000000..68f10fe3 --- /dev/null +++ b/workflows/autofix/autofix.md @@ -0,0 +1,660 @@ +--- +description: > + Addresses the PR reviewer's own feedback on demand, one run per arming. Arm it + with an `/autofix [blocking|nits]` comment, or with an `autofix: blocking` / + `autofix: nits` label; the two are peers. The run fixes the reviewer's open + threads in that scope, pushes one commit, replies in each thread, and clears + the label if one armed it. + +on: + # Two arming surfaces, and they are PEERS — neither is a shorthand for the + # other. A label is state you click; a command is an event you type and can + # pass arguments to. Both resolve through one shared resolver (`scope.ts`), so + # a value can never mean one thing as a label and another as a command. + pull_request: + types: [labeled] + issue_comment: + types: [created] + # Acknowledge an `/autofix` comment immediately, the same way the reviewer + # acknowledges `/review`. Without this the author has no signal between typing + # the command and the summary comment several minutes later. + reaction: eyes + # No status comment: the run posts exactly one summary comment of its own + # (Step 7), and a gh-aw "started/completed" comment on top of that would + # double the noise on a PR that is already carrying a full review. + status-comment: false + # Autofix writes code to someone's branch, so the actor who armed it must be + # able to write to the repo themselves. This is deliberately NOT the + # reviewer's `roles: all` override: the reviewer only reads and comments, and + # its gate is relaxed so a collaborator's push still triggers a review. On the + # comment path this role check is the PRIMARY gate — see below. + roles: [admin, maintainer, write] + +# One gate per surface. Double-quoted YAML so the `\n`/`\r`/`\t` escapes below +# become real characters in the expression rather than literal backslashes. +# +# LABEL PATH. Two cheap gates before the agent starts: +# 1. Same-repo branches only. A fork PR gets no secrets, so the push would +# fail anyway. +# 2. The label that fired this event is an autofix label. Every other label +# addition on the PR is a run we never pay for. +# +# `skip-ai-review` is deliberately NOT a gate here, and that is a decision, not +# an omission. The label stops the reviewer from running again; it does not +# dismiss a review already posted (the reviewer's own `if:` says so: "adding it +# prevents the *next* run"). The reviewer even suggests the label from inside a +# review body it just posted, so "labelled" and "has current findings" is a +# state the workflow steers users into, not a corner case. Reading the label as +# "no AI may act on this PR" would silently swallow an explicit `autofix:` label +# from someone with write access; that opt-in IS the authorisation, while +# autofix only ever runs when a human arms it. Revisit when autofix runs +# automatically: that is when a push nobody asked for becomes possible, and when +# autofix should get its own opt-out rather than borrowing the reviewer's. +# +# COMMAND PATH — deliberately weaker, and worth understanding before you touch +# it. `issue_comment` carries no `github.event.pull_request`, so the fork guard +# CANNOT be evaluated here at all. It is instead enforced in `plan.ts`, which +# refuses a fork from the staged `context.json`. That is real code, not an +# aspiration: an earlier version of this comment claimed the check "moves into +# the plan" while the plan did not implement it, and Khan/actions#298's review +# caught it. The cost is that an `/autofix` on a fork PR the label path would +# have rejected for free still burns a job before refusing. +# +# The gate that actually matters is unaffected: gh-aw's `roles` check above +# still runs, so a comment from someone without write access never reaches the +# agent. +# +# ONE STRUCTURAL LIMIT. `issue_comment` is a repository-level event, so GitHub +# reads the workflow file from the DEFAULT BRANCH, never from the PR's head. +# `/autofix` therefore cannot fire for an install that only exists on a branch, +# which is why every run of the Khan/webapp#41140 trial was `pull_request` and a +# reviewer's `/autofix` comment there did nothing. The command surface can only +# be exercised once this workflow is on the consuming repo's default branch. +# +# The command match is written out longhand rather than using gh-aw's +# `slash_command` trigger. gh-aw's compiled gate only matches the command +# followed by a space, a bare `\n`, or end-of-body, so a comment saved with a +# trailing CRLF — which the GitHub web UI produces when you press Enter after +# the command — never activates the workflow. That silently killed `/review` in +# Khan/webapp#40943. `scope.ts`'s parser tolerates the same shapes; keep the two +# in step. +if: "(github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository && startsWith(github.event.label.name, 'autofix: ')) || (github.event_name == 'issue_comment' && github.event.issue.pull_request != null && (github.event.comment.body == '/autofix' || startsWith(github.event.comment.body, '/autofix ') || startsWith(github.event.comment.body, '/autofix\n') || startsWith(github.event.comment.body, '/autofix\r') || startsWith(github.event.comment.body, '/autofix\t')))" + +permissions: + contents: read + pull-requests: read + +tools: + github: + lockdown: false + min-integrity: none + toolsets: [pull_requests, repos] + edit: + # NOTE THE `:*` SUFFIX. gh-aw's schema documents `"npx *"` (space-star) as + # "command with any args", but it compiles that form to the Claude Code + # permission `Bash(npx)`, which matches ONLY a bare `npx` with no arguments — + # so `npx -y tsx …` is denied and the plan CLI never runs. `"npx:*"` compiles + # to `Bash(npx:*)`, which is the form gh-aw's own defaults use + # (`Bash(git add:*)`). Observed on gh-aw v0.83.4; verify the compiled + # `--allowed-tools` list still carries the `:*` suffix after any gh-aw bump. + # + # Declaring this list at all NARROWS the agent: a workflow with no `bash:` key + # (the reviewer, for one) compiles to unrestricted `Bash`. That is the + # trade being made here deliberately, which is why the list must be right. + bash: + - "git:*" + - "npx:*" + - "node:*" + - "cat:*" + - "ls:*" + - "date:*" + - "mkdir:*" + +safe-outputs: + allowed-domains: + - github.com + - khanacademy.org + - khanacademy.dev + - khanacademy.atlassian.net + + # The commit. `KHAN_ACTIONS_BOT_TOKEN` rather than the default GITHUB_TOKEN is + # load-bearing, not incidental: GitHub does not create workflow runs for + # events triggered by GITHUB_TOKEN, so a push made with it would emit no + # `synchronize` and the reviewer would never re-review the fix. + # + # That re-review is the INTENDED verification for an autofix commit, and it is + # best-effort rather than guaranteed: the chain from this push to a posted + # review has several links and any of them can fail. Khan/webapp#41194 lost + # one to a gh-aw setup failure that was invisible on the PR. So the reason for + # the token is the stronger one: GITHUB_TOKEN guarantees zero re-review, the + # bot token buys a best-effort one. Step 7 states the pending status on the PR + # so the human re-arm can act as the backstop, and the README's "Verification + # is best-effort" carries the measured numbers. + # + # `if-no-changes: ignore` because "the agent decided nothing needed changing" + # is a legitimate outcome that Step 7 already reports in prose; failing the + # job on it would turn a correct no-op into a red X on the PR. + push-to-pull-request-branch: + target: "triggering" + max: 1 + if-no-changes: "ignore" + github-token: ${{ secrets.KHAN_ACTIONS_BOT_TOKEN }} + + # One reply per fixed thread (Step 6). Replying rather than resolving is + # deliberate: `thread-reconciler` already weighs author replies when the next + # review decides whether a thread is settled, and the re-review accountability + # section links threads that are still open. Resolving here would bypass both + # and destroy the record of whether the fix actually worked. + reply-to-pull-request-review-comment: + max: 20 + target: "triggering" + footer: false + + # The run summary (Step 7): what was fixed, what was skipped and why. Exactly + # one per run, and older ones collapse, so a PR armed several times keeps only + # the latest visible. + add-comment: + target: "triggering" + max: 1 + discussions: false + hide-older-comments: true + footer: false + + # The label is a button, not a mode: it is removed on EVERY outcome, including + # a refusal. A label left on after the run reads as "still queued" when + # nothing is, and re-arming is one click. + remove-labels: + allowed: + - "autofix: blocking" + - "autofix: nits" + - "autofix: loop" + - "autofix: human" + - "autofix: author" + + # The plan artifact. `plan.json` is the only record of what the run was asked + # to do versus what it did, which is what the trial reads to score fix quality + # against the next review's findings. + upload-artifact: + max-uploads: 1 + retention-days: 30 + allowed-paths: + - "out/**" + - "/tmp/gh-aw/autofix/out/**" + +network: + allowed: + - defaults + - github + +# HELD AT OPUS 4.8. This should be `claude-opus-5`, matching the roster +# Khan/actions#294 moves the reviewer to. Three live runs on Khan/webapp#41140 +# failed to get there, and the cause is not yet established, so the model stays +# where it demonstrably works rather than where we want it. +# +# WHAT WAS OBSERVED. The api-proxy's AI-credits guard rejects an un-priced model +# with a 400 before the request reaches the model, and `claude-opus-5` is in no +# firewall release's curated pricing table. +# - 30416237794: no fallback configured. 400, as expected. +# - 30421726630: `models.default-ai-credits-pricing` configured, firewall +# v0.27.42 (the compiler default). Staged awf-config.json confirmed to carry +# `apiProxy.defaultAiCreditsPricing: {input: 5, output: 25}`. Still 400. +# - 30422315631: same, firewall pinned to v0.27.27. Config confirmed present, +# image confirmed pulled. Still 400. +# +# WHAT IS ESTABLISHED. The mechanism exists and is documented: awf-config-spec +# 10.7.3 says a configured fallback makes an unresolvable model "proceed +# normally", and `config-mapper.ts` maps the field to +# `AWF_DEFAULT_AI_CREDITS_PRICING` in BOTH v0.27.27 and v0.27.42. So the pin to +# v0.27.27 above was pointless and has been removed; version is not the +# variable. Note also that #294's own `review.lock.yml` contains zero +# occurrences of `claude-opus-5` (only the shared review.md was edited, never +# recompiled), so its "verified" claim is a reading of the spec rather than a +# run, which is consistent with these three failures. +# +# WHAT IS UNTESTED, and the next thing to try. Spec 10.7.1 applies the fallback +# only when the model "cannot be resolved from the curated table or the bundled +# models.dev catalog". These runs supplied BOTH the fallback AND a +# `models.providers.anthropic.models.claude-opus-5.cost` entry (copied from +# #294). If that providers entry makes the model *resolve* to something carrying +# no AI-credits pricing, it would short-circuit the fallback and reject, which +# is exactly the symptom. The untried combination is: keep +# `default-ai-credits-pricing`, DROP the `models.providers` block, leave the +# firewall at the default. One run settles it. +# +# Until then this is a one-line change plus its `models:` block. Do not restore +# them without a run that reaches the model. +engine: + id: claude +model: claude-opus-4-8 + +timeout-minutes: 20 + +# Autofix reads the reviewer's staged artifacts and the reviewer's own label +# taxonomy, so it checks out Khan/actions for both libs at once: one tag, one +# tree, `workflows/autofix/lib` and `workflows/review/lib` guaranteed to be the +# versions that were released together. The ref is rewritten by +# utils/sync-workflow-versions.ts during the release, and +# workflows/autofix/version-sync.test.ts fails CI if it ever drifts from the +# `autofix` package version. +pre-agent-steps: + - name: Check out shared workflow lib (Khan/actions) + uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5 + with: + repository: Khan/actions + ref: autofix-v0.0.0 + path: gh-aw-autofix-lib + persist-credentials: false + + # Staging is deterministic and runs BEFORE the agent, so it costs zero + # assistant turns. This follows the reviewer's orchestrator slice 1 (#280): + # anything that never needed model output belongs in a pre-agent step. + # + # The first live run measured why. Staging by prose cost roughly fifteen of + # that run's 131 turns (seven creating a directory, five hand-assembling JSON + # through repeated `node -e` scripts, three reading this workflow's own lib + # source), and turns are what autofix costs: each re-reads the whole context, + # so 61% of the bill was cache reads. Caching was already near-optimal at a + # 40:1 read-to-write ratio; there were simply too many turns. + # + # A staging failure fails this step before any AI credits are spent. + # + # The comment body is passed through `env:` rather than interpolated into + # `run:`. It is attacker-controlled text, and `env:` keeps it out of the + # shell's parse. + - name: Stage the plan's inputs (deterministic) + working-directory: gh-aw-autofix-lib + env: + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + GITHUB_REPOSITORY: ${{ github.repository }} + AUTOFIX_PR_NUMBER: ${{ github.event.pull_request.number || github.event.issue.number }} + AUTOFIX_COMMAND_BODY: ${{ github.event_name == 'issue_comment' && github.event.comment.body || '' }} + run: npx -y tsx workflows/autofix/lib/stage.ts + +# A fix run is a fraction of a review run: no reviewer roster, no lenses, one +# agent editing a bounded set of files. 1000 credits ($10) is the gh-aw default +# and is generous for that shape; the daily ceiling stays on, because unlike +# reviews (which must never be skipped) a deferred autofix costs nothing but a +# re-click. +max-ai-credits: 1000 + +# ───────────────────────────────────────────────────────────────────────────── +# WORKAROUND for a gh-aw bug. Remove once it is fixed upstream. +# +# THE BUG. The safe-outputs job checks out with `actions/checkout` and no +# `fetch-depth`, so it gets a depth-1 shallow clone of `refs/pull/N/merge` — the +# merge commit alone, without its parents. `push_to_pull_request_branch.cjs:935` +# then fetches the PR branch with NO `--depth`: +# +# git fetch origin :refs/remotes/origin/ +# +# The branch tip is a parent of the merge commit and therefore absent, and its +# history never reaches the existing shallow boundary, so git walks the branch +# all the way back: a full-history fetch. On a small repo this is invisible +# (the branch's parent usually IS the boundary). On Khan/webapp it is fatal. +# +# MEASURED, reproducing the exact command against Khan/webapp from a faithful +# depth-1 checkout of refs/pull/41130/merge: +# - as gh-aw runs it: >14 min, 5.6 GB and still climbing, never finished +# (this is what cancelled the safe_outputs job on Khan/webapp#41130) +# - with the filter below: 91 s, ~557 MB, exit 0 +# +# THE FIX WE CANNOT APPLY. `--depth=1` on that fetch. git has no `fetch.depth` +# config, so it cannot be injected; there is no gh-aw option for it (none of the +# 20 `push-to-pull-request-branch` keys touch fetch or checkout); `checkout:` in +# frontmatter configures the AGENT job only; and `pre-agent-steps`/`post-steps` +# cannot add steps to the safe-outputs job. gh-aw's own +# `checkout_pr_branch.cjs:229` does pass `--depth`, and this same file passes +# `--depth=1` at :1001 and `--filter=blob:none` at :1065, so the omission at +# :935 is an oversight rather than a design choice. +# +# WHAT THIS DOES. git honours `GIT_CONFIG_COUNT` / `GIT_CONFIG_KEY_` / +# `GIT_CONFIG_VALUE_` as if passed via `-c`. The handler runs the fetch with +# `env: {...process.env, ...gitAuthEnv}` (`:934`) and `gitAuthEnv` is empty +# (`:343`), so these reach it. We cannot bound the DEPTH, but we can make the +# fetch a partial clone that skips blobs, which is where the bulk sits. +# +# SAFE AGAINST gh-aw's OWN USE. `ensureSafeDirectoryTrust` +# (`git_helpers.cjs:64-76`) reads the existing `GIT_CONFIG_COUNT` and APPENDS +# its `safe.directory` entry at the next free index, so indices 0 and 1 here +# compose with it rather than clobbering it. +# +# KNOWN TRADE-OFFS. +# - Workflow-level `env:` reaches every job, so the agent job's checkout also +# becomes a partial clone and file reads lazily fetch blobs. For an agent +# that reads a handful of files that is fine and probably faster; gh-aw +# itself notes the lazy-fetch cost at `push_to_pull_request_branch.cjs:1062`. +# gh-aw exposes no per-job env, so this cannot be scoped more tightly. +# - The fetch still pulls all tags. `remote.origin.tagOpt=--no-tags` would +# trim more, but it is NOT set here because it has not been measured; add it +# only with a number behind it. +# ───────────────────────────────────────────────────────────────────────────── +env: + GIT_CONFIG_COUNT: "2" + GIT_CONFIG_KEY_0: remote.origin.promisor + GIT_CONFIG_VALUE_0: "true" + GIT_CONFIG_KEY_1: remote.origin.partialclonefilter + GIT_CONFIG_VALUE_1: "blob:none" + +source: Khan/actions/workflows/autofix/autofix.md@autofix-v0.0.0 +--- + +# PR Autofixer + +You address feedback the PR reviewer left on this pull request. You are not a +reviewer: you do not judge whether a finding is worth fixing, and you do not +look for problems nobody raised. Your scope is exactly the threads the plan +hands you in Step 2. + +## Current Context + +- **Repository**: ${{ github.repository }} +- **Pull Request**: #${{ github.event.pull_request.number || github.event.issue.number }} + +This workflow fires on two events, so the PR number is read from whichever one +carries it. What armed the run is not interpolated here on purpose: gh-aw's +expression allowlist excludes `github.event.label.name` from prompt bodies, and +the plan resolves the arming itself in Step 2, which is the authoritative answer +either way. + +## Cost: turns are the expensive thing + +This workflow's cost is dominated by the number of assistant turns, not by how +much data any one of them moves. Every turn re-reads the whole accumulated +context. The first live run took 131 turns and 460 AI credits to land a +six-line fix, with only ~93 KB of tool output in the entire run: the payload was +trivial and the turn count was not. + +So, concretely: + +- **Batch shell work.** One command with `&&` beats three round trips. Create + every directory you need in a single `mkdir -p` and do not verify it + afterwards; `mkdir -p` does not fail on an existing directory, and `ls` to + confirm is a wasted turn. +- **Do not read the lib source.** `gh-aw-autofix-lib` holds released, + tested code. Reading `plan.ts` or `staleness.ts` to work out what the plan + will say costs turns and context, and tempts you to re-derive a decision the + plan has already made. Run the CLI and read its output. +- **Do not explore the `safeoutputs` CLI.** Its invocations are written out at + each step below. Calling `--help` first is a wasted turn. +- **Do not re-verify your own writes.** If a command exits zero, it worked. +- **Prefer a heredoc to a chain of `node -e` scripts.** If you find yourself + writing the same inline script twice with small edits, write it to a file once + and run it. + +## Step 1: Read the staged inputs + +**There is nothing to stage.** A deterministic pre-agent step +(`workflows/autofix/lib/stage.ts`) has already fetched everything and written it +to `/tmp/gh-aw/autofix/` before you started. Do not fetch any of it again, and +do not rewrite any of these files: + +| file | what it holds | +| --- | --- | +| `labels.json` | the PR's current label names | +| `threads.json` | the reviewer's unresolved threads, each with its full reply chain, verbatim | +| `prior-reviews.json` | every review by the reviewer bot, whatever its state | +| `pr.diff` | the PR's unified diff | +| `commits.json` | commit messages on the head, the autofix cycle ledger | +| `head-sha.txt` | the head SHA when the run started; Step 5 compares against it | +| `command.txt` | the `/autofix` comment body, **only** on a command-armed run | + +You do not need to read most of these. Step 2's CLI parses them; the only ones +you will open yourself are `plan.json` (which Step 2 produces) and +`head-sha.txt` (Step 5). Reading `pr.diff` or `threads.json` in full is a waste +of context: the plan already extracted what matters into `plan.items`. + +If a file is missing, the staging step failed and the run should not have +reached you. Say so in Step 7 and stop; do not reconstruct it by hand. + +## Step 2: Build the plan (deterministic code) + +Run the plan CLI once, from the shared lib checkout: + +``` +cd gh-aw-autofix-lib && npx -y tsx workflows/autofix/lib/plan.ts +``` + +It writes `/tmp/gh-aw/autofix/plan.json` and prints a summary. Copy +`plan.json` to `/tmp/gh-aw/autofix/out/plan.json` now, so the run artifact +records what was planned even if a later step fails. + +**The plan is final.** It decided the scope, which threads are in it, which were +skipped and why, and what the commit trailer says. Do not widen it, narrow it, +re-classify a skipped thread, or act on a finding it did not hand you. If you +disagree with the plan, say so in the Step 7 comment; do not act on the +disagreement. + +**If the CLI does not run, the run is over.** If `npx` is unavailable, the +command errors, or the tool call is denied, do **not** reconstruct the plan by +reading the library source and reasoning about it. A hand-simulated plan is +exactly the thing this workflow's determinism boundary exists to prevent, and +it produces a confident-looking result nobody can audit. Instead: change +nothing, post a Step 7 comment saying the plan CLI could not be executed and +quoting the error, remove the labels (Step 8), and stop. + +The plan resolves the arming itself, from whichever surface triggered the run: +`command.txt` when it exists, the PR's labels otherwise. **The trigger decides, +and the two never union** — a stale `autofix: nits` label must not silently +widen someone's `/autofix blocking`. `plan.surface` records which one won. + +`plan.json` has a `status`: + +- **`refused`** — the run cannot proceed safely (an autofix label on an axis + this version does not implement, no reviewer feedback to act on, or a review + that cannot be matched to the current diff). Skip to Step 7, post the comment + with the plan's `reason` verbatim, remove the labels, and stop. Change + nothing. +- **`no-op`** — the labels were understood and there is nothing to fix. Skip to + Step 7 the same way. This is a success, not a failure; say so plainly. +- **`armed`** — continue to Step 3. + +## Step 3: Understand each finding before changing anything + +For every item in `plan.items`, read the file at `path` around `line` from the +Actions workspace (the PR head is checked out; read from disk, not through the +API). The item's `body` is the reviewer's statement of the problem, verbatim. + +Work out what the reviewer meant. If a finding is ambiguous, or you cannot +determine what a correct fix would be, **leave it alone** and record it as +not-fixed for Step 7. A wrong fix on a blocking finding is worse than no fix: +the author now has to review a change they did not write, on a problem they had +not yet looked at. + +## Step 4: Make the fixes + +Edit the files directly in the workspace. Rules, all hard: + +- **Fix the finding, nothing else.** No drive-by refactors, no reformatting, no + fixing things you noticed on the way. Every hunk you write must be traceable + to an item in `plan.items`. +- **Never weaken a test to satisfy a finding.** If the honest fix makes a test + fail, fix the source. If you believe the test itself encodes the bug, leave + the finding unfixed and explain why in Step 7. Deleting an assertion, loosening + a matcher, adding a skip, or widening an expected range to make something pass + is never an acceptable outcome of this workflow. +- **Do not touch files no item points at.** The one exception is a change that + is mechanically forced by a fix (a caller that must be updated for a changed + signature); note any such file in Step 7. +- **Do not amend, rebase, or force-push.** You produce working-tree changes; + the push is a safe output. +- If a fix would require a design decision the reviewer did not make for you, + leave it unfixed and say so. + +**Verify what you can, and be honest about what you cannot.** Nothing between +your edit and the push checks that the code still compiles or that its tests +still pass; the re-review and the repo's CI both run only after the commit is on +the author's branch. If this repository has a cheap check you can run from the +allowlisted commands, run it. If you cannot verify (no allowlisted runner, or +the check needs a toolchain that is not installed), that is expected, not a +failure; say so in the Step 7 summary rather than implying the fix was +validated. An unverified fix presented as a verified one is the thing to +avoid. + +## Step 5: Push one commit + +Compare the live head SHA against `/tmp/gh-aw/autofix/head-sha.txt`, which +staging captured from the job's checkout (`git rev-parse HEAD`) rather than from +the API, because the checkout is what your edits are actually against. +**If it changed, do not push.** The author pushed while you were working, and +your edits are against a base that no longer exists. Skip to Step 7, report that +the run was abandoned for that reason, and remove the labels; the author can +re-label once their push settles. + +Otherwise commit your working-tree changes locally, then emit a single +`push-to-pull-request-branch`. The engine builds its patch and bundle from that +commit, so both steps are needed: + +``` +git -C "$GITHUB_WORKSPACE" add -- +git -C "$GITHUB_WORKSPACE" commit -F /tmp/gh-aw/autofix/commitmsg.txt +printf '{"message":%s}' "$(...)" | safeoutputs push_to_pull_request_branch . +``` + +Write the message to `commitmsg.txt` first rather than passing it inline; it is +multi-paragraph and shell-quoting it is a reliable way to lose the trailer. + +**The commit message must stand on its own.** Someone reading `git log` a year +from now, with no PR open and no reviewer thread to click through to, should be +able to tell what changed and why. Write it as you would any commit; the fact +that a bot wrote it is not the interesting part. + +``` +autofix: + + + +:: `, only when +there is more than one finding; with a single finding the paragraph above has +already said it.> + + +``` + +Rules for the subject line, which is the part that ages worst: + +- **Never** use a generic subject. `autofix: address reviewer feedback` is + banned: it is identical on every run, so a branch with several autofix + commits becomes a wall of indistinguishable `git log --oneline` entries. +- Name the change, not the process. `autofix: clamp Page start into range`, + not `autofix: fix blocking finding` or `autofix: apply review comments`. +- Under 60 characters, imperative mood, no trailing period. + +Do not reference thread ids, run urls, or the reviewer by name in the prose; +that is what the trailer is for. + +The trailer block must be the last paragraph and must be copied exactly as +`plan.json` renders it. It is what a later run reads to know this one happened. +It is machine metadata: never describe it, expand it, or move it into the body. + +## Step 6: Reply in each thread + +For every item you fixed, emit one `reply-to-pull-request-review-comment` on +that item's thread, stating what you changed in one or two sentences. Be +specific: "Renamed to `parsedConfig` and updated the three call sites" beats +"Fixed". + +Do **not** resolve any thread. The next review decides whether the fix settled +the finding; that is the whole verification story for this workflow, it is +best-effort (see Step 7), and resolving here would erase it. + +For an item you deliberately left unfixed (Step 3 or Step 4), reply saying so +and why, in one sentence. A finding that was handed to you and silently skipped +is the one outcome an author cannot debug. + +## Step 7: Post the run summary + +**Post exactly one `add-comment`, on every path through this workflow.** A +refusal, a no-op, a finding left unfixed, an abandoned push, a run that fixed +everything cleanly: all of them get one comment, and never more than one. + +An earlier version of this step stayed quiet when a run was unremarkable, on the +reasoning that a clean run already tells its own story in three other places: +the thread reply on each fixed finding, the commit in the PR timeline, and the +engine's own "Commit pushed" comment, so a fourth notification repeating them is +noise. That reasoning was right about noise and wrong about what still needed +saying. All three of those places record that something *changed*; none of them +records that nothing has *checked* it. Item 8 below is the only place a reader +learns that, so the comment carrying it cannot be optional. + +The quiet branch was therefore retracted deliberately, not lost. It was also +unreachable in practice: it required `plan.degradedNote` to be empty, and the +reviewer's hidden fingerprint stamp is stripped from every posted review (the +README's "Degrading when there is no fingerprint" documents it), so that note is +essentially always set. Do not reintroduce the branch without first answering +where the pending-verification statement goes instead. + +Do **not** try to add a hidden HTML-comment marker of your own. gh-aw's +safe-output ingest strips every XML/HTML comment before posting +(`removeXmlComments` in `sanitize_content_core.cjs`, a depth-tracking scan with +no allowlist), so such a marker is silently deleted. An earlier version of this +step asked for ``; the posted comments never +carried it. Collapsing older comments still works, because the engine adds its +own `gh-aw-workflow-call-id` marker after sanitisation. + +Write the body directly, in this order, including only the parts that apply: + +1. One sentence of plain past-tense prose saying what happened. Take the + substance from the plan's `reason` but write it as a sentence to a person: + `Fixed 1 blocking finding.`, not `fixing 1 blocking finding(s).` Get the + tense and the plural right; the work is already done by the time anyone + reads this. +2. If anything was fixed: a list, one line per finding, `path:line` plus what + changed. Link each to its thread `url` when the item has one. +3. If anything was left unfixed: a list, one line each, with the reason. This + is the most important section in the comment; never omit or soften it. +4. If `plan.skipped` contains entries whose reason is **not** `out-of-scope`: + one line each with the reason (`outdated-anchor`, `unparseable-label`, + `stale-path`). Put any `out-of-scope` entries in a collapsed + `
N thread(s) outside this run's scope` block, or + omit them entirely when the comment already has more urgent content: they + are the expected consequence of the scope the author picked. +5. If `plan.stalePaths` is non-empty, one line: `Files changed since the last + review, so findings in them were not acted on: .` +6. If you could not run any check against your own edit, one line saying so, in + plain terms: `Not verified locally: no test or build command is available to + this workflow.` Never imply a fix was validated when it was not. +7. If `plan.degradedNote` is non-empty, that note **verbatim** on its own line. + Never omit it and never soften it: a weaker check that goes unmentioned is + indistinguishable from the full one. +8. When anything was pushed, last two lines, exactly: + + `Not verified: nothing has checked this commit. Autofix does not resolve its + own threads; whether these fixes settled the findings is decided by the + reviewer's next review of this branch, not by this run.` + + `That re-review can fail, or never trigger at all; a /review comment asks for + one at any time.` + +Items 6 and 8 are different claims and both can appear in the same comment: item +6 is about what this run could check *before* pushing, item 8 about what checks +the commit *after*. Do not merge them or drop one as a duplicate. + +Both of item 8's lines have to still be true a week later, once the re-review +has landed and approved. The first is tensed to the moment of writing, which is +what makes it safe. The second states a standing fact and a standing capability, +not a conditional instruction, because a one-shot run can never come back and +retract one: `If no review appears, comment /review` would leave every verified +PR permanently carrying an instruction to go and trigger a review. The same +objection rules out putting verification state in the commit trailer, where +nothing could ever update it either. + +Write nothing else. No preamble, no summary of the PR, no opinion on the code. +Do not use em dashes; a semicolon, colon, or full stop reads better and matches +the rest of this repo's bot output. + +## Step 8: Remove the labels + +Emit `remove-labels` for every label in the plan's `labelsToRemove`. Do this on +every path through this workflow, including refusals and no-ops. The label is a +button: once the run is over it must be off, so that its presence always means +"queued" and never "already done". + +On a command-armed run `labelsToRemove` is empty, and that is correct, not an +oversight: a comment is self-clearing, and any autofix label sitting on the PR +was not what armed this run. Removing it would clear an intent nobody acted on. +Emit nothing in that case. + +## Step 9: Upload the artifact + +Upload `/tmp/gh-aw/autofix/out/` with `upload-artifact` in one call. diff --git a/workflows/autofix/lib/plan.test.ts b/workflows/autofix/lib/plan.test.ts new file mode 100644 index 00000000..a03c41dd --- /dev/null +++ b/workflows/autofix/lib/plan.test.ts @@ -0,0 +1,515 @@ +import {describe, expect, it} from "vitest"; + +import {buildPlan, runPlanCli} from "./plan.ts"; +import type {PlanCliFs, PlanInput} from "./plan.ts"; +import {REFUSAL_REASONS} from "./staleness.ts"; +import {parseTrailer} from "./trailer.ts"; +import { + computeHunkSignature, + renderRereviewStamp, + STAMP_SCHEMA_VERSION, +} from "../../review/lib/rereview-mode.ts"; +import type {StagedThread} from "../../review/lib/rereview.ts"; + +const DIFF = + "diff --git a/src/a.ts b/src/a.ts\n--- a/src/a.ts\n+++ b/src/a.ts\n" + + "@@ -1,1 +1,2 @@\n context\n+added line\n"; + +const OTHER_DIFF = + "diff --git a/src/b.ts b/src/b.ts\n--- a/src/b.ts\n+++ b/src/b.ts\n" + + "@@ -1,1 +1,2 @@\n context\n+other line\n"; + +const reviewStamped = (diff: string) => ({ + body: renderRereviewStamp({ + schemaVersion: STAMP_SCHEMA_VERSION, + depth: "full" as const, + verdict: "REQUEST_CHANGES", + anchorDraft: false, + anchorHunks: computeHunkSignature(diff), + }), + submittedAt: "2026-07-01T00:00:00Z", +}); + +const thread = ( + over: Partial & {body: string}, +): StagedThread => ({ + thread_id: over.thread_id ?? "T1", + path: over.path ?? "src/a.ts", + line: over.line === undefined ? 2 : over.line, + comments: [{author: "github-actions[bot]", body: over.body}], +}); + +const input = (over: Partial = {}): PlanInput => ({ + labels: ["autofix: blocking"], + threads: [thread({body: "**issue (blocking):** guard is inverted"})], + priorReviews: [reviewStamped(DIFF)], + diffText: DIFF, + commitMessages: [], + isFork: false, + ...over, +}); + +describe("buildPlan", () => { + it("arms with the in-scope work and a rendered trailer", () => { + const plan = buildPlan(input()); + expect(plan.status).toBe("armed"); + expect(plan.scopes).toEqual(["blocking"]); + expect(plan.items.map((i) => i.threadId)).toEqual(["T1"]); + expect(plan.labelsToRemove).toEqual(["autofix: blocking"]); + expect(parseTrailer(`x\n\n${plan.trailer}`)).toMatchObject({ + scopes: ["blocking"], + cycle: 1, + threadIds: ["T1"], + }); + }); + + it("refuses an unimplemented axis and still clears the labels", () => { + const plan = buildPlan(input({labels: ["autofix: loop"]})); + expect(plan.status).toBe("refused"); + expect(plan.reason).toContain("cadence"); + // A label left on after a refusal reads as "still queued". + expect(plan.labelsToRemove).toEqual(["autofix: loop"]); + expect(plan.trailer).toBe(""); + }); + + it("refuses only when the PR has never been reviewed", () => { + const plan = buildPlan(input({priorReviews: []})); + expect(plan.status).toBe("refused"); + expect(plan.reason).toContain("nothing to autofix"); + expect(plan.items).toEqual([]); + }); + + it("still arms against an unstamped review, with a degraded note", () => { + // The Khan/webapp#41130 shape: real blocking feedback, no fingerprint. + const plan = buildPlan( + input({ + priorReviews: [ + { + body: "Changes requested — see inline comments.", + submittedAt: "2026-07-01T00:00:00Z", + }, + ], + }), + ); + expect(plan.status).toBe("armed"); + expect(plan.items.map((i) => i.threadId)).toEqual(["T1"]); + expect(plan.degradedNote).toContain("thread anchors only"); + expect(plan.stalePaths).toEqual([]); + }); + + it("still arms when the fingerprint overflowed", () => { + const plan = buildPlan( + input({ + priorReviews: [ + { + body: renderRereviewStamp({ + schemaVersion: STAMP_SCHEMA_VERSION, + depth: "full", + verdict: "REQUEST_CHANGES", + anchorDraft: false, + anchorHunks: "overflow", + }), + submittedAt: "2026-07-01T00:00:00Z", + }, + ], + }), + ); + expect(plan.status).toBe("armed"); + expect(plan.degradedNote).toContain("overflowed"); + }); + + it("leaves the degraded note empty when the fingerprint check ran", () => { + expect(buildPlan(input()).degradedNote).toBe(""); + }); + + it("still drops an outdated thread when running degraded", () => { + // The anchor check is what the degraded path leans on, so it has to + // keep working when the fingerprint is gone. + const plan = buildPlan( + input({ + priorReviews: [ + { + body: "Changes requested — see inline comments.", + submittedAt: "2026-07-01T00:00:00Z", + }, + ], + threads: [ + thread({body: "**issue (blocking):** x", line: null}), + ], + }), + ); + expect(plan.status).toBe("no-op"); + expect(plan.skipped[0].reason).toBe("outdated-anchor"); + }); + + it("drops a finding whose file changed after the review", () => { + // The review saw src/a.ts; the head now also carries an edit to it. + const movedOn = + DIFF + + "diff --git a/src/a.ts b/src/a.ts\n--- a/src/a.ts\n+++ b/src/a.ts\n" + + "@@ -9,1 +9,2 @@\n context\n+later edit\n"; + const plan = buildPlan(input({diffText: movedOn})); + expect(plan.status).toBe("no-op"); + expect(plan.stalePaths).toEqual(["src/a.ts"]); + expect(plan.skipped[0]).toMatchObject({ + threadId: "T1", + reason: "stale-path", + }); + }); + + it("fixes findings in untouched files when another file went stale", () => { + // The degradation the per-path guard buys: partial work, not refusal. + const reviewed = DIFF + OTHER_DIFF; + const head = + DIFF + + "diff --git a/src/b.ts b/src/b.ts\n--- a/src/b.ts\n+++ b/src/b.ts\n" + + "@@ -1,1 +1,2 @@\n context\n+other line CHANGED\n"; + const plan = buildPlan( + input({ + priorReviews: [reviewStamped(reviewed)], + diffText: head, + threads: [ + thread({thread_id: "T1", body: "**issue (blocking):** a"}), + thread({ + thread_id: "T2", + path: "src/b.ts", + body: "**issue (blocking):** b", + }), + ], + }), + ); + expect(plan.status).toBe("armed"); + expect(plan.items.map((i) => i.threadId)).toEqual(["T1"]); + expect(plan.skipped).toEqual([ + { + threadId: "T2", + path: "src/b.ts", + reason: "stale-path", + label: "issue (blocking)", + }, + ]); + }); + + it("is a no-op, not a refusal, when nothing is in scope", () => { + const plan = buildPlan( + input({ + threads: [ + thread({body: "**nitpick (non-blocking):** rename this"}), + ], + }), + ); + expect(plan.status).toBe("no-op"); + expect(plan.reason).toContain("no open blocking findings"); + expect(plan.labelsToRemove).toEqual(["autofix: blocking"]); + }); + + it("unions both scopes when both labels are on", () => { + const plan = buildPlan( + input({ + labels: ["autofix: blocking", "autofix: nits"], + threads: [ + thread({thread_id: "T1", body: "**issue (blocking):** a"}), + thread({ + thread_id: "T2", + body: "**nitpick (non-blocking):** b", + }), + ], + }), + ); + expect(plan.status).toBe("armed"); + expect(plan.items.map((i) => i.threadId)).toEqual(["T1", "T2"]); + expect(plan.labelsToRemove.sort()).toEqual([ + "autofix: blocking", + "autofix: nits", + ]); + }); + + it("takes its cycle number from the branch ledger", () => { + const prior = + "autofix: address reviewer feedback\n\n" + + "Autofix-Version: 1\nAutofix-Scope: blocking\n" + + "Autofix-Cycle: 1\nAutofix-Threads: T9\n"; + const plan = buildPlan(input({commitMessages: ["feat: x", prior]})); + expect(plan.cycle).toBe(2); + expect(parseTrailer(`x\n\n${plan.trailer}`)?.cycle).toBe(2); + }); + + it("reports no-op when the label is absent entirely", () => { + const plan = buildPlan(input({labels: ["bug"]})); + expect(plan.status).toBe("no-op"); + expect(plan.labelsToRemove).toEqual([]); + }); +}); + +describe("runPlanCli", () => { + const fsFor = (files: Record) => { + const written: Record = {}; + const fs: PlanCliFs = { + existsSync: (path) => files[path] !== undefined, + readFileSync: (path) => files[path], + writeFileSync: (path, data) => { + written[path] = data; + }, + }; + return {fs, written}; + }; + + it("reads the staged inputs and writes plan.json", () => { + const {fs, written} = fsFor({ + "/tmp/gh-aw/autofix/labels.json": JSON.stringify([ + "autofix: blocking", + ]), + "/tmp/gh-aw/autofix/threads.json": JSON.stringify([ + thread({body: "**issue (blocking):** x"}), + ]), + "/tmp/gh-aw/autofix/prior-reviews.json": JSON.stringify([ + reviewStamped(DIFF), + ]), + "/tmp/gh-aw/autofix/pr.diff": DIFF, + "/tmp/gh-aw/autofix/commits.json": JSON.stringify([]), + "/tmp/gh-aw/autofix/context.json": JSON.stringify({isFork: false}), + }); + const plan = runPlanCli(fs); + expect(plan.status).toBe("armed"); + const onDisk = JSON.parse(written["/tmp/gh-aw/autofix/plan.json"]); + expect(onDisk.items).toHaveLength(1); + expect(written["/tmp/gh-aw/autofix/plan.json"].endsWith("\n")).toBe( + true, + ); + }); + + it("degrades a missing input into an ordinary refusal, not a crash", () => { + const {fs} = fsFor({ + "/tmp/gh-aw/autofix/labels.json": JSON.stringify([ + "autofix: blocking", + ]), + }); + expect(runPlanCli(fs).status).toBe("refused"); + }); + + it("degrades malformed JSON the same way", () => { + const {fs} = fsFor({ + "/tmp/gh-aw/autofix/labels.json": "{not json", + "/tmp/gh-aw/autofix/pr.diff": DIFF, + }); + expect(runPlanCli(fs).status).toBe("no-op"); + }); +}); + +describe("buildPlan across both surfaces", () => { + it("arms from an /autofix command with no label present", () => { + const plan = buildPlan( + input({labels: [], command: "/autofix blocking"}), + ); + expect(plan.status).toBe("armed"); + expect(plan.surface).toBe("command"); + expect(plan.scopes).toEqual(["blocking"]); + expect(plan.items.map((i) => i.threadId)).toEqual(["T1"]); + }); + + it("never removes labels on a command-armed run", () => { + // A stale label the author never acted on must survive an /autofix. + const plan = buildPlan( + input({ + labels: ["autofix: nits"], + command: "/autofix blocking", + }), + ); + expect(plan.status).toBe("armed"); + expect(plan.labelsToRemove).toEqual([]); + }); + + it("never removes labels on a command-armed refusal either", () => { + const plan = buildPlan( + input({labels: ["autofix: nits"], command: "/autofix loop"}), + ); + expect(plan.status).toBe("refused"); + expect(plan.surface).toBe("command"); + expect(plan.labelsToRemove).toEqual([]); + }); + + it("lets the command decide, never unioning it with stale labels", () => { + // `autofix: nits` on the PR must not widen an explicit /autofix + // blocking into both scopes. + const plan = buildPlan( + input({ + labels: ["autofix: nits"], + command: "/autofix blocking", + threads: [ + thread({thread_id: "T1", body: "**issue (blocking):** a"}), + thread({ + thread_id: "T2", + body: "**nitpick (non-blocking):** b", + }), + ], + }), + ); + expect(plan.scopes).toEqual(["blocking"]); + expect(plan.items.map((i) => i.threadId)).toEqual(["T1"]); + }); + + it("falls back to labels when the comment is not an autofix command", () => { + const plan = buildPlan( + input({labels: ["autofix: blocking"], command: "/review"}), + ); + expect(plan.status).toBe("armed"); + expect(plan.surface).toBe("label"); + expect(plan.labelsToRemove).toEqual(["autofix: blocking"]); + }); + + it("falls back to labels for an empty or whitespace command", () => { + for (const command of ["", " "]) { + const plan = buildPlan( + input({labels: ["autofix: blocking"], command}), + ); + expect(plan.surface).toBe("label"); + } + }); + + it("still removes every autofix label on a label-armed refusal", () => { + const plan = buildPlan( + input({labels: ["autofix: blocking", "autofix: loop"]}), + ); + expect(plan.status).toBe("refused"); + expect(plan.labelsToRemove.sort()).toEqual([ + "autofix: blocking", + "autofix: loop", + ]); + }); + + it("produces an identical plan from equivalent label and command armings", () => { + const viaLabel = buildPlan(input({labels: ["autofix: blocking"]})); + const viaCommand = buildPlan( + input({labels: [], command: "/autofix blocking"}), + ); + expect(viaCommand.items).toEqual(viaLabel.items); + expect(viaCommand.scopes).toEqual(viaLabel.scopes); + expect(viaCommand.trailer).toEqual(viaLabel.trailer); + expect(viaCommand.status).toEqual(viaLabel.status); + }); +}); + +describe("runPlanCli command staging", () => { + it("reads command.txt when the comment path staged it", () => { + const files: Record = { + "/tmp/gh-aw/autofix/labels.json": JSON.stringify([]), + "/tmp/gh-aw/autofix/threads.json": JSON.stringify([ + thread({body: "**issue (blocking):** x"}), + ]), + "/tmp/gh-aw/autofix/prior-reviews.json": JSON.stringify([ + reviewStamped(DIFF), + ]), + "/tmp/gh-aw/autofix/pr.diff": DIFF, + "/tmp/gh-aw/autofix/commits.json": JSON.stringify([]), + "/tmp/gh-aw/autofix/context.json": JSON.stringify({isFork: false}), + "/tmp/gh-aw/autofix/command.txt": "/autofix blocking\r\n", + }; + const plan = runPlanCli({ + existsSync: (path) => files[path] !== undefined, + readFileSync: (path) => files[path], + writeFileSync: () => {}, + }); + expect(plan.status).toBe("armed"); + expect(plan.surface).toBe("command"); + }); + + it("falls back to labels when command.txt is absent", () => { + const files: Record = { + "/tmp/gh-aw/autofix/labels.json": JSON.stringify([ + "autofix: blocking", + ]), + "/tmp/gh-aw/autofix/threads.json": JSON.stringify([ + thread({body: "**issue (blocking):** x"}), + ]), + "/tmp/gh-aw/autofix/prior-reviews.json": JSON.stringify([ + reviewStamped(DIFF), + ]), + "/tmp/gh-aw/autofix/pr.diff": DIFF, + "/tmp/gh-aw/autofix/commits.json": JSON.stringify([]), + "/tmp/gh-aw/autofix/context.json": JSON.stringify({isFork: false}), + }; + const plan = runPlanCli({ + existsSync: (path) => files[path] !== undefined, + readFileSync: (path) => files[path], + writeFileSync: () => {}, + }); + expect(plan.status).toBe("armed"); + expect(plan.surface).toBe("label"); + }); +}); + +describe("guards the command path cannot express in the workflow if:", () => { + // Khan/actions#298 review, blocking: the issue_comment branch of the gate + // carried no fork check, and both the workflow comment and the README said + // it "moves into the plan" while the plan did not implement it. + it("refuses a fork", () => { + const plan = buildPlan(input({isFork: true})); + expect(plan.status).toBe("refused"); + expect(plan.reason).toContain("fork"); + }); + + it("refuses when the fork status is unknown", () => { + // Fail closed: an unreadable context must not authorise a push. + const plan = buildPlan(input({isFork: undefined})); + expect(plan.status).toBe("refused"); + expect(plan.reason).toContain("could not be"); + }); + + it("proceeds on a same-repo PR without the label", () => { + expect(buildPlan(input({isFork: false})).status).toBe("armed"); + }); + + it("enforces it on the command path too", () => { + const plan = buildPlan( + input({labels: [], command: "/autofix blocking", isFork: true}), + ); + expect(plan.status).toBe("refused"); + expect(plan.reason).toContain("fork"); + }); +}); + +describe("skip-ai-review does not disarm autofix", () => { + // The label stops the reviewer's NEXT run; it does not withdraw a review + // already posted, so a labelled PR can still carry current findings. The + // reviewer even suggests the label from inside a review body it just posted. + // An explicit `autofix:` label from someone with write access is the + // authorisation to act on those findings, and swallowing it silently was the + // behaviour that made the Khan/webapp#41177 docs trial unable to run both a + // suppressed reviewer and autofix. Revisit if autofix ever runs unarmed. + it("arms a labelled PR that still has a current review", () => { + const plan = buildPlan( + input({ + isFork: false, + labels: ["autofix: blocking", "skip-ai-review"], + }), + ); + expect(plan.status).toBe("armed"); + expect(plan.reason).not.toContain("skip-ai-review"); + }); + + it("arms on the command path too", () => { + const plan = buildPlan( + input({ + isFork: false, + labels: ["skip-ai-review"], + command: "/autofix blocking", + }), + ); + expect(plan.status).toBe("armed"); + }); + + it("still refuses a labelled PR with no review, on the review guard", () => { + // The old gate was justified as "with no review there is nothing to + // fix". That case is real; it is just already covered here, by the + // guard that actually checks for a review. + const plan = buildPlan( + input({ + isFork: false, + labels: ["autofix: blocking", "skip-ai-review"], + priorReviews: [], + }), + ); + expect(plan.status).toBe("refused"); + expect(plan.reason).toBe(REFUSAL_REASONS["no-review"]); + }); +}); diff --git a/workflows/autofix/lib/plan.ts b/workflows/autofix/lib/plan.ts new file mode 100644 index 00000000..1a815ae4 --- /dev/null +++ b/workflows/autofix/lib/plan.ts @@ -0,0 +1,353 @@ +/** + * The autofix plan: every decision this run makes, made in code before the + * agent is asked to edit anything. + * + * This module is the determinism boundary for autofix, mirroring the split the + * reviewer draws. CODE decides whether the run is armed, which findings are in + * scope, which are refused and why, and what the commit trailer says. The MODEL + * decides only how to change the code. Nothing here composes a sentence about + * the code under review, and nothing downstream re-opens a decision made here: + * the plan is final, and the prompt's contract is to execute it or stop. + * + * The plan has three outcomes and the distinction matters to the PR comment the + * run posts: + * - `armed` — there is work; the agent runs. + * - `no-op` — the request was understood and nothing needed fixing. Not a + * failure; the run says so and clears any label that armed it. + * - `refused` — the run cannot safely proceed (unimplemented token, no + * review, unreadable fingerprint). A label that armed it is still removed, + * because a label left on after a refusal reads as "still queued" when + * nothing is. + * + * Autofix is armed from either of two peer surfaces, a label or an `/autofix` + * comment. {@link resolveRequest} owns the rule that reconciles them, and the + * rule is that the trigger decides: they never union with each other. + */ + +import {AUTOFIX_LABEL_PREFIX, resolveCommand, resolveScope} from "./scope.ts"; +import type {RequestSurface, ScopeResolution} from "./scope.ts"; +import {buildWorkList} from "./worklist.ts"; +import type {SkippedThread, WorkItem} from "./worklist.ts"; +import { + assessReviewCurrency, + DEGRADED_NOTES, + REFUSAL_REASONS, +} from "./staleness.ts"; +import { + renderTrailer, + summariseLedger, + TRAILER_SCHEMA_VERSION, +} from "./trailer.ts"; +import type {StagedThread} from "../../review/lib/rereview.ts"; +import type {PriorReview} from "../../review/lib/rereview-mode.ts"; + +export type AutofixPlan = { + status: "armed" | "no-op" | "refused"; + /** One sentence, rendered verbatim into the run's PR comment. */ + reason: string; + /** + * Autofix labels to remove. Every autofix label on the PR, whatever the + * status, on a label-armed run; empty on a command-armed one, which has no + * label state of its own to tidy. + */ + labelsToRemove: string[]; + scopes: string[]; + items: WorkItem[]; + skipped: SkippedThread[]; + /** 1 in v1; the field a continual cadence would increment. */ + cycle: number; + /** Pre-rendered trailer block for the commit message; empty unless armed. */ + trailer: string; + /** Paths carrying hunks no stamped review has seen. */ + stalePaths: string[]; + /** Which surface armed the run; empty when nothing did. */ + surface: RequestSurface | ""; + /** + * Set when the file-level currency check could not run and the plan fell + * back to per-thread anchors. Rendered into the summary verbatim; empty + * when the fingerprint check ran normally. + */ + degradedNote: string; +}; + +export type PlanInput = { + labels: readonly string[]; + threads: readonly StagedThread[]; + priorReviews: readonly PriorReview[]; + /** The stripped diff of the current head (`full-stripped.diff` shape). */ + diffText: string; + /** Commit messages on the PR head, for the cycle ledger. */ + commitMessages: readonly string[]; + /** + * The body of the `/autofix` comment that triggered this run, when one did. + * Absent on a label-triggered run. + */ + command?: string; + /** + * Whether the PR head is a fork. Defaults to treating unknown as a fork, so + * an unreadable context refuses rather than proceeds. + */ + isFork?: boolean; + /** Login the reviewer posts as; threaded to the ownership guard. */ + botLogin?: string; +}; + +/** Every autofix label present, so a label-armed refusal still clears the PR. */ +const autofixLabelsOn = (labels: readonly string[]): string[] => + labels.filter((label) => label.startsWith(AUTOFIX_LABEL_PREFIX)); + +/** + * Resolve the request from whichever surface armed the run. + * + * **The trigger decides, and the two surfaces never union.** A run triggered by + * a comment resolves the comment; a run triggered by a label resolves the + * labels. Unioning them would mean a stale `autofix: nits` label silently + * widening someone's `/autofix blocking`, and would make the request depend on + * PR state the author was not looking at when they typed the command. + */ +const resolveRequest = (input: PlanInput): ScopeResolution => { + if (input.command !== undefined && input.command.trim() !== "") { + const fromCommand = resolveCommand(input.command); + if (fromCommand.status !== "none") { + return fromCommand; + } + } + return resolveScope(input.labels); +}; + +export const buildPlan = (input: PlanInput): AutofixPlan => { + const ledger = summariseLedger(input.commitMessages); + const base = { + labelsToRemove: [] as string[], + scopes: [] as string[], + items: [] as WorkItem[], + skipped: [] as SkippedThread[], + cycle: ledger.nextCycle, + trailer: "", + stalePaths: [] as string[], + surface: "" as RequestSurface | "", + degradedNote: "", + }; + + const resolution = resolveRequest(input); + if (resolution.status === "none") { + return { + ...base, + status: "no-op", + reason: "no autofix label or `/autofix` command armed this run.", + }; + } + + // A command is self-clearing, so only a label-armed run has label state to + // tidy. On that path every autofix label present comes off, not just the + // ones that resolved, so a refusal never leaves one reading as "queued". + const surface = + resolution.status === "armed" + ? resolution.request.surface + : resolution.surface; + base.surface = surface; + base.labelsToRemove = + surface === "command" ? [] : autofixLabelsOn(input.labels); + if (resolution.status === "rejected") { + return {...base, status: "refused", reason: resolution.reason}; + } + + // The `pull_request` branch of the workflow's `if:` gates on this before the + // job starts. The `issue_comment` branch CANNOT: that event carries no + // `github.event.pull_request`, so a command-armed run reaches here ungated. + // An earlier version of the workflow comment and the README both said this + // check "moves into the plan" while the plan did not implement it, which is + // how Khan/actions#298's review found it. + // + // Enforced on every path, not just the command one: a duplicated guard is + // cheap, and a missing one authorises a code push. + // + // `skip-ai-review` is NOT checked here, deliberately. It stops the reviewer + // from running again; it does not withdraw a review already posted, so a + // labelled PR can still carry current findings, and an explicit `autofix:` + // label or `/autofix` from someone with write access is the authorisation + // for acting on them. See the reasoning in `autofix.md`'s gate comment, and + // revisit it if autofix ever runs without a human arming it. + if (input.isFork !== false) { + return { + ...base, + status: "refused", + scopes: resolution.request.scopes, + reason: + "this PR's head is a fork, or its origin could not be " + + "determined, and autofix does not push to forks.", + }; + } + + const currency = assessReviewCurrency(input.priorReviews, input.diffText); + if (currency.status === "no-review") { + return { + ...base, + status: "refused", + scopes: resolution.request.scopes, + reason: REFUSAL_REASONS["no-review"], + }; + } + + // No fingerprint means the file-level check cannot run, not that the run + // must stop: the per-thread anchor check in `buildWorkList` still applies, + // and it is the signal that covers "the author edited the flagged code". + // The note is carried into the summary so the weaker check is never silent. + const degradedNote = + currency.status === "unverifiable" ? DEGRADED_NOTES[currency.why] : ""; + base.degradedNote = degradedNote; + + const stalePaths = currency.status === "current" ? currency.stalePaths : []; + const stale = new Set(stalePaths); + const {items, skipped} = buildWorkList( + input.threads, + resolution.request.findingLabels, + input.botLogin, + ); + + // Drop findings whose file moved on after the review that raised them. The + // finding may already be fixed, or may describe code that no longer exists; + // either way the statement is no longer known to be true of this head. + const actionable: WorkItem[] = []; + const allSkipped = [...skipped]; + for (const item of items) { + if (stale.has(item.path)) { + allSkipped.push({ + threadId: item.threadId, + path: item.path, + reason: "stale-path", + label: item.label, + }); + continue; + } + actionable.push(item); + } + + const common = { + ...base, + scopes: resolution.request.scopes, + skipped: allSkipped, + stalePaths, + }; + + if (actionable.length === 0) { + return { + ...common, + status: "no-op", + reason: + `no open ${resolution.request.scopes.join(" or ")} findings ` + + `are actionable on this head` + + (allSkipped.length === 0 + ? "." + : ` (${allSkipped.length} thread(s) skipped; see below).`), + }; + } + + return { + ...common, + status: "armed", + items: actionable, + reason: `fixing ${actionable.length} ${resolution.request.scopes.join( + " and ", + )} finding(s).`, + trailer: renderTrailer({ + schemaVersion: TRAILER_SCHEMA_VERSION, + scopes: resolution.request.scopes, + cycle: ledger.nextCycle, + threadIds: actionable.map((item) => item.threadId), + }), + }; +}; + +/* -------------------------------------------------------------------------- */ +/* CLI */ +/* -------------------------------------------------------------------------- */ + +/** Injected filesystem, so the CLI is unit-testable without touching disk. */ +export type PlanCliFs = { + readFileSync: (path: string) => string; + writeFileSync: (path: string, data: string) => void; + existsSync: (path: string) => boolean; +}; + +export const AUTOFIX_DIR = "/tmp/gh-aw/autofix"; + +const readJson = (fs: PlanCliFs, path: string, fallback: T): T => { + if (!fs.existsSync(path)) { + return fallback; + } + try { + return JSON.parse(fs.readFileSync(path)) as T; + } catch { + return fallback; + } +}; + +/** + * Read the staged inputs, write `plan.json`, and return the plan. + * + * A missing or malformed input degrades to its empty value rather than + * throwing, which routes it into the ordinary refusal path (no reviews reads as + * `no-review`) instead of failing the job with a stack trace the PR author + * cannot act on. + */ +export const runPlanCli = (fs: PlanCliFs, dir = AUTOFIX_DIR): AutofixPlan => { + const plan = buildPlan({ + labels: readJson(fs, `${dir}/labels.json`, []), + threads: readJson(fs, `${dir}/threads.json`, []), + priorReviews: readJson( + fs, + `${dir}/prior-reviews.json`, + [], + ), + diffText: fs.existsSync(`${dir}/pr.diff`) + ? fs.readFileSync(`${dir}/pr.diff`) + : "", + commitMessages: readJson(fs, `${dir}/commits.json`, []), + // Staged only on the comment-triggered path. Absent on a label run, + // and absent for any consumer still on an install that predates the + // command surface, which falls back to labels unchanged. + command: fs.existsSync(`${dir}/command.txt`) + ? fs.readFileSync(`${dir}/command.txt`) + : undefined, + // Missing context reads as a fork, so a staging gap refuses. + isFork: + readJson<{isFork?: boolean}>(fs, `${dir}/context.json`, {}) + .isFork !== false, + botLogin: process.env.AUTOFIX_BOT_LOGIN?.trim() || undefined, + }); + fs.writeFileSync(`${dir}/plan.json`, `${JSON.stringify(plan, null, 2)}\n`); + return plan; +}; + +// Run only when executed directly (autofix.md), never on import (tests). +if (typeof require !== "undefined" && require.main === module) { + const nodeFs = require("node:fs"); + // Optional staging directory argument. autofix.md passes nothing and gets + // AUTOFIX_DIR; a human reproducing a run points it at a copy of the staged + // inputs, which is the only way to re-run a plan outside the workflow. + const plan = runPlanCli( + { + readFileSync: (path: string) => nodeFs.readFileSync(path, "utf-8"), + writeFileSync: (path: string, data: string) => + nodeFs.writeFileSync(path, data), + existsSync: (path: string) => nodeFs.existsSync(path), + }, + process.argv[2] || AUTOFIX_DIR, + ); + // stdout is the prompt's read surface: status and reason drive the + // comment, the item count drives whether the agent edits anything. + // eslint-disable-next-line no-console + console.log( + JSON.stringify({ + status: plan.status, + reason: plan.reason, + surface: plan.surface, + scopes: plan.scopes, + cycle: plan.cycle, + degraded: plan.degradedNote !== "", + itemCount: plan.items.length, + skippedCount: plan.skipped.length, + }), + ); +} diff --git a/workflows/autofix/lib/scope.test.ts b/workflows/autofix/lib/scope.test.ts new file mode 100644 index 00000000..8c576c2a --- /dev/null +++ b/workflows/autofix/lib/scope.test.ts @@ -0,0 +1,281 @@ +import {describe, expect, it} from "vitest"; + +import { + AUTOFIX_SCOPES, + findingLabelsForScope, + isLoopEligible, + labelForToken, + resolveCommand, + resolveScope, + SCOPE_LABELS, + SCOPE_TOKENS, + UNIMPLEMENTED_LABELS, + UNIMPLEMENTED_TOKENS, +} from "./scope.ts"; +import { + BLOCKING_LABELS, + NON_BLOCKING_LABELS, +} from "../../review/lib/render-comment.ts"; + +describe("resolveScope", () => { + it("reports none when no autofix label is present", () => { + expect(resolveScope(["skip-ai-review", "bug"])).toEqual({ + status: "none", + }); + }); + + it("arms a single scope", () => { + const result = resolveScope(["autofix: blocking"]); + expect(result.status).toBe("armed"); + if (result.status !== "armed") { + return; + } + expect(result.request.scopes).toEqual(["blocking"]); + expect(result.request.labels).toEqual(["autofix: blocking"]); + expect(result.request.findingLabels).toEqual([...BLOCKING_LABELS]); + }); + + it("unions both scopes and orders them canonically", () => { + // Labels supplied in the reverse of AUTOFIX_SCOPES order. + const result = resolveScope(["autofix: nits", "autofix: blocking"]); + expect(result.status).toBe("armed"); + if (result.status !== "armed") { + return; + } + expect(result.request.scopes).toEqual(["blocking", "nits"]); + expect(result.request.findingLabels).toEqual([ + ...BLOCKING_LABELS, + ...NON_BLOCKING_LABELS, + ]); + }); + + it("ignores non-autofix labels alongside an autofix one", () => { + const result = resolveScope(["bug", "autofix: nits", "priority"]); + expect(result.status).toBe("armed"); + if (result.status !== "armed") { + return; + } + expect(result.request.scopes).toEqual(["nits"]); + }); + + it("rejects a label on an axis this version does not implement", () => { + const result = resolveScope(["autofix: blocking", "autofix: loop"]); + expect(result.status).toBe("rejected"); + if (result.status !== "rejected") { + return; + } + expect(result.labels).toEqual(["autofix: loop"]); + expect(result.reason).toContain("cadence"); + }); + + it("rejects rather than ignores an unrecognised autofix label", () => { + const result = resolveScope(["autofix: everything"]); + expect(result.status).toBe("rejected"); + if (result.status !== "rejected") { + return; + } + expect(result.reason).toContain("unrecognised"); + // The message tells the author what they could have used instead. + expect(result.reason).toContain("autofix: blocking"); + }); + + it("rejects when an unimplemented label is present even with a valid one", () => { + // Guards the silent-drop failure: honouring the blocking half of + // `blocking + loop` would look like a loop that stopped after one run. + expect(resolveScope(["autofix: nits", "autofix: human"]).status).toBe( + "rejected", + ); + }); +}); + +describe("the label vocabulary", () => { + it("keeps every known label inside the shared namespace", () => { + for (const label of [ + ...Object.keys(SCOPE_LABELS), + ...Object.keys(UNIMPLEMENTED_LABELS), + ]) { + expect(label.startsWith("autofix: ")).toBe(true); + } + }); + + it("never lets a label mean two things", () => { + for (const label of Object.keys(SCOPE_LABELS)) { + expect(UNIMPLEMENTED_LABELS[label]).toBeUndefined(); + } + }); + + it("covers every scope with exactly one label", () => { + expect(Object.values(SCOPE_LABELS).sort()).toEqual( + [...AUTOFIX_SCOPES].sort(), + ); + }); +}); + +describe("isLoopEligible", () => { + // The constraint most likely to be violated by whoever adds the cadence + // axis: non-blocking findings have no fixed point, so they cannot loop. + it("allows blocking scope to loop", () => { + expect(isLoopEligible("blocking")).toBe(true); + }); + + it("forbids nits from ever looping", () => { + expect(isLoopEligible("nits")).toBe(false); + }); +}); + +describe("findingLabelsForScope", () => { + it("maps each scope onto the reviewer's own taxonomy", () => { + expect(findingLabelsForScope("blocking")).toEqual(BLOCKING_LABELS); + expect(findingLabelsForScope("nits")).toEqual(NON_BLOCKING_LABELS); + }); + + it("keeps the two classes disjoint", () => { + const blocking = new Set(findingLabelsForScope("blocking")); + for (const label of findingLabelsForScope("nits")) { + expect(blocking.has(label)).toBe(false); + } + }); +}); + +describe("resolveCommand", () => { + it("reports none for a comment that is not an autofix command", () => { + expect(resolveCommand("looks good to me").status).toBe("none"); + expect(resolveCommand("/review").status).toBe("none"); + }); + + it("does not match a command that is only a prefix of another word", () => { + expect(resolveCommand("/autofixer please").status).toBe("none"); + }); + + it("arms blocking scope for a bare command", () => { + const result = resolveCommand("/autofix"); + expect(result.status).toBe("armed"); + if (result.status !== "armed") { + return; + } + expect(result.request.scopes).toEqual(["blocking"]); + expect(result.request.surface).toBe("command"); + }); + + it("arms the scopes named as arguments", () => { + const result = resolveCommand("/autofix nits"); + if (result.status !== "armed") { + throw new Error("expected armed"); + } + expect(result.request.scopes).toEqual(["nits"]); + }); + + it("unions multiple arguments and orders them canonically", () => { + const result = resolveCommand("/autofix nits blocking"); + if (result.status !== "armed") { + throw new Error("expected armed"); + } + expect(result.request.scopes).toEqual(["blocking", "nits"]); + }); + + it("tolerates a trailing CRLF", () => { + // The exact shape that silently killed /review in Khan/webapp#40943: + // the GitHub web UI appends \r\n when you press Enter after a command. + const result = resolveCommand("/autofix blocking\r\n"); + if (result.status !== "armed") { + throw new Error("expected armed"); + } + expect(result.request.scopes).toEqual(["blocking"]); + }); + + it("tolerates a bare command with a trailing CRLF", () => { + expect(resolveCommand("/autofix\r\n").status).toBe("armed"); + }); + + it("tolerates leading whitespace and tabs between arguments", () => { + const result = resolveCommand(" /autofix\tblocking nits "); + if (result.status !== "armed") { + throw new Error("expected armed"); + } + expect(result.request.scopes).toEqual(["blocking", "nits"]); + }); + + it("reads arguments from the command line only, ignoring later prose", () => { + const result = resolveCommand( + "/autofix blocking\n\nbut please keep the existing naming", + ); + if (result.status !== "armed") { + throw new Error("expected armed"); + } + expect(result.request.scopes).toEqual(["blocking"]); + }); + + it("rejects an unimplemented axis and quotes the command form", () => { + const result = resolveCommand("/autofix loop"); + expect(result.status).toBe("rejected"); + if (result.status !== "rejected") { + return; + } + expect(result.surface).toBe("command"); + expect(result.reason).toContain("cadence"); + expect(result.labels).toEqual(["/autofix loop"]); + }); + + it("rejects an unrecognised argument", () => { + const result = resolveCommand("/autofix everything"); + expect(result.status).toBe("rejected"); + if (result.status !== "rejected") { + return; + } + expect(result.reason).toContain("unrecognised"); + expect(result.reason).toContain("/autofix blocking"); + }); + + it("never asks a command-armed run to remove labels", () => { + const result = resolveCommand("/autofix blocking"); + if (result.status !== "armed") { + throw new Error("expected armed"); + } + expect(result.request.labels).toEqual([]); + }); +}); + +describe("the two surfaces agree", () => { + // The whole reason both funnel through one resolver: a token must not mean + // one thing as a label and another as a command. + it("resolves the same scopes from either surface", () => { + for (const [labels, command] of [ + [["autofix: blocking"], "/autofix blocking"], + [["autofix: nits"], "/autofix nits"], + [["autofix: blocking", "autofix: nits"], "/autofix blocking nits"], + ] as const) { + const fromLabel = resolveScope(labels); + const fromCommand = resolveCommand(command); + if ( + fromLabel.status !== "armed" || + fromCommand.status !== "armed" + ) { + throw new Error(`expected both armed for ${command}`); + } + expect(fromCommand.request.scopes).toEqual( + fromLabel.request.scopes, + ); + expect(fromCommand.request.findingLabels).toEqual( + fromLabel.request.findingLabels, + ); + } + }); + + it("rejects the same tokens on either surface", () => { + for (const token of Object.keys(UNIMPLEMENTED_TOKENS)) { + expect(resolveScope([labelForToken(token)]).status).toBe( + "rejected", + ); + expect(resolveCommand(`/autofix ${token}`).status).toBe("rejected"); + } + }); + + it("derives the label tables from the token tables", () => { + expect(Object.keys(SCOPE_LABELS)).toEqual( + Object.keys(SCOPE_TOKENS).map(labelForToken), + ); + expect(Object.keys(UNIMPLEMENTED_LABELS)).toEqual( + Object.keys(UNIMPLEMENTED_TOKENS).map(labelForToken), + ); + }); +}); diff --git a/workflows/autofix/lib/scope.ts b/workflows/autofix/lib/scope.ts new file mode 100644 index 00000000..78955813 --- /dev/null +++ b/workflows/autofix/lib/scope.ts @@ -0,0 +1,284 @@ +/** + * The autofix request vocabulary, and the scope it resolves to. + * + * Autofix is armed two ways, and they are peers: a namespaced `autofix: ` + * label, or an `/autofix [value …]` PR comment. Both are first-class; neither is + * a shorthand for the other. What they share is the **token** — the value after + * the namespace — which is the single currency this module resolves. Both + * surfaces funnel through {@link resolveTokens}, so a value can never mean one + * thing as a label and another as a command. + * + * The token space is flat while the semantics are not. Three axes exist, and a + * token names a value on exactly one of them: + * + * - **scope** — which review findings are in scope. Tokens union: `blocking` + * and `nits` together fix both classes in one run. + * - **cadence** — how many autofix runs one arming authorises. Absent means + * `once`, which is the only cadence this version implements. + * - **source** — whose feedback is fixed. Absent means the reviewer bot, the + * only source this version implements. + * + * The combination rule is: union within the scope axis, union within the source + * axis, cadence is a flag. Because unioning happens *within* an axis, the token + * space stays bounded by the axes (five tokens across all three) rather than + * growing as their product — there is no `blocking-loop` token, and there must + * never be one. + * + * A token on an axis this version does not implement is REJECTED, not ignored + * ({@link UNIMPLEMENTED_TOKENS}). A silently-dropped `loop` would look like a + * working loop that stopped after one cycle, which is the worst of both + * behaviours. + * + * One constraint outlives this version and is enforced here rather than left to + * convention: **`nits` is never loop-eligible**. Non-blocking findings have no + * fixed point (the reviewer will always find something cosmetic in the + * autofixer's own output), so a nits-scoped loop cannot converge and must not be + * offered. See {@link isLoopEligible}; when the cadence axis lands, the loop + * token has to consult it. + */ + +import { + BLOCKING_LABELS, + NON_BLOCKING_LABELS, +} from "../../review/lib/render-comment.ts"; + +/** Namespace every autofix label shares. */ +export const AUTOFIX_LABEL_PREFIX = "autofix: "; + +/** The command form, at the start of a PR comment. */ +export const AUTOFIX_COMMAND = "/autofix"; + +/** + * The scope axis: which class of review finding a token puts in scope. + * `blocking` is the default a repo reaches for (it terminates naturally at the + * merge gate); `nits` is the deliberate one-shot tidy-up. + */ +export const AUTOFIX_SCOPES = ["blocking", "nits"] as const; + +export type AutofixScope = typeof AUTOFIX_SCOPES[number]; + +/** + * Scope-axis tokens. The only tokens this version acts on. + */ +export const SCOPE_TOKENS: Readonly> = { + blocking: "blocking", + nits: "nits", +}; + +/** + * Tokens reserved for axes a later version implements. Listed so the resolver + * can fail loudly and specifically ("not implemented yet") instead of treating + * them as typos, and so nobody reuses one of these strings for something else. + */ +export const UNIMPLEMENTED_TOKENS: Readonly> = { + loop: "the cadence axis (continual autofix) is not implemented", + human: "the source axis (human feedback) is not implemented", + author: "the source axis (author feedback) is not implemented", +}; + +/** The label form of a token. */ +export const labelForToken = (token: string): string => + `${AUTOFIX_LABEL_PREFIX}${token}`; + +const byLabel = (tokens: Readonly>): Record => + Object.fromEntries( + Object.entries(tokens).map(([token, value]) => [ + labelForToken(token), + value, + ]), + ); + +/** Scope-axis label -> scope. Derived from {@link SCOPE_TOKENS}. */ +export const SCOPE_LABELS: Readonly> = + byLabel(SCOPE_TOKENS); + +/** Reserved labels -> why they are refused. Derived from the token table. */ +export const UNIMPLEMENTED_LABELS: Readonly> = + byLabel(UNIMPLEMENTED_TOKENS); + +/** + * The scope a bare `/autofix` means. Blocking, because it is the scope that + * terminates at the merge gate and the one people reach for; a bare command + * must not silently do the open-ended thing. + * + * There is deliberately NO bare `autofix` label equivalent. A label carries no + * arguments, so a bare label and a scoped one would be two ways to say the same + * thing with nothing distinguishing them at a glance. + */ +export const DEFAULT_COMMAND_SCOPE: AutofixScope = "blocking"; + +/** + * Whether a scope may ever be driven by a loop cadence. Blocking findings + * terminate at the merge gate; non-blocking ones have no fixed point. This + * version has no loop, but the rule is encoded now because it is the constraint + * most likely to be violated by whoever adds one later. + */ +export const isLoopEligible = (scope: AutofixScope): boolean => + scope === "blocking"; + +/** The Conventional-Comment labels a given autofix scope covers. */ +export const findingLabelsForScope = (scope: AutofixScope): readonly string[] => + scope === "blocking" ? BLOCKING_LABELS : NON_BLOCKING_LABELS; + +/** How this run was armed. Recorded so the summary can say which surface. */ +export type RequestSurface = "label" | "command"; + +/** A resolved arming request: what this run was asked to do. */ +export type AutofixRequest = { + /** Which surface armed it. */ + surface: RequestSurface; + /** Scopes in effect, in {@link AUTOFIX_SCOPES} order, deduplicated. */ + scopes: AutofixScope[]; + /** + * The autofix labels this run must remove. Populated only for the `label` + * surface: a comment is self-clearing, so a command-armed run has no label + * state to tidy and must not go removing labels nobody acted on. + */ + labels: string[]; + /** Every Conventional-Comment label the union of scopes covers. */ + findingLabels: string[]; +}; + +export type ScopeResolution = + | {status: "none"} + /** + * `surface` is carried on the rejection too, not just on success: the + * caller decides whether to clear labels from it, and a command-armed + * rejection must not go removing labels nobody acted on. + */ + | { + status: "rejected"; + surface: RequestSurface; + reason: string; + labels: string[]; + } + | {status: "armed"; request: AutofixRequest}; + +/** + * The shared core both surfaces resolve through. + * + * `render` turns a token back into the form the user actually typed, so a + * rejection quotes their input rather than a normalised version of it. + */ +const resolveTokens = ( + tokens: readonly string[], + surface: RequestSurface, + render: (token: string) => string, +): ScopeResolution => { + const unimplemented = tokens.filter( + (token) => UNIMPLEMENTED_TOKENS[token] !== undefined, + ); + if (unimplemented.length > 0) { + return { + status: "rejected", + surface, + labels: unimplemented.map(render), + reason: unimplemented + .map( + (token) => + `\`${render(token)}\`: ${UNIMPLEMENTED_TOKENS[token]}`, + ) + .join("; "), + }; + } + + const unknown = tokens.filter((token) => SCOPE_TOKENS[token] === undefined); + if (unknown.length > 0) { + return { + status: "rejected", + surface, + labels: unknown.map(render), + reason: + `unrecognised autofix ${surface}(s): ` + + `${unknown.map((t) => `\`${render(t)}\``).join(", ")}. ` + + `Known: ${Object.keys(SCOPE_TOKENS) + .map((t) => `\`${render(t)}\``) + .join(", ")}`, + }; + } + + // Order by AUTOFIX_SCOPES, not by input order, so the request (and every + // artifact rendered from it) is stable regardless of the order GitHub + // returns labels in or the order someone typed the arguments. + const selected = new Set(tokens.map((token) => SCOPE_TOKENS[token])); + const scopes = AUTOFIX_SCOPES.filter((scope) => selected.has(scope)); + + return { + status: "armed", + request: { + surface, + scopes: [...scopes], + labels: surface === "label" ? scopes.map(labelForToken) : [], + findingLabels: scopes.flatMap((scope) => [ + ...findingLabelsForScope(scope), + ]), + }, + }; +}; + +/** + * Resolve the PR's labels into an autofix request. + * + * `none` means no autofix label is present. `rejected` means an + * autofix-namespaced label is present that this version cannot honour — an + * unimplemented axis, or an unrecognised value — and the run must stop and say + * so rather than guess at an intent. + */ +export const resolveScope = (labels: readonly string[]): ScopeResolution => { + const namespaced = labels.filter((label) => + label.startsWith(AUTOFIX_LABEL_PREFIX), + ); + if (namespaced.length === 0) { + return {status: "none"}; + } + return resolveTokens( + namespaced.map((label) => label.slice(AUTOFIX_LABEL_PREFIX.length)), + "label", + labelForToken, + ); +}; + +/** + * Match `/autofix` at the start of a comment body, requiring the command to be + * followed by whitespace or end-of-body. + * + * The trailing-whitespace tolerance is not incidental. gh-aw's own + * `slash_command` gate only matches a bare `\n` or end-of-body, so a comment + * saved with a trailing CRLF — which the GitHub web UI produces when you press + * Enter after the command — never activates the workflow. That cost Khan/webapp + * a silently-dead `/review` (Khan/webapp#40943), which is why the reviewer there + * uses a raw `issue_comment` trigger with its own gate, and why this parser and + * the workflow's `if:` do the same. + */ +const COMMAND_RE = new RegExp(`^${AUTOFIX_COMMAND}(?=\\s|$)`); + +/** + * Resolve an `/autofix` comment body into an autofix request. + * + * Returns `none` when the body is not an autofix command at all, so the caller + * can fall through to the label surface. + * + * Arguments are read from the command's own line only. Prose on later lines is + * ignored rather than parsed as tokens, so + * `/autofix blocking\n\nkeep the existing naming please` arms cleanly and the + * context is still there for a human reading the thread. + */ +export const resolveCommand = (body: string): ScopeResolution => { + const firstLine = body.trimStart().split(/\r?\n/, 1)[0] ?? ""; + if (!COMMAND_RE.test(firstLine)) { + return {status: "none"}; + } + + const args = firstLine + .slice(AUTOFIX_COMMAND.length) + .trim() + .split(/\s+/) + .filter((token) => token !== ""); + + const tokens = args.length === 0 ? [DEFAULT_COMMAND_SCOPE] : args; + return resolveTokens( + tokens, + "command", + (token) => `${AUTOFIX_COMMAND} ${token}`, + ); +}; diff --git a/workflows/autofix/lib/stage.test.ts b/workflows/autofix/lib/stage.test.ts new file mode 100644 index 00000000..75f7cdc1 --- /dev/null +++ b/workflows/autofix/lib/stage.test.ts @@ -0,0 +1,554 @@ +import {describe, expect, it} from "vitest"; + +import { + assertNoGraphqlErrors, + collectInputs, + collectThreads, + writeInputs, +} from "./stage.ts"; +import type {StageCliFs, StagePort} from "./stage.ts"; +import {computeHunkSignature} from "../../review/lib/rereview-mode.ts"; + +const BOT = "github-actions[bot]"; + +const threadNode = (over: Record = {}) => ({ + id: "PRRT_1", + isResolved: false, + path: "src/a.ts", + line: 12, + comments: { + nodes: [ + { + author: {login: BOT}, + body: "**issue (blocking):** boom", + url: "https://github.com/o/r/pull/1#discussion_r1", + }, + ], + }, + ...over, +}); + +const portFor = (opts: { + threadPages?: unknown[]; + rest?: Record; + paged?: Record; + checkoutSha?: string; +}): StagePort => { + let page = 0; + return { + checkoutHeadSha: () => opts.checkoutSha ?? "", + rest: async () => opts.rest ?? {}, + restPaged: async (path) => { + for (const [key, value] of Object.entries(opts.paged ?? {})) { + if (path.endsWith(key)) { + return value; + } + } + return []; + }, + // Falls back to a well-formed EMPTY page, not `{}`. Staging now throws + // on a body it cannot parse, so the default has to be a valid response + // that simply carries no threads; `{}` would make every test that does + // not care about threads fail as a rate-limit. Evaluated lazily, after + // `onePage` is initialised. + graphql: async () => (opts.threadPages ?? [])[page++] ?? onePage([]), + }; +}; + +const onePage = (nodes: unknown[], hasNextPage = false) => ({ + data: { + repository: { + pullRequest: { + reviewThreads: { + pageInfo: {hasNextPage, endCursor: "c1"}, + nodes, + }, + }, + }, + }, +}); + +describe("collectThreads", () => { + it("keeps an unresolved thread opened by the bot", async () => { + const port = portFor({threadPages: [onePage([threadNode()])]}); + const threads = await collectThreads(port, "o", "r", 1, BOT); + expect(threads).toEqual([ + { + thread_id: "PRRT_1", + path: "src/a.ts", + line: 12, + url: "https://github.com/o/r/pull/1#discussion_r1", + comments: [{author: BOT, body: "**issue (blocking):** boom"}], + }, + ]); + }); + + it("drops resolved threads", async () => { + const port = portFor({ + threadPages: [onePage([threadNode({isResolved: true})])], + }); + expect(await collectThreads(port, "o", "r", 1, BOT)).toEqual([]); + }); + + it("drops threads a human started", async () => { + // Somebody else's conversation; autofix stays out of it, the same line + // the reviewer draws with human-threads.json. + const human = threadNode({ + comments: {nodes: [{author: {login: "alice"}, body: "hmm"}]}, + }); + expect( + await collectThreads( + portFor({threadPages: [onePage([human])]}), + "o", + "r", + 1, + BOT, + ), + ).toEqual([]); + }); + + it("keeps a bot thread that a human replied to, with the full chain", async () => { + const withReply = threadNode({ + comments: { + nodes: [ + {author: {login: BOT}, body: "**issue (blocking):** boom"}, + {author: {login: "alice"}, body: "already handled"}, + ], + }, + }); + const threads = await collectThreads( + portFor({threadPages: [onePage([withReply])]}), + "o", + "r", + 1, + BOT, + ); + expect(threads[0].comments).toHaveLength(2); + expect(threads[0].comments[1]).toEqual({ + author: "alice", + body: "already handled", + }); + }); + + it("copies comment bodies verbatim", async () => { + // The label parser reads `**label:**` off this string; normalising it + // is exactly how a finding becomes unclassifiable. + const body = "**issue (blocking):** boom\r\n\r\n indented\t"; + const node = threadNode({ + comments: {nodes: [{author: {login: BOT}, body}]}, + }); + const threads = await collectThreads( + portFor({threadPages: [onePage([node])]}), + "o", + "r", + 1, + BOT, + ); + expect(threads[0].comments[0].body).toBe(body); + }); + + it("carries a null line for an outdated thread", async () => { + const node = threadNode({line: null}); + const threads = await collectThreads( + portFor({threadPages: [onePage([node])]}), + "o", + "r", + 1, + BOT, + ); + expect(threads[0].line).toBeNull(); + }); + + it("omits url when the API returned none", async () => { + const node = threadNode({ + comments: { + nodes: [ + {author: {login: BOT}, body: "**note (non-blocking):** x"}, + ], + }, + }); + const threads = await collectThreads( + portFor({threadPages: [onePage([node])]}), + "o", + "r", + 1, + BOT, + ); + expect("url" in threads[0]).toBe(false); + }); + + it("follows pagination", async () => { + const port = portFor({ + threadPages: [ + onePage([threadNode({id: "A"})], true), + onePage([threadNode({id: "B"})]), + ], + }); + const threads = await collectThreads(port, "o", "r", 1, BOT); + expect(threads.map((t) => t.thread_id)).toEqual(["A", "B"]); + }); + + it("stops rather than looping when a page omits its cursor", async () => { + const noCursor = { + data: { + repository: { + pullRequest: { + reviewThreads: { + pageInfo: {hasNextPage: true}, + nodes: [threadNode()], + }, + }, + }, + }, + }; + const threads = await collectThreads( + portFor({threadPages: [noCursor, noCursor, noCursor]}), + "o", + "r", + 1, + BOT, + ); + expect(threads).toHaveLength(1); + }); + + // These four pin the fail-CLOSED direction. Staging zero threads on a PR + // that has open findings is the one failure this module cannot absorb: the + // plan becomes a no-op, the no-op removes the arming label, and the author + // is told there is nothing to fix. Nothing is left to re-arm from, so the + // findings stay open silently. Throwing instead fails the staging step + // before any AI spend, and the label survives for a retry. + it("throws when GraphQL reports a rate limit as HTTP 200 with errors", async () => { + await expect( + collectThreads( + portFor({threadPages: [{errors: [{type: "RATE_LIMITED"}]}]}), + "o", + "r", + 1, + BOT, + ), + ).rejects.toThrow(/RATE_LIMITED/); + }); + + it("throws on partial data carrying errors, rather than staging the subset", async () => { + await expect( + collectThreads( + portFor({ + threadPages: [ + { + ...onePage([threadNode()]), + errors: [{type: "FORBIDDEN"}], + }, + ], + }), + "o", + "r", + 1, + BOT, + ), + ).rejects.toThrow(/FORBIDDEN/); + }); + + it("throws for a malformed GraphQL body", async () => { + await expect( + collectThreads( + portFor({threadPages: [{errors: []}]}), + "o", + "r", + 1, + BOT, + ), + ).rejects.toThrow(/no reviewThreads connection/); + }); + + it("still returns an empty list for a well-formed PR with no threads", async () => { + expect( + await collectThreads( + portFor({threadPages: [onePage([])]}), + "o", + "r", + 1, + BOT, + ), + ).toEqual([]); + }); +}); + +describe("assertNoGraphqlErrors", () => { + it("passes a clean body", () => { + expect(() => assertNoGraphqlErrors(onePage([]))).not.toThrow(); + }); + + it("passes an empty errors array", () => { + // GitHub omits `errors` on success; an empty array is not a failure. + expect(() => + assertNoGraphqlErrors({data: {}, errors: []}), + ).not.toThrow(); + }); + + it("throws on any error entry, and names it", () => { + expect(() => + assertNoGraphqlErrors({errors: [{type: "RATE_LIMITED"}]}), + ).toThrow(/RATE_LIMITED/); + }); + + it("ignores a non-object body", () => { + expect(() => assertNoGraphqlErrors(null)).not.toThrow(); + }); +}); + +describe("collectInputs", () => { + const port = portFor({ + rest: { + labels: [{name: "autofix: blocking"}, {name: "bug"}], + head: {sha: "abc123"}, + }, + paged: { + "/reviews": [ + { + user: {login: BOT}, + body: "Changes requested.", + submitted_at: "2026-07-01T00:00:00Z", + }, + { + user: {login: "alice"}, + body: "lgtm", + submitted_at: "2026-07-02T00:00:00Z", + }, + ], + "/files": [{filename: "a.ts", patch: "@@ -1,1 +1,2 @@\n c\n+x"}], + "/commits": [ + {commit: {message: "feat: a"}}, + {commit: {message: "fix: b"}}, + ], + }, + threadPages: [onePage([threadNode()])], + }); + + it("stages every input the plan needs", async () => { + const inputs = await collectInputs(port, "o", "r", 1, BOT); + expect(inputs.labels).toEqual(["autofix: blocking", "bug"]); + expect(inputs.headSha).toBe("abc123"); + expect(inputs.commitMessages).toEqual(["feat: a", "fix: b"]); + expect(inputs.threads).toHaveLength(1); + }); + + it("keeps only the reviewer bot's reviews", async () => { + const inputs = await collectInputs(port, "o", "r", 1, BOT); + expect(inputs.priorReviews).toEqual([ + {body: "Changes requested.", submittedAt: "2026-07-01T00:00:00Z"}, + ]); + }); +}); + +describe("writeInputs", () => { + const fsFor = () => { + const written: Record = {}; + const dirs: string[] = []; + const fs: StageCliFs = { + mkdirSync: (p) => { + dirs.push(p); + }, + writeFileSync: (p, d) => { + written[p] = d; + }, + }; + return {fs, written, dirs}; + }; + + const inputs = { + labels: ["autofix: blocking"], + isFork: false, + threads: [], + priorReviews: [], + diffText: "diff --git a/a b/a\n", + commitMessages: ["feat: a"], + headSha: "abc123", + }; + + it("creates the out directory and writes every file", () => { + const {fs, written, dirs} = fsFor(); + writeInputs(fs, inputs, "/d"); + expect(dirs).toEqual(["/d/out"]); + expect(Object.keys(written).sort()).toEqual([ + "/d/commits.json", + "/d/context.json", + "/d/head-sha.txt", + "/d/labels.json", + "/d/pr.diff", + "/d/prior-reviews.json", + "/d/threads.json", + ]); + }); + + it("omits command.txt on a label-armed run", () => { + // Its ABSENCE is what tells plan.ts to resolve labels, so an empty file + // would silently switch the resolver's surface. + const {fs, written} = fsFor(); + writeInputs(fs, inputs, "/d"); + expect("/d/command.txt" in written).toBe(false); + writeInputs(fs, inputs, "/d", " "); + expect("/d/command.txt" in written).toBe(false); + }); + + it("writes command.txt verbatim, trailing CRLF included", () => { + const {fs, written} = fsFor(); + writeInputs(fs, inputs, "/d", "/autofix blocking\r\n"); + expect(written["/d/command.txt"]).toBe("/autofix blocking\r\n"); + }); + + it("writes the diff raw, not JSON-wrapped", () => { + const {fs, written} = fsFor(); + writeInputs(fs, inputs, "/d"); + expect(written["/d/pr.diff"]).toBe("diff --git a/a b/a\n"); + }); +}); + +describe("the head SHA comes from the checkout", () => { + const base = { + rest: {labels: [], head: {sha: "from-api"}}, + threadPages: [onePage([])], + }; + + it("prefers the checked-out HEAD over the API", () => { + // Khan/actions#298 review: the edits are made against the checkout, so + // comparing an API read leaves a window where both reads agree while + // the working tree is already stale. + return collectInputs( + portFor({...base, checkoutSha: "from-checkout"}), + "o", + "r", + 1, + BOT, + ).then((i) => expect(i.headSha).toBe("from-checkout")); + }); + + it("falls back to the API when no checkout is available", async () => { + const inputs = await collectInputs(portFor(base), "o", "r", 1, BOT); + expect(inputs.headSha).toBe("from-api"); + }); +}); + +describe("the staged diff round-trips into the currency check", () => { + // autofix rebuilds the diff with the reviewer's `buildUnifiedDiff` and then + // parses it with the reviewer's `computeHunkSignature`. Both live in the + // other package, so this pins the contract between them from this side: if + // either changes shape, the currency guard silently stops seeing files. + it("produces a signature keyed by path", async () => { + const inputs = await collectInputs( + portFor({ + rest: {labels: [], head: {sha: "s"}}, + paged: { + "/files": [ + { + filename: "src/a.ts", + status: "modified", + patch: "@@ -1,1 +1,2 @@\n context\n+added", + }, + ], + }, + threadPages: [onePage([])], + }), + "o", + "r", + 1, + BOT, + ); + expect(Object.keys(computeHunkSignature(inputs.diffText))).toEqual([ + "src/a.ts", + ]); + }); + + it("keeps a renamed file keyed by its new path", async () => { + // The reason for using the shared builder: a local copy emitted the new + // name on both sides, which is wrong for a rename. + const inputs = await collectInputs( + portFor({ + rest: {labels: [], head: {sha: "s"}}, + paged: { + "/files": [ + { + filename: "src/new.ts", + previous_filename: "src/old.ts", + status: "renamed", + patch: "@@ -1,1 +1,2 @@\n c\n+x", + }, + ], + }, + threadPages: [onePage([])], + }), + "o", + "r", + 1, + BOT, + ); + expect(inputs.diffText).toContain( + "diff --git a/src/old.ts b/src/new.ts", + ); + expect(Object.keys(computeHunkSignature(inputs.diffText))).toEqual([ + "src/new.ts", + ]); + }); +}); + +describe("the REST/GraphQL bot-suffix split", () => { + // Khan/webapp#41140 run 30416237794 staged threadCount: 0 on a PR with five + // reviewer threads: GraphQL reports the App as `github-actions`, REST as + // `github-actions[bot]`, and one configured spelling cannot match both. + it("matches a GraphQL thread author that carries no [bot] suffix", async () => { + const node = threadNode({ + comments: { + nodes: [ + { + author: {login: "github-actions"}, + body: "**issue (blocking):** boom", + }, + ], + }, + }); + const threads = await collectThreads( + portFor({threadPages: [onePage([node])]}), + "o", + "r", + 1, + "github-actions[bot]", + ); + expect(threads).toHaveLength(1); + }); + + it("matches a REST review author that carries the suffix", async () => { + const inputs = await collectInputs( + portFor({ + rest: {labels: [], head: {sha: "s"}}, + paged: { + "/reviews": [ + { + user: {login: "github-actions[bot]"}, + body: "b", + submitted_at: "2026-07-01T00:00:00Z", + }, + ], + }, + threadPages: [onePage([])], + }), + "o", + "r", + 1, + "github-actions", + ); + expect(inputs.priorReviews).toHaveLength(1); + }); + + it("still excludes a genuinely different author", async () => { + const node = threadNode({ + comments: {nodes: [{author: {login: "alice"}, body: "hi"}]}, + }); + const threads = await collectThreads( + portFor({threadPages: [onePage([node])]}), + "o", + "r", + 1, + "github-actions[bot]", + ); + expect(threads).toEqual([]); + }); +}); diff --git a/workflows/autofix/lib/stage.ts b/workflows/autofix/lib/stage.ts new file mode 100644 index 00000000..97fe9e48 --- /dev/null +++ b/workflows/autofix/lib/stage.ts @@ -0,0 +1,502 @@ +/** + * Deterministic staging: fetch everything the plan needs, in code, in one step. + * + * This module exists because the first live run measured the cost of NOT having + * it. Step 1 used to be prose telling the agent which five files to write; the + * agent improvised, and staging alone burned roughly fifteen of the run's 131 + * assistant turns (seven creating a directory, five hand-assembling JSON through + * repeated `node -e` scripts, three reading this workflow's own library source + * to work out what the plan would decide). Turns are what autofix costs: each + * one re-reads the whole accumulated context, so 131 turns over a ~43k-token + * context read 5.6M cached tokens and 61% of the run's bill was cache reads. + * Caching was working near-optimally (a 40:1 read-to-write ratio); there were + * simply too many turns. + * + * So staging moves across the determinism boundary to join the plan. CODE + * fetches and writes; the MODEL reads the result. That is the same split + * `plan.ts` already draws, and the same one the reviewer draws throughout; it + * was an oversight that staging sat on the wrong side of it. + * + * Correctness matters here as much as cost. The old prose had to ask the agent + * to stage each comment body "verbatim as the tool returned it", because a + * reformatted body breaks the `**label:**` parse that decides whether a finding + * is in scope. An instruction like that is a hope. Code copying a string is a + * guarantee. + * + * No new runtime dependency: this talks to the GitHub API with the global + * `fetch` Node provides, unlike the thumbs sweep, which predates that and pulls + * in octokit. Network access sits behind {@link StagePort} so the whole module + * is unit-testable without a socket. + * + * The unified diff is rebuilt by the reviewer's own `buildUnifiedDiff` + * (`stage-pr.ts`, the orchestrator's staging slice) rather than a local copy. + * An earlier local copy emitted `diff --git a/ b/` from `filename` + * alone, which is wrong for a rename and for an add or delete; the shared one + * carries `previous_filename` and the `/dev/null` sides. One rebuilder also + * means one thing for `splitUnifiedDiff` to keep parsing. + */ + +import {buildUnifiedDiff} from "../../review/lib/stage-pr.ts"; +import type {StagedThread} from "../../review/lib/rereview.ts"; +import type {PriorReview} from "../../review/lib/rereview-mode.ts"; + +/** The five files the plan CLI reads, plus the head SHA the prompt re-checks. */ +export type StagedInputs = { + labels: string[]; + threads: StagedThread[]; + priorReviews: PriorReview[]; + diffText: string; + commitMessages: string[]; + /** + * The SHA the agent's edits are actually made against. + * + * Read from the job's checkout, not from the API. Khan/actions#298 review: + * comparing an API read against a checkout taken earlier leaves a window in + * which a push lands between the two, and both reads then agree while the + * working tree is already stale. The checkout's own HEAD closes it. + */ + headSha: string; + /** + * Whether the PR's head is a fork. + * + * Staged because the command path cannot gate on it: `issue_comment` + * carries no `github.event.pull_request`, so the workflow's `if:` cannot + * check the fork there, and it has to be enforced after the job starts. + */ + isFork: boolean; +}; + +/** Everything this module needs from the outside world. */ +export type StagePort = { + /** The checked-out HEAD (`git rev-parse HEAD`), or "" when unavailable. */ + checkoutHeadSha: () => string; + /** A REST GET, following pagination; returns the concatenated array. */ + restPaged: (path: string) => Promise; + /** A single REST GET returning one object. */ + rest: (path: string) => Promise; + /** A GraphQL POST. */ + graphql: ( + query: string, + variables: Record, + ) => Promise; +}; + +const isRecord = (v: unknown): v is Record => + typeof v === "object" && v !== null && !Array.isArray(v); + +const str = (v: unknown): string => (typeof v === "string" ? v : ""); + +/** + * Compare two GitHub logins across the REST/GraphQL bot-suffix split. + * + * REST reports an App's login as `github-actions[bot]`; GraphQL reports the + * same actor as `github-actions`. Staging reads threads over GraphQL and + * reviews over REST, so a single spelling cannot match both. Comparing on the + * suffix-stripped form does. + * + * This is not hypothetical: the first run with deterministic staging staged + * `threadCount: 0` on a PR carrying five reviewer threads, because every + * GraphQL author was `github-actions` and the configured login was + * `github-actions[bot]` (Khan/webapp#41140, run 30416237794). Unit tests could + * not have caught it; the fixtures were written in the REST spelling. + */ +const baseLogin = (login: string): string => + login.endsWith("[bot]") ? login.slice(0, -"[bot]".length) : login; + +const sameLogin = (a: string, b: string): boolean => + baseLogin(a).toLowerCase() === baseLogin(b).toLowerCase(); + +/** + * Review threads with their full reply chain. + * + * `comments(first: 100)` rather than the sweep's `first: 1`: the reconciler + * contract wants the whole chain, because an author's reply is often what says + * a finding is already handled. `isResolved` is fetched so resolved threads can + * be dropped here rather than downstream. + */ +const THREADS_QUERY = ` +query ($owner: String!, $repo: String!, $number: Int!, $cursor: String) { + repository(owner: $owner, name: $repo) { + pullRequest(number: $number) { + reviewThreads(first: 100, after: $cursor) { + pageInfo { hasNextPage endCursor } + nodes { + id + isResolved + path + line + comments(first: 100) { + nodes { author { login } body url } + } + } + } + } + } +}`; + +/** + * Throw when a GraphQL body carries errors, mirroring the REST paths' `throw`. + * + * GraphQL does not use HTTP status to report failure. GitHub answers + * `RATE_LIMITED`, node-access failures, and partial field failures with HTTP + * **200** and an `errors` array, `data` absent or partial. A transport that + * only checks `res.ok` therefore reads a throttled response as a successful + * one, and every downstream reader sees a PR with no threads. + * + * Any `errors` entry is fatal here, including the partial-data case. Staging a + * subset of the reviewer's threads is worse than refusing: the plan would fix + * the threads that happened to arrive, clear the arming label, and report a + * clean run, leaving the rest silently unaddressed with nothing left to + * re-arm. + */ +export const assertNoGraphqlErrors = (body: unknown): void => { + if (!isRecord(body)) { + return; + } + const errors = body["errors"]; + if (Array.isArray(errors) && errors.length > 0) { + throw new Error(`GraphQL errors: ${JSON.stringify(errors)}`); + } +}; + +const threadsConnectionOf = ( + body: unknown, +): Record | undefined => { + if (!isRecord(body)) { + return undefined; + } + const data = body["data"]; + if (!isRecord(data)) { + return undefined; + } + const repository = data["repository"]; + if (!isRecord(repository)) { + return undefined; + } + const pullRequest = repository["pullRequest"]; + if (!isRecord(pullRequest)) { + return undefined; + } + const threads = pullRequest["reviewThreads"]; + return isRecord(threads) ? threads : undefined; +}; + +/** + * Collect the bot's unresolved threads, newest page last. + * + * A thread is kept when it is unresolved and its FIRST comment is the bot's: + * that opener is the finding. A thread a human started is somebody else's + * conversation and autofix stays out of it, which is the same line the reviewer + * draws with its `human-threads.json`. + */ +export const collectThreads = async ( + port: StagePort, + owner: string, + repo: string, + number: number, + botLogin: string, +): Promise => { + const out: StagedThread[] = []; + let cursor: string | null = null; + + for (;;) { + const body = await port.graphql(THREADS_QUERY, { + owner, + repo, + number, + cursor, + }); + // Fail closed on both shapes a failed query can take. `errors` is the + // throttled/partial case; a missing connection is any other malformed + // body. Neither means "this PR has no threads", and reading them that + // way is the one mistake this module cannot afford: an empty + // threads.json makes the plan a no-op, the no-op populates + // `labelsToRemove`, and the run tells the author there is nothing to + // fix while the blocking findings it was armed for sit open. The label + // is gone, so nothing survives to re-arm from. The REST paths already + // throw on `!res.ok` for exactly this reason; this is that contract + // applied to the transport that does not signal failure by status. + assertNoGraphqlErrors(body); + const conn = threadsConnectionOf(body); + if (conn === undefined) { + throw new Error( + `GraphQL returned no reviewThreads connection for ` + + `${owner}/${repo}#${number}`, + ); + } + + const nodes = Array.isArray(conn["nodes"]) ? conn["nodes"] : []; + for (const node of nodes) { + if (!isRecord(node) || node["isResolved"] === true) { + continue; + } + + const commentsConn = isRecord(node["comments"]) + ? node["comments"] + : {}; + const rawComments = Array.isArray(commentsConn["nodes"]) + ? commentsConn["nodes"] + : []; + const comments = rawComments.filter(isRecord).map((c) => ({ + author: isRecord(c["author"]) ? str(c["author"]["login"]) : "", + // Verbatim. The label parser reads the leading `**label:**` off + // this string; normalising it here is how a finding becomes + // unclassifiable. + body: str(c["body"]), + })); + if ( + comments.length === 0 || + !sameLogin(comments[0].author, botLogin) + ) { + continue; + } + + const firstUrl = isRecord(rawComments[0]) + ? str(rawComments[0]["url"]) + : ""; + out.push({ + thread_id: str(node["id"]), + path: str(node["path"]), + line: typeof node["line"] === "number" ? node["line"] : null, + ...(firstUrl === "" ? {} : {url: firstUrl}), + comments, + }); + } + + const pageInfo = isRecord(conn["pageInfo"]) ? conn["pageInfo"] : {}; + if (pageInfo["hasNextPage"] !== true) { + break; + } + const next = pageInfo["endCursor"]; + if (typeof next !== "string" || next === "") { + break; + } + cursor = next; + } + + return out; +}; + +/** Fetch everything the plan needs. */ +export const collectInputs = async ( + port: StagePort, + owner: string, + repo: string, + number: number, + botLogin: string, +): Promise => { + const base = `/repos/${owner}/${repo}/pulls/${number}`; + + const checkoutSha = port.checkoutHeadSha(); + const pr = await port.rest(base); + const prRec = isRecord(pr) ? pr : {}; + const rawLabels = Array.isArray(prRec["labels"]) ? prRec["labels"] : []; + const head = isRecord(prRec["head"]) ? prRec["head"] : {}; + const headRepo = isRecord(head["repo"]) ? head["repo"] : {}; + // Absent or unreadable repo data reads as a fork, so the guard fails closed. + const headRepoName = str(headRepo["full_name"]); + const isFork = headRepoName === "" || headRepoName !== `${owner}/${repo}`; + + const [reviews, files, commits, threads] = await Promise.all([ + port.restPaged(`${base}/reviews`), + port.restPaged(`${base}/files`), + port.restPaged(`${base}/commits`), + collectThreads(port, owner, repo, number, botLogin), + ]); + + return { + labels: rawLabels + .filter(isRecord) + .map((l) => str(l["name"])) + .filter((n) => n !== ""), + threads, + // Every review by the bot, whatever its state: the fingerprint stamp + // lives in the body, and a dismissed or comment-only review still + // carries one. + priorReviews: reviews + .filter(isRecord) + .filter( + (r) => + isRecord(r["user"]) && + sameLogin(str(r["user"]["login"]), botLogin), + ) + .map((r) => ({ + body: str(r["body"]), + submittedAt: str(r["submitted_at"]), + })) as PriorReview[], + diffText: buildUnifiedDiff( + files as Parameters[0], + ), + commitMessages: commits + .filter(isRecord) + .map((c) => + isRecord(c["commit"]) ? str(c["commit"]["message"]) : "", + ) + .filter((m) => m !== ""), + isFork, + // Prefer the checkout; fall back to the API only when the working tree + // is unavailable, in which case a stale-base push is still possible. + headSha: checkoutSha !== "" ? checkoutSha : str(head["sha"]), + }; +}; + +/* -------------------------------------------------------------------------- */ +/* CLI */ +/* -------------------------------------------------------------------------- */ + +export type StageCliFs = { + mkdirSync: (path: string, opts: {recursive: true}) => void; + writeFileSync: (path: string, data: string) => void; +}; + +export const AUTOFIX_DIR = "/tmp/gh-aw/autofix"; + +const json = (value: unknown): string => `${JSON.stringify(value, null, 2)}\n`; + +/** + * Write the staged files. + * + * `command` is the triggering `/autofix` comment body when one exists. It is + * written only on that path: `plan.ts` treats the file's ABSENCE as "this run + * was armed by a label", so writing an empty file would silently switch the + * resolver's surface. + */ +export const writeInputs = ( + fs: StageCliFs, + inputs: StagedInputs, + dir = AUTOFIX_DIR, + command?: string, +): void => { + fs.mkdirSync(`${dir}/out`, {recursive: true}); + fs.writeFileSync(`${dir}/labels.json`, json(inputs.labels)); + fs.writeFileSync(`${dir}/threads.json`, json(inputs.threads)); + fs.writeFileSync(`${dir}/prior-reviews.json`, json(inputs.priorReviews)); + fs.writeFileSync(`${dir}/pr.diff`, inputs.diffText); + fs.writeFileSync(`${dir}/commits.json`, json(inputs.commitMessages)); + fs.writeFileSync(`${dir}/head-sha.txt`, `${inputs.headSha}\n`); + fs.writeFileSync(`${dir}/context.json`, json({isFork: inputs.isFork})); + if (command !== undefined && command.trim() !== "") { + // Verbatim, trailing CRLF included: the parser's tolerance of the shape + // the GitHub web UI produces is only meaningful if the shape survives. + fs.writeFileSync(`${dir}/command.txt`, command); + } +}; + +// Run only when executed directly (autofix.md), never on import (tests). +if (typeof require !== "undefined" && require.main === module) { + const nodeFs = require("node:fs"); + const env = (name: string): string => { + const v = process.env[name]; + if (v === undefined || v.trim() === "") { + throw new Error(`${name} must be set`); + } + return v.trim(); + }; + + const token = env("GITHUB_TOKEN"); + const [owner, repo] = env("GITHUB_REPOSITORY").split("/"); + const number = Number(env("AUTOFIX_PR_NUMBER")); + const botLogin = + process.env.AUTOFIX_BOT_LOGIN?.trim() || "github-actions[bot]"; + const api = process.env.GITHUB_API_URL?.trim() || "https://api.github.com"; + + const headers = { + authorization: `Bearer ${token}`, + accept: "application/vnd.github+json", + "user-agent": "khan-autofix", + }; + + const {execFileSync} = require("node:child_process"); + const port: StagePort = { + checkoutHeadSha: () => { + try { + return String( + execFileSync("git", ["rev-parse", "HEAD"], { + cwd: process.env.GITHUB_WORKSPACE || process.cwd(), + encoding: "utf-8", + }), + ).trim(); + } catch { + return ""; + } + }, + rest: async (path) => { + const res = await fetch(`${api}${path}`, {headers}); + if (!res.ok) { + throw new Error(`GET ${path} failed: ${res.status}`); + } + return res.json(); + }, + restPaged: async (path) => { + const out: unknown[] = []; + // 100 is the API maximum; the cap bounds a pathological PR rather + // than silently truncating a normal one. + for (let page = 1; page <= 20; page++) { + const sep = path.includes("?") ? "&" : "?"; + const res = await fetch( + `${api}${path}${sep}per_page=100&page=${page}`, + {headers}, + ); + if (!res.ok) { + throw new Error(`GET ${path} failed: ${res.status}`); + } + const batch = await res.json(); + if (!Array.isArray(batch) || batch.length === 0) { + break; + } + out.push(...batch); + if (batch.length < 100) { + break; + } + } + return out; + }, + graphql: async (query, variables) => { + const res = await fetch(`${api}/graphql`, { + method: "POST", + headers: {...headers, "content-type": "application/json"}, + body: JSON.stringify({query, variables}), + }); + if (!res.ok) { + throw new Error(`GraphQL failed: ${res.status}`); + } + // Duplicated in `collectThreads`, deliberately. The guard is cheap + // and its absence clears the arming label on a PR with open + // findings, so it belongs both at the transport (any future + // GraphQL caller inherits it) and at the reader (which is the one + // the unit tests can reach). + const body = await res.json(); + assertNoGraphqlErrors(body); + return body; + }, + }; + + collectInputs(port, owner, repo, number, botLogin) + .then((inputs) => { + writeInputs( + { + mkdirSync: (p, o) => nodeFs.mkdirSync(p, o), + writeFileSync: (p, d) => nodeFs.writeFileSync(p, d), + }, + inputs, + AUTOFIX_DIR, + process.env.AUTOFIX_COMMAND_BODY, + ); + // eslint-disable-next-line no-console + console.log( + JSON.stringify({ + labels: inputs.labels, + threadCount: inputs.threads.length, + reviewCount: inputs.priorReviews.length, + commitCount: inputs.commitMessages.length, + diffBytes: inputs.diffText.length, + headSha: inputs.headSha, + }), + ); + }) + .catch((error) => { + // eslint-disable-next-line no-console + console.error(`staging failed: ${error?.message ?? error}`); + process.exit(1); + }); +} diff --git a/workflows/autofix/lib/staleness.test.ts b/workflows/autofix/lib/staleness.test.ts new file mode 100644 index 00000000..f740001a --- /dev/null +++ b/workflows/autofix/lib/staleness.test.ts @@ -0,0 +1,230 @@ +import {describe, expect, it} from "vitest"; + +import { + assessReviewCurrency, + DEGRADED_NOTES, + REFUSAL_REASONS, +} from "./staleness.ts"; +import { + computeHunkSignature, + renderRereviewStamp, + STAMP_SCHEMA_VERSION, +} from "../../review/lib/rereview-mode.ts"; +import type {HunkSignature} from "../../review/lib/rereview-mode.ts"; + +const diffFor = (files: Record): string => + Object.entries(files) + .map( + ([path, added]) => + `diff --git a/${path} b/${path}\n` + + `--- a/${path}\n+++ b/${path}\n` + + `@@ -1,1 +1,${added.length + 1} @@\n` + + ` context\n` + + added.map((line) => `+${line}`).join("\n") + + `\n`, + ) + .join(""); + +const stampedReview = (anchorHunks: HunkSignature | "overflow") => ({ + body: + "Approved.\n\n" + + renderRereviewStamp({ + schemaVersion: STAMP_SCHEMA_VERSION, + depth: "full", + verdict: "APPROVE", + anchorDraft: false, + anchorHunks, + }), + submittedAt: "2026-07-01T00:00:00Z", +}); + +describe("assessReviewCurrency", () => { + it("refuses when no review has ever stamped the PR", () => { + expect(assessReviewCurrency([], diffFor({"a.ts": ["x"]}))).toEqual({ + status: "no-review", + }); + }); + + it("degrades, not refuses, when a review carries no readable stamp", () => { + // Khan/webapp#41130: the reviewer posted a correct blocking finding + // under a body of exactly "Changes requested — see inline comments." + // and no stamp. Reporting that as "no review" was wrong twice over: + // wrong message, and a refusal on a PR that had real feedback. + const result = assessReviewCurrency( + [ + { + body: "Changes requested — see inline comments.", + submittedAt: "2026-07-01T00:00:00Z", + }, + ], + diffFor({"a.ts": ["x"]}), + ); + expect(result).toEqual({status: "unverifiable", why: "unstamped"}); + }); + + it("degrades when the fingerprint overflowed", () => { + const result = assessReviewCurrency( + [stampedReview("overflow")], + diffFor({"a.ts": ["x"]}), + ); + expect(result).toEqual({status: "unverifiable", why: "overflow"}); + }); + + it("distinguishes no reviews at all from unstamped reviews", () => { + expect(assessReviewCurrency([], diffFor({"a.ts": ["x"]})).status).toBe( + "no-review", + ); + expect( + assessReviewCurrency( + [{body: "anything", submittedAt: "2026-07-01T00:00:00Z"}], + diffFor({"a.ts": ["x"]}), + ).status, + ).toBe("unverifiable"); + }); + + it("reports current with no stale paths when the head matches the review", () => { + const diff = diffFor({"a.ts": ["x"], "b.ts": ["y"]}); + const result = assessReviewCurrency( + [stampedReview(computeHunkSignature(diff))], + diff, + ); + expect(result.status).toBe("current"); + if (result.status !== "current") { + return; + } + expect(result.stalePaths).toEqual([]); + expect(result.divergence.unreviewedHunks).toBe(0); + }); + + it("marks only the files that changed after the review", () => { + // The common case the per-path guard exists for: the author pushed one + // unrelated fix after the review, and findings in other files are + // still perfectly actionable. + const reviewed = diffFor({"a.ts": ["x"], "b.ts": ["y"]}); + const now = diffFor({"a.ts": ["x"], "b.ts": ["y CHANGED"]}); + const result = assessReviewCurrency( + [stampedReview(computeHunkSignature(reviewed))], + now, + ); + expect(result.status).toBe("current"); + if (result.status !== "current") { + return; + } + expect(result.stalePaths).toEqual(["b.ts"]); + expect(result.divergence.unreviewedHunks).toBe(1); + }); + + it("marks a file the review never saw at all as stale", () => { + const reviewed = diffFor({"a.ts": ["x"]}); + const now = diffFor({"a.ts": ["x"], "new.ts": ["z"]}); + const result = assessReviewCurrency( + [stampedReview(computeHunkSignature(reviewed))], + now, + ); + if (result.status !== "current") { + throw new Error("expected current"); + } + expect(result.stalePaths).toEqual(["new.ts"]); + }); + + it("reads the most recent stamp when several reviews exist", () => { + const older = diffFor({"a.ts": ["old"]}); + const current = diffFor({"a.ts": ["new"]}); + const result = assessReviewCurrency( + [ + {...stampedReview(computeHunkSignature(older))}, + { + ...stampedReview(computeHunkSignature(current)), + submittedAt: "2026-07-02T00:00:00Z", + }, + ], + current, + ); + if (result.status !== "current") { + throw new Error("expected current"); + } + expect(result.stalePaths).toEqual([]); + }); + + it("sorts stale paths so the plan artifact is stable", () => { + const reviewed = diffFor({"a.ts": ["x"]}); + const now = diffFor({"z.ts": ["1"], "b.ts": ["2"], "a.ts": ["x"]}); + const result = assessReviewCurrency( + [stampedReview(computeHunkSignature(reviewed))], + now, + ); + if (result.status !== "current") { + throw new Error("expected current"); + } + expect(result.stalePaths).toEqual(["b.ts", "z.ts"]); + }); + + it("is unaffected by a rebase that only moves the code", () => { + // The signature hashes added-line content, so identical content at a + // different offset is still "reviewed". + const reviewed = + "diff --git a/a.ts b/a.ts\n--- a/a.ts\n+++ b/a.ts\n" + + "@@ -1,1 +1,2 @@\n context\n+added line\n"; + const rebased = + "diff --git a/a.ts b/a.ts\n--- a/a.ts\n+++ b/a.ts\n" + + "@@ -40,1 +40,2 @@\n context\n+added line\n"; + const result = assessReviewCurrency( + [stampedReview(computeHunkSignature(reviewed))], + rebased, + ); + if (result.status !== "current") { + throw new Error("expected current"); + } + expect(result.stalePaths).toEqual([]); + }); +}); + +describe("REFUSAL_REASONS", () => { + it("gives the author an action for each refusal", () => { + for (const reason of Object.values(REFUSAL_REASONS)) { + expect(reason.length).toBeGreaterThan(40); + } + }); + + it("no longer claims there is no feedback when there is", () => { + // The exact wording that misreported Khan/webapp#41130. + expect(REFUSAL_REASONS["no-review"]).not.toContain( + "no reviewer feedback has been posted", + ); + }); +}); + +describe("DEGRADED_NOTES", () => { + it("names the weaker check for each degraded cause", () => { + for (const note of Object.values(DEGRADED_NOTES)) { + expect(note).toContain("thread anchors only"); + } + }); +}); + +describe("an unreadable diff must not read as a clean check", () => { + const stamped = [ + stampedReview(computeHunkSignature(diffFor({"a.ts": ["x"]}))), + ]; + + it("degrades on an empty diff rather than reporting current", () => { + // Khan/actions#298 review, blocking: computeHunkSignature("") is {}, + // the stale-path loop is vacuous, and the old code returned `current` + // with no note. The guard would report a clean full check having + // performed none. + expect(assessReviewCurrency(stamped, "")).toEqual({ + status: "unverifiable", + why: "unreadable-diff", + }); + }); + + it("degrades on a patch with no file headers", () => { + // Raw `get_files` patches carry no diff --git/---/+++ headers, so + // splitUnifiedDiff recognises no file section in them. + const headerless = "@@ -1,1 +1,9 @@\n context\n+totally different"; + expect(assessReviewCurrency(stamped, headerless)).toEqual({ + status: "unverifiable", + why: "unreadable-diff", + }); + }); +}); diff --git a/workflows/autofix/lib/staleness.ts b/workflows/autofix/lib/staleness.ts new file mode 100644 index 00000000..2003a0d0 --- /dev/null +++ b/workflows/autofix/lib/staleness.ts @@ -0,0 +1,173 @@ +/** + * The review-currency guard: is there a review recent enough to act on? + * + * The failure this prevents is the one most likely to make autofix look broken + * for reasons unrelated to fix quality. Someone labels a PR whose review + * predates the current head; the findings then describe code that has already + * changed, and an agent acting on them edits on the strength of a stale + * statement. Nothing about the label tells you this happened, so it has to be + * checked before any work is planned. + * + * The check reuses the reviewer's own fingerprint rather than inventing a + * freshness signal. Every review stamps the hunk signature it reviewed into a + * hidden comment in its body, precisely because that record survives when cache + * memory does not (`review.md` Step 6; `rereview-mode.ts`). Comparing the + * current diff's signature against that stamp answers "has this code changed + * since a review saw it" exactly, across force-pushes and rebases, because the + * hashes cover added-line content rather than commit SHAs or line numbers. + * + * Granularity is per-path, not per-PR, and that is deliberate. An all-or-nothing + * gate would refuse the whole run because the author pushed one unrelated fix + * after the review, which is both common and harmless for findings in other + * files. Reporting the stale paths lets the caller drop just the affected work + * items and fix the rest, so the guard degrades to partial work instead of + * refusal. + * + * **The unstamped path is the NORMAL path, not an edge case.** gh-aw's + * safe-output ingest sanitizer strips every XML/HTML comment before a review + * posts (`removeXmlComments` in `sanitize_content_core.cjs`, a depth-tracking + * scan with no allowlist), so the reviewer's body stamp is deleted on the way + * out and has never reached a posted review. Khan/actions#287 documents this + * end to end and gives the reviewer a second carrier, its cache-memory record. + * + * That carrier is not available here: cache memory is scoped per workflow, and + * autofix is a different workflow from the reviewer, so it cannot read the + * reviewer's. Until that changes, {@link assessReviewCurrency} will return + * `unverifiable` on essentially every real run, and the per-thread anchor check + * is what autofix actually runs on. Treat the fingerprint branch below as the + * optimisation, not the main path. + * + * **Degrading, and why it is not a weakened guard.** An earlier version refused + * outright whenever no fingerprint could be read, which given the above made + * autofix refuse every real run: on Khan/webapp#41130 the reviewer posted a + * correct blocking finding under a body of exactly "Changes requested + * — see inline comments." and no stamp, and autofix refused every time. + * + * The fingerprint is not the only currency signal, and it is not even the + * primary one. GitHub marks a review comment outdated when the diff hunk it + * anchors to changes, which `worklist.ts` already reads (a null anchor is + * dropped as `outdated-anchor`). That per-thread signal covers the case that + * actually matters: the author edited the flagged code. The fingerprint adds + * coarser file-level detection — "something else in this file moved" — whose + * failure mode is a redundant fix that the following re-review catches. + * + * So the policy is: use the fingerprint when it is there, fall back to anchors + * when it is not, and **say so in the summary** so a weaker check is never + * silent. Refuse only when there is no review at all, which is the one state + * where there is genuinely nothing to act on. + */ + +import { + computeDivergence, + computeHunkSignature, + findLatestStamp, +} from "../../review/lib/rereview-mode.ts"; +import type {Divergence, PriorReview} from "../../review/lib/rereview-mode.ts"; + +export type CurrencyAssessment = + | { + /** The reviewer has never reviewed this PR; there is nothing to act on. */ + status: "no-review"; + } + | { + /** + * Reviews exist but carry no usable fingerprint, so the file-level + * check cannot run. This is NOT a refusal — see the note on degrading + * below. + */ + status: "unverifiable"; + why: "unstamped" | "overflow" | "unreadable-diff"; + } + | { + status: "current"; + divergence: Divergence; + /** Paths carrying hunks no stamped review has seen. */ + stalePaths: string[]; + }; + +/** + * Compare the current diff against the most recent stamped review. + * + * `diffText` is the PR's full diff, NOT the reviewer's generated-file-stripped + * copy, so a lockfile or bundle churning after the review does appear in the + * signature and does land in `stalePaths`. That is harmless here and the + * asymmetry is deliberate: `plan.ts` consults `stalePaths` only for paths that + * carry a finding, and the reviewer does not raise findings on generated files, + * so a stale generated path can never drop real work. Stripping would mean + * threading the router's `generatedFiles` through staging to buy nothing. + */ +export const assessReviewCurrency = ( + reviews: readonly PriorReview[], + diffText: string, +): CurrencyAssessment => { + // "No review at all" and "reviews exist but none carry a fingerprint" are + // different facts and must not be collapsed. Collapsing them told the author + // of Khan/webapp#41130 that "no reviewer feedback has been posted on this + // PR", on a PR carrying a blocking finding that the reviewer had just + // posted. The message was wrong because the state was wrong. + if (reviews.length === 0) { + return {status: "no-review"}; + } + + const stamp = findLatestStamp(reviews); + if (stamp === null) { + return {status: "unverifiable", why: "unstamped"}; + } + if (stamp.anchorHunks === "overflow") { + return {status: "unverifiable", why: "overflow"}; + } + + const current = computeHunkSignature(diffText); + + // A signature with no paths means the diff told us nothing: it was empty, + // or it carried no `diff --git`/`---`/`+++` headers for `splitUnifiedDiff` + // to recognise a file section from. Falling through would make the + // stale-path loop vacuous and return `current` with zero stale paths and no + // note, i.e. the guard would report a clean full check having performed + // none. That is the one failure direction this module must not have, so an + // unreadable diff degrades like any other unusable input. + if (Object.keys(current).length === 0) { + return {status: "unverifiable", why: "unreadable-diff"}; + } + + const divergence = computeDivergence(current, stamp.anchorHunks); + + const stalePaths: string[] = []; + for (const [path, hashes] of Object.entries(current)) { + const seen = new Set(stamp.anchorHunks[path] ?? []); + if (hashes.some((hash) => !seen.has(hash))) { + stalePaths.push(path); + } + } + + return {status: "current", divergence, stalePaths: stalePaths.sort()}; +}; + +/** Human-readable reason a refusal is a refusal; rendered into the PR comment. */ +export const REFUSAL_REASONS: Readonly> = { + "no-review": + "the reviewer has not reviewed this PR yet, so there is nothing to " + + "autofix. Re-run the reviewer and arm autofix again once a review " + + "exists.", +}; + +/** + * What to tell the author when the file-level check could not run. Rendered into + * the summary so a weaker check is never silent. + */ +export const DEGRADED_NOTES: Readonly< + Record<"unstamped" | "overflow" | "unreadable-diff", string> +> = { + "unreadable-diff": + "The PR's diff could not be parsed, so the file-level currency check " + + "could not run; findings were checked against their thread anchors " + + "only.", + unstamped: + "The reviewer's review carries no diff fingerprint, so the file-level " + + "currency check could not run; findings were checked against their " + + "thread anchors only.", + overflow: + "The reviewer's diff fingerprint overflowed (very large diff), so the " + + "file-level currency check could not run; findings were checked " + + "against their thread anchors only.", +}; diff --git a/workflows/autofix/lib/trailer.test.ts b/workflows/autofix/lib/trailer.test.ts new file mode 100644 index 00000000..fd8bbcef --- /dev/null +++ b/workflows/autofix/lib/trailer.test.ts @@ -0,0 +1,147 @@ +import {describe, expect, it} from "vitest"; + +import { + parseTrailer, + renderTrailer, + summariseLedger, + TRAILER_SCHEMA_VERSION, +} from "./trailer.ts"; + +const trailer = { + schemaVersion: TRAILER_SCHEMA_VERSION, + scopes: ["blocking"], + cycle: 1, + threadIds: ["PRRT_a", "PRRT_b"], +}; + +const commit = (body: string): string => + `autofix: address reviewer feedback\n\n${body}`; + +describe("renderTrailer / parseTrailer", () => { + it("round-trips", () => { + expect(parseTrailer(commit(renderTrailer(trailer)))).toEqual(trailer); + }); + + it("renders git-trailer syntax, one key per line", () => { + const lines = renderTrailer(trailer).split("\n"); + expect(lines).toEqual([ + "Autofix-Version: 1", + "Autofix-Scope: blocking", + "Autofix-Cycle: 1", + "Autofix-Threads: PRRT_a,PRRT_b", + ]); + }); + + it("round-trips a multi-scope run", () => { + const both = {...trailer, scopes: ["blocking", "nits"]}; + expect(parseTrailer(commit(renderTrailer(both)))?.scopes).toEqual([ + "blocking", + "nits", + ]); + }); + + it("round-trips an empty thread ledger", () => { + const empty = {...trailer, threadIds: []}; + expect(parseTrailer(commit(renderTrailer(empty)))?.threadIds).toEqual( + [], + ); + }); + + it("returns null for a commit with no trailer", () => { + expect(parseTrailer("fix: unrelated human commit")).toBeNull(); + }); + + it("returns null for a schema version it does not understand", () => { + const future = renderTrailer(trailer).replace( + "Autofix-Version: 1", + "Autofix-Version: 99", + ); + expect(parseTrailer(commit(future))).toBeNull(); + }); + + it("returns null when the cycle is missing or not a positive integer", () => { + for (const bad of ["", "0", "-1", "two"]) { + const broken = renderTrailer(trailer).replace( + "Autofix-Cycle: 1", + `Autofix-Cycle: ${bad}`, + ); + expect(parseTrailer(commit(broken))).toBeNull(); + } + }); + + it("tolerates extra whitespace after the key", () => { + expect( + parseTrailer(commit("Autofix-Version: 1\nAutofix-Cycle:\t2")), + ).toMatchObject({cycle: 2}); + }); +}); + +describe("summariseLedger", () => { + it("reports an empty ledger for a branch with no autofix commits", () => { + expect(summariseLedger(["feat: a", "fix: b"])).toEqual({ + cycles: 0, + nextCycle: 1, + attemptedThreadIds: [], + }); + }); + + it("counts cycles and unions attempted threads", () => { + const first = commit( + renderTrailer({...trailer, cycle: 1, threadIds: ["A", "B"]}), + ); + const second = commit( + renderTrailer({...trailer, cycle: 2, threadIds: ["B", "C"]}), + ); + expect(summariseLedger(["feat: x", first, "fix: y", second])).toEqual({ + cycles: 2, + nextCycle: 3, + attemptedThreadIds: ["A", "B", "C"], + }); + }); + + it("derives the next cycle from the highest recorded, not the count", () => { + // A squashed or dropped intermediate commit must not hand out a cycle + // number that was already used. + const third = commit(renderTrailer({...trailer, cycle: 3})); + expect(summariseLedger([third]).nextCycle).toBe(4); + }); + + it("ignores unreadable trailers rather than counting them", () => { + // Fail-closed direction: under-reporting past work can only make a + // cycle cap trip earlier, never later. + const unreadable = commit("Autofix-Version: 99\nAutofix-Cycle: 5"); + expect(summariseLedger([unreadable])).toEqual({ + cycles: 0, + nextCycle: 1, + attemptedThreadIds: [], + }); + }); + + it("sorts attempted ids so the ledger is comparable across runs", () => { + const one = commit(renderTrailer({...trailer, threadIds: ["C", "A"]})); + expect(summariseLedger([one]).attemptedThreadIds).toEqual(["A", "C"]); + }); +}); + +describe("trailers are read from the final paragraph only", () => { + it("ignores a quoted trailer in the body", () => { + // Khan/actions#298 review: a revert or doc commit quoting the trailer + // parsed as an autofix commit and inflated the cycle count. + const quoting = + "revert: back out the autofix commit\n\n" + + "It carried `Autofix-Version: 1` and `Autofix-Cycle: 4`, which\n" + + "we no longer want.\n\n" + + "Reverts: abc123\n"; + expect(parseTrailer(quoting)).toBeNull(); + expect(summariseLedger([quoting]).cycles).toBe(0); + }); + + it("still reads a real trailer in the final paragraph", () => { + const real = commit(renderTrailer({...trailer, cycle: 4})); + expect(parseTrailer(real)?.cycle).toBe(4); + }); + + it("reads a trailer that is the whole message", () => { + expect(parseTrailer(renderTrailer(trailer))?.cycle).toBe(1); + }); +}); diff --git a/workflows/autofix/lib/trailer.ts b/workflows/autofix/lib/trailer.ts new file mode 100644 index 00000000..aa81a2e8 --- /dev/null +++ b/workflows/autofix/lib/trailer.ts @@ -0,0 +1,159 @@ +/** + * The autofix commit trailer: the run's durable, machine-readable record. + * + * v1 runs once per arming, so nothing in v1 needs to know what a previous run + * did. This module exists anyway, because the cycle counter a later continual + * mode needs is nearly free to write now and genuinely awkward to backfill: the + * only way to reconstruct it later would be to re-parse prose from commit + * messages that were never written to be parsed. + * + * The branch itself is the store. Counting autofix commits on the head branch + * gives a cycle count that survives cache eviction, needs no external state, + * and is visible to a human reading the PR — the same reasoning that put the + * reviewer's authoritative fingerprint in the review body rather than in cache + * memory (`review.md` Step 6: "cache memory can be evicted, the review body + * cannot"). It also fails in the right direction: a trailer that cannot be read + * yields a lower cycle count, and the caller's rule is to stop when the count + * cannot be established, not to continue. + * + * `Autofix-Threads` is the attempted-fingerprint ledger. It is what lets the + * trial answer the question that actually matters — did the fix clear the + * finding — by diffing attempted thread ids against the ones the next review + * still reports open. A later no-progress guard reads the same field. + */ + +/** Bumped when a field is added, removed, or retyped. */ +export const TRAILER_SCHEMA_VERSION = 1; + +export type AutofixTrailer = { + schemaVersion: number; + /** Scopes this run acted under, in resolver order. */ + scopes: string[]; + /** 1-based; always 1 in v1, the field the cadence axis will increment. */ + cycle: number; + /** Thread ids this run attempted to address. */ + threadIds: string[]; +}; + +const KEYS = { + version: "Autofix-Version", + scope: "Autofix-Scope", + cycle: "Autofix-Cycle", + threads: "Autofix-Threads", +} as const; + +/** + * Render the trailer block appended to the autofix commit message. Git trailers + * are `Key: value` lines in the final paragraph, so the caller must place this + * last, separated from the subject/body by a blank line. + */ +export const renderTrailer = (trailer: AutofixTrailer): string => + [ + `${KEYS.version}: ${trailer.schemaVersion}`, + `${KEYS.scope}: ${trailer.scopes.join(",")}`, + `${KEYS.cycle}: ${trailer.cycle}`, + `${KEYS.threads}: ${trailer.threadIds.join(",")}`, + ].join("\n"); + +/** + * The final paragraph of a commit message, which is where git looks for + * trailers. + * + * Scoping to it matters: an earlier version matched keys anywhere in the + * message via a multiline regex, so a revert, or a doc commit quoting + * `Autofix-Version: 1`, parsed as an autofix commit and inflated the cycle + * count. Harmless while the count is only reported, load-bearing the moment a + * cadence axis caps cycles on it. + */ +const lastParagraph = (message: string): string => { + const paragraphs = message + .split(/\r?\n\s*\r?\n/) + .map((p) => p.trim()) + .filter((p) => p !== ""); + return paragraphs.length === 0 ? "" : paragraphs[paragraphs.length - 1]; +}; + +const valueOf = (message: string, key: string): string | null => { + const re = new RegExp(`^${key}:[ \\t]*(.*)$`, "m"); + const match = re.exec(lastParagraph(message)); + return match === null ? null : match[1].trim(); +}; + +const splitList = (value: string | null): string[] => + value === null || value === "" + ? [] + : value + .split(",") + .map((entry) => entry.trim()) + .filter((entry) => entry !== ""); + +/** + * Parse a commit message's trailer. Returns null when the message carries no + * autofix trailer or one stamped with a schema version this code does not + * understand — both of which the caller treats as "not a readable autofix + * commit". + */ +export const parseTrailer = (message: string): AutofixTrailer | null => { + const rawVersion = valueOf(message, KEYS.version); + if (rawVersion === null) { + return null; + } + const schemaVersion = Number(rawVersion); + if ( + !Number.isInteger(schemaVersion) || + schemaVersion !== TRAILER_SCHEMA_VERSION + ) { + return null; + } + const cycle = Number(valueOf(message, KEYS.cycle) ?? ""); + if (!Number.isInteger(cycle) || cycle < 1) { + return null; + } + return { + schemaVersion, + scopes: splitList(valueOf(message, KEYS.scope)), + cycle, + threadIds: splitList(valueOf(message, KEYS.threads)), + }; +}; + +export type Ledger = { + /** Readable autofix commits found on the branch. */ + cycles: number; + /** The cycle number a new run would take. */ + nextCycle: number; + /** Union of every thread id previously attempted, sorted. */ + attemptedThreadIds: string[]; +}; + +/** + * Summarise the autofix history recorded on a branch. + * + * `messages` is every commit message on the PR head. Commits without a readable + * trailer are ignored rather than counted, which is the fail-closed direction + * for the caller's stop rule: an unreadable history under-reports work already + * done, so a cycle cap derived from it can only trip earlier, never later. + */ +export const summariseLedger = (messages: readonly string[]): Ledger => { + const attempted = new Set(); + let cycles = 0; + let highest = 0; + + for (const message of messages) { + const trailer = parseTrailer(message); + if (trailer === null) { + continue; + } + cycles++; + highest = Math.max(highest, trailer.cycle); + for (const id of trailer.threadIds) { + attempted.add(id); + } + } + + return { + cycles, + nextCycle: highest + 1, + attemptedThreadIds: [...attempted].sort(), + }; +}; diff --git a/workflows/autofix/lib/worklist.test.ts b/workflows/autofix/lib/worklist.test.ts new file mode 100644 index 00000000..23523fba --- /dev/null +++ b/workflows/autofix/lib/worklist.test.ts @@ -0,0 +1,189 @@ +import {describe, expect, it} from "vitest"; + +import {buildWorkList} from "./worklist.ts"; +import type {StagedThread} from "../../review/lib/rereview.ts"; +import { + BLOCKING_LABELS, + NON_BLOCKING_LABELS, +} from "../../review/lib/render-comment.ts"; + +const thread = ( + over: Partial & {body: string}, +): StagedThread => ({ + thread_id: over.thread_id ?? "T1", + path: over.path ?? "src/a.ts", + line: over.line === undefined ? 12 : over.line, + url: over.url, + comments: over.comments ?? [ + {author: "github-actions[bot]", body: over.body}, + ], +}); + +const BLOCKING = [...BLOCKING_LABELS]; +const NITS = [...NON_BLOCKING_LABELS]; + +describe("buildWorkList", () => { + it("selects threads whose label is in scope", () => { + const {items, skipped} = buildWorkList( + [thread({body: "**issue (blocking):** null deref here"})], + BLOCKING, + ); + expect(skipped).toEqual([]); + expect(items).toHaveLength(1); + expect(items[0]).toMatchObject({ + threadId: "T1", + path: "src/a.ts", + line: 12, + label: "issue (blocking)", + }); + }); + + it("skips threads whose label is out of scope", () => { + const {items, skipped} = buildWorkList( + [thread({body: "**nitpick (non-blocking):** rename this"})], + BLOCKING, + ); + expect(items).toEqual([]); + expect(skipped).toEqual([ + { + threadId: "T1", + path: "src/a.ts", + reason: "out-of-scope", + label: "nitpick (non-blocking)", + }, + ]); + }); + + it("reads the markdown-stripped label form the staging can produce", () => { + // Khan/webapp#40561: staged openers arrived without the ** wrapping. + const {items} = buildWorkList( + [thread({body: "issue (blocking): unbounded read"})], + BLOCKING, + ); + expect(items).toHaveLength(1); + expect(items[0].label).toBe("issue (blocking)"); + }); + + it("excludes a thread whose label will not parse", () => { + // Fail-closed, inverted from rereview.ts: an unclassifiable finding is + // the last thing an agent should be editing code on the strength of. + const {items, skipped} = buildWorkList( + [thread({body: "please just fix this thanks"})], + [...BLOCKING, ...NITS], + ); + expect(items).toEqual([]); + expect(skipped[0]).toMatchObject({reason: "unparseable-label"}); + expect(skipped[0].label).toBeUndefined(); + }); + + it("excludes an outdated thread whose anchor is gone", () => { + const {items, skipped} = buildWorkList( + [thread({body: "**issue (blocking):** stale", line: null})], + BLOCKING, + ); + expect(items).toEqual([]); + expect(skipped[0]).toMatchObject({ + reason: "outdated-anchor", + label: "issue (blocking)", + }); + }); + + it("excludes a malformed line rather than passing it downstream", () => { + const staged = { + ...thread({body: "**issue (blocking):** x"}), + line: "12" as unknown as number, + }; + const {items, skipped} = buildWorkList([staged], BLOCKING); + expect(items).toEqual([]); + expect(skipped[0].reason).toBe("outdated-anchor"); + }); + + it("keeps the bot's opening comment verbatim as the finding statement", () => { + const body = "**issue (blocking):** the `x` guard is inverted\n\nmore"; + const {items} = buildWorkList([thread({body})], BLOCKING); + expect(items[0].body).toBe(body); + }); + + it("carries the thread url when staged and omits it when not", () => { + const withUrl = buildWorkList( + [ + thread({ + body: "**issue (blocking):** x", + url: "https://github.com/o/r/pull/1#discussion_r1", + }), + ], + BLOCKING, + ); + expect(withUrl.items[0].url).toBe( + "https://github.com/o/r/pull/1#discussion_r1", + ); + const withoutUrl = buildWorkList( + [thread({body: "**issue (blocking):** x"})], + BLOCKING, + ); + expect("url" in withoutUrl.items[0]).toBe(false); + }); + + it("preserves staged order so the plan is stable across runs", () => { + const threads = ["T1", "T2", "T3"].map((id) => + thread({thread_id: id, body: "**issue (blocking):** x"}), + ); + const {items} = buildWorkList(threads, BLOCKING); + expect(items.map((i) => i.threadId)).toEqual(["T1", "T2", "T3"]); + }); + + it("selects both classes when both scopes are unioned", () => { + const {items} = buildWorkList( + [ + thread({thread_id: "T1", body: "**issue (blocking):** a"}), + thread({ + thread_id: "T2", + body: "**suggestion (non-blocking):** b", + }), + ], + [...BLOCKING, ...NITS], + ); + expect(items.map((i) => i.threadId)).toEqual(["T1", "T2"]); + }); +}); + +describe("thread ownership", () => { + it("excludes a thread a human opened, even with a blocking-looking label", () => { + // Khan/actions#298 review: nothing downstream would stop a human thread + // whose first line quotes the reviewer's label from becoming a work + // item that an agent then edits code for. + const {items, skipped} = buildWorkList( + [ + { + ...thread({body: "x"}), + comments: [ + {author: "alice", body: "**issue (blocking):** boom"}, + ], + }, + ], + BLOCKING, + ); + expect(items).toEqual([]); + expect(skipped[0].reason).toBe("not-reviewer-thread"); + }); + + it("excludes a thread with no comments at all", () => { + const {items, skipped} = buildWorkList( + [{...thread({body: ""}), comments: []}], + BLOCKING, + ); + expect(items).toEqual([]); + expect(skipped[0].reason).toBe("not-reviewer-thread"); + }); + + it("honours a configured bot login", () => { + const staged = { + ...thread({body: "x"}), + comments: [{author: "other-bot", body: "**issue (blocking):** b"}], + }; + expect(buildWorkList([staged], BLOCKING).items).toEqual([]); + expect( + buildWorkList([staged], BLOCKING, "other-bot").items, + ).toHaveLength(1); + }); +}); diff --git a/workflows/autofix/lib/worklist.ts b/workflows/autofix/lib/worklist.ts new file mode 100644 index 00000000..825ac1b2 --- /dev/null +++ b/workflows/autofix/lib/worklist.ts @@ -0,0 +1,151 @@ +/** + * The work list: which of the reviewer's open threads this autofix run acts on. + * + * Input is the same staged shape the reviewer already produces — the unresolved + * `github-actions[bot]` threads of `review.md` Step 3 Phase 2 ({@link + * StagedThread}) — so autofix reads the reviewer's own artifact rather than + * re-deriving one, and the Conventional-Comment label on each thread is parsed + * with the reviewer's own parser ({@link parseLeadingLabel}). The label + * taxonomy has exactly one owner (`render-comment.ts`) and this module is not + * it. + * + * **The fail-closed direction is inverted from the reviewer's.** In + * `rereview.ts` an unparseable label is treated as blocking, because there the + * safe default is to KEEP a thread the bot cannot classify. Here the safe + * default is the opposite: a thread whose label will not parse is EXCLUDED, + * because the risk being managed is an agent editing code on the strength of a + * finding it could not classify. Same principle, opposite outcome; both are + * "do the conservative thing", and conflating them would have autofix acting on + * exactly the threads the reviewer flagged as least trustworthy. + * + * Outdated threads are excluded for the same reason: a null anchor means the + * line the finding was written about is gone from the diff, so there is nothing + * to act on and any edit would be guesswork. + */ + +import {parseLeadingLabel} from "../../review/lib/rereview.ts"; +import type {StagedThread} from "../../review/lib/rereview.ts"; + +/** One thread selected for fixing, flattened for the prompt and the ledger. */ +export type WorkItem = { + /** GraphQL thread id; the stable identity recorded in the commit trailer. */ + threadId: string; + path: string; + line: number; + /** The Conventional-Comment label parsed off the bot's opening comment. */ + label: string; + /** The bot's opening comment, verbatim; the statement of the finding. */ + body: string; + /** Permalink to the thread's opening comment, when staged. */ + url?: string; +}; + +/** A thread that will NOT be fixed, and why; surfaced in the run summary. */ +export type SkippedThread = { + threadId: string; + path: string; + reason: + | "out-of-scope" + | "outdated-anchor" + | "unparseable-label" + /** The file changed after the review that raised this finding. */ + | "stale-path" + /** Somebody other than the reviewer opened it; not v1's to act on. */ + | "not-reviewer-thread"; + /** The parsed label when there was one; absent for unparseable. */ + label?: string; +}; + +export type WorkList = { + items: WorkItem[]; + skipped: SkippedThread[]; +}; + +/** + * Select the threads in scope for this run. + * + * `findingLabels` is the union the scope resolver computed; a thread is in + * scope when its parsed label is in that set. Selection is stable: threads keep + * their staged order, so the plan artifact and the prompt agree run to run. + */ +export const buildWorkList = ( + threads: readonly StagedThread[], + findingLabels: readonly string[], + botLogin: string | undefined = "github-actions[bot]", +): WorkList => { + const inScope = new Set(findingLabels); + const items: WorkItem[] = []; + const skipped: SkippedThread[] = []; + + for (const thread of threads) { + const first = thread.comments?.[0]; + // Whose thread this is decides whether autofix may touch it at all, and + // that must be enforced here rather than left to staging. A human can + // open a thread whose first line happens to read `**issue (blocking):**` + // (quoting the reviewer, for instance), and nothing else downstream + // would stop it becoming a work item that an agent then edits code for. + // v1 acts on reviewer feedback only; the source axis is not implemented. + // Suffix-insensitive for the same reason staging is: REST spells an + // App's login `github-actions[bot]`, GraphQL spells it + // `github-actions`, and a staged thread can carry either. + const strip = (login: string): string => + login.endsWith("[bot]") + ? login.slice(0, -"[bot]".length).toLowerCase() + : login.toLowerCase(); + if ( + first === undefined || + strip(first.author) !== strip(botLogin ?? "github-actions[bot]") + ) { + skipped.push({ + threadId: thread.thread_id, + path: thread.path, + reason: "not-reviewer-thread", + }); + continue; + } + const opener = first.body; + const label = parseLeadingLabel(opener); + + if (label === null) { + skipped.push({ + threadId: thread.thread_id, + path: thread.path, + reason: "unparseable-label", + }); + continue; + } + if (!inScope.has(label)) { + skipped.push({ + threadId: thread.thread_id, + path: thread.path, + reason: "out-of-scope", + label, + }); + continue; + } + // `line` is null for a thread GitHub marks outdated (the anchored line + // left the diff) and for file-level threads, which have no line to fix. + // Checked by type rather than against null so a malformed staging (the + // JSON is read off disk) lands here instead of downstream. + if (typeof thread.line !== "number") { + skipped.push({ + threadId: thread.thread_id, + path: thread.path, + reason: "outdated-anchor", + label, + }); + continue; + } + + items.push({ + threadId: thread.thread_id, + path: thread.path, + line: thread.line, + label, + body: opener, + ...(thread.url === undefined ? {} : {url: thread.url}), + }); + } + + return {items, skipped}; +}; diff --git a/workflows/autofix/package.json b/workflows/autofix/package.json new file mode 100644 index 00000000..a1368133 --- /dev/null +++ b/workflows/autofix/package.json @@ -0,0 +1,4 @@ +{ + "name": "autofix", + "version": "0.0.0" +} diff --git a/workflows/autofix/version-sync.test.ts b/workflows/autofix/version-sync.test.ts new file mode 100644 index 00000000..88db0025 --- /dev/null +++ b/workflows/autofix/version-sync.test.ts @@ -0,0 +1,45 @@ +/** + * CI backstop for the autofix.md version surface. + * + * autofix.md checks out Khan/actions at a pinned `ref: autofix-v` tag + * to fetch the lib the prompt invokes at runtime, and names the same tag in its + * `source:`. Both must name the release the file ships in, or a consumer gets a + * prompt from one version running code from another. The release flow keeps + * them true by running utils/sync-workflow-versions.ts alongside `changeset + * version`; this test fails any PR (the Version Packages PR included) where the + * literals and the `autofix` package version diverge. + * + * Mirrors workflows/review/version-sync.test.ts; see its header for the failure + * this class of test was written in response to. + */ +import * as fs from "fs"; +import {describe, expect, it} from "vitest"; + +const autofixMd = fs.readFileSync( + new URL("./autofix.md", import.meta.url), + "utf-8", +); +const pkg = JSON.parse( + fs.readFileSync(new URL("./package.json", import.meta.url), "utf-8"), +); + +describe("autofix.md version surface", () => { + it("pins the Khan/actions checkout ref to this release's version", () => { + const refs = [...autofixMd.matchAll(/^\s*ref:\s*(\S+)\s*$/gm)].map( + (m) => m[1], + ); + expect(refs).toEqual([`autofix-v${pkg.version}`]); + }); + + it("matches every autofix-v literal to the package version", () => { + const literals = autofixMd.match(/autofix-v\d+\.\d+\.\d+/g) ?? []; + expect(literals.length).toBeGreaterThan(0); + expect(new Set(literals)).toEqual(new Set([`autofix-v${pkg.version}`])); + }); + + it("names the pinned tag in `source:` too", () => { + expect(autofixMd).toContain( + `source: Khan/actions/workflows/autofix/autofix.md@autofix-v${pkg.version}`, + ); + }); +});