diff --git a/.github/workflows/ci-guards.yml b/.github/workflows/ci-guards.yml index bc0a888e5edd..0b3cda064ce5 100644 --- a/.github/workflows/ci-guards.yml +++ b/.github/workflows/ci-guards.yml @@ -360,6 +360,10 @@ jobs: if: ${{ matrix.group == 'ci' }} run: python3 tests/test_ci_main_full_suite.py + - name: Validate main full-suite failure attribution + if: ${{ matrix.group == 'ci' }} + run: python3 tests/test_ci_main_regression_attribution.py + - name: Validate CI queue janitor policy if: ${{ matrix.group == 'ci' }} run: python3 tests/test_ci_queue_janitor.py diff --git a/.github/workflows/ci-main-full-suite.yml b/.github/workflows/ci-main-full-suite.yml index d30f735930a7..f6ba84c5983f 100644 --- a/.github/workflows/ci-main-full-suite.yml +++ b/.github/workflows/ci-main-full-suite.yml @@ -7,7 +7,8 @@ name: CI main full suite # completion dispatches the next on the newest HEAD. A red result therefore # covers only the commits that landed during one run. The schedule is a # backstop for a missed event. It keeps one tracking issue open while main is -# red. +# red, and names the merged pull requests suspected of each new failure, on +# the issue and on those pull requests (main_regression_attribution.py). # # It dispatches ci.yml instead of calling it so the run keeps the identity the # app-host product transport trusts (path ci.yml, event workflow_dispatch), and @@ -128,7 +129,9 @@ jobs: # report. Reports are idempotent. if: ${{ (github.event_name != 'workflow_run' && github.event_name != 'push' && github.ref == 'refs/heads/main') || (github.event.workflow_run.event == 'workflow_dispatch' && github.event.workflow_run.head_branch == 'main' && github.event.workflow_run.path == '.github/workflows/ci.yml') }} runs-on: ${{ github.repository_owner != 'manaflow-ai' && 'ubuntu-24.04' || vars.LINUX_RUNNER || 'blacksmith-4vcpu-ubuntu-2404' }} - timeout-minutes: 5 + # Attribution reads the failed app-host shard logs of two runs (about + # three minutes) and diffs the pull requests merged between them. + timeout-minutes: 20 concurrency: group: ci-main-full-suite-report cancel-in-progress: false @@ -136,13 +139,41 @@ jobs: actions: read contents: read issues: write + # Comments on the pull requests suspected of a new failure. + pull-requests: write steps: - name: Checkout uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 with: fetch-depth: 1 persist-credentials: false - sparse-checkout: scripts/ci + # The trees attribution ranks pull requests against, besides the scripts. + sparse-checkout: | + scripts/ci + cmuxTests + Sources + Packages/macOS + Packages/Shared + CLI + + - name: Attribute new failures to merged pull requests + # A report-only heuristic: its failure must not stop the issue sync. + continue-on-error: true + env: + GH_TOKEN: ${{ github.token }} + RUN_ID: ${{ github.event.workflow_run.id }} + run: | + set -euo pipefail + # Main's recent history, without blobs, for the commit range between + # two runs and each merged pull request's diff. Continuous runs keep + # that range to a few dozen commits. + git fetch --no-tags --filter=blob:none --depth=1000 origin main + run_args=() + if [ -n "$RUN_ID" ]; then + run_args=(--run-id "$RUN_ID") + fi + python3 scripts/ci/main_regression_attribution.py report ${run_args[@]+"${run_args[@]}"} \ + --section-output "$RUNNER_TEMP/new-failures.md" - name: Open, update or close the tracking issue env: @@ -154,4 +185,5 @@ jobs: if [ -n "$RUN_ID" ]; then run_args=(--run-id "$RUN_ID") fi - python3 scripts/ci/main_full_suite.py report ${run_args[@]+"${run_args[@]}"} + python3 scripts/ci/main_full_suite.py report ${run_args[@]+"${run_args[@]}"} \ + --extra-section "$RUNNER_TEMP/new-failures.md" diff --git a/scripts/ci/main_full_suite.py b/scripts/ci/main_full_suite.py index 94d71d95759a..a85b9d6e844a 100644 --- a/scripts/ci/main_full_suite.py +++ b/scripts/ci/main_full_suite.py @@ -22,6 +22,8 @@ `report` syncs the single tracking issue with a completed dispatch run on main: a red run opens the issue or comments on it once, and a green run closes it. +A red report carries the "New since" section main_regression_attribution.py +writes: which tests newly fail and the pull requests suspected of it. """ from __future__ import annotations @@ -135,7 +137,7 @@ def issue_plan(conclusion: str, has_open_issue: bool, already_reported: bool) -> return "none" -def failure_body(run: Mapping[str, object], jobs: list[Mapping[str, object]]) -> str: +def failure_body(run: Mapping[str, object], jobs: list[Mapping[str, object]], extra: str = "") -> str: lines = [ f"Full-suite CI on `main` failed at {run.get('head_sha')}: {run.get('html_url')}", "", @@ -148,6 +150,8 @@ def failure_body(run: Mapping[str, object], jobs: list[Mapping[str, object]]) -> lines.append(f"- ...and {len(jobs) - MAX_LISTED_JOBS} more") else: lines.append("No individual job reported failure; see the run summary.") + if extra.strip(): + lines += ["", extra.strip()] lines += [ "", "Pull requests run a subset of the suite, so this run is the first place " @@ -233,6 +237,17 @@ def ensure_label(repo: str) -> None: ], check=True, capture_output=True, text=True) +def read_extra_section(path: str | None) -> str: + """The new-failure attribution, when main_regression_attribution.py wrote one.""" + if not path: + return "" + try: + with open(path, encoding="utf-8") as handle: + return handle.read() + except OSError: + return "" + + def command_report(args: argparse.Namespace) -> int: if args.run_id: run = gh_json_lines([f"repos/{args.repo}/actions/runs/{args.run_id}", "--jq", "tojson"])[0] @@ -258,7 +273,7 @@ def command_report(args: argparse.Namespace) -> int: "-X", "GET", "-f", "filter=latest", "-f", "per_page=100", "--jq", ".jobs[] | {name, conclusion, html_url} | tojson", ])) - body = failure_body(run, jobs) + body = failure_body(run, jobs, read_extra_section(args.extra_section)) if plan == "open": ensure_label(args.repo) subprocess.run([ @@ -296,6 +311,7 @@ def main(argv: list[str]) -> int: report = commands.add_parser("report", help="sync the tracking issue with a completed run") report.add_argument("--run-id", help="defaults to the newest green or red full-suite run") + report.add_argument("--extra-section", help="markdown to add to a failure report, e.g. new-failure attribution") report.set_defaults(handler=command_report) args = parser.parse_args(argv) diff --git a/scripts/ci/main_regression_attribution.py b/scripts/ci/main_regression_attribution.py new file mode 100644 index 000000000000..56a8c99d9a8d --- /dev/null +++ b/scripts/ci/main_regression_attribution.py @@ -0,0 +1,645 @@ +#!/usr/bin/env python3 +"""Name the merged pull requests behind tests that newly fail on main. + +main_full_suite.py keeps one issue open while main's full suite is red, but a +red run lists failing jobs, not what broke them, and nobody is told. This +reads a red full-suite run and finds its new failures: app-host tests the +shard ratchet reported as RATCHET_NEW_FAILURE, or xcodebuild listed under +"Failing tests:" in a batch the ratchet does not grade, that are not in the +known-failures catalog and did not fail in the previous full-suite run whose app-host +shards all finished. A failure in a shard that run did not fully grade (a +dedicated lane failed first, or the batch stopped early) is listed as having +no baseline instead. Such a run is only used while the test list and shard +packing are unchanged; otherwise an older, fully graded run is the baseline, +and tests the skipped runs saw failing are not new either. + +Each new failure is attributed to the commits between the two runs' head +SHAs, mapped to the pull requests merged into main by those commits. One pull +request in the range is the suspect. With several, each is ranked by whether +its diff reaches the failing test's suite: 2 when it edits the suite +(test_impact.py), 1 when a changed app declaration or string is named by the +suite (reverse_test_impact.py), 0 otherwise. The top score names the +suspects; when every score is 0 the failure is left unattributed rather than +blaming the whole range. + +`report` writes a "New since" markdown section for the tracking issue (read +by main_full_suite.py report --extra-section) and comments once on each +suspect pull request, idempotent through a hidden marker keyed on the pull +request and its failing test set, and once per commit range. A test tied between more than +MAX_PINGED_SUSPECTS pull requests is listed in the issue only. Nothing is +reverted or re-run here. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import re +import subprocess +import sys +from collections.abc import Iterable, Mapping +from dataclasses import dataclass, field +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent)) +import main_full_suite as suite_run # noqa: E402 + +APP_HOST_JOB_RE = re.compile(r"app-host unit tests \((\d+)/\d+\)") +ANSI_RE = re.compile(r"\x1b\[[0-9;]*m") +TIMESTAMP_RE = re.compile(r"^\d{4}-\d\d-\d\dT[\d:.]+Z ?") +# One ratchet verdict per line. +RATCHET_RE = re.compile(r"^RATCHET_NEW_FAILURE (\S+)\s*$") +# xcodebuild's closing "Failing tests:" block, one tab-indented `Suite.test()` +# per line. Batches the ratchet does not grade (dedicated lanes such as the +# global-search shortcuts batch) name their failures only here. +FAILING_TESTS_HEADER = "Failing tests:" +FAILING_TEST_RE = re.compile(r"^\t(?:cmuxTests\.)?([A-Za-z_][\w.]*)\.([A-Za-z_]\w*\(.*\))\s*$") +EXECUTION_FAILED_MARKERS = ("** TEST EXECUTE FAILED **", "** TEST FAILED **") +RAN_CONCLUSIONS = frozenset({"success", "failure"}) +# app_host_result_accounting.py closes every graded batch with one of these. +# Only the accounting's own lines count: a dedicated lane's xcodebuild failure +# stops the shard before its graded batches run. +VERDICT_MARKERS = ( + "RATCHET_NEW_FAILURE ", "typed app-host run passed", "known-main failures tolerated", + "recorded verdicts:", +) +# ...and prints one of these when a batch's tests did not all report, so a +# test that already failed may be missing from its RATCHET_NEW_FAILURE lines. +INCOMPLETE_MARKERS = ( + "incomplete app-host run:", + "typed xcresult is incomplete", + "typed xcresult contains zero Test Case nodes", + "No typed xcresult test JSON found", + "is not ratchetable", + "selector matched zero built tests", + "nonterminal or unknown result", +) +CATALOG = Path(__file__).resolve().parent / "app-host-known-failures.json" +MARKER_PREFIX = "") +# Bounds on one report, so a long red streak cannot fan out into a comment storm. +MAX_RANKED_PRS = 40 +MAX_COMMENTED_PRS = 5 +# A test that ties more pull requests than this is listed in the issue but +# pings none of them: that is a guess, and slice 2's bisect should settle it. +MAX_PINGED_SUSPECTS = 3 +# Earlier runs tried as the baseline before giving up on a comparison. +MAX_BASELINE_CANDIDATES = 8 +MAX_LISTED_TESTS = 30 + + +@dataclass +class PullRequest: + number: int + title: str + url: str + merge_sha: str + author: str = "" + # Suites the diff edits, and suites that name what the diff changes. + edited_suites: set[str] = field(default_factory=set) + reached_suites: set[str] = field(default_factory=set) + + +def log_failures(log_text: str, known: Iterable[str] = ()) -> set[str]: + """Failing test ids in one app-host job log that the known-failures catalog does not list. + + Ratchet verdicts are already graded against the catalog; the xcodebuild + block is not, so both pass through the catalog here. + """ + found = set() + in_block = False + for raw in log_text.splitlines(): + line = TIMESTAMP_RE.sub("", ANSI_RE.sub("", raw)).rstrip() + match = RATCHET_RE.match(line.strip()) + if match: + found.add(match.group(1)) + if line.strip() == FAILING_TESTS_HEADER: + in_block = True + continue + if in_block: + listed = FAILING_TEST_RE.match(line) + if listed: + found.add(f"{listed.group(1).replace('.', '/')}/{listed.group(2)}") + else: + in_block = False + return found - set(known) + + +def shard_log_complete(log_text: str) -> bool: + """True when a failed shard graded every batch, so its failures are the full set.""" + return any(text in log_text for text in VERDICT_MARKERS) and not any( + text in log_text for text in INCOMPLETE_MARKERS + ) + + +def app_host_jobs(jobs: Iterable[Mapping[str, object]]) -> list[Mapping[str, object]]: + return [job for job in jobs if APP_HOST_JOB_RE.search(str(job.get("name") or ""))] + + +def app_host_ran(jobs: Iterable[Mapping[str, object]]) -> bool: + """True when every app-host shard finished, so its failures are a full picture.""" + shards = app_host_jobs(jobs) + return bool(shards) and all(job.get("conclusion") in RAN_CONCLUSIONS for job in shards) + + +def earlier_tested_runs( + runs: Iterable[Mapping[str, object]], current: Mapping[str, object], branch: str = "main", +) -> list[Mapping[str, object]]: + """Completed green or red full-suite runs created before `current`, newest first.""" + created = str(current.get("created_at") or "") + earlier = [ + run for run in runs + if suite_run.is_main_full_suite_run(run, branch) + and run.get("status") == "completed" + and run.get("conclusion") in suite_run.TESTED_CONCLUSIONS + and run.get("id") != current.get("id") + and str(run.get("created_at") or "") < created + ] + earlier.sort(key=lambda run: str(run.get("created_at") or ""), reverse=True) + return earlier + + +def shard_of(job: Mapping[str, object]) -> str: + match = APP_HOST_JOB_RE.search(str(job.get("name") or "")) + return match.group(1) if match else "" + + +def new_failures( + current: Mapping[str, list[str]], + current_shards: Mapping[str, set[str]], + previous: set[str], + ungraded: set[str], +) -> tuple[dict[str, list[str]], list[str]]: + """(new failure -> job URLs, failures with no baseline) against the previous run. + + A test that failed only in shards the previous run did not grade has no + baseline: it may have been failing there unseen, so it is not called new. + """ + new: dict[str, list[str]] = {} + unknown: list[str] = [] + for test, jobs in sorted(current.items()): + if test in previous: + continue + if current_shards.get(test, set()) <= ungraded: + unknown.append(test) + else: + new[test] = jobs + return new, unknown + + +def merged_prs( + range_shas: Iterable[str], associated: Mapping[str, list[Mapping[str, object]]], branch: str = "main", +) -> tuple[list[PullRequest], list[str]]: + """Pull requests merged into `branch` by a commit in the range, oldest first, and direct commits. + + A commit also lists open or unrelated pull requests that contain it, so a + pull request counts only when its merge commit is itself in the range. + """ + ordered = list(range_shas) + in_range = set(ordered) + found: dict[int, PullRequest] = {} + covered: set[str] = set() + for sha in ordered: + for pr in associated.get(sha, []): + merge_sha = str(((pr.get("mergeCommit") or {}) or {}).get("oid") or "") + if pr.get("state") != "MERGED" or pr.get("baseRefName") != branch or merge_sha not in in_range: + continue + covered.add(sha) + number = int(pr["number"]) + if number not in found: + found[number] = PullRequest( + number=number, + title=str(pr.get("title") or ""), + url=str(pr.get("url") or ""), + merge_sha=merge_sha, + author=str(((pr.get("author") or {}) or {}).get("login") or ""), + ) + merge_order = {sha: index for index, sha in enumerate(reversed(ordered))} + prs = sorted(found.values(), key=lambda pr: merge_order.get(pr.merge_sha, 0)) + direct = [sha for sha in ordered if sha not in covered] + return prs, direct + + +def suite_of(test: str) -> str: + return test.split("/", 1)[0] + + +def score(test: str, pr: PullRequest) -> int: + name = suite_of(test) + if name in pr.edited_suites: + return 2 + if name in pr.reached_suites: + return 1 + return 0 + + +def suspects_for( + test: str, prs: list[PullRequest], direct: Iterable[str] = (), +) -> tuple[list[PullRequest], str]: + """(suspects, how) for one failing test; no suspects when the range gives no signal.""" + if len(prs) == 1 and not list(direct): + return prs, "only pull request in the range" + if not prs: + return [], "no merged pull request in the range" + scored = [(score(test, pr), pr) for pr in prs] + best = max((value for value, _ in scored), default=0) + if best == 0: + return [], "no pull request in the range reaches this suite" + how = "edits the suite" if best == 2 else "changes code the suite names" + return [pr for value, pr in scored if value == best], how + + +def tests_digest(tests: Iterable[str]) -> str: + return hashlib.sha256("\n".join(sorted(tests)).encode()).hexdigest()[:16] + + +def commit_range(previous: Mapping[str, object], run: Mapping[str, object]) -> str: + return f"{short(str(previous.get('head_sha') or ''))}..{short(str(run.get('head_sha') or ''))}" + + +def marker(pr_number: int, tests: Iterable[str], range_: str) -> str: + return f"{MARKER_PREFIX} pr={pr_number} tests={tests_digest(tests)} range={range_} -->" + + +def already_told(bodies: Iterable[str], pr_number: int, tests: Iterable[str], range_: str) -> bool: + """A pull request hears once per failing test set, and once per commit range. + + The range covers a re-run of the same red run whose failing set shifted. + """ + digest = tests_digest(tests) + for body in bodies: + for number, seen_digest, seen_range in MARKER_RE.findall(body or ""): + if int(number) == pr_number and (seen_digest == digest or seen_range == range_): + return True + return False + + +def short(sha: str) -> str: + return sha[:10] + + +def issue_section( + *, + repo: str, + run: Mapping[str, object], + previous: Mapping[str, object] | None, + failures: Mapping[str, list[str]], + attributions: Mapping[str, tuple[list[PullRequest], str]], + prs: list[PullRequest], + direct: list[str], + no_baseline: Iterable[str] = (), +) -> str: + if previous is None: + return "### New failures\n\nNo earlier full-suite run with every app-host shard finished to compare against." + prev_sha = str(previous.get("head_sha") or "") + head_sha = str(run.get("head_sha") or "") + lines = [ + f"### New since `{short(prev_sha)}`", + "", + f"Compared with [the previous full-suite run]({previous.get('html_url')}) " + f"({previous.get('conclusion')}); commits: " + f"https://github.com/{repo}/compare/{prev_sha}...{head_sha}", + "", + ] + no_baseline = list(no_baseline) + if no_baseline: + lines += [ + "Not compared, because that run's shard stopped before grading them: " + + ", ".join(f"`{test}`" for test in no_baseline[:MAX_LISTED_TESTS]), + "", + ] + if not failures: + lines.append("No app-host test fails here that did not already fail in that run.") + return "\n".join(lines) + lines += ["Test | Suspect | Jobs", "--- | --- | ---"] + for test in list(failures)[:MAX_LISTED_TESTS]: + suspects, how = attributions[test] + named = ", ".join(f"#{pr.number}" for pr in suspects) or "unattributed" + jobs = " ".join(f"[job]({url})" for url in failures[test][:3]) + lines.append(f"`{test}` | {named} ({how}) | {jobs}") + if len(failures) > MAX_LISTED_TESTS: + lines.append(f"...and {len(failures) - MAX_LISTED_TESTS} more | |") + lines += ["", f"Pull requests merged in the range: " + (", ".join(f"#{pr.number}" for pr in prs) or "none")] + if direct and not prs: + lines.append("Commits without a merged pull request: " + ", ".join(short(sha) for sha in direct[:10])) + return "\n".join(lines) + + +def pr_comment( + *, + repo: str, + pr: PullRequest, + tests: list[str], + how: Mapping[str, str], + run: Mapping[str, object], + previous: Mapping[str, object], + failures: Mapping[str, list[str]], + others: Mapping[str, list[int]], +) -> str: + prev_sha = str(previous.get("head_sha") or "") + head_sha = str(run.get("head_sha") or "") + lines = [ + marker(pr.number, tests, commit_range(previous, run)), + f"These app-host tests newly fail in [main's full suite]({run.get('html_url')}) " + f"at `{short(head_sha)}`, after this pull request merged. They did not fail in " + f"[the previous full-suite run]({previous.get('html_url')}) at `{short(prev_sha)}`, " + "and are not in `scripts/ci/app-host-known-failures.json`.", + "", + ] + for test in tests[:MAX_LISTED_TESTS]: + jobs = " ".join(f"[job]({url})" for url in failures[test][:3]) + shared = others.get(test) or [] + also = f"; also suspected: {', '.join(f'#{n}' for n in shared)}" if shared else "" + lines.append(f"- `{test}` ({how[test]}{also}) {jobs}") + if len(tests) > MAX_LISTED_TESTS: + lines.append(f"- ...and {len(tests) - MAX_LISTED_TESTS} more") + lines += [ + "", + f"Commits in the range: https://github.com/{repo}/compare/{prev_sha}...{head_sha}", + "", + "Pull requests run only the suites their diff reaches, so main's full suite is where " + "this shows first. If this pull request is the cause, please fix forward or revert; if " + "it is not, say so here. This is an automated attribution and can be wrong, most often " + "for a flaky test.", + ] + return "\n".join(lines) + + +def comment_plan( + failures: Mapping[str, list[str]], attributions: Mapping[str, tuple[list[PullRequest], str]], +) -> list[tuple[PullRequest, list[str], dict[str, str], dict[str, list[int]]]]: + """(pr, its tests, how each was attributed, co-suspects per test) per suspect pull request.""" + by_pr: dict[int, tuple[PullRequest, list[str], dict[str, str], dict[str, list[int]]]] = {} + for test in failures: + suspects, how = attributions[test] + if len(suspects) > MAX_PINGED_SUSPECTS: + continue + for pr in suspects: + entry = by_pr.setdefault(pr.number, (pr, [], {}, {})) + entry[1].append(test) + entry[2][test] = how + others = [other.number for other in suspects if other.number != pr.number] + if others: + entry[3][test] = others + return list(by_pr.values())[:MAX_COMMENTED_PRS] + + +# ---- I/O --------------------------------------------------------------------------------- + + +def gh(args: list[str]) -> str: + return subprocess.run(["gh", *args], check=True, capture_output=True, text=True).stdout + + +def git(root: Path, *args: str) -> str: + return subprocess.run( + ["git", "-C", str(root), *args], check=True, capture_output=True, text=True, errors="replace", + ).stdout + + +def run_jobs(repo: str, run_id: object) -> list[dict]: + return suite_run.gh_json_lines([ + f"repos/{repo}/actions/runs/{run_id}/jobs", "--paginate", + "-X", "GET", "-f", "filter=latest", "-f", "per_page=100", + "--jq", ".jobs[] | {id, name, conclusion, html_url} | tojson", + ]) + + +def job_failures( + repo: str, jobs: list[Mapping[str, object]], known: Iterable[str], +) -> tuple[dict[str, list[str]], dict[str, set[str]], set[str]]: + """(failing test -> job URLs, failing test -> shards, shards that did not grade every test) for one run.""" + from app_host_failure_census import _gh_api_escape_flag + + failures: dict[str, list[str]] = {} + shards: dict[str, set[str]] = {} + ungraded: set[str] = set() + for job in app_host_jobs(jobs): + if job.get("conclusion") != "failure": + continue + log = gh(["api", *_gh_api_escape_flag(), f"repos/{repo}/actions/jobs/{job['id']}/logs"]) + if not shard_log_complete(ANSI_RE.sub("", log)): + ungraded.add(shard_of(job)) + for test in log_failures(log, known): + failures.setdefault(test, []).append(str(job.get("html_url") or "")) + shards.setdefault(test, set()).add(shard_of(job)) + return failures, shards, ungraded + + +# What decides which physical shard runs a test. The lane env vars in +# ci-macos.yml also do, but that file changes too often to gate on. +SHARD_INPUTS = ( + "cmuxTests", + "scripts/ci/cmux-unit-test-timings.json", + "scripts/ci/cmux_unit_test_shard.py", + "scripts/ci/run-app-host-unit-batches.sh", +) + + +def shard_map_changed(root: Path, base: str, head: str) -> bool: + """True unless the test list and shard packing are the same at both commits.""" + result = subprocess.run( + ["git", "-C", str(root), "diff", "--quiet", base, head, "--", *SHARD_INPUTS], + capture_output=True, text=True, + ) + return result.returncode != 0 + + +def associated_prs(repo: str, shas: list[str]) -> dict[str, list[dict]]: + owner, name = repo.split("/", 1) + result: dict[str, list[dict]] = {} + for start in range(0, len(shas), 40): + chunk = shas[start:start + 40] + fields = " ".join( + f'c{index}: object(oid: "{sha}") {{ ... on Commit {{ associatedPullRequests(first: 5) ' + "{ nodes { number title url state baseRefName author { login } mergeCommit { oid } } } } }" + for index, sha in enumerate(chunk) + ) + query = f'query {{ repository(owner: "{owner}", name: "{name}") {{ {fields} }} }}' + data = json.loads(gh(["api", "graphql", "-f", f"query={query}"]))["data"]["repository"] + for index, sha in enumerate(chunk): + node = data.get(f"c{index}") or {} + result[sha] = ((node.get("associatedPullRequests") or {}).get("nodes")) or [] + return result + + +def overlay(files: Mapping[str, str], changes: Mapping[str, str | None]) -> dict[str, str]: + """`files` with changed paths replaced by their text, or removed when None.""" + result = dict(files) + for path, text in changes.items(): + if text is None: + result.pop(path, None) + else: + result[path] = text + return result + + +def rank_inputs(root: Path, prs: list[PullRequest]) -> None: + """Fill each pull request's suite sets from its merge commit's diff. + + The trees are read once from the checkout. Each pull request's own changed + files are read at its merge commit, so its hunks' line numbers match the + text they are resolved against. + """ + import tempfile + + import reverse_test_impact + import test_impact + + head_files = reverse_test_impact.read_root(root) + for pr in prs[:MAX_RANKED_PRS]: + base = f"{pr.merge_sha}^1" + try: + status = git(root, "diff", "--no-renames", "--name-status", base, pr.merge_sha).splitlines() + test_diff = git(root, "diff", "--no-renames", "-U0", base, pr.merge_sha, "--", "cmuxTests") + app_diff = git( + root, "diff", "--no-renames", "-U0", base, pr.merge_sha, + "--", "Sources", "Packages/macOS", "Packages/Shared", "CLI", + ) + changes: dict[str, str | None] = {} + for line in status: + code, _, path = line.partition("\t") + if path.endswith(".swift") and path.startswith(reverse_test_impact.TREE_PREFIXES): + changes[path] = None if code == "D" else git(root, "show", f"{pr.merge_sha}:{path}") + except subprocess.CalledProcessError as error: + print(f"::warning::Could not diff #{pr.number}: {(error.stderr or '').strip()}", file=sys.stderr) + continue + files = overlay(head_files, changes) + paths = [line.partition("\t")[2] for line in status] + if any(path.startswith("cmuxTests/") for path in paths): + with tempfile.TemporaryDirectory(prefix="attribution-") as scratch: + for path, text in files.items(): + if path.startswith("cmuxTests/"): + target = Path(scratch) / path + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(text, encoding="utf-8") + edited = test_impact.affected_suites(Path(scratch), paths, test_diff) or [] + pr.edited_suites = {suite.removeprefix("cmuxTests/") for suite in edited} + pr.reached_suites = set(reverse_test_impact.select(files, app_diff).suites) + + +def pr_comment_bodies(repo: str, number: int) -> list[str]: + owner, name = repo.split("/", 1) + query = ( + 'query($owner: String!, $name: String!, $number: Int!) { repository(owner: $owner, name: $name) ' + "{ pullRequest(number: $number) { comments(last: 100) { nodes { body } } } } }" + ) + data = json.loads(gh([ + "api", "graphql", "-f", f"query={query}", "-f", f"owner={owner}", "-f", f"name={name}", + "-F", f"number={number}", + ])) + nodes = data["data"]["repository"]["pullRequest"]["comments"]["nodes"] + return [str(node.get("body") or "") for node in nodes] + + +def resolve_run(args: argparse.Namespace) -> dict | None: + if args.run_id: + run = suite_run.gh_json_lines([f"repos/{args.repo}/actions/runs/{args.run_id}", "--jq", "tojson"])[0] + if not suite_run.is_main_full_suite_run(run, args.branch) or run.get("status") != "completed": + return None + return run + return suite_run.latest_tested_run( + suite_run.list_runs(args.repo, args.branch, ["-f", "status=completed"]), args.branch, + ) + + +def command_report(args: argparse.Namespace) -> int: + run = resolve_run(args) + if run is None or run.get("conclusion") != "failure": + print("No red full-suite run to attribute.") + return 0 + jobs = run_jobs(args.repo, run["id"]) + if not app_host_ran(jobs): + print(f"Run {run['id']} did not finish every app-host shard; nothing to compare.") + return 0 + known = set(json.loads(CATALOG.read_text(encoding="utf-8")).get("tests") or {}) + current, current_shards, _ = job_failures(args.repo, jobs, known) + + # The baseline is the newest earlier run whose app-host shards all + # finished. A shard of it that stopped before grading every test cannot + # show a test was already failing, so failures in that shard get no verdict. + previous = None + previous_ungraded: set[str] = set() + earlier = earlier_tested_runs( + suite_run.list_runs(args.repo, args.branch, ["-f", "status=completed"]), run, args.branch, + ) + # A run with an ungraded shard is a usable baseline only while shard + # numbers still name the same tests: shards are packed from the test list + # and timings, so a change to either can move a test into a shard it never + # ran in. Otherwise look further back for a fully graded run, keeping what + # the skipped runs saw fail, since a test failing there is not new either. + seen_failing: set[str] = set() + for candidate in earlier[:MAX_BASELINE_CANDIDATES]: + candidate_jobs = run_jobs(args.repo, candidate["id"]) + if not app_host_ran(candidate_jobs): + continue + failed: dict[str, list[str]] = {} + ungraded: set[str] = set() + if candidate.get("conclusion") == "failure": + failed, _, ungraded = job_failures(args.repo, candidate_jobs, known) + seen_failing |= set(failed) + if ungraded and shard_map_changed(args.root, str(candidate["head_sha"]), str(run["head_sha"])): + continue + previous, previous_ungraded = candidate, ungraded + break + previous_failures = seen_failing + + failures: dict[str, list[str]] = {} + no_baseline: list[str] = [] + if previous: + failures, no_baseline = new_failures(current, current_shards, previous_failures, previous_ungraded) + prs: list[PullRequest] = [] + direct: list[str] = [] + attributions: dict[str, tuple[list[PullRequest], str]] = {} + if previous and failures: + shas = git(args.root, "rev-list", f"{previous['head_sha']}..{run['head_sha']}").split() + prs, direct = merged_prs(shas, associated_prs(args.repo, shas), args.branch) + if len(prs) > 1 or (prs and direct): + rank_inputs(args.root, prs) + attributions = {test: suspects_for(test, prs, direct) for test in failures} + + section = issue_section( + repo=args.repo, run=run, previous=previous, failures=failures, + attributions=attributions, prs=prs, direct=direct, no_baseline=no_baseline, + ) + print(section) + if args.section_output: + Path(args.section_output).write_text(section + "\n", encoding="utf-8") + + for pr, tests, how, others in comment_plan(failures, attributions): + if already_told(pr_comment_bodies(args.repo, pr.number), pr.number, tests, commit_range(previous, run)): + print(f"#{pr.number} already told about these tests.") + continue + body = pr_comment( + repo=args.repo, pr=pr, tests=tests, how=how, run=run, previous=previous, + failures=failures, others=others, + ) + if args.dry_run: + print(f"--- would comment on #{pr.number} ---\n{body}") + continue + gh(["api", f"repos/{args.repo}/issues/{pr.number}/comments", "-f", f"body={body}"]) + print(f"Commented on #{pr.number}.") + return 0 + + +def main(argv: list[str]) -> int: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--repo", default=os.environ.get("GITHUB_REPOSITORY", "")) + parser.add_argument("--branch", default="main") + commands = parser.add_subparsers(dest="command", required=True) + report = commands.add_parser("report", help="attribute a red run's new failures and tell the suspects") + report.add_argument("--run-id", help="defaults to the newest green or red full-suite run") + report.add_argument("--root", type=Path, default=Path.cwd(), help="a main checkout with history") + report.add_argument("--section-output", help="write the issue's markdown section here") + report.add_argument("--dry-run", action="store_true", help="print pull request comments instead of posting") + report.set_defaults(handler=command_report) + args = parser.parse_args(argv) + if not args.repo: + parser.error("--repo or GITHUB_REPOSITORY is required") + return args.handler(args) + + +if __name__ == "__main__": + sys.exit(main(sys.argv[1:])) diff --git a/tests/test-execution.toml b/tests/test-execution.toml index 4513493c5321..6733ebce9513 100644 --- a/tests/test-execution.toml +++ b/tests/test-execution.toml @@ -1211,3 +1211,7 @@ lane = "macos-shell" [[test]] path = "tests/test_shell_surface_scoped_keys.py" lane = "linux-guard" + +[[test]] +path = "tests/test_ci_main_regression_attribution.py" +lane = "linux-guard" diff --git a/tests/test_ci_main_full_suite.py b/tests/test_ci_main_full_suite.py index 9b42669db9d2..9ed3aa984696 100644 --- a/tests/test_ci_main_full_suite.py +++ b/tests/test_ci_main_full_suite.py @@ -184,6 +184,13 @@ def test_failure_body_lists_failing_jobs_and_run(self): self.assertIn("macos / packages", body) self.assertNotIn("app-host (2)", body) + def test_failure_body_carries_the_attribution_section(self): + body = MODULE.failure_body(run(), [], "### New since `abc`\n\nrow\n") + self.assertIn("### New since `abc`", body) + self.assertIn("closes itself on the next green run", body.split("### New since", 1)[1]) + self.assertEqual(MODULE.read_extra_section(None), "") + self.assertEqual(MODULE.read_extra_section("/nonexistent/new-failures.md"), "") + class SuiteSelectionTests(unittest.TestCase): def test_dispatched_ci_runs_the_full_suite_under_compile_only_policy(self): diff --git a/tests/test_ci_main_regression_attribution.py b/tests/test_ci_main_regression_attribution.py new file mode 100644 index 000000000000..d59f44cd4785 --- /dev/null +++ b/tests/test_ci_main_regression_attribution.py @@ -0,0 +1,301 @@ +"""New failures on main's full suite, their suspect pull requests, and the report text.""" + +import importlib.util +import pathlib +import sys +import unittest + +ROOT = pathlib.Path(__file__).resolve().parents[1] +SCRIPT = ROOT / "scripts/ci/main_regression_attribution.py" +WORKFLOW = ROOT / ".github/workflows/ci-main-full-suite.yml" +sys.path.insert(0, str(SCRIPT.parent)) +SPEC = importlib.util.spec_from_file_location("main_regression_attribution", SCRIPT) +MODULE = importlib.util.module_from_spec(SPEC) +assert SPEC.loader is not None +sys.modules[SPEC.name] = MODULE # dataclasses resolve annotations through it +SPEC.loader.exec_module(MODULE) + +REPO = "manaflow-ai/cmux" +PREV = "p" * 40 +HEAD = "h" * 40 +LOG = """\ +2026-09-25T07:44:10.4990000Z Test Case '-[cmuxTests.FooTests testBar]' failed (0.1 seconds). +2026-09-25T07:44:10.4990110Z RATCHET_NEW_FAILURE AgentSessionAutoResumeSwiftTests/splitAfterRestore() +2026-09-25T07:44:10.4990120Z \x1b[31mRATCHET_NEW_FAILURE FooTests/testBar\x1b[0m +2026-09-25T07:44:10.4990130Z RATCHET_KNOWN_FAILURE SidebarHiddenPresentationTests/visibility() +2026-09-25T07:44:10.4990140Z echo "RATCHET_NEW_FAILURE $identifier" +""" +# A dedicated batch the ratchet does not grade reports only through xcodebuild. +XCODEBUILD_LOG = """\ +2026-09-25T05:55:55.4893920Z Failing tests: +2026-09-25T05:55:55.4894300Z \tGlobalSearchLocalMonitorChainTests.visibleSearchCloses() +2026-09-25T05:55:55.4894300Z \tGlobalSearchLocalMonitorChainTests.visibleSearchCloses() +2026-09-25T05:55:55.4894400Z \tcmuxTests.LegacyTests.testOld() +2026-09-25T05:55:55.4894500Z \tSidebarHiddenPresentationTests.visibility() +2026-09-25T05:55:55.4907210Z +2026-09-25T05:55:55.4907400Z \x1b[1m\x1b[31m** TEST EXECUTE FAILED ** +2026-09-25T05:55:55.4907500Z \tNotATest.after() +""" + + +def run(**overrides): + base = { + "id": 2, "event": "workflow_dispatch", "head_branch": "main", "path": ".github/workflows/ci.yml", + "head_sha": HEAD, "status": "completed", "conclusion": "failure", + "created_at": "2026-09-25T07:00:00Z", "html_url": "https://github.com/x/runs/2", + } + base.update(overrides) + return base + + +def pr_node(number, merge_sha, state="MERGED", base="main"): + return { + "number": number, "title": f"PR {number}", "url": f"https://github.com/{REPO}/pull/{number}", + "state": state, "baseRefName": base, "author": {"login": "someone"}, + "mergeCommit": {"oid": merge_sha} if merge_sha else None, + } + + +def pr(number, edited=(), reached=()): + return MODULE.PullRequest( + number=number, title=f"PR {number}", url=f"u/{number}", merge_sha=f"m{number}", + edited_suites=set(edited), reached_suites=set(reached), + ) + + +class ExtractionTests(unittest.TestCase): + def test_reads_only_ratchet_new_failure_verdict_lines(self): + self.assertEqual( + MODULE.log_failures(LOG), + {"AgentSessionAutoResumeSwiftTests/splitAfterRestore()", "FooTests/testBar"}, + ) + + def test_reads_the_xcodebuild_failing_tests_block_minus_the_catalog(self): + self.assertEqual( + MODULE.log_failures(XCODEBUILD_LOG, {"SidebarHiddenPresentationTests/visibility()"}), + {"GlobalSearchLocalMonitorChainTests/visibleSearchCloses()", "LegacyTests/testOld()"}, + ) + # A dedicated lane's xcodebuild failure stops the shard before its + # graded batches, so it is not a verdict on the shard. + self.assertFalse(MODULE.shard_log_complete(MODULE.ANSI_RE.sub("", XCODEBUILD_LOG))) + + def test_catalog_ids_match_what_the_log_names(self): + known = set(MODULE.json.loads(MODULE.CATALOG.read_text())["tests"]) + listed = "Failing tests:\n" + "".join("\t" + t.replace("/", ".", 1) + "\n" for t in known) + self.assertEqual(MODULE.log_failures(listed, known), set()) + + def test_a_failed_shard_is_complete_only_when_every_batch_was_graded(self): + self.assertTrue(MODULE.shard_log_complete(LOG)) + self.assertTrue(MODULE.shard_log_complete("typed app-host run passed: 796 test cases\n")) + # Failed before any batch was graded: no verdict at all. + self.assertFalse(MODULE.shard_log_complete("##[error]Process completed with exit code 1.\n")) + for stop in ( + "incomplete app-host run: app host restarted after test execution", + "typed xcresult is incomplete: 3 selected Test Case(s) have no terminal result", + "No typed xcresult test JSON found for unit-physical-3", + "xcodebuild status 70 is not ratchetable", + ): + with self.subTest(stop=stop): + self.assertFalse(MODULE.shard_log_complete(LOG + stop + "\n")) + + def test_app_host_ran_needs_every_shard_finished(self): + shards = [{"name": f"macos / app-host unit tests ({n}/7)", "conclusion": "failure"} for n in range(1, 8)] + rollup = {"name": "ci-status", "conclusion": "failure"} + self.assertTrue(MODULE.app_host_ran(shards + [rollup])) + self.assertFalse(MODULE.app_host_ran(shards[:-1] + [{**shards[-1], "conclusion": "cancelled"}])) + # A compile break skips every shard, which says nothing about tests. + self.assertFalse(MODULE.app_host_ran([{"name": "macos / macOS compile admission", "conclusion": "failure"}])) + + +class BaselineTests(unittest.TestCase): + def test_previous_run_is_an_earlier_tested_full_suite_run(self): + current = run() + runs = [ + current, + run(id=5, created_at="2026-09-25T08:00:00Z"), # later + run(id=3, created_at="2026-09-25T06:00:00Z", conclusion="cancelled"), + run(id=4, created_at="2026-09-25T05:00:00Z", event="pull_request"), + run(id=1, created_at="2026-09-25T04:00:00Z", conclusion="success"), + run(id=0, created_at="2026-09-25T03:00:00Z"), + ] + self.assertEqual([r["id"] for r in MODULE.earlier_tested_runs(runs, current)], [1, 0]) + + def test_new_failures_drop_what_failed_before(self): + current = {"A/a()": ["j1"], "B/b()": ["j2"]} + shards = {"A/a()": {"1"}, "B/b()": {"2"}} + self.assertEqual(MODULE.new_failures(current, shards, {"A/a()"}, set()), ({"B/b()": ["j2"]}, [])) + self.assertEqual(MODULE.new_failures(current, shards, set(), set()), (current, [])) + + def test_a_shard_the_baseline_did_not_grade_gives_no_verdict(self): + current = {"A/a()": ["j1"], "B/b()": ["j2"], "C/c()": ["j3", "j4"]} + shards = {"A/a()": {"1"}, "B/b()": {"2"}, "C/c()": {"2", "3"}} + self.assertEqual( + MODULE.new_failures(current, shards, set(), {"2"}), + ({"A/a()": ["j1"], "C/c()": ["j3", "j4"]}, ["B/b()"]), + ) + + def test_shard_map_changed_compares_the_packing_inputs(self): + import subprocess, tempfile + with tempfile.TemporaryDirectory() as repo: + def git(*args): + return subprocess.run(["git", "-C", repo, *args], check=True, capture_output=True, text=True).stdout.strip() + git("init", "-q") + git("config", "user.email", "t@t"); git("config", "user.name", "t") + (pathlib.Path(repo) / "cmuxTests").mkdir() + (pathlib.Path(repo) / "cmuxTests/A.swift").write_text("a") + git("add", "-A"); git("commit", "-qm", "a"); first = git("rev-parse", "HEAD") + (pathlib.Path(repo) / "Sources").mkdir() + (pathlib.Path(repo) / "Sources/B.swift").write_text("b") + git("add", "-A"); git("commit", "-qm", "b"); second = git("rev-parse", "HEAD") + (pathlib.Path(repo) / "cmuxTests/A.swift").write_text("a2") + git("add", "-A"); git("commit", "-qm", "c"); third = git("rev-parse", "HEAD") + self.assertFalse(MODULE.shard_map_changed(pathlib.Path(repo), first, second)) + self.assertTrue(MODULE.shard_map_changed(pathlib.Path(repo), second, third)) + self.assertTrue(MODULE.shard_map_changed(pathlib.Path(repo), first, "0" * 40)) + + def test_shard_of_reads_the_job_name(self): + self.assertEqual(MODULE.shard_of({"name": "macos / app-host unit tests (4/7)"}), "4") + + +class MergedPullRequestTests(unittest.TestCase): + def test_only_pull_requests_merged_by_a_commit_in_the_range_count(self): + # rev-list order: newest first. + shas = ["m2", "branchcommit", "m1", "direct"] + associated = { + "m2": [pr_node(2, "m2")], + "branchcommit": [pr_node(2, "m2"), pr_node(9, None, state="OPEN")], + "m1": [pr_node(1, "m1"), pr_node(7, "elsewhere")], + "direct": [pr_node(8, "m8", base="release")], + } + prs, direct = MODULE.merged_prs(shas, associated) + self.assertEqual([p.number for p in prs], [1, 2]) # oldest merge first + self.assertEqual(direct, ["direct"]) + + +class RankingTests(unittest.TestCase): + def test_one_pull_request_is_the_suspect(self): + only = pr(1) + self.assertEqual(MODULE.suspects_for("Suite/test()", [only]), ([only], "only pull request in the range")) + + def test_a_direct_push_in_the_range_needs_the_pull_request_to_reach_the_suite(self): + self.assertEqual(MODULE.suspects_for("Suite/test()", [pr(1)], ["abc"])[0], []) + reaches = pr(1, reached={"Suite"}) + self.assertEqual(MODULE.suspects_for("Suite/test()", [reaches], ["abc"])[0], [reaches]) + + def test_overlay_reads_changed_files_at_the_merge(self): + files = {"Sources/A.swift": "old", "Sources/B.swift": "b", "cmuxTests/T.swift": "t"} + self.assertEqual( + MODULE.overlay(files, {"Sources/A.swift": "new", "cmuxTests/T.swift": None, "Sources/C.swift": "c"}), + {"Sources/A.swift": "new", "Sources/B.swift": "b", "Sources/C.swift": "c"}, + ) + self.assertEqual(files["Sources/A.swift"], "old") + + def test_editing_the_suite_beats_reaching_it(self): + edits, reaches, neither = pr(1, edited={"Suite"}), pr(2, reached={"Suite"}), pr(3) + suspects, how = MODULE.suspects_for("Suite/test()", [neither, reaches, edits]) + self.assertEqual([p.number for p in suspects], [1]) + self.assertEqual(how, "edits the suite") + suspects, how = MODULE.suspects_for("Suite/test()", [neither, reaches]) + self.assertEqual([p.number for p in suspects], [2]) + + def test_ties_name_every_top_pull_request(self): + a, b = pr(1, reached={"Suite"}), pr(2, reached={"Suite"}) + self.assertEqual([p.number for p in MODULE.suspects_for("Suite/t()", [a, b, pr(3)])[0]], [1, 2]) + + def test_no_signal_blames_nobody(self): + self.assertEqual(MODULE.suspects_for("Suite/t()", [pr(1), pr(2)])[0], []) + self.assertEqual(MODULE.suspects_for("Suite/t()", [])[0], []) + + +class ReportTests(unittest.TestCase): + def setUp(self): + self.a, self.b = pr(1, reached={"S"}), pr(2, reached={"S", "T"}) + self.failures = {"S/x()": ["https://job/1"], "T/y()": ["https://job/2"], "U/z()": ["https://job/3"]} + self.attributions = {test: MODULE.suspects_for(test, [self.a, self.b]) for test in self.failures} + self.previous = run(id=1, head_sha=PREV, conclusion="success", html_url="https://github.com/x/runs/1") + + def test_issue_section_lists_each_new_failure_with_suspects_and_jobs(self): + text = MODULE.issue_section( + repo=REPO, run=run(), previous=self.previous, failures=self.failures, + attributions=self.attributions, prs=[self.a, self.b], direct=[], + ) + self.assertIn(f"### New since `{PREV[:10]}`", text) + self.assertIn(f"https://github.com/{REPO}/compare/{PREV}...{HEAD}", text) + self.assertIn("`S/x()` | #1, #2 (changes code the suite names) | [job](https://job/1)", text) + self.assertIn("`T/y()` | #2 (changes code the suite names)", text) + self.assertIn("`U/z()` | unattributed", text) + + def test_issue_section_lists_failures_without_a_baseline(self): + text = MODULE.issue_section( + repo=REPO, run=run(), previous=self.previous, failures={}, attributions={}, prs=[], direct=[], + no_baseline=["B/b()"], + ) + self.assertIn("Not compared, because that run's shard stopped before grading them: `B/b()`", text) + + def test_issue_section_without_new_failures_or_baseline(self): + text = MODULE.issue_section( + repo=REPO, run=run(), previous=self.previous, failures={}, attributions={}, prs=[], direct=[], + ) + self.assertIn("No app-host test fails here", text) + text = MODULE.issue_section( + repo=REPO, run=run(), previous=None, failures={}, attributions={}, prs=[], direct=[], + ) + self.assertIn("No earlier full-suite run", text) + + def test_one_comment_per_suspect_with_its_own_tests(self): + plan = MODULE.comment_plan(self.failures, self.attributions) + self.assertEqual([(p.number, tests) for p, tests, _, _ in plan], [(1, ["S/x()"]), (2, ["S/x()", "T/y()"])]) + pr2, tests, how, others = plan[1] + body = MODULE.pr_comment( + repo=REPO, pr=pr2, tests=tests, how=how, run=run(), previous=self.previous, + failures=self.failures, others=others, + ) + self.assertTrue(body.startswith(MODULE.marker(2, ["S/x()", "T/y()"], f"{PREV[:10]}..{HEAD[:10]}"))) + self.assertTrue(MODULE.already_told([body], 2, ["T/y()", "S/x()"], "other..range")) + self.assertIn("- `S/x()` (changes code the suite names; also suspected: #1) [job](https://job/1)", body) + self.assertIn("- `T/y()` (changes code the suite names) [job](https://job/2)", body) + self.assertNotIn("U/z()", body) + self.assertNotIn("—", body) + + def test_a_wide_tie_pings_nobody(self): + tied = [pr(n, reached={"S"}) for n in range(1, MODULE.MAX_PINGED_SUSPECTS + 2)] + failures = {"S/x()": ["https://job/1"]} + attributions = {"S/x()": MODULE.suspects_for("S/x()", tied)} + self.assertEqual(len(attributions["S/x()"][0]), len(tied)) + self.assertEqual(MODULE.comment_plan(failures, attributions), []) + + def test_a_pull_request_hears_once_per_test_set_and_once_per_range(self): + told = ["intro", MODULE.marker(2, ["a", "b"], "p..h")] + self.assertEqual(MODULE.marker(2, ["b", "a"], "p..h"), MODULE.marker(2, ["a", "b"], "p..h")) + self.assertTrue(MODULE.already_told(told, 2, ["b", "a"], "p2..h2")) # same tests, later range + self.assertTrue(MODULE.already_told(told, 2, ["a"], "p..h")) # re-run of the same range + self.assertFalse(MODULE.already_told(told, 2, ["a"], "p2..h2")) + self.assertFalse(MODULE.already_told(told, 3, ["a", "b"], "p..h")) + self.assertFalse(MODULE.already_told([], 2, ["a"], "p..h")) + + +class WorkflowTests(unittest.TestCase): + text = WORKFLOW.read_text(encoding="utf-8") + report = text.split("\n report:\n", 1)[1] + + def test_report_job_can_comment_on_pull_requests(self): + self.assertIn("pull-requests: write", self.report) + self.assertIn("issues: write", self.report) + + def test_attribution_feeds_the_issue_section_and_cannot_block_it(self): + self.assertIn("scripts/ci/main_regression_attribution.py", self.report) + step = self.report.split("main_regression_attribution.py", 1)[0].rsplit("- name:", 1)[1] + self.assertIn("continue-on-error: true", step) + self.assertIn("--extra-section", self.report) + + def test_the_issue_sync_checkout_stays_shallow_and_attribution_deepens_it(self): + checkout = self.report.split(" - name: Checkout\n", 1)[1].split("\n - name:", 1)[0] + self.assertIn("fetch-depth: 1", checkout) + for tree in ("scripts/ci", "cmuxTests", "Sources", "Packages/macOS", "Packages/Shared", "CLI"): + self.assertIn(f" {tree}\n", checkout) + step = self.report.split("main_regression_attribution.py", 1)[0].rsplit("- name:", 1)[1] + self.assertIn("git fetch --no-tags --filter=blob:none --depth=", step) + + +if __name__ == "__main__": + unittest.main()