diff --git a/docs/ci/health-report.md b/docs/ci/health-report.md index 6c40bd855d5b..093741aba79c 100644 --- a/docs/ci/health-report.md +++ b/docs/ci/health-report.md @@ -33,6 +33,9 @@ Related: #13095 (CI cost and capacity) and #13325 (CI waste). Minutes are `completed_at - started_at` per job, summed, and split by conclusion into success / failure / cancelled / skipped. `timed_out` and `startup_failure` count as failure, because they cost what a failure costs. +A job cancelled before any runner picked it up (`runner_id: 0`, no steps) has +zero minutes; the API stamps its `started_at` at creation, so its whole wait +until the cancel counts as queue wait instead. These tables are **sampled**: job listings are the expensive API call, so the report reads jobs for a bounded number of runs spread across workflows (macOS- @@ -41,11 +44,18 @@ which workflow, which job, which pool, which conclusion is eating the minutes not as a billing total. The header states how many runs were sampled. Failure and cancelled minutes are the interesting columns. Success minutes are -the price of CI; the rest is the price of CI not working. +the price of CI; the rest is the price of CI not working. One exception: +`cmux-tui-testbox-warmup.yml` holds a Testbox VM for a maintainer session, and +the session's cleanup (`scripts/blacksmith-testbox-demo.sh`) cancels the run on +purpose, so its cancelled minutes are the session itself. ### Queue wait (created → started) per runner label `started_at - created_at` per job, as p50 / p90 / p99 / worst, per runner label. +A job cancelled while still queued counts with its wait up to the cancel, a +lower bound on what it would have waited. These used to count as zero, so the +first reports after that change show a higher macOS p90 without any real +regression. Labels are joined when a job asks for several, so `self-hosted+macos` is not silently pooled with `macos`. diff --git a/scripts/ci/ci_health_report.py b/scripts/ci/ci_health_report.py index 8815012d2444..dd964e936076 100644 --- a/scripts/ci/ci_health_report.py +++ b/scripts/ci/ci_health_report.py @@ -246,6 +246,16 @@ class JobRow: fork: bool +def never_got_a_runner(job: Mapping[str, Any]) -> bool: + """A job that finished without ever being assigned a runner. + + The Actions API reports such a job (cancelled while queued) with + `runner_id: 0`, an empty runner name and no steps. A skipped job has + `runner_id: null` instead and no start time, so it is not matched here. + """ + return job.get("runner_id") == 0 and not job.get("runner_name") and not job.get("steps") + + def job_rows(run: Mapping[str, Any], jobs: Iterable[Mapping[str, Any]], repo: str) -> list[JobRow]: """Flatten one run's jobs into rows the aggregations read. @@ -258,6 +268,10 @@ def job_rows(run: Mapping[str, Any], jobs: Iterable[Mapping[str, Any]], repo: st created = parse_time(job.get("created_at")) started = parse_time(job.get("started_at")) completed = parse_time(job.get("completed_at")) + if never_got_a_runner(job): + # GitHub stamps started_at = created_at on a job cancelled while + # still queued, so started -> completed would be the whole wait. + started = completed minutes = 0.0 if started and completed and completed > started: minutes = (completed - started).total_seconds() / 60.0 diff --git a/tests/test_ci_health_report.py b/tests/test_ci_health_report.py index a6d6b31c0833..ec999f92891c 100644 --- a/tests/test_ci_health_report.py +++ b/tests/test_ci_health_report.py @@ -85,6 +85,30 @@ def test_a_job_that_never_started_has_no_wait_and_no_minutes(self): self.assertEqual(skipped.minutes, 0.0) self.assertEqual(skipped.bucket, "skipped") + def test_a_job_cancelled_while_queued_is_wait_not_minutes(self): + # Shape of iroh-v2 run 35941162262's client job: superseded after 53 + # minutes in the macOS queue, never assigned a runner. The API stamps + # started_at = created_at on it. + queued = { + "name": "client", + "conclusion": "cancelled", + "created_at": "2026-09-24T01:02:41Z", + "started_at": "2026-09-24T01:02:41Z", + "completed_at": "2026-09-24T01:55:37Z", + "runner_id": 0, + "runner_name": "", + "steps": [], + "labels": ["blacksmith-6vcpu-macos-26"], + } + ran = dict(queued, runner_id=1195080, runner_name="mac-runner-1", steps=[{"name": "Set up job"}]) + run = self.fixture["runs"][0] + [queued_row, ran_row] = report.job_rows(run, [queued, ran], REPO) + self.assertEqual(queued_row.minutes, 0.0) + self.assertAlmostEqual(queued_row.queue_seconds, (52 * 60) + 56) + self.assertEqual(queued_row.bucket, "cancelled") + self.assertEqual(report.cancelled_macos_waste([queued_row]), []) + self.assertAlmostEqual(ran_row.minutes, 52 + 56 / 60) + def test_fork_head_repository_marks_the_row(self): fork = next(row for row in self.rows if row.run_id == 1013) self.assertTrue(fork.fork)