Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
29 changes: 26 additions & 3 deletions scripts/ci/pr_runner_pool.py
Original file line number Diff line number Diff line change
Expand Up @@ -274,9 +274,13 @@
take an owned pool like a same-repository pull request, and ci-macos.yml
already routes a `workflow_dispatch` on `refs/heads/main` through the same
inputs. It is placed like a pull request, split and queue rounds
(CI_PR_POOL_QUEUE_ROUNDS) included, on the owned pools only; what does not
fit keeps its own route (MACOS_RUNNER_PR), since only an owned pool is a
candidate for it. With no Blacksmith pool to compare against, its jobs may
(CI_PR_POOL_QUEUE_ROUNDS) included, on the owned pools only. With the
split and queue rounds, an owned pool with machines and root runners for
its whole run holds all of it (queued there), not just the jobs that fit:
the rest would take retry_runner and wait behind every overflowed pull
request. A smaller pool (light) still splits, since the excess would wait
past the owned-pool rescue's budget. With no owned pool it keeps its own route
(MACOS_RUNNER_PR), since only an owned pool is a candidate for it. With no Blacksmith pool to compare against, its jobs may
wait up to the queue rounds and the bound (owned_room()). CI_OWNED_MAIN_RESERVE (0 when unset) holds that many
machines and root runners back for pull requests; with a reserve it takes
an owned pool only whole, and only while its peak is free now (no queue
Expand Down Expand Up @@ -1876,6 +1880,25 @@ def root_charge(machines: Mapping[str, int]) -> dict[str, int]:
# Main only ever takes an owned pool; the replay still
# spreads newer runs over the whole order.
choose_from=tuple(label for label in limits.order if persistent(label)) if main else None)
if (main and limits.queue_rounds and persistent(choice.runner)
and (choice.owned_budget < jobs or choice.root_runner and choice.root_budget < jobs)
# Only a pool that holds the whole run at once: on a small one the
# excess would wait rounds past the rescue's budget.
and owned_capacity.get(choice.runner, 0) >= jobs
and (not choice.root_runner or owned_capacity.get(choice.root_runner, 0) >= jobs)):
# Main queues its whole run on the owned pool instead of splitting:
# the jobs that did not fit took the retry runner and waited behind
# every overflowed pull request there (run 36402943637, 2026-09-28:
# five jobs queued over an hour behind 76 others on 3 running
# machines), so main gave no verdict. The owned queue drains in
# rounds, and the rescue still bounds the wait. With the rounds at 0
# (no queueing) it splits as before.
choice = dataclasses.replace(
# A root budget of `jobs` holds every job whether or not the pool
# has gui runners (root_held() never exceeds the machines held).
choice, owned_budget=jobs, root_budget=max(choice.root_budget, jobs),
reason=choice.reason.replace("the jobs that fit run there, the rest on the retry runner",
"the whole run queues there"))
if main and not persistent(choice.runner):
# A Blacksmith pick would move main off MACOS_RUNNER_PR; keep its route.
choice = Choice("", "", f"main's full-suite dispatch: no owned pool fits its whole run with "
Expand Down
33 changes: 33 additions & 0 deletions tests/test_ci_pr_runner_pool.py
Original file line number Diff line number Diff line change
Expand Up @@ -2730,6 +2730,39 @@ def test_the_reserve_holds_root_runners_and_machines_back_for_pull_requests(self
pull = owned_choice(self.snap(roots_busy=5), owned_slots=self.SLOTS, jobs=9, root_jobs=9)
self.assertEqual(pull.runner, MINI)

def test_a_split_run_queues_whole_on_the_owned_pool(self):
# Every root runner busy and 22 jobs queued there: 6 queue places, the
# run needs 9. A pull request would put the rest on the retry runner;
# main queues them on the minis instead, so they do not wait behind
# every overflowed pull request on Blacksmith (run 36402943637).
snap = self.snap(roots_busy=14)
snap["pools"][ROOT_MINI]["queued"] = 22
choice = self.main_choice(snap, split="1", queue_rounds="2")
self.assertEqual((choice.runner, choice.root_runner), (MINI, ROOT_MINI))
self.assertEqual(pool.place(self.PLAN, choice.owned_budget, root_budget=choice.root_budget)[0],
("admission", *(f"shard-{index}" for index in range(1, 8)), "lag", "cli-product"))
self.assertIn("0 of 14 root runners free and 6 queue places", choice.reason)
self.assertIn("the whole run queues there", choice.reason)
self.assertNotIn("the rest on the retry runner", choice.reason)
# A pull request with the same load still splits.
pull = owned_choice(snap, owned_slots=self.SLOTS, jobs=9, root_jobs=9, split="1", queue_rounds="2")
self.assertEqual((pull.runner, pull.root_budget), (MINI, 6))
# A pool too small for the whole run (light: 4 machines, 2 root
# runners) keeps the split: the excess would wait past the rescue.
light_root = "glaeda-root-light-xcode-26.6"
snap = self.snap(roots_busy=14)
snap["pools"][ROOT_MINI]["queued"] = 30
snap["pools"][LIGHT] = {"queued": 0, "running": 0}
snap["pools"][light_root] = {"queued": 0, "running": 0}
small = self.main_choice(snap, split="1", queue_rounds="2",
owned_slots=json.dumps({MINI: 36, ROOT_MINI: 14, LIGHT: 4, light_root: 2}))
self.assertEqual(small.runner, LIGHT, small.reason)
self.assertNotIn("the whole run queues there", small.reason)
self.assertLess(small.root_budget, pool.root_peak(self.PLAN))
# With the rounds at 0 (no queueing) main splits as before.
self.assertIn("the rest on the retry runner",
self.main_choice(self.snap(roots_busy=6), split="1", queue_rounds="0").reason)

def test_queues_a_round_like_a_pull_request_unless_a_reserve_is_set(self):
# Every mini and root runner busy, Blacksmith backed up (12vcpu's wait
# for 9 jobs is 15 minutes): a pull request queues its 9 root jobs
Expand Down
Loading