Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .github/workflows/ci-owned-pool-rescue.yml
Original file line number Diff line number Diff line change
Expand Up @@ -157,6 +157,9 @@ jobs:
GH_TOKEN: ${{ github.token }}
RESCUE_SECONDS: ${{ vars.CI_OWNED_POOL_RESCUE_SECONDS }}
POOL_OWNED: ${{ vars.CI_PR_POOL_OWNED }}
# The cloud switch owns this durable outage state. Do not cancel an
# owned run into Blacksmith while its probe says Blacksmith is dead.
CI_CLOUD_OVERFLOW_SAVED: ${{ vars.CI_CLOUD_OVERFLOW_SAVED }}
WATCH_RUN_ID: ${{ inputs.run_id }}
SWEEP: ${{ (github.event_name == 'schedule' || github.event_name == 'workflow_dispatch' && !inputs.run_id) && '1' || '' }}
# 1: wait for ci-macos.yml's late-placement before calling the run ephemeral.
Expand Down
3 changes: 3 additions & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -723,6 +723,9 @@ jobs:
EVENT_NAME: ${{ github.event_name }}
HEAD_REPO: ${{ github.event.pull_request.head.repo.full_name }}
DEFAULT_RUNNER: ${{ vars.MACOS_RUNNER_PR }}
# Durable cloud-outage state. While present, the picker keeps its
# owned placement logic but excludes Blacksmith candidates.
CI_CLOUD_OVERFLOW_SAVED: ${{ vars.CI_CLOUD_OVERFLOW_SAVED }}
# Empty unless the step above minted it: idle owned runners, live.
ROUTE_TOKEN: ${{ steps.route-token.outputs.token || steps.route-token-repo.outputs.token }}
POOL_OVERFLOW: ${{ vars.CI_PR_POOL_OVERFLOW }}
Expand Down
7 changes: 7 additions & 0 deletions scripts/ci/owned_pool_rescue.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,10 @@
below. A dispatch with a run's id (WATCH_RUN_ID) watches that run alone,
checked as a workflow_run event's run would be.

While CI_CLOUD_OVERFLOW_SAVED is present, Blacksmith is known not to start
jobs. The cloud switch keeps overflow on owned pools, so this watcher pauses
instead of cancelling a run into a dead Blacksmith retry.

The script waits for ci.yml's `changes` job, which runs the picker. When the
picker chose a persistent pool, that job uploads a marker artifact
(`macos-pool-persistent-<run id>-<attempt>-<jobs>p<placed>-<pool>`, the counts and pool
Expand Down Expand Up @@ -227,6 +231,7 @@
import ui_tests_dispatch # noqa: E402

CI_WORKFLOW_PATH = ".github/workflows/ci.yml"
CLOUD_OVERFLOW_RECORD_VARIABLE = "CI_CLOUD_OVERFLOW_SAVED"
E2E_WORKFLOW_PATH = ".github/workflows/test-e2e.yml"
IOS_TEST_WORKFLOW_PATH = ".github/workflows/test-ios.yml"
IOS_SCREENSHOTS_WORKFLOW_PATH = ".github/workflows/ios-screenshots.yml"
Expand Down Expand Up @@ -1091,6 +1096,8 @@ def finish(outcome: str) -> int:

if (env.get("POOL_OWNED") or "").strip() != "1":
return finish("owned pools are off (CI_PR_POOL_OWNED is not 1); nothing to watch")
if (env.get(CLOUD_OVERFLOW_RECORD_VARIABLE) or "").strip():
return finish("cloud overflow is off; pausing owned-pool rescue to avoid a Blacksmith retry")
seconds = budget(env.get("RESCUE_SECONDS"))
light_retry = (env.get("OWNED_LIGHT_RETRY") or "").strip() == "1"
if seconds is None:
Expand Down
49 changes: 44 additions & 5 deletions scripts/ci/pr_runner_pool.py
Original file line number Diff line number Diff line change
Expand Up @@ -372,6 +372,26 @@
# outputs and takes the owned label, and a full re-run picks here like attempt 1 (without queueing).
# The light tier, which the rescue's full re-run may claim, stays the rescue's.
RESCUE_ACTOR = "github-actions[bot]"
CLOUD_OVERFLOW_RECORD_VARIABLE = "CI_CLOUD_OVERFLOW_SAVED"


def cloud_overflow_active(record: str | None, default_runner: str | None) -> bool:
"""Whether the cloud switch moved this run's PR lane off Blacksmith.

The record is durable state written before the switch changes repository
variables. Requiring its recorded `after` value to equal the live lane
keeps a stale or hand-edited record from changing picker behavior.
"""
if not (record or "").strip() or not (default_runner or "").strip():
return False
try:
data = json.loads(str(record))
except (TypeError, ValueError):
return False
changed = data.get("changed") if isinstance(data, Mapping) else None
lane = changed.get("MACOS_RUNNER_PR") if isinstance(changed, Mapping) else None
return (isinstance(lane, Mapping) and isinstance(lane.get("after"), str)
and lane["after"].strip() == default_runner.strip())


def host_fault_retry(run_attempt: int, triggering_actor: str | None) -> bool:
Expand Down Expand Up @@ -1540,6 +1560,7 @@ def decide(
owned_now: Mapping[str, int] | None = None,
root_since: Mapping[str, int] | None = None,
root_now: Mapping[str, int] | None = None,
outage: bool = False,
) -> Choice:
"""The preference rule over a janitor snapshot. Uncertainty keeps today's route.

Expand Down Expand Up @@ -1696,10 +1717,13 @@ def idle(counts: Mapping[str, int], added_jobs: int, peaks: Mapping[str, int],
retry = ""
if persistent(label):
# A re-run of failed jobs keeps this run's outputs, so it needs a pool
# named now: the Blacksmith pool this rule would take on the lane's
# own Xcode, which is also the Xcode the owned label names.
# named now: normally the Blacksmith pool this rule would take on the
# lane's own Xcode, which is also the Xcode the owned label names. A
# cloud outage leaves it empty so the workflow falls back to the
# owned route instead of naming a dead pool.
lane = [pool_label for pool_label in usable if not persistent(pool_label) and not POOLS.get(pool_label)]
retry = pick(load, added, lane, limits.max_queued, queue_rounds=queue_rounds).label if lane else DEFAULT_RUNNER
retry = pick(load, added, lane, limits.max_queued, queue_rounds=queue_rounds).label \
if lane else ("" if outage else DEFAULT_RUNNER)
shard = spread_shards(load, added, usable, label, shards)
if shard:
note += f"; its {shards} app-host shards take {shard}, which has more room for them"
Expand Down Expand Up @@ -1767,6 +1791,7 @@ def choose(
shards: int = 0,
queue_rounds: str | None = None,
ref: str = "",
cloud_overflow: str | None = None,
) -> tuple[Choice, Mapping[str, Any] | None]:
"""The pool for this run and the snapshot it was read from (None when none was read).

Expand Down Expand Up @@ -1799,12 +1824,22 @@ def choose(
return Choice("", "", "pull request head repository unknown"), None
fork = head_repo != repo
if not fork:
if (default_runner or "").strip() != DEFAULT_RUNNER:
outage = cloud_overflow_active(cloud_overflow, default_runner)
if not outage and (default_runner or "").strip() != DEFAULT_RUNNER:
return Choice("", "", f"MACOS_RUNNER_PR is {default_runner or 'unset'}, not {DEFAULT_RUNNER}"), None
limits = settings(overflow, order, max_queued, owned, xcode_pins.get(PR_XCODE_VARIABLE), queue_rounds)
if limits is None:
return Choice("", "", f"{OVERFLOW_VARIABLE} is 0, or {ORDER_VARIABLE}/{MAX_QUEUED_VARIABLE}/"
f"{QUEUE_ROUNDS_VARIABLE} is invalid"), None
if outage:
# Blacksmith is unavailable while the record exists. Keep the
# normal owned split/root/GUI placement, but never emit an
# ephemeral candidate or a retry target back onto the dead pool.
limits = dataclasses.replace(limits, order=tuple(label for label in limits.order if persistent(label)))
if not limits.order:
return Choice("", "", "cloud overflow is off; no owned pool is in the picker order"), None
else:
outage = False
unreadable = ""
try:
snapshot = fetch()
Expand Down Expand Up @@ -1932,7 +1967,8 @@ def root_charge(machines: Mapping[str, int]) -> dict[str, int]:
root_since=root_since, root_now=root_now,
# Main only ever takes an owned pool; the replay still
# spreads newer runs over the whole order.
choose_from=tuple(label for label in limits.order if persistent(label)) if main else None)
choose_from=tuple(label for label in limits.order if persistent(label)) if main else None,
outage=outage)
if (main and limits.queue_rounds and persistent(choice.runner)
and (choice.owned_budget < jobs or choice.root_runner and choice.root_budget < jobs)
# Only a pool that holds the whole run at once: on a small one the
Expand Down Expand Up @@ -1971,6 +2007,8 @@ def root_charge(machines: Mapping[str, int]) -> dict[str, int]:
choice = dataclasses.replace(choice, reason=f"retry attempt {run_attempt}; {choice.reason}")
if main and choice.runner:
choice = dataclasses.replace(choice, reason=f"main's full-suite dispatch; {choice.reason}")
if outage and choice.runner:
choice = dataclasses.replace(choice, reason=f"cloud overflow off; {choice.reason}")
return choice, snapshot


Expand Down Expand Up @@ -2368,6 +2406,7 @@ def count_routed(since: str) -> int:
live_online=online,
shards=sum(1 for key in plan.after if key.startswith("shard-")),
queue_rounds=env.get("POOL_QUEUE_ROUNDS") or "",
cloud_overflow=env.get(CLOUD_OVERFLOW_RECORD_VARIABLE),
)
pr_xcode_app = env.get(PR_XCODE_VARIABLE)
# Only a same-repository pull request (and main's dispatch) reads the slots;
Expand Down
11 changes: 11 additions & 0 deletions tests/test_ci_owned_pool_rescue.py
Original file line number Diff line number Diff line change
Expand Up @@ -434,6 +434,16 @@ def test_invalid_budget_watches_nothing(self):
self.assertEqual((code, api.calls), (0, []), value)
self.assertIn("must be 30 to 600", summary)

def test_cloud_overflow_pauses_rescue_without_api_requests(self):
clock = Clock()
api = FakeAPI(clock, persistent_run())
code, summary = run_main(api, clock, env_extra={
"CI_CLOUD_OVERFLOW_SAVED": json.dumps({"changed": {"MACOS_RUNNER_PR": {
"before": BLACKSMITH, "after": MINI}}})
})
self.assertEqual((code, api.calls), (0, []))
self.assertIn("pausing owned-pool rescue", summary)

def test_budget_defaults_to_90(self):
self.assertEqual(rescue.budget(""), 90)
self.assertEqual(rescue.budget(" 120 "), 120)
Expand Down Expand Up @@ -1763,6 +1773,7 @@ def test_runs_the_rescue_script(self):
self.assertEqual(step["run"], "python3 scripts/ci/owned_pool_rescue.py")
self.assertEqual(step["env"]["RESCUE_SECONDS"], "${{ vars.CI_OWNED_POOL_RESCUE_SECONDS }}")
self.assertEqual(step["env"]["POOL_OWNED"], "${{ vars.CI_PR_POOL_OWNED }}")
self.assertEqual(step["env"]["CI_CLOUD_OVERFLOW_SAVED"], "${{ vars.CI_CLOUD_OVERFLOW_SAVED }}")

def test_polls_from_a_github_hosted_runner(self):
self.assertEqual(self.doc["jobs"]["rescue"]["runs-on"], "ubuntu-24.04")
Expand Down
21 changes: 21 additions & 0 deletions tests/test_ci_pr_runner_pool.py
Original file line number Diff line number Diff line change
Expand Up @@ -259,6 +259,26 @@ def test_only_moves_off_the_6vcpu_macos_26_lane(self):
self.assert_default(choose(backlog(), default=""))
self.assert_default(choose(backlog(), default=OLD))

def test_cloud_overflow_record_keeps_owned_picker_active(self):
record = json.dumps({"changed": {"MACOS_RUNNER_PR": {"before": SMALL, "after": MINI}}})
choice = owned_choice(fleet(busy=0), default=MINI, order=f"{MINI},{LARGE},{SMALL}",
cloud_overflow=record)
self.assertEqual(choice.runner, MINI)
self.assertEqual(choice.retry_runner, "")

def test_cloud_overflow_record_never_selects_blacksmith(self):
record = json.dumps({"changed": {"MACOS_RUNNER_PR": {"before": SMALL, "after": MINI}}})
choice = owned_choice(fleet(busy=11), default=MINI, order=f"{MINI},{LARGE},{SMALL}",
cloud_overflow=record, split="1")
self.assertTrue(not choice.runner or choice.runner.startswith("glaeda-"), choice)
self.assertTrue(not choice.retry_runner or choice.retry_runner.startswith("glaeda-"), choice)

def test_unrelated_or_malformed_cloud_record_does_not_enable_picker(self):
unrelated = json.dumps({"changed": {"LINUX_RUNNER": {"before": "blacksmith-4vcpu-ubuntu-2404",
"after": "ubuntu-24.04"}}})
self.assert_default(choose(backlog(), default=MINI, cloud_overflow=unrelated))
self.assert_default(choose(backlog(), default=MINI, cloud_overflow="{"))

def test_kill_switch_and_invalid_settings(self):
self.assert_default(choose(backlog(), overflow="0"))
for bad in ({"order": "warp-macos-26-arm64-12x"}, {"order": f"{LARGE},{LARGE}"},
Expand Down Expand Up @@ -2374,6 +2394,7 @@ def test_changes_job_chooses_once(self):
self.assertIs(step["continue-on-error"], True)
self.assertEqual(step["run"], "python3 scripts/ci/pr_runner_pool.py")
self.assertEqual(step["env"]["DEFAULT_RUNNER"], "${{ vars.MACOS_RUNNER_PR }}")
self.assertEqual(step["env"]["CI_CLOUD_OVERFLOW_SAVED"], "${{ vars.CI_CLOUD_OVERFLOW_SAVED }}")

def test_a_persistent_choice_publishes_the_rescue_marker(self):
steps = self.workflow("ci.yml")["jobs"]["changes"]["steps"]
Expand Down
Loading