Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
35 changes: 35 additions & 0 deletions scripts/ci/app_host_result_accounting.py
Original file line number Diff line number Diff line change
Expand Up @@ -285,6 +285,35 @@ def run_is_complete(log_text: str) -> tuple[bool, str]:
return True, "no interruption marker"


def recorded_failure_diagnostics(
results: dict[str, str],
known: dict[str, dict[str, Any]],
) -> list[str]:
"""Name every recorded failure without deciding the run's verdict.

The ratchet below fails fast: it reports new failures and returns without
mentioning known ones, because the verdict is already decided. A run that
is being reported for some other reason wants the opposite -- the complete
picture, since its verdict does not depend on what this finds.
"""
failures = {
identifier for identifier, result in results.items() if result == "Failed"
}
new_failures = sorted(failures - set(known))
known_failures = sorted(failures & set(known))
messages = [f"RATCHET_NEW_FAILURE {identifier}" for identifier in new_failures]
messages += [f"RATCHET_KNOWN_FAILURE {identifier}" for identifier in known_failures]
if messages:
# Mirror the summary the complete path prints. Without it, a reader
# scanning shard output for "the accounting ran" sees the same silence
# here that the missing verdicts themselves used to produce.
messages.append(
f"recorded verdicts: {len(new_failures)} new, "
f"{len(known_failures)} known-main; typed test cases: {len(results)}"
)
return messages


def check_run(
*,
inventory: set[str],
Expand Down Expand Up @@ -324,6 +353,12 @@ def check_run(
messages.append(
f"... {len(missing_execution) - 20} additional selected Test Case(s) missing"
)
# An incomplete result set still carries a verdict for everything that
# did finish. Naming those costs nothing and is the only way to tell a
# shard whose remaining tests regressed from one whose remaining tests
# went green -- without it both print the same "incomplete" line, and a
# full suite can be red while naming no regression at all.
messages.extend(recorded_failure_diagnostics(results, known))
return False, messages

if xcode_status not in {0, 65}:
Expand Down
40 changes: 40 additions & 0 deletions tests/test_ci_app_host_result_accounting.py
Original file line number Diff line number Diff line change
Expand Up @@ -185,6 +185,46 @@ def test_missing_selected_test_result_never_passes() -> None:
]


def test_incomplete_run_still_names_the_failures_it_recorded() -> None:
"""A shard with one missing result must still report what actually failed.

On main's full suite at f3d204a462 all six app-host shards returned at the
incompleteness gate, so not one RATCHET_NEW_FAILURE line was printed across
the whole run even though the logs carried real assertion failures. A red
suite that names no regression cannot tell anyone whether a fix landed.
"""
passed, messages = accounting.check_run(
inventory={"FooTests/testOne()", "BarTests/testTwo()", "BazTests/testThree()"},
selectors=["FooTests", "BarTests", "BazTests"],
results={
"FooTests/testOne()": "Failed",
"BazTests/testThree()": "Failed",
},
known={"BazTests/testThree()": "known on main"},
log_text="",
xcode_status=65,
)
assert passed is False
assert "typed xcresult is incomplete: 1 selected Test Case(s) have no terminal result" in messages
assert "missing typed test result: BarTests/testTwo()" in messages
assert "RATCHET_NEW_FAILURE FooTests/testOne()" in messages
assert "RATCHET_KNOWN_FAILURE BazTests/testThree()" in messages
assert "recorded verdicts: 1 new, 1 known-main; typed test cases: 2" in messages


def test_incomplete_run_without_failures_adds_no_ratchet_noise() -> None:
passed, messages = accounting.check_run(
inventory={"FooTests/testOne()", "BarTests/testTwo()"},
selectors=["FooTests", "BarTests"],
results={"FooTests/testOne()": "Passed"},
known={},
log_text="",
xcode_status=0,
)
assert passed is False
assert not [m for m in messages if m.startswith("RATCHET_")]


def test_partial_suite_result_never_passes() -> None:
passed, messages = accounting.check_run(
inventory={"FooTests/testOne()", "FooTests/testTwo()"},
Expand Down