diff --git a/scripts/ci/app_host_result_accounting.py b/scripts/ci/app_host_result_accounting.py index dc5a20cbe92c..7f5d9a8940eb 100644 --- a/scripts/ci/app_host_result_accounting.py +++ b/scripts/ci/app_host_result_accounting.py @@ -285,6 +285,35 @@ def run_is_complete(log_text: str) -> tuple[bool, str]: return True, "no interruption marker" +def recorded_failure_diagnostics( + results: dict[str, str], + known: dict[str, dict[str, Any]], +) -> list[str]: + """Name every recorded failure without deciding the run's verdict. + + The ratchet below fails fast: it reports new failures and returns without + mentioning known ones, because the verdict is already decided. A run that + is being reported for some other reason wants the opposite -- the complete + picture, since its verdict does not depend on what this finds. + """ + failures = { + identifier for identifier, result in results.items() if result == "Failed" + } + new_failures = sorted(failures - set(known)) + known_failures = sorted(failures & set(known)) + messages = [f"RATCHET_NEW_FAILURE {identifier}" for identifier in new_failures] + messages += [f"RATCHET_KNOWN_FAILURE {identifier}" for identifier in known_failures] + if messages: + # Mirror the summary the complete path prints. Without it, a reader + # scanning shard output for "the accounting ran" sees the same silence + # here that the missing verdicts themselves used to produce. + messages.append( + f"recorded verdicts: {len(new_failures)} new, " + f"{len(known_failures)} known-main; typed test cases: {len(results)}" + ) + return messages + + def check_run( *, inventory: set[str], @@ -324,6 +353,12 @@ def check_run( messages.append( f"... {len(missing_execution) - 20} additional selected Test Case(s) missing" ) + # An incomplete result set still carries a verdict for everything that + # did finish. Naming those costs nothing and is the only way to tell a + # shard whose remaining tests regressed from one whose remaining tests + # went green -- without it both print the same "incomplete" line, and a + # full suite can be red while naming no regression at all. + messages.extend(recorded_failure_diagnostics(results, known)) return False, messages if xcode_status not in {0, 65}: diff --git a/tests/test_ci_app_host_result_accounting.py b/tests/test_ci_app_host_result_accounting.py index 70e974110c60..c57c39aea6e6 100644 --- a/tests/test_ci_app_host_result_accounting.py +++ b/tests/test_ci_app_host_result_accounting.py @@ -185,6 +185,46 @@ def test_missing_selected_test_result_never_passes() -> None: ] +def test_incomplete_run_still_names_the_failures_it_recorded() -> None: + """A shard with one missing result must still report what actually failed. + + On main's full suite at f3d204a462 all six app-host shards returned at the + incompleteness gate, so not one RATCHET_NEW_FAILURE line was printed across + the whole run even though the logs carried real assertion failures. A red + suite that names no regression cannot tell anyone whether a fix landed. + """ + passed, messages = accounting.check_run( + inventory={"FooTests/testOne()", "BarTests/testTwo()", "BazTests/testThree()"}, + selectors=["FooTests", "BarTests", "BazTests"], + results={ + "FooTests/testOne()": "Failed", + "BazTests/testThree()": "Failed", + }, + known={"BazTests/testThree()": "known on main"}, + log_text="", + xcode_status=65, + ) + assert passed is False + assert "typed xcresult is incomplete: 1 selected Test Case(s) have no terminal result" in messages + assert "missing typed test result: BarTests/testTwo()" in messages + assert "RATCHET_NEW_FAILURE FooTests/testOne()" in messages + assert "RATCHET_KNOWN_FAILURE BazTests/testThree()" in messages + assert "recorded verdicts: 1 new, 1 known-main; typed test cases: 2" in messages + + +def test_incomplete_run_without_failures_adds_no_ratchet_noise() -> None: + passed, messages = accounting.check_run( + inventory={"FooTests/testOne()", "BarTests/testTwo()"}, + selectors=["FooTests", "BarTests"], + results={"FooTests/testOne()": "Passed"}, + known={}, + log_text="", + xcode_status=0, + ) + assert passed is False + assert not [m for m in messages if m.startswith("RATCHET_")] + + def test_partial_suite_result_never_passes() -> None: passed, messages = accounting.check_run( inventory={"FooTests/testOne()", "FooTests/testTwo()"},