From eca34965b3f1a53bee4ca4a9dd6dd204c3598fd4 Mon Sep 17 00:00:00 2001 From: Shannon Sands Date: Sun, 16 Aug 2026 00:34:08 +1000 Subject: [PATCH 1/2] fix(readiness): surface unrepaired state.db corruption on the state_db probe (OOF-106) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The state_db readiness probe only read sqlite_master (page 1), so page-level corruption in the sessions table b-tree kept /api/status green — in the OOF-106 incident a page-corrupt state.db reported "ok" for 10+ days while sessions silently failed to persist and repair attempts churned in the background. Two additions, both bounded and read-only: * Consult the persistent repair-attempt ledger (#86747's state.db.repair-attempts.json): when its fingerprint still matches the current file bytes, automatic repair has already failed on this exact database and the probe reports "degraded / unrepaired corruption" without opening the DB at all. Import-light (reads the sidecar JSON directly); malformed/stale ledgers read as "no signal". * Deepen the read probe one step past page 1: fetch a single row from the sessions table (SELECT * so the table b-tree, not an index, is walked). Still O(1) pages, still read-only, catches root-page damage of the table every session write depends on. Tests: ledger-match degrades, stale ledger ignored, garbage ledger ignored, and a real zeroed-root-page corruption fixture that passes the schema-only probe but fails the deepened one. Complements #86747 (bounded repair loop + backup dedup), which covers the repair/backup side of OOF-106 but leaves the readiness surface falsely green. Supersedes the readiness part of #82940. --- gateway/readiness.py | 56 ++++++++++++++++ tests/gateway/test_readiness.py | 111 ++++++++++++++++++++++++++++++++ 2 files changed, 167 insertions(+) diff --git a/gateway/readiness.py b/gateway/readiness.py index d49ea820482c5..badba7b3a0786 100644 --- a/gateway/readiness.py +++ b/gateway/readiness.py @@ -2,6 +2,7 @@ from __future__ import annotations +import json import shutil import sqlite3 from contextlib import closing @@ -24,10 +25,50 @@ def _check(status: str, detail: str | None = None, **extra: Any) -> dict[str, An return result +def _unrepaired_corruption_marker(path: Path) -> bool: + """True when the repair-attempt ledger matches the current file bytes. + + ``hermes_state`` writes ``state.db.repair-attempts.json`` beside the + database when automatic schema surgery fails, keyed by a size+mtime_ns + fingerprint of the exact bytes attempted, and deletes it on success + (#86747). A ledger whose fingerprint still matches therefore means: this + database is corrupt, automatic repair already failed on these bytes, and + nothing has changed since — the strongest cheap corruption signal + available to a bounded probe, and it works across processes (the repair + may have been attempted by the CLI or a cron worker, not this gateway). + + Import-light on purpose: reads the sidecar JSON directly instead of + importing ``hermes_state``. Any failed attempt on the current + fingerprint is enough to report degraded — waiting for the attempt + budget to exhaust would keep the probe green while repair retries churn. + """ + ledger_path = path.with_name(path.name + ".repair-attempts.json") + try: + data = json.loads(ledger_path.read_text(encoding="utf-8")) + st = path.stat() + except (OSError, ValueError): + return False + if not isinstance(data, dict): + return False + fingerprint = data.get("fingerprint") + try: + attempts = int(data.get("failed_attempts", 0)) + except (TypeError, ValueError): + return False + return attempts >= 1 and fingerprint == f"{st.st_size}:{st.st_mtime_ns}" + + def _probe_state_db(home: Path) -> dict[str, Any]: path = home / "state.db" if not path.exists(): return _check("ok", "not initialized") + if _unrepaired_corruption_marker(path): + # Report the corruption class without paths or messages (this feeds + # public component rollups). "degraded" — not an error state that + # would trip restart loops — but no longer a false green (OOF-106: + # a page-corrupt state.db kept /api/status "ok" for 10+ days while + # sessions silently failed to persist). + return _check("degraded", "unrepaired corruption") try: # A readiness probe must never compete with normal state writers. A # read-only schema query still catches unreadable/corrupt databases @@ -40,6 +81,21 @@ def _probe_state_db(home: Path) -> dict[str, Any]: with closing(sqlite3.connect(uri, uri=True, timeout=1.0)) as conn: conn.execute("PRAGMA query_only = ON") conn.execute("SELECT name FROM sqlite_master LIMIT 1").fetchone() + # The schema read only touches page 1; page-level damage in the + # canonical table b-trees sails past it (the OOF-106 false-green + # gap). Walking one row of ``sessions`` descends its b-tree root + # — still O(1) pages, still read-only, but it catches root-page + # corruption of the table every session write depends on. + # ``SELECT *`` on purpose: a narrower projection (e.g. ``id``) + # can be satisfied from an index b-tree without ever touching + # the table's pages. The row is fetched and discarded — probes + # expose status only, never data. Guarded for pre-schema + # databases where the table doesn't exist yet. + has_sessions = conn.execute( + "SELECT 1 FROM sqlite_master WHERE type='table' AND name='sessions'" + ).fetchone() + if has_sessions: + conn.execute("SELECT * FROM sessions LIMIT 1").fetchone() return _check("ok") except Exception as exc: return _check("degraded", type(exc).__name__) diff --git a/tests/gateway/test_readiness.py b/tests/gateway/test_readiness.py index ef5f7848d402f..da139f4a86c7e 100644 --- a/tests/gateway/test_readiness.py +++ b/tests/gateway/test_readiness.py @@ -59,3 +59,114 @@ def test_collect_runtime_readiness_degrades_on_invalid_config_and_stopped_gatewa assert (home / "config.yaml").read_text(encoding="utf-8") == "model: [unterminated" + + +def test_state_db_probe_degrades_on_unrepaired_corruption_ledger(tmp_path, monkeypatch): + """A repair-attempts ledger matching the current file bytes must flip the + state_db probe to degraded even though the schema page still reads fine + (OOF-106: page-corrupt state.db stayed "ok" for 10+ days).""" + home = tmp_path / ".hermes" + home.mkdir() + db_path = home / "state.db" + with sqlite3.connect(db_path) as conn: + conn.execute("CREATE TABLE probe (id INTEGER PRIMARY KEY)") + st = db_path.stat() + (home / "state.db.repair-attempts.json").write_text( + json.dumps( + { + "fingerprint": f"{st.st_size}:{st.st_mtime_ns}", + "failed_attempts": 1, + "last_attempt": "2026-08-15T00:00:00", + } + ), + encoding="utf-8", + ) + monkeypatch.setenv("HERMES_HOME", str(home)) + + result = collect_runtime_readiness( + configured_model="test/model", + runtime_status={"gateway_state": "running", "platforms": {}}, + active_api_runs=0, + ) + + assert result["checks"]["state_db"]["status"] == "degraded" + assert result["checks"]["state_db"]["detail"] == "unrepaired corruption" + + +def test_state_db_probe_ignores_stale_corruption_ledger(tmp_path, monkeypatch): + """A ledger whose fingerprint no longer matches (file repaired/replaced + since) must NOT degrade the probe.""" + home = tmp_path / ".hermes" + home.mkdir() + db_path = home / "state.db" + with sqlite3.connect(db_path) as conn: + conn.execute("CREATE TABLE probe (id INTEGER PRIMARY KEY)") + (home / "state.db.repair-attempts.json").write_text( + json.dumps({"fingerprint": "1:1", "failed_attempts": 3}), + encoding="utf-8", + ) + monkeypatch.setenv("HERMES_HOME", str(home)) + + result = collect_runtime_readiness( + configured_model="test/model", + runtime_status={"gateway_state": "running", "platforms": {}}, + active_api_runs=0, + ) + + assert result["checks"]["state_db"]["status"] == "ok" + + +def test_state_db_probe_ignores_malformed_corruption_ledger(tmp_path, monkeypatch): + """Garbage in the ledger file must read as "no signal", never crash the + probe or degrade a healthy database.""" + home = tmp_path / ".hermes" + home.mkdir() + db_path = home / "state.db" + with sqlite3.connect(db_path) as conn: + conn.execute("CREATE TABLE probe (id INTEGER PRIMARY KEY)") + (home / "state.db.repair-attempts.json").write_text( + "not json at all", encoding="utf-8" + ) + monkeypatch.setenv("HERMES_HOME", str(home)) + + result = collect_runtime_readiness( + configured_model="test/model", + runtime_status={"gateway_state": "running", "platforms": {}}, + active_api_runs=0, + ) + + assert result["checks"]["state_db"]["status"] == "ok" + + +def test_state_db_probe_catches_sessions_root_page_corruption(tmp_path, monkeypatch): + """Page-level damage inside the sessions table b-tree (schema page intact) + must degrade the probe — the exact OOF-106 false-green failure mode.""" + home = tmp_path / ".hermes" + home.mkdir() + db_path = home / "state.db" + with sqlite3.connect(db_path) as conn: + conn.execute("PRAGMA page_size = 4096") + conn.execute("CREATE TABLE sessions (id TEXT PRIMARY KEY, data TEXT)") + conn.executemany( + "INSERT INTO sessions VALUES (?, ?)", + [(f"s{i}", "x" * 3500) for i in range(40)], + ) + # Find the sessions table's root page and zero it out: sqlite_master + # (page 1) stays valid, so the schema probe alone would still pass. + with sqlite3.connect(db_path) as conn: + rootpage = conn.execute( + "SELECT rootpage FROM sqlite_master WHERE type='table' AND name='sessions'" + ).fetchone()[0] + page_size = conn.execute("PRAGMA page_size").fetchone()[0] + with open(db_path, "r+b") as fh: + fh.seek((rootpage - 1) * page_size) + fh.write(b"\x00" * page_size) + monkeypatch.setenv("HERMES_HOME", str(home)) + + result = collect_runtime_readiness( + configured_model="test/model", + runtime_status={"gateway_state": "running", "platforms": {}}, + active_api_runs=0, + ) + + assert result["checks"]["state_db"]["status"] == "degraded" From 2c2f0b8def5caeeaa1562e1ec5e780a402f04695 Mon Sep 17 00:00:00 2001 From: Shannon Sands Date: Sun, 16 Aug 2026 04:39:07 +1000 Subject: [PATCH 2/2] test(readiness): pin the repair-ledger contract to hermes_state's real writer MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review follow-up (#87052): the probe hand-parses the sidecar ledger that hermes_state writes — an unenforced cross-module contract. New parity test drives hermes_state._record_repair_outcome() directly and asserts the probe degrades on a recorded failure and recovers when the writer clears the ledger, so any schema drift (filename, fingerprint format, failed_attempts key) fails CI instead of silently re-opening the false-green gap. Also: docstring note on size:mtime_ns granularity (coarse-mtime filesystems can hold a stale match until the next write — pessimistic, never falsely green) and a stray blank line in the test file. --- gateway/readiness.py | 6 +++++ tests/gateway/test_readiness.py | 44 +++++++++++++++++++++++++++++++-- 2 files changed, 48 insertions(+), 2 deletions(-) diff --git a/gateway/readiness.py b/gateway/readiness.py index badba7b3a0786..90d6713cc8e77 100644 --- a/gateway/readiness.py +++ b/gateway/readiness.py @@ -41,6 +41,12 @@ def _unrepaired_corruption_marker(path: Path) -> bool: importing ``hermes_state``. Any failed attempt on the current fingerprint is enough to report degraded — waiting for the attempt budget to exhaust would keep the probe green while repair retries churn. + + The ``size:mtime_ns`` fingerprint can false-match on filesystems with + coarse mtime granularity if a repair rewrites the file to the same size + within one timestamp tick — the probe then stays degraded until the next + successful write bumps the mtime. Acceptable: the failure mode is a + briefly pessimistic health signal, never a false green. """ ledger_path = path.with_name(path.name + ".repair-attempts.json") try: diff --git a/tests/gateway/test_readiness.py b/tests/gateway/test_readiness.py index da139f4a86c7e..6c6add4d77c13 100644 --- a/tests/gateway/test_readiness.py +++ b/tests/gateway/test_readiness.py @@ -59,8 +59,6 @@ def test_collect_runtime_readiness_degrades_on_invalid_config_and_stopped_gatewa assert (home / "config.yaml").read_text(encoding="utf-8") == "model: [unterminated" - - def test_state_db_probe_degrades_on_unrepaired_corruption_ledger(tmp_path, monkeypatch): """A repair-attempts ledger matching the current file bytes must flip the state_db probe to degraded even though the schema page still reads fine @@ -170,3 +168,45 @@ def test_state_db_probe_catches_sessions_root_page_corruption(tmp_path, monkeypa ) assert result["checks"]["state_db"]["status"] == "degraded" + + +def test_corruption_ledger_contract_parity_with_hermes_state(tmp_path, monkeypatch): + """Guard the cross-module contract: the probe hand-parses the sidecar + ledger that ``hermes_state`` writes (filename, ``fingerprint`` format, + ``failed_attempts`` key). Drive the REAL writer here so any schema change + in ``hermes_state`` fails this test instead of silently re-opening the + false-green gap this probe exists to close.""" + import hermes_state + + home = tmp_path / ".hermes" + home.mkdir() + db_path = home / "state.db" + with sqlite3.connect(db_path) as conn: + conn.execute("CREATE TABLE probe (id INTEGER PRIMARY KEY)") + monkeypatch.setenv("HERMES_HOME", str(home)) + + def _probe_status() -> str: + result = collect_runtime_readiness( + configured_model="test/model", + runtime_status={"gateway_state": "running", "platforms": {}}, + active_api_runs=0, + ) + return result["checks"]["state_db"]["status"] + + # Failed repair recorded by the real writer -> probe must degrade. + hermes_state._record_repair_outcome(db_path, repaired=False) + ledger_path = hermes_state._repair_ledger_path(db_path) + assert ledger_path.exists(), "writer no longer produces the sidecar ledger" + assert ledger_path == db_path.with_name(db_path.name + ".repair-attempts.json"), ( + "ledger filename contract changed — update gateway/readiness.py" + ) + assert _probe_status() == "degraded", ( + "probe no longer recognises hermes_state's ledger schema — " + "the fingerprint/failed_attempts contract has drifted" + ) + + # Successful repair recorded by the real writer -> ledger cleared, + # probe must return to ok. + hermes_state._record_repair_outcome(db_path, repaired=True) + assert not ledger_path.exists() + assert _probe_status() == "ok"