Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
14 commits
Select commit Hold shift + click to select a range
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 19 additions & 9 deletions agent/turn_explainers.py
Original file line number Diff line number Diff line change
Expand Up @@ -139,10 +139,10 @@
"reported structural corruption (the transcript would "
"have been lost on restart). Freeing disk space will "
"not help. Recovery options:\n"
"1. Run `hermes doctor --fix`\n"
"1. Run `hermes {profile_arg}doctor --fix`\n"
"2. Stop the gateway, then recover with:\n"
" hermes sessions recover --source {db_path} --inspect-only\n"
" (if it reports recoverable) hermes sessions recover "
" hermes {profile_arg}sessions recover --source {db_path} --inspect-only\n"
" (if it reports recoverable) hermes {profile_arg}sessions recover "
"--source {db_path} --output recovered-state.db\n"
" — recovery snapshots the damaged file first; do NOT "
"run `sqlite3 ... \".recover\"` against the live "
Expand All @@ -151,6 +151,16 @@
"3. Restore from a backup in {backups_dir}/\n"
"Then send your message again."
),
# SQLite scoped the corruption to the FTS index and the derived indexes could not be
# detached, so this write did not land; the message store itself is intact (#97794).
"fts_index": (
"the turn was stopped because the session search index (FTS5) "
"is corrupt and could not be detached, so this message was not "
"saved. The message store itself is not damaged: do not run "
"recovery tools or restore a backup. Run `hermes {profile_arg}doctor --fix` "
"(or restart Hermes, which repairs the index on open), then "
"send your message again."
),
"disk": (
"the turn was stopped because session storage could not "
"be written (the transcript would have been lost on "
Expand Down Expand Up @@ -318,14 +328,14 @@ def _format_turn_completion_explanation(
body = _PERSISTENCE_CAUSE_EXPLANATIONS.get(
persistence_cause or "unknown", _PERSISTENCE_DEFAULT_EXPLANATION
)
if persistence_cause == "corrupt":
# Copy-pasteable, so name the store that actually failed: the agent's own
# SessionDB. A multi-profile backend (Desktop serve) hosts sessions whose
# state.db is NOT the process default, so the default would send the operator
# to inspect/repair the wrong profile's database (#105887).
from hermes_constants import get_default_hermes_root
if persistence_cause in ("corrupt", "fts_index"):
# Copy-pasteable, so name the store that actually failed and pin the profile:
# a multi-profile backend (Desktop serve) hosts sessions whose state.db is NOT
# the process default, and a bare `hermes` follows active_profile (#105887).
from hermes_constants import get_default_hermes_root, profile_cli_selector
from hermes_state import _default_db_path

body = body.replace("{profile_arg}", profile_cli_selector())
body = body.replace("{db_path}", str(db_path or _default_db_path()))
body = body.replace(
"{backups_dir}", str(get_default_hermes_root() / "backups")
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
jangomango76
# PR #103321 salvage
2 changes: 2 additions & 0 deletions contributors/emails/devops@77hub.com
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
TaoMasterCoder
# PR #102808 salvage (header-damaged state.db in lost_and_found lane; #106667)
2 changes: 2 additions & 0 deletions contributors/emails/gunwoo@wustl.edu
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
leegunwoo98
# PR #91413 salvage (skip destination-rejected rows in partial recovery; #102240)
2 changes: 2 additions & 0 deletions contributors/emails/zsulthan9@gmail.com
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
SulthanZahran1
# PR #97843 salvage
26 changes: 19 additions & 7 deletions gateway/run_notifications.py
Original file line number Diff line number Diff line change
Expand Up @@ -833,26 +833,38 @@ async def _send_session_db_warning_notifications(self) -> None:
if not error:
logger.info("state.db recovered before the home-channel warning went out; not broadcasting")
return
from hermes_constants import get_default_hermes_root
from hermes_constants import get_default_hermes_root, profile_cli_selector
from hermes_state import _default_db_path, classify_persistence_error, format_session_db_unavailable
if classify_persistence_error(error) == "corrupt":
# Copy-pasteable, so name the real store (profiles / HERMES_HOME do not live under ~/.hermes).
cause = classify_persistence_error(error)
# Copy-pasteable, so name the real store and pin the profile: a bare `hermes` follows
# active_profile, which may be a different database (#105887).
profile_arg = profile_cli_selector()
if cause == "corrupt":
db_path = _default_db_path()
backups_dir = get_default_hermes_root() / "backups"
message = (
"⚠️ Session database corruption detected. Messages may not be "
"persisted. Recovery options:\n"
"1. Run `hermes doctor --fix`\n"
f"1. Run `hermes {profile_arg}doctor --fix`\n"
"2. Stop the gateway, then recover with:\n"
f" hermes sessions recover --source {db_path} "
f" hermes {profile_arg}sessions recover --source {db_path} "
"--inspect-only\n"
" (if it reports recoverable) hermes sessions recover "
f" (if it reports recoverable) hermes {profile_arg}sessions recover "
f"--source {db_path} --output recovered-state.db\n"
" — recovery snapshots the damaged file first; do NOT run "
"`sqlite3 ... \".recover\"` against the live state.db, a "
"vulnerable sqlite3 CLI can corrupt it further\n"
f"3. Restore from a backup in {backups_dir}/\n"
"Run `hermes doctor` for sanitized diagnostics."
f"Run `hermes {profile_arg}doctor` for sanitized diagnostics."
)
elif cause == "fts_index":
# Index-scoped corruption: the message tables are not damaged, so the recover /
# restore advice above would be destructive on a healthy file (#97794).
message = (
"⚠️ Session database reported a corruption error confined to the search index "
"(FTS5); the message tables are not damaged. Messages may not be persisted until "
f"it is repaired: run `hermes {profile_arg}doctor --fix`, then restart the gateway. Do not run "
"recovery tools or restore a backup unless `hermes doctor` confirms damage."
)
else:
message = (
Expand Down
4 changes: 2 additions & 2 deletions hermes_cli/console_engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -705,7 +705,7 @@ def _sessions_optimize(_engine: HermesConsoleEngine, args: list[str]) -> None:


@_captured
def _sessions_repair(_engine: HermesConsoleEngine, args: list[str]) -> None:
def _sessions_repair(_engine: HermesConsoleEngine, args: list[str]) -> int | None:
ns = _parse(
"sessions repair", args, (("--check-only",), dict(action="store_true")),
(("--no-backup",), dict(action="store_true")))
Expand All @@ -721,7 +721,7 @@ def _sessions_repair(_engine: HermesConsoleEngine, args: list[str]) -> None:
return
print(f"{db_path} does not open cleanly: {reason}")
if ns.check_only:
return
return 1 # _capture_output turns a non-zero status into a ConsoleCommandError carrying the printed reason
report = repair_state_db_schema(db_path, backup=not ns.no_backup)
if not report.get("repaired"):
raise ConsoleCommandError(f"Repair failed: {report.get('error')}")
Expand Down
19 changes: 18 additions & 1 deletion hermes_cli/doctor_state.py
Original file line number Diff line number Diff line change
Expand Up @@ -162,6 +162,8 @@ def _session_count(state_db_path: Path):


# Corruption class -> (ok label, not-fixed label, failed issue, fix hint). ``{count}`` = recovered sessions.
# ``structural`` has no in-place repair: an FTS rebuild cannot fix a canonical b-tree, and the
# ``.malformed-backup`` the repair path would leave beside state.db is a copy of the same damage (#88587).
_STATE_DB_REPAIRS = {
"fts": ("Repaired state.db FTS write health",
"state.db FTS write-health repair did not recover automatically",
Expand All @@ -172,10 +174,21 @@ def _session_count(state_db_path: Path):
"state.db schema malformed and auto-repair failed — restore from the backup copy beside state.db",
"state.db schema malformed — run 'hermes doctor --fix' (or 'hermes sessions repair') to recover hidden sessions"),
}
_STATE_DB_STRUCTURAL_ISSUE = (
"state.db structural corruption (canonical tables/indexes damaged, not the FTS index) — an FTS rebuild "
"cannot repair it. Stop the gateway, then run 'hermes {profile_arg}sessions recover --source {db_path} "
"--inspect-only' and, if it reports recoverable, 'hermes {profile_arg}sessions recover --source {db_path} "
"--output recovered-state.db'. Do NOT restore a .malformed-backup copy beside state.db: it is a snapshot "
"of the same corrupt file."
)


def _repair_state_db(f: Finding, should_fix: bool, state_db_path: Path, kind: str) -> None:
"""Shared --fix path for both state.db corruption classes (FTS write health, malformed schema)."""
if kind == "structural":
from hermes_constants import profile_cli_selector
return f.manual_issues.append(_STATE_DB_STRUCTURAL_ISSUE.format(
profile_arg=profile_cli_selector(), db_path=state_db_path))
ok_label, not_fixed_label, failed_issue, fix_hint = _STATE_DB_REPAIRS[kind]
if not should_fix:
return f.issues.append(fix_hint)
Expand All @@ -200,11 +213,15 @@ def _state_db_health(f: Finding, should_fix: bool, state_db_path: Path, _DHH: st
check_ok(f"{_DHH}/state.db exists ({_session_count(state_db_path)} sessions)")
# COUNT(*) succeeds even when the FTS index is corrupt and every write fails through the triggers;
# _db_opens_cleanly drives a rolled-back write to surface that.
from hermes_state_repair import _db_opens_cleanly
from hermes_state_repair import _db_opens_cleanly, state_db_has_structural_damage
# `_db_opens_cleanly` now drives a rolled-back write so this otherwise-silent corruption class is
# surfaced (and repaired in place with --fix). See #50502.
_write_reason = _db_opens_cleanly(state_db_path)
if _write_reason is not None:
if state_db_has_structural_damage(state_db_path):
check_warn(f"{_DHH}/state.db has structural corruption (canonical tables/indexes damaged, "
"not the FTS index)", f"({_write_reason})")
return _repair_state_db(f, should_fix, state_db_path, "structural")
check_warn(f"{_DHH}/state.db fails a write-health probe (FTS index may be corrupt)", f"({_write_reason})")
_repair_state_db(f, should_fix, state_db_path, "fts")
except Exception as e:
Expand Down
42 changes: 34 additions & 8 deletions hermes_cli/session_lost_and_found.py
Original file line number Diff line number Diff line change
Expand Up @@ -180,10 +180,41 @@ def _cli_supports_recover(binary: str) -> bool:
shutil.rmtree(scratch_dir, ignore_errors=True)


SQLITE_HEADER_LENGTH = 100


def run_cli_lost_and_found_recover(
source: Path, lf_path: Path, sqlite3_bin: str, *, timeout: float = 3600.0,
) -> dict[str, Any]:
"""Run ``sqlite3 <source> .recover`` streamed into a fresh scratch DB."""
"""Run ``sqlite3 <source> .recover`` streamed into a fresh scratch DB.

A file whose page-1 header is garbage (SIGKILL mid-write) is refused outright by the shell
(``file is not a database``, rc 26) although the data pages after it survive. ``.recover``
walks pages via sqlite_dbpage and only trips on the magic check, so on that refusal the
100-byte header of the private snapshot is zeroed and the attempts rerun; a zeroed header
makes .recover infer page size and layout from the pages themselves (a spliced donor header
would instead report a database size/freelist that contradicts the file). ``source`` is
the caller's snapshot copy, never the user's file (#106667).
"""
attempts = _cli_recover_attempts(source, lf_path, sqlite3_bin, timeout=timeout)
if attempts[-1]["usable"]:
return {"binary": sqlite3_bin, "attempts": attempts}
if any("not a database" in a["dump_stderr_tail"] for a in attempts):
with source.open("r+b") as handle:
handle.write(bytes(SQLITE_HEADER_LENGTH))
attempts += _cli_recover_attempts(source, lf_path, sqlite3_bin, timeout=timeout)
if attempts[-1]["usable"]:
return {"binary": sqlite3_bin, "attempts": attempts, "header_zeroed": True}
details = "; ".join(
f"[{a['command']}] dump rc={a['dump_returncode']} load rc={a['load_returncode']} "
f"{a['dump_stderr_tail'] or a['load_stderr_tail']}".strip()
for a in attempts
)
raise LostAndFoundError(f"sqlite3 .recover did not produce a usable lost_and_found database: {details}")


def _cli_recover_attempts(source: Path, lf_path: Path, sqlite3_bin: str, *, timeout: float) -> list[dict[str, Any]]:
"""``--ignore-freelist`` first (no resurrected deleted rows), plain ``.recover`` for older shells."""
attempts: list[dict[str, Any]] = []
for command in (".recover --ignore-freelist", ".recover"):
if lf_path.exists():
Expand Down Expand Up @@ -211,13 +242,8 @@ def run_cli_lost_and_found_recover(
"usable": _lost_and_found_db_usable(lf_path),
})
if attempts[-1]["usable"]:
return {"binary": sqlite3_bin, "attempts": attempts}
details = "; ".join(
f"[{a['command']}] dump rc={a['dump_returncode']} load rc={a['load_returncode']} "
f"{a['dump_stderr_tail'] or a['load_stderr_tail']}".strip()
for a in attempts
)
raise LostAndFoundError(f"sqlite3 .recover did not produce a usable lost_and_found database: {details}")
break
return attempts


def _lost_and_found_db_usable(lf_path: Path) -> bool:
Expand Down
36 changes: 33 additions & 3 deletions hermes_cli/session_recovery.py
Original file line number Diff line number Diff line change
Expand Up @@ -414,6 +414,22 @@ def _salvage_rowid_bounds(source: sqlite3.Connection, table: str) -> dict[str, A
result["empty" if not result["errors"] else "unavailable"] = True
return result

# An ordered LIMIT 1 walks the table b-tree and dies on a damaged edge leaf, while the
# aggregate lets the planner answer from any covering index (every Hermes table has at
# least a PRIMARY KEY autoindex). Ask it before falling back to the synthetic domain:
# bisecting from INT64_MIN burned the whole query budget on a 4-row table (#98050).
missing = [edge for edge in ("low", "high") if rows[edge] is None]
if missing:
try:
aggregate = source.execute(f'SELECT min(rowid), max(rowid) FROM "{table}"').fetchone()
except sqlite3.DatabaseError as exc:
result["errors"].append(f"aggregate rowid bounds: {exc}")
else:
for edge, value in zip(("low", "high"), aggregate):
if rows[edge] is None and value is not None:
rows[edge] = int(value)
result.setdefault("aggregate_edges", []).append(edge)

# A damaged edge can stop one ordered probe. Keep the readable edge and bound the other side by the
# SQLite rowid domain, so bisection never assumes user databases hold only positive ids.
if rows["low"] is None:
Expand Down Expand Up @@ -511,8 +527,15 @@ def recover_exact_rowid(self, rowid: int) -> bool:
if not self._keep([value]):
result["excluded_rows"] += 1
return True
with _immediate_transaction(self.destination):
self.destination.execute(self.insert_sql, value)
try:
with _immediate_transaction(self.destination):
self.destination.execute(self.insert_sql, value)
except sqlite3.IntegrityError as exc:
# A phantom row from a damaged page (NULL in a NOT NULL column, FK to nothing) is rejected
# by the destination schema; report it as a skipped singleton instead of aborting the table.
result["destination_rejected_rows"] += 1
self._skip(rowid, rowid, f"destination constraint rejected row: {exc}")
return True
result["copied_rows"] += 1
result["exact_lookup_recovered"] += 1
return True
Expand Down Expand Up @@ -568,7 +591,8 @@ def _copy_table_salvage(
"""Best-effort rowid-range copy that continues past damaged source pages."""
result: dict[str, Any] = {
"mode": "rowid_range_salvage", "source_rows": source_rows, "copied_rows": 0, "excluded_rows": 0,
"columns": [], "range_queries": 0, "exact_lookup_recovered": 0, "skipped_rowid_ranges": [],
"columns": [], "range_queries": 0, "exact_lookup_recovered": 0, "destination_rejected_rows": 0,
"skipped_rowid_ranges": [],
}
columns = _compatible_columns(source, destination, table, result)
if columns is None:
Expand Down Expand Up @@ -1016,6 +1040,12 @@ def _recover_via_lost_and_found(
"BEST-EFFORT page-level salvage: the source table schemas were unreadable, so rows were rebuilt from raw "
"pages and mapped heuristically. Review every count before trusting this output."
)
if cli_report.get("header_zeroed"):
verification["warnings"].append(
"header salvage: SQLite refused the source outright (page-1 header damaged, 'file is not a "
"database'); the header of the private snapshot copy was zeroed so .recover could walk the "
"surviving pages. Rows written only to a -wal after the last checkpoint are not included."
)
verification.update(loss_detected=True, complete=False)
# Structural checks cannot see a positional mis-mapping: every row still inserts, so integrity/FK/FTS
# stay green. A systematic timestamp violation is the semantic tell — never report such a salvage as verified.
Expand Down
2 changes: 1 addition & 1 deletion hermes_cli/sessions_cmd.py
Original file line number Diff line number Diff line change
Expand Up @@ -97,7 +97,7 @@ def _cmd_repair(args):
return
print(f"✗ {db_path} does not open cleanly: {reason}")
if getattr(args, "check_only", False):
return
return 1
print("Repairing (a backup copy is made first)…")
report = repair_state_db_schema(db_path, backup=not getattr(args, "no_backup", False))
if report.get("repaired"):
Expand Down
40 changes: 39 additions & 1 deletion hermes_cli/web_routers/_common.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,9 @@
import asyncio
import contextlib
import logging
from typing import Any, Callable, Optional
import sqlite3
import time
from typing import Any, Callable, Dict, Optional

from fastapi import HTTPException

Expand Down Expand Up @@ -76,3 +78,39 @@ def require(value: Optional[str], detail: str) -> str:
if not stripped:
raise HTTPException(status_code=400, detail=detail)
return stripped


# Corrupt-store reporting for polled read endpoints. The dashboard polls analytics every few
# seconds; a persistently malformed state.db once produced ~520K identical tracebacks in 24 h
# (#96591). One WARNING per store per interval, then debug; the caller gets an explicit status
# instead of a 500. The file is never quarantined or renamed from here — that is `hermes doctor`'s job.
_CORRUPT_STORE_WARN_INTERVAL_S = 300.0
_corrupt_store_warned_at: Dict[str, float] = {} # {db path: monotonic}

CORRUPT_STORE_DETAIL = {
"error": "state_db_corrupt",
"message": "state.db corrupt — run `hermes doctor` (then `hermes doctor --fix` or `hermes sessions repair`).",
}


@contextlib.contextmanager
def corrupt_store_as_status(db_path):
"""Map a corrupt-image ``sqlite3.DatabaseError`` from a state.db read to a 503 status
payload, warning once per store per :data:`_CORRUPT_STORE_WARN_INTERVAL_S`.
Busy/locked and every other error propagate unchanged."""
from hermes_state_errors import is_malformed_db_error

try:
yield
except sqlite3.DatabaseError as exc:
if not is_malformed_db_error(exc):
raise
key, now = str(db_path), time.monotonic()
last = _corrupt_store_warned_at.get(key)
if last is None or now - last >= _CORRUPT_STORE_WARN_INTERVAL_S:
_corrupt_store_warned_at[key] = now
log.warning("state.db at %s is corrupt (%s); dashboard reads return a status payload until it is "
"repaired — run `hermes doctor`", db_path, exc)
else:
log.debug("state.db at %s still corrupt: %s", db_path, exc)
raise HTTPException(status_code=503, detail={**CORRUPT_STORE_DETAIL, "path": key}) from exc
Loading
Loading