From 2d32c82734b4b74024637d0141bf8d2f84249562 Mon Sep 17 00:00:00 2001 From: Scout Date: Tue, 29 Sep 2026 23:55:19 -0400 Subject: [PATCH 01/25] Currency due-file writer and its daily user timer (drafted units) scripts/currency_due.py runs receipt_staleness.py --json, adoption_status.py --pinned-versions --json and saturation_ledger.py --report --json (plus runtime_skill_freshness.py with --network, off by default) as subprocesses with the workflows' arguments, and writes ${XDG_STATE_HOME:-~/.local/state}/native-agent-stack/currency-due.json atomically (mode 0600, os.replace) only while pins_behind, stale_receipts, due_layers or reopen_triggers is nonzero; otherwise it removes the file. due_layers follows the sweep recipe's monthly cadence (--sweep-cadence-days, default 30; 0 gives the raw count). Exit 2 on an internal error leaves the state directory as it was. adoption/templates/systemd/stack-currency.{service,timer}: OnCalendar=daily, Persistent=true, RandomizedDelaySec=15m; oneshot with an explicit PATH so the version probes resolve under the user manager. tests/test_currency_due.py: fake checks in a temporary checkout, a dry run of this checkout's real checks, and the units' settings. Co-Authored-By: Claude Opus 5.5 --- .../templates/systemd/stack-currency.service | 42 ++ .../templates/systemd/stack-currency.timer | 20 + scripts/currency_due.py | 396 ++++++++++++++ tests/test_currency_due.py | 515 ++++++++++++++++++ 4 files changed, 973 insertions(+) create mode 100644 adoption/templates/systemd/stack-currency.service create mode 100644 adoption/templates/systemd/stack-currency.timer create mode 100644 scripts/currency_due.py create mode 100644 tests/test_currency_due.py diff --git a/adoption/templates/systemd/stack-currency.service b/adoption/templates/systemd/stack-currency.service new file mode 100644 index 000000000..306aa19f9 --- /dev/null +++ b/adoption/templates/systemd/stack-currency.service @@ -0,0 +1,42 @@ +# Drafted, not run: no host has loaded, enabled or started this unit (the worktree that authored it made no host +# changes). stack-currency.timer starts it once a day. It has no [Install] section, matching this directory's +# timer-triggered oneshot convention (codex-broker-reaper.service, host-requests-workstation.service). Why a timer +# and a due-file rather than a check at session start: docs/decisions/2026-09-30-session-currency-notice.md. +# +# @REPOSITORY@ is the checkout that holds scripts/currency_due.py. On the workstation that is the live clone, +# %h/code/native-agent-stack-live, as for credential-boot-receipt.service; never the working checkout, whose +# revision moves, and never a build worktree, which a merge deletes. Substitute it before installing, as for the +# other @REPOSITORY@ templates in this directory, and verify both units together (adoption/lifecycle.md): +# sed 's#@REPOSITORY@#%h/code/native-agent-stack-live#g' adoption/templates/systemd/stack-currency.service > ~/.config/systemd/user/stack-currency.service +# cp adoption/templates/systemd/stack-currency.timer ~/.config/systemd/user/ +# systemd-analyze --user verify ~/.config/systemd/user/stack-currency.service ~/.config/systemd/user/stack-currency.timer +# +# The run makes no network call and writes nothing into the checkout. scripts/currency_due.py runs +# receipt_staleness.py, adoption_status.py --pinned-versions and saturation_ledger.py --report, then writes +# $XDG_STATE_HOME/native-agent-stack/currency-due.json (~/.local/state/... when XDG_STATE_HOME is unset; mode 0600) +# when something is due and removes it otherwise. Its --network option (runtime-worker skill pins through gh api) +# stays off here. +# +# PATH: adoption_status.py --pinned-versions finds each version probe with shutil.which, and the user manager's own +# PATH lacks the ecosystem bin directory, so without this line every probe would be reported unchecked and +# pins_behind would read 0 (on the workstation on 2026-09-30 all six exec probes of the default profile resolved +# only there). On another host, use the directories that hold its pinned tools. +# PYTHONDONTWRITEBYTECODE=1 keeps this interpreter and the three it starts from writing __pycache__ into the +# checkout; -B on ExecStart would not reach the child interpreters. +# TimeoutStartSec= covers the script's own per-check bounds (120 s + 120 s + 600 s for the version probes); a +# oneshot has no start timeout by default (systemd.service(5)). UMask=0077 and NoNewPrivileges=true follow this +# directory's other oneshots. +[Unit] +Description=Write the stack currency due-file for the next session (read-only checks, no network) + +[Service] +Type=oneshot +UMask=0077 +NoNewPrivileges=true +Nice=10 +IOSchedulingClass=best-effort +IOSchedulingPriority=7 +TimeoutStartSec=900 +Environment=PATH=%h/.local/share/codex-ecosystem/bin:%h/.local/bin:/usr/local/bin:/usr/bin:/bin +Environment=PYTHONDONTWRITEBYTECODE=1 +ExecStart=/usr/bin/python3 @REPOSITORY@/scripts/currency_due.py diff --git a/adoption/templates/systemd/stack-currency.timer b/adoption/templates/systemd/stack-currency.timer new file mode 100644 index 000000000..ebda67991 --- /dev/null +++ b/adoption/templates/systemd/stack-currency.timer @@ -0,0 +1,20 @@ +# Drafted, not run: see stack-currency.service in this directory for the placeholder, PATH and deployment notes +# that also apply here. +[Unit] +Description=Daily trigger for the stack currency due-file + +[Timer] +# daily is *-*-* 00:00:00 local time (systemd.time(7)). Persistent=true catches up once after a day on which the +# host or the user manager was down: the service "is triggered immediately if it would have been triggered at +# least once during the time when the timer was inactive", and the setting "only has an effect on timers +# configured with OnCalendar=" (systemd.timer(5), systemd 255.4-1ubuntu8.17 on the workstation). On WSL, where the +# distribution often stops, that catch-up is what keeps the due-file at most a day old. RandomizedDelaySec=15m +# spreads the start, the catch-up run included ("Such triggering is nonetheless subject to the delay imposed by +# RandomizedDelaySec="), so it does not start together with every other catch-up at boot. +OnCalendar=daily +Persistent=true +RandomizedDelaySec=15m +Unit=stack-currency.service + +[Install] +WantedBy=timers.target diff --git a/scripts/currency_due.py b/scripts/currency_due.py new file mode 100644 index 000000000..65212b79d --- /dev/null +++ b/scripts/currency_due.py @@ -0,0 +1,396 @@ +#!/usr/bin/env python3 +"""Write a one-line currency notice for the next session when a pin, receipt or layer is due. + +The daily user timer adoption/templates/systemd/stack-currency.timer runs this. It runs this checkout's own +read-only checks as subprocesses, with the arguments their weekly workflows use, and aggregates four counts: + +- pins_behind: components whose platform pin's version probe did not observe the pinned version + (scripts/adoption_status.py --pinned-versions --json, the "mismatched" ids of each selected profile, each id + once); with --network also the runtime-worker skill pins that tools/adoption/runtime_skill_freshness.py reports + as drifted from upstream HEAD (skill-drift, repository-drift, removed-at-head) and a drifted skills CLI pin; +- stale_receipts: the component x platform buckets scripts/receipt_staleness.py --json flags ("flagged"); +- due_layers: the layers scripts/saturation_ledger.py --report --json marks due (not a saturation candidate) whose + last completed sweep is at least --sweep-cadence-days old, or that have none or an undatable one. The default, + 30, follows recipes/saturation-sweep.md ("Sweep only the due layers, at most monthly"); 0 counts every layer + the report marks due; +- reopen_triggers: the layers with a current reopen trigger in that report (the recipe's "or sooner when a reopen + trigger fires"). + +When any count is nonzero it writes ${XDG_STATE_HOME:-~/.local/state}/native-agent-stack/currency-due.json +atomically (a temporary file in the same directory, fsync, mode 0600, os.replace): + + {"generated_at": "YYYY-MM-DDTHH:MM:SSZ", "due": {the four counts}, + "summary_line": "at most 160 characters, ending with the command below", "details": [...]} + +and otherwise removes that file. A SessionStart hook, a separate change, prints summary_line when the file exists +and nothing when it does not (docs/decisions/2026-09-30-session-currency-notice.md). + + python3 scripts/currency_due.py # write or remove the due-file; one line for the journal + python3 scripts/currency_due.py --dry-run # the report as text; writes and removes nothing + python3 scripts/currency_due.py --dry-run --json # the due-file document; writes and removes nothing + python3 scripts/currency_due.py --network # also compare runtime-worker skill pins through gh api + +No network call unless --network is given. It exits 0 whether or not anything is due, and 2 on an internal error: +a check that fails, times out or prints something other than its JSON report, an unreadable saturation ledger or +a failed write. An error leaves the state directory as it was. +""" + +from __future__ import annotations + +import argparse +import contextlib +import json +import os +import re +import subprocess +import sys +import tempfile +from datetime import date, datetime, timezone +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +STATE_NAME = "native-agent-stack" +DUE_FILE = "currency-due.json" +DUE_KEYS = ("pins_behind", "stale_receipts", "due_layers", "reopen_triggers") +LABELS = {"pins_behind": ("pin behind", "pins behind"), + "stale_receipts": ("stale receipt", "stale receipts"), + "due_layers": ("layer due", "layers due"), + "reopen_triggers": ("layer with reopen triggers", "layers with reopen triggers")} +SUMMARY_LIMIT = 160 +DETAILS_COMMAND = "python3 scripts/currency_due.py --dry-run" +# recipes/saturation-sweep.md: "Sweep only the due layers, at most monthly"; 30 days is also +# scripts/receipt_staleness.py's DEFAULT_MAX_AGE_DAYS. +DEFAULT_SWEEP_CADENCE_DAYS = 30 +# tools/adoption/runtime_skill_freshness.py main(): the states its own summary counts as drift. +SKILL_DRIFT_STATES = ("skill-drift", "repository-drift", "removed-at-head") +ISO_UTC = re.compile(r"[0-9]{4}-[0-9]{2}-[0-9]{2}T[0-9]{2}:[0-9]{2}:[0-9]{2}Z") # host_receipts.ISO_UTC_PATTERN +LEDGER = "catalogs/saturation/ledger.json" # scripts/saturation_ledger.py LEDGER +SKILLS_MANIFEST = "blueprints/runtime-workers/skills/manifest.json" # runtime_skill_freshness.py DEFAULT_MANIFEST + +# (script, exit codes that still carry its JSON report, timeout in seconds). adoption_status.py exits 2 whenever +# prerequisites are missing and bounds each version probe at 30 s; runtime_skill_freshness.py exits 1 whenever its +# report is not ok (drift or fetch errors), and its gh api calls time out at 60 s each. +RECEIPTS = ("scripts/receipt_staleness.py", frozenset({0}), 120) +LAYERS = ("scripts/saturation_ledger.py", frozenset({0}), 120) +PINS = ("scripts/adoption_status.py", frozenset({0, 2}), 600) +SKILLS = ("tools/adoption/runtime_skill_freshness.py", frozenset({0, 1}), 900) + + +class CheckError(Exception): + """A check failed or printed something other than its report; nothing is written or removed.""" + + +def default_state_dir(environ=os.environ) -> Path: + """$XDG_STATE_HOME/native-agent-stack, or ~/.local/state/native-agent-stack when XDG_STATE_HOME is unset, empty or + relative (the XDG Base Directory specification treats a relative value as invalid).""" + configured = environ.get("XDG_STATE_HOME") or "" + if os.path.isabs(configured): + return Path(configured) / STATE_NAME + home = environ.get("HOME") or str(Path.home()) + return Path(home) / ".local/state" / STATE_NAME + + +def run_check(root: Path, check: tuple, arguments: list[str]) -> str: + """Run one check of ``root`` with this interpreter; return its stdout when it exits with an accepted code.""" + script, accepted, timeout = check + name = Path(script).name + try: + result = subprocess.run([sys.executable, str(root / script), *arguments], cwd=root, capture_output=True, + text=True, timeout=timeout, stdin=subprocess.DEVNULL, check=False) + except subprocess.TimeoutExpired: + raise CheckError(f"{name} timed out after {timeout} s") from None + except OSError as error: + raise CheckError(f"{name} could not start ({type(error).__name__})") from None + if result.returncode not in accepted: + last = next((line.strip() for line in reversed(result.stderr.splitlines()) if line.strip()), "") + raise CheckError(f"{name} exited {result.returncode}" + (f": {last[:300]}" if last else "")) + return result.stdout + + +def parse_report(name: str, text: str) -> dict: + try: + report = json.loads(text) + except json.JSONDecodeError: + raise CheckError(f"{name} printed malformed JSON") from None + if not isinstance(report, dict): + raise CheckError(f"{name} printed JSON that is not an object") + return report + + +def field(report: dict, key: str, kind: type, name: str): + value = report.get(key) + if not isinstance(value, kind) or (kind is int and (isinstance(value, bool) or value < 0)): + raise CheckError(f"{name} report has no valid {key!r} ({kind.__name__} expected)") + return value + + +def strings(value, label: str, name: str) -> list[str]: + if not isinstance(value, list) or not all(isinstance(item, str) for item in value): + raise CheckError(f"{name} report has no valid {label!r} (list of strings expected)") + return value + + +def sweep_dates(root: Path) -> dict[str, str]: + """{sweep_id: date} for the ledger's completed sweeps: a layer's ``last_sweep`` in the saturation report names + only completed sweeps (saturation_ledger.derive() skips stopped ones).""" + try: + ledger = json.loads((root / LEDGER).read_text(encoding="utf-8")) + except (OSError, UnicodeError, json.JSONDecodeError): + raise CheckError(f"{LEDGER} is unreadable or not JSON") from None + sweeps = ledger.get("sweeps") if isinstance(ledger, dict) else None + if not isinstance(sweeps, list): + raise CheckError(f"{LEDGER} has no sweeps list") + return {sweep["sweep_id"]: sweep["date"] for sweep in sweeps + if isinstance(sweep, dict) and sweep.get("status") == "completed" + and isinstance(sweep.get("sweep_id"), str) and isinstance(sweep.get("date"), str)} + + +def collect(root: Path, now_text: str, network: bool) -> dict: + """Each check's report. The receipt report reaches saturation_ledger.py as a file, the way + .github/workflows/saturation-tracking.yml composes them, in a temporary directory outside the checkout.""" + with tempfile.TemporaryDirectory(prefix="currency-due-") as scratch: + receipts_text = run_check(root, RECEIPTS, ["--root", str(root), "--json", "--now", now_text]) + receipts = parse_report(Path(RECEIPTS[0]).name, receipts_text) + handed = Path(scratch) / "receipt-staleness.json" + handed.write_text(receipts_text, encoding="utf-8") + layers = parse_report(Path(LAYERS[0]).name, run_check( + root, LAYERS, ["--root", str(root), "--report", "--json", "--staleness", str(handed)])) + pins = parse_report(Path(PINS[0]).name, run_check( + root, PINS, ["--manifest", str(root / "adoption/manifest.json"), "--pinned-versions", "--json"])) + skills = None + if network: + output = Path(scratch) / "runtime-skill-freshness.json" + run_check(root, SKILLS, ["--manifest", str(root / SKILLS_MANIFEST), "--output", str(output)]) + try: + skills_text = output.read_text(encoding="utf-8") + except (OSError, UnicodeError): + raise CheckError(f"{Path(SKILLS[0]).name} wrote no report") from None + skills = parse_report(Path(SKILLS[0]).name, skills_text) + return {"receipts": receipts, "layers": layers, "pins": pins, "skills": skills, "sweep_dates": sweep_dates(root)} + + +def summary_line(due: dict) -> str: + """The nonzero counts and the command that prints the details, in at most SUMMARY_LIMIT characters.""" + parts = [f"{due[key]} {LABELS[key][0] if due[key] == 1 else LABELS[key][1]}" for key in DUE_KEYS if due[key]] + if not parts: + return "stack currency: nothing due" + prefix, suffix = "stack currency: ", f"; details: {DETAILS_COMMAND}" + counts, room = ", ".join(parts), SUMMARY_LIMIT - len(prefix) - len(suffix) + if len(counts) > room: + counts = counts[:room - 3] + "..." + return prefix + counts + suffix + + +def aggregate(reports: dict, now: datetime, now_text: str, cadence_days: int) -> dict: + """The due-file document from the checks' reports (their JSON shapes; see the module docstring).""" + details: list[dict] = [] + + name = Path(PINS[0]).name + pins = reports["pins"] + errors = field(pins, "errors", list, name) + if errors: + raise CheckError(f"{name} reported: {'; '.join(str(item) for item in errors)[:300]}") + mismatched: dict[str, dict] = {} + unchecked: set[str] = set() + for profile in field(pins, "profiles", list, name): + if not isinstance(profile, dict): + raise CheckError(f"{name} report has a profile that is not an object") + summary = field(profile, "pinned_versions_summary", dict, name) + versions = {item.get("id"): item.get("pinned_version") for item in profile.get("pinned_versions") or [] + if isinstance(item, dict)} + for component in strings(summary.get("mismatched"), "mismatched", name): + entry = mismatched.setdefault(component, {"kind": "pin_mismatch", "component_id": component, + "pinned_version": versions.get(component), "profiles": []}) + entry["profiles"].append(profile.get("id")) + unchecked.update(strings(summary.get("unchecked"), "unchecked", name)) + details += [mismatched[component] for component in sorted(mismatched)] + pins_behind = len(mismatched) + + skills, skills_errors = reports["skills"], None + if skills is not None: + name = Path(SKILLS[0]).name + for entry in field(skills, "skills", list, name): + if isinstance(entry, dict) and entry.get("state") in SKILL_DRIFT_STATES: + details.append({"kind": "skill_drift", "skill": entry.get("name"), "source": entry.get("source"), + "state": entry.get("state"), "pinned_ref": entry.get("pinned_ref"), + "head_ref": entry.get("head_ref")}) + pins_behind += 1 + cli = skills.get("cli") if isinstance(skills.get("cli"), dict) else {} + if cli.get("drift") is True: + details.append({"kind": "skills_cli_drift", "pinned": cli.get("pinned"), "latest": cli.get("latest")}) + pins_behind += 1 + skills_errors = len(field(skills, "errors", list, name)) + + name = Path(RECEIPTS[0]).name + receipts = reports["receipts"] + stale_receipts = field(receipts, "flagged", int, name) + for row in field(receipts, "rows", list, name): + if not isinstance(row, dict): + raise CheckError(f"{name} report has a row that is not an object") + flags = strings(row.get("flags"), "flags", name) + if flags: + details.append({"kind": "stale_receipt", "platform_id": row.get("platform_id"), + "component_id": row.get("component_id"), "flags": flags, + "pin_moved_hosts": row.get("pin_moved_hosts") or []}) + + name = Path(LAYERS[0]).name + layers = reports["layers"] + due_total = len(field(layers, "due", list, name)) + triggers = field(layers, "current_reopen_triggers", dict, name) + due_layers = 0 + for layer in field(layers, "layers", list, name): + if not isinstance(layer, dict): + raise CheckError(f"{name} report has a layer that is not an object") + if layer.get("due") is not True: + continue + last = layer.get("last_sweep") + swept = reports["sweep_dates"].get(last) if isinstance(last, str) else None + age = None + if swept is not None: + with contextlib.suppress(ValueError): + age = (now.date() - date.fromisoformat(swept)).days + if cadence_days == 0 or age is None or age >= cadence_days: + due_layers += 1 + details.append({"kind": "due_layer", "layer": f"{layer.get('catalog')}/{layer.get('layer_id')}", + "last_sweep": last, "last_sweep_date": swept, "age_days": age}) + for key, items in triggers.items(): + if not isinstance(items, list): + raise CheckError(f"{name} report has current_reopen_triggers[{key!r}] that is not a list") + details.append({"kind": "reopen_trigger", "layer": key, "triggers": items}) + reopen_triggers = len(triggers) + + details.append({"kind": "coverage", "pins_unchecked": len(unchecked), "due_layers_total": due_total, + "sweep_cadence_days": cadence_days, "network": skills is not None, + "skills_fetch_errors": skills_errors}) + due = {"pins_behind": pins_behind, "stale_receipts": stale_receipts, "due_layers": due_layers, + "reopen_triggers": reopen_triggers} + return {"generated_at": now_text, "due": due, "summary_line": summary_line(due), "details": details} + + +def write_due_file(directory: Path, document: dict) -> Path: + """Replace the due-file atomically: os.replace of a fsynced, mode-0600 temporary file in the same directory + (the pattern of saturation_ledger.write_ledger). A failure before the rename leaves the earlier file.""" + directory.mkdir(mode=0o700, parents=True, exist_ok=True) + target = directory / DUE_FILE + handle, temporary = tempfile.mkstemp(dir=directory, prefix=".currency-due-", suffix=".tmp") + try: + with os.fdopen(handle, "w", encoding="utf-8") as stream: + stream.write(json.dumps(document, indent=1) + "\n") + stream.flush() + os.fsync(stream.fileno()) + os.chmod(temporary, 0o600) + os.replace(temporary, target) + except BaseException: + with contextlib.suppress(FileNotFoundError): + os.unlink(temporary) + raise + return target + + +def remove_due_file(directory: Path) -> bool: + try: + (directory / DUE_FILE).unlink() + except FileNotFoundError: + return False + return True + + +def render_text(document: dict) -> str: + lines = [document["summary_line"]] + for item in document["details"]: + kind = item["kind"] + if kind == "pin_mismatch": + lines.append(f" pin: {item['component_id']} did not report its pin {item['pinned_version']} " + f"({', '.join(str(profile) for profile in item['profiles'])})") + elif kind == "skill_drift": + lines.append(f" skill pin: {item['skill']} ({item['source']}): {item['state']}") + elif kind == "skills_cli_drift": + lines.append(f" skills CLI pin: {item['pinned']}, latest release {item['latest']}") + elif kind == "stale_receipt": + lines.append(f" receipt: {item['platform_id']} {item['component_id']}: {', '.join(item['flags'])}") + elif kind == "due_layer": + since = ("never swept" if item["last_sweep"] is None else + f"last swept {item['last_sweep_date']} ({item['age_days']} days ago)" if item["age_days"] is not None + else f"last sweep {item['last_sweep']} has no completed date in the ledger") + lines.append(f" layer due: {item['layer']}: {since}") + elif kind == "reopen_trigger": + names = sorted({str(trigger.get("trigger")) for trigger in item["triggers"] if isinstance(trigger, dict)}) + lines.append(f" reopen trigger: {item['layer']}: {len(item['triggers'])} ({', '.join(names)})") + elif kind == "coverage": + network = "off" if not item["network"] else ( + "on" + (f", {item['skills_fetch_errors']} fetch error(s) left unknown" + if item["skills_fetch_errors"] else "")) + lines.append(f"coverage: {item['pins_unchecked']} pinned component(s) unchecked on this host; " + f"{item['due_layers_total']} layer(s) not yet saturation candidates, due " + f"{item['sweep_cadence_days']} days after their last sweep; network checks {network}") + due = document["due"] + actions = [] + if due["pins_behind"]: + actions.append("pins: python3 scripts/adoption_status.py --pinned-versions") + if due["stale_receipts"]: + actions.append("receipts: python3 scripts/receipt_staleness.py and adoption/update.md") + if due["due_layers"] or due["reopen_triggers"]: + actions.append("layers: recipes/saturation-sweep.md") + if actions: + lines.append("next: " + "; ".join(actions)) + return "\n".join(lines) + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--root", type=Path, default=ROOT, help="the checkout whose checks run (default: this one)") + parser.add_argument("--state-dir", type=Path, + help="directory of currency-due.json, outside the checkout (default: " + "$XDG_STATE_HOME/native-agent-stack, else ~/.local/state/native-agent-stack)") + parser.add_argument("--dry-run", action="store_true", help="print the report; write and remove nothing") + parser.add_argument("--json", action="store_true", help="print the due-file document instead of text") + parser.add_argument("--network", action="store_true", + help="also run tools/adoption/runtime_skill_freshness.py (gh api calls; off by default)") + parser.add_argument("--now", help="evaluate at this UTC time (YYYY-MM-DDTHH:MM:SSZ); default: the clock") + parser.add_argument("--sweep-cadence-days", type=int, default=DEFAULT_SWEEP_CADENCE_DAYS, + help=f"count a due layer once its last sweep is this many days old " + f"(default {DEFAULT_SWEEP_CADENCE_DAYS}; 0 counts every due layer)") + return parser + + +def main(argv: list[str] | None = None) -> int: + parser = build_parser() + args = parser.parse_args(argv) + root = args.root.resolve() + state = (args.state_dir if args.state_dir is not None else default_state_dir()).expanduser().resolve() + if state == root or root in state.parents: + parser.error(f"--state-dir must be outside the checkout ({root}); the due-file never goes into it") + if args.sweep_cadence_days < 0: + parser.error("--sweep-cadence-days must be zero or more") + if args.now is None: + now = datetime.now(timezone.utc).replace(microsecond=0) + else: + try: + if not ISO_UTC.fullmatch(args.now): + raise ValueError + now = datetime.strptime(args.now, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=timezone.utc) + except ValueError: + parser.error(f"--now must look like 2026-09-30T00:00:00Z, got {args.now!r}") + now_text = now.strftime("%Y-%m-%dT%H:%M:%SZ") + try: + document = aggregate(collect(root, now_text, args.network), now, now_text, args.sweep_cadence_days) + action = "dry run" + if not args.dry_run: + if any(document["due"].values()): + action = f"wrote {write_due_file(state, document)}" + else: + action = f"removed {state / DUE_FILE}" if remove_due_file(state) else "no due-file" + except (CheckError, OSError) as error: + print(f"currency_due.py: {error}", file=sys.stderr) + return 2 + if args.json: + print(json.dumps(document, indent=1)) + elif args.dry_run: + print(render_text(document)) + else: + print(f"{document['summary_line']} ({action})") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/test_currency_due.py b/tests/test_currency_due.py new file mode 100644 index 000000000..88abeb1e0 --- /dev/null +++ b/tests/test_currency_due.py @@ -0,0 +1,515 @@ +"""scripts/currency_due.py against fake checks in a temporary checkout (synthetic fixtures, not host evidence). + +Each fake check is a small script that prints canned JSON in the shape the real script prints (receipt_staleness.py +assess(), adoption_status.py inspect_adoption(), saturation_ledger.py build_report() and runtime_skill_freshness.py +build_report()) and records the arguments it received. ThisCheckoutTests runs this checkout's real checks with +--dry-run into a temporary state directory. None of the fixtures is a recorded observation. +""" + +from __future__ import annotations + +import contextlib +import io +import json +import stat +import subprocess +import sys +import tempfile +import time +import unittest +from pathlib import Path +from unittest import mock + +from scripts import currency_due as cd + +ROOT = Path(__file__).resolve().parents[1] +SYSTEMD_DIR = ROOT / "adoption/templates/systemd" +NOW = "2026-09-30T12:00:00Z" +COMMAND = "python3 scripts/currency_due.py --dry-run" + +STALENESS = "scripts/receipt_staleness.py" +PINNED = "scripts/adoption_status.py" +SATURATION = "scripts/saturation_ledger.py" +SKILLS = "tools/adoption/runtime_skill_freshness.py" + +# A fake check. It records its arguments beside itself, copies a --staleness input it was given (so a test can +# see what reached saturation_ledger.py), writes spec["output_file"] to the path after --output (where +# runtime_skill_freshness.py writes its report), prints spec["stdout"] and exits spec["code"]. +FAKE_CHECK = """\ +import json, pathlib, sys +here = pathlib.Path(__file__) +spec = json.loads(here.with_name(here.name + ".spec.json").read_text(encoding="utf-8")) +argv = sys.argv[1:] +here.with_name(here.name + ".argv.json").write_text(json.dumps(argv), encoding="utf-8") +if "--staleness" in argv: + received = pathlib.Path(argv[argv.index("--staleness") + 1]).read_text(encoding="utf-8") + here.with_name(here.name + ".staleness.json").write_text(received, encoding="utf-8") +if "output_file" in spec: + pathlib.Path(argv[argv.index("--output") + 1]).write_text(spec["output_file"], encoding="utf-8") +sys.stdout.write(spec.get("stdout", "")) +sys.exit(spec.get("code", 0)) +""" + + +def staleness_report(*flagged): + """receipt_staleness.py --json (assess(), receipt_staleness.py:156-164); ``flagged`` is (component, flags) pairs. + One unflagged row is always present.""" + rows = [{"platform_id": "linux-wsl2-x86_64", "component_id": component, "current_pins": ["1.0.0"], + "pin_source": "landscape_winner", "alias_of": [], "grandfathered": False, "receipts": 1, + "bound_receipts": 0, "latest_bound": None, "pin_moved_hosts": ["synthetic-host/use"], "unbound": [], + "flags": list(flags), "info": []} for component, flags in flagged] + rows.append({"platform_id": "linux-wsl2-x86_64", "component_id": "widget", "current_pins": ["1.0.0"], + "pin_source": "stack_manifest", "alias_of": [], "grandfathered": False, "receipts": 1, + "bound_receipts": 1, "latest_bound": {"path": "evidence/hosts/synthetic.json", + "observed_at_utc": NOW, "age_days": 0, "result": "pass", + "stage": "use", "component_version": "1.0.0"}, + "pin_moved_hosts": [], "unbound": [], "flags": [], "info": []}) + flags = ("stale", "pin_moved", "no_bound_receipt", "no_current_pin", "stack_alias") + flagged_rows = sum(1 for row in rows if row["flags"]) + return {"generated_at_utc": NOW, "max_age_days": 30, "rows": rows, + "flag_counts": {flag: sum(1 for row in rows if flag in row["flags"]) for flag in flags}, + "info_counts": {"stack_alias_grandfathered": 0}, "flagged": flagged_rows, + "status": "flagged" if flagged_rows else "current"} + + +def pinned_report(mismatched=(), matched=("rtk",), unchecked=("context-mode",), profiles=("foundation-cpu",), + status="prerequisites_present"): + """adoption_status.py --pinned-versions --json (inspect_adoption(), adoption_status.py:1282-1344).""" + def profile(identifier): + results = ([{"id": item, "pinned_version": "1.0.0", "checked": True, "matches_pin": True} for item in matched] + + [{"id": item, "pinned_version": "2.0.0", "checked": True, "matches_pin": False} + for item in mismatched] + + [{"id": item, "pinned_version": None, "checked": False, "matches_pin": None} + for item in unchecked]) + return {"id": identifier, "commands": [], "recipes": [], "status": status, "pinned_versions": results, + "pinned_versions_summary": {"matched": list(matched), "mismatched": list(mismatched), + "unchecked": list(unchecked)}} + return {"schema_version": 1, "status": status, + "platform": {"os": "linux", "architecture": "x86_64", "python": "3.13.0", "supported": True}, + "manifest": {"status": "valid", "schema_version": 1}, "profiles": [profile(item) for item in profiles], + "errors": [], "runtime_acceptance_verified": False, "limitations": [], + "pinned_versions_match": (not mismatched) if (matched or mismatched) else None, + "git": {"baseline_commit": "0" * 40, "current_commit": "1" * 40, "comparison": "baseline_differs"}} + + +def pinned_error_report(): + """adoption_status.py's early return for an unreadable manifest (adoption_status.py:1311-1313).""" + return {"schema_version": 1, "status": "prerequisites_missing", + "platform": {"os": "linux", "architecture": "x86_64", "python": "3.13.0", "supported": False}, + "manifest": {"status": "invalid"}, "profiles": [], + "errors": ["manifest is unavailable or not valid UTF-8 JSON"], "runtime_acceptance_verified": False, + "limitations": []} + + +def saturation_report(layers=(), triggers=None): + """saturation_ledger.py --report --json (build_report(), saturation_ledger.py:966-1002). + ``layers`` is (catalog/layer_id, due, last_sweep) triples; ``triggers`` maps a layer to its current triggers.""" + triggers = triggers or {} + rows = [{"catalog": key.split("/")[0], "layer_id": key.split("/")[1], "research_status": "on_requirement_change", + "clean_count": 0 if due else 3, "saturation_candidate": not due, "due": due, "last_sweep": last, + "last_counted": None, "reset": list(triggers.get(key, [])), + "current_triggers": list(triggers.get(key, []))} for key, due, last in layers] + return {"policy": {"K": 3, "min_gap_days": 7}, "sweeps": 1, "completed_sweeps": 1, + "last_completed": {"sweep_id": "sweep-a", "date": "2026-09-29"}, + "inputs": {"staleness": True, "freshness": "not available", "baseline_manifest": None}, "notes": [], + "due": [key for key, due, _ in layers if due], + "saturation_candidates": [key for key, due, _ in layers if not due], + "current_reopen_triggers": {key: value for key, value in triggers.items() if value}, "layers": rows} + + +def skills_report(states=(), cli_drift=False, errors=()): + """runtime_skill_freshness.py --output (build_report(), runtime_skill_freshness.py:106-116).""" + skills = [{"name": f"skill-{index}", "source": "example/skills", "path": f"skills/skill-{index}", + "status": "selected", "pinned_ref": "a" * 40, "head_ref": "b" * 40, "pinned_tree": "c" * 40, + "head_tree": "d" * 40, "manifest_tree_matches_pin": True, "state": state, "native_check_ref": "a" * 40, + "native_check_advances_commit_pin": False} for index, state in enumerate(states)] + return {"schema_version": 1, "kind": "runtime_skill_freshness_report", "checked_at": "2026-09-30T12:00:00+00:00", + "report_only": True, + "cli": {"pinned": "1.5.0", "latest": "v1.6.0" if cli_drift else "v1.5.0", "drift": cli_drift}, + "native_check": {"executed": False, "source": None, "pinned_version": "1.5.0", "pinned_ref": "a" * 40, + "reason": "synthetic"}, + "skills": skills, "errors": list(errors), "ok": not errors} + + +def ledger(*sweeps): + """catalogs/saturation/ledger.json; only sweeps[].{sweep_id, date, status} matter here.""" + return {"schema_version": 1, "policy": {"K": 3, "min_gap_days": 7}, + "sweeps": [{"sweep_id": sweep_id, "date": day, "status": status} for sweep_id, day, status in sweeps]} + + +class Checkout: + """A temporary checkout whose checks are fakes, with a state directory beside it (outside the checkout). + By default nothing is due.""" + + def __init__(self, test: unittest.TestCase): + temporary = tempfile.TemporaryDirectory() + test.addCleanup(temporary.cleanup) + base = Path(temporary.name).resolve() + self.root = base / "checkout" + self.state = base / "state" + self.set(STALENESS, staleness_report()) + self.set(PINNED, pinned_report()) + self.set(SATURATION, saturation_report([("foundation/workers", False, "sweep-a")])) + self.write_ledger(ledger(("sweep-a", "2026-09-29", "completed"))) + + def set(self, relative: str, report=None, *, stdout: str | None = None, code: int = 0, + output_file: str | None = None) -> None: + path = self.root / relative + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(FAKE_CHECK, encoding="utf-8") + spec = {"stdout": json.dumps(report) if stdout is None else stdout, "code": code} + if output_file is not None: + spec["output_file"] = output_file + path.with_name(path.name + ".spec.json").write_text(json.dumps(spec), encoding="utf-8") + + def something_due(self) -> None: + self.set(STALENESS, staleness_report(("codex", ["pin_moved"]), ("qmd", ["stale", "pin_moved"]))) + self.set(PINNED, pinned_report(mismatched=("codex",))) + self.set(SATURATION, saturation_report( + [("foundation/workers", True, "sweep-a")], + {"foundation/workers": [{"trigger": "pin_moved", "ref": "receipt_staleness:linux-wsl2-x86_64/codex"}]})) + + def write_ledger(self, data) -> None: + path = self.root / "catalogs/saturation/ledger.json" + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(data) if not isinstance(data, str) else data, encoding="utf-8") + + def recorded(self, relative: str, suffix: str = ".argv.json"): + path = self.root / relative + marker = path.with_name(path.name + suffix) + return json.loads(marker.read_text(encoding="utf-8")) if marker.exists() else None + + def run(self, *extra: str) -> tuple[int, str, str]: + stdout, stderr = io.StringIO(), io.StringIO() + with contextlib.redirect_stdout(stdout), contextlib.redirect_stderr(stderr): + code = cd.main(["--root", str(self.root), "--state-dir", str(self.state), "--now", NOW, *extra]) + return code, stdout.getvalue(), stderr.getvalue() + + @property + def due_file(self) -> Path: + return self.state / "currency-due.json" + + +class DueFileTests(unittest.TestCase): + def test_nothing_due_writes_no_file(self): + checkout = Checkout(self) + code, _, stderr = checkout.run() + self.assertEqual(code, 0, stderr) + self.assertFalse(checkout.due_file.exists()) + + def test_nothing_due_removes_an_earlier_file(self): + checkout = Checkout(self) + checkout.state.mkdir(mode=0o700) + checkout.due_file.write_text('{"earlier": true}\n', encoding="utf-8") + code, _, stderr = checkout.run() + self.assertEqual(code, 0, stderr) + self.assertFalse(checkout.due_file.exists()) + + def test_something_due_writes_the_document_privately_with_a_short_summary(self): + checkout = Checkout(self) + checkout.something_due() + code, _, stderr = checkout.run() + self.assertEqual(code, 0, stderr) + document = json.loads(checkout.due_file.read_text(encoding="utf-8")) + self.assertEqual(list(document), ["generated_at", "due", "summary_line", "details"]) + self.assertEqual(document["generated_at"], NOW) + self.assertEqual(document["due"], {"pins_behind": 1, "stale_receipts": 2, "due_layers": 0, + "reopen_triggers": 1}) + self.assertEqual(document["summary_line"], + f"stack currency: 1 pin behind, 2 stale receipts, 1 layer with reopen triggers; " + f"details: {COMMAND}") + self.assertLessEqual(len(document["summary_line"]), 160) + self.assertEqual(stat.S_IMODE(checkout.due_file.stat().st_mode), 0o600) + self.assertEqual(stat.S_IMODE(checkout.state.stat().st_mode), 0o700) + # The temporary file was renamed into place, not left beside it. + self.assertEqual([path.name for path in checkout.state.iterdir()], ["currency-due.json"]) + + def test_a_write_that_fails_before_the_rename_keeps_the_earlier_file(self): + checkout = Checkout(self) + checkout.something_due() + checkout.state.mkdir(mode=0o700) + checkout.due_file.write_text("earlier\n", encoding="utf-8") + with mock.patch.object(cd.os, "replace", side_effect=OSError("simulated rename failure")): + code, _, stderr = checkout.run() + self.assertEqual(code, 2) + self.assertIn("simulated rename failure", stderr) + self.assertEqual(checkout.due_file.read_text(encoding="utf-8"), "earlier\n") + self.assertEqual([path.name for path in checkout.state.iterdir()], ["currency-due.json"]) + + def test_the_state_directory_must_be_outside_the_checkout(self): + checkout = Checkout(self) + with contextlib.redirect_stderr(io.StringIO()), self.assertRaises(SystemExit) as raised: + cd.main(["--root", str(checkout.root), "--state-dir", str(checkout.root / "state"), "--now", NOW]) + self.assertEqual(raised.exception.code, 2) + self.assertFalse((checkout.root / "state").exists()) + + +class FailureTests(unittest.TestCase): + def test_malformed_check_output_exits_2_and_writes_nothing(self): + outputs = {"not JSON": "{not json", "a JSON array": "[]", "an object without the report's keys": "{}"} + for relative in (STALENESS, PINNED, SATURATION): + for label, text in outputs.items(): + with self.subTest(check=relative, output=label): + checkout = Checkout(self) + checkout.something_due() + checkout.set(relative, stdout=text) + code, _, stderr = checkout.run() + self.assertEqual(code, 2, stderr) + self.assertIn(Path(relative).name, stderr) + self.assertFalse(checkout.state.exists()) + + def test_a_malformed_ledger_exits_2(self): + checkout = Checkout(self) + checkout.something_due() + checkout.write_ledger("{not json") + code, _, _ = checkout.run() + self.assertEqual(code, 2) + self.assertFalse(checkout.state.exists()) + + def test_a_failed_run_leaves_an_earlier_file_byte_identical(self): + checkout = Checkout(self) + checkout.state.mkdir(mode=0o700) + checkout.due_file.write_bytes(b'{"earlier": true}\n') + checkout.set(SATURATION, stdout="{not json") + code, _, _ = checkout.run() + self.assertEqual(code, 2) + self.assertEqual(checkout.due_file.read_bytes(), b'{"earlier": true}\n') + + def test_exit_codes_follow_each_check_s_own_contract(self): + # receipt_staleness.py and saturation_ledger.py exit 2 only when their inputs are unreadable + # (receipt_staleness.py:236-238, saturation_ledger.py:1342-1344). adoption_status.py exits 2 whenever + # prerequisites are missing (adoption_status.py:1400), which is not a failure of this report; its errors + # list (adoption_status.py:1311-1316) is. + missing = pinned_report(mismatched=("codex",), status="prerequisites_missing") + cases = [ + ("receipt_staleness exit 2", STALENESS, {"stdout": json.dumps({"status": "error", "error": "x"}), + "code": 2}, 2), + ("saturation_ledger exit 2", SATURATION, {"stdout": "", "code": 2}, 2), + ("adoption_status exit 2, prerequisites missing", PINNED, {"stdout": json.dumps(missing), "code": 2}, 0), + ("adoption_status errors", PINNED, {"stdout": json.dumps(pinned_error_report()), "code": 2}, 2), + ("adoption_status exit 1", PINNED, {"stdout": json.dumps(pinned_report()), "code": 1}, 2), + ] + for label, relative, spec, expected in cases: + with self.subTest(case=label): + checkout = Checkout(self) + checkout.set(relative, **spec) + code, _, stderr = checkout.run("--dry-run") + self.assertEqual(code, expected, stderr) + + +class DryRunTests(unittest.TestCase): + def test_dry_run_prints_the_document_and_writes_nothing(self): + checkout = Checkout(self) + checkout.something_due() + code, stdout, stderr = checkout.run("--dry-run", "--json") + self.assertEqual(code, 0, stderr) + self.assertEqual(json.loads(stdout)["due"]["pins_behind"], 1) + self.assertFalse(checkout.state.exists()) + + def test_dry_run_keeps_an_earlier_file_when_nothing_is_due(self): + checkout = Checkout(self) + checkout.state.mkdir(mode=0o700) + checkout.due_file.write_text("earlier\n", encoding="utf-8") + code, stdout, _ = checkout.run("--dry-run") + self.assertEqual(code, 0) + self.assertEqual(stdout.splitlines()[0], "stack currency: nothing due") + self.assertEqual(checkout.due_file.read_text(encoding="utf-8"), "earlier\n") + + def test_text_output_starts_with_the_summary_line(self): + checkout = Checkout(self) + checkout.something_due() + _, text, _ = checkout.run("--dry-run") + _, raw, _ = checkout.run("--dry-run", "--json") + self.assertEqual(text.splitlines()[0], json.loads(raw)["summary_line"]) + + +class CountTests(unittest.TestCase): + def document(self, checkout: Checkout, *extra: str) -> dict: + code, stdout, stderr = checkout.run("--dry-run", "--json", *extra) + self.assertEqual(code, 0, stderr) + return json.loads(stdout) + + def test_layers_are_due_once_the_monthly_sweep_cadence_has_passed(self): + # recipes/saturation-sweep.md: "Sweep only the due layers, at most monthly ..., or sooner when a + # reopen trigger fires"; a layer that is not a saturation candidate counts once its last completed + # sweep is at least 30 days old, or when it has none or it cannot be dated. + checkout = Checkout(self) + checkout.set(SATURATION, saturation_report([ + ("foundation/thirty-days", True, "sweep-thirty"), ("foundation/twenty-nine-days", True, "sweep-29"), + ("foundation/never-swept", True, None), ("foundation/unknown-sweep", True, "sweep-missing"), + ("foundation/saturated", False, "sweep-thirty")])) + checkout.write_ledger(ledger(("sweep-thirty", "2026-08-31", "completed"), + ("sweep-29", "2026-09-01", "completed"), + ("sweep-missing", "2026-08-01", "stopped"))) + document = self.document(checkout) + self.assertEqual(document["due"]["due_layers"], 3) + due = [item for item in document["details"] if item["kind"] == "due_layer"] + self.assertEqual([item["layer"] for item in due], + ["foundation/thirty-days", "foundation/never-swept", "foundation/unknown-sweep"]) + self.assertEqual(due[0]["age_days"], 30) + coverage = document["details"][-1] + self.assertEqual(coverage["kind"], "coverage") + self.assertEqual((coverage["due_layers_total"], coverage["sweep_cadence_days"]), (4, 30)) + # --sweep-cadence-days 0 is the plain reading: every layer the report marks due. + self.assertEqual(self.document(checkout, "--sweep-cadence-days", "0")["due"]["due_layers"], 4) + + def test_reopen_triggers_count_layers(self): + checkout = Checkout(self) + trigger = {"trigger": "pin_moved", "ref": "receipt_staleness:linux-wsl2-x86_64/codex"} + checkout.set(SATURATION, saturation_report( + [("foundation/a", True, "sweep-a"), ("us-equities/b", True, "sweep-a")], + {"foundation/a": [trigger, dict(trigger, ref="receipt_staleness:linux-wsl2-x86_64/qmd")], + "us-equities/b": [trigger]})) + document = self.document(checkout) + self.assertEqual(document["due"]["reopen_triggers"], 2) + reopened = {item["layer"]: item["triggers"] for item in document["details"] if item["kind"] == "reopen_trigger"} + self.assertEqual(len(reopened["foundation/a"]), 2) + + def test_stale_receipts_are_the_report_s_flagged_rows(self): + checkout = Checkout(self) + checkout.set(STALENESS, staleness_report(("codex", ["pin_moved", "no_bound_receipt"]), ("qmd", ["stale"]))) + document = self.document(checkout) + self.assertEqual(document["due"]["stale_receipts"], 2) + flagged = [(item["component_id"], item["flags"]) for item in document["details"] + if item["kind"] == "stale_receipt"] + self.assertEqual(flagged, [("codex", ["pin_moved", "no_bound_receipt"]), ("qmd", ["stale"])]) + + def test_pins_behind_counts_each_mismatched_component_once(self): + checkout = Checkout(self) + checkout.set(PINNED, pinned_report(mismatched=("codex", "rtk"), matched=(), + profiles=("foundation-cpu", "token-efficiency"))) + document = self.document(checkout) + self.assertEqual(document["due"]["pins_behind"], 2) + codex = next(item for item in document["details"] + if item["kind"] == "pin_mismatch" and item["component_id"] == "codex") + self.assertEqual(codex["profiles"], ["foundation-cpu", "token-efficiency"]) + self.assertEqual(codex["pinned_version"], "2.0.0") + self.assertEqual(document["details"][-1]["pins_unchecked"], 1) + + def test_every_check_gets_its_documented_arguments_and_the_same_clock(self): + checkout = Checkout(self) + checkout.something_due() + self.assertEqual(checkout.run("--dry-run")[0], 0) + root = str(checkout.root) + self.assertEqual(checkout.recorded(STALENESS), ["--root", root, "--json", "--now", NOW]) + self.assertEqual(checkout.recorded(PINNED), + ["--manifest", str(checkout.root / "adoption/manifest.json"), "--pinned-versions", "--json"]) + saturation = checkout.recorded(SATURATION) + self.assertEqual(saturation[:5], ["--root", root, "--report", "--json", "--staleness"]) + # saturation_ledger.py read receipt_staleness.py's report, as .github/workflows/saturation-tracking.yml + # composes them, from a temporary file outside the checkout and the state directory that is gone now. + self.assertEqual(checkout.recorded(SATURATION, ".staleness.json")["flagged"], 2) + handed = Path(saturation[5]) + self.assertFalse(handed.exists()) + self.assertNotIn(checkout.root, handed.parents) + self.assertNotIn(checkout.state, handed.parents) + + +class NetworkTests(unittest.TestCase): + def test_network_checks_are_off_by_default(self): + checkout = Checkout(self) + checkout.set(SKILLS, stdout="{}", output_file=json.dumps(skills_report(("skill-drift",), cli_drift=True))) + code, stdout, _ = checkout.run("--dry-run", "--json") + self.assertEqual(code, 0) + self.assertIsNone(checkout.recorded(SKILLS)) + self.assertEqual(json.loads(stdout)["due"]["pins_behind"], 0) + + def test_network_adds_drifted_skill_pins_and_keeps_fetch_errors_unknown(self): + # runtime_skill_freshness.py:152 counts skill-drift, repository-drift and removed-at-head as drift, and + # :103 sets cli.drift; it exits 1 whenever its report is not ok (:154), which is not a failure here. + checkout = Checkout(self) + states = ("current", "skill-drift", "repository-drift", "removed-at-head", "unfetched", "invalid-pin") + checkout.set(SKILLS, stdout='{"ok": false}', code=1, output_file=json.dumps( + skills_report(states, cli_drift=True, errors=["gh api failed (exit 1): repos/example/skills"]))) + code, stdout, stderr = checkout.run("--dry-run", "--json", "--network") + self.assertEqual(code, 0, stderr) + document = json.loads(stdout) + self.assertEqual(document["due"]["pins_behind"], 4) + self.assertEqual(sorted(item["state"] for item in document["details"] if item["kind"] == "skill_drift"), + ["removed-at-head", "repository-drift", "skill-drift"]) + self.assertEqual(document["details"][-1]["skills_fetch_errors"], 1) + arguments = checkout.recorded(SKILLS) + self.assertEqual(arguments[:2], + ["--manifest", str(checkout.root / "blueprints/runtime-workers/skills/manifest.json")]) + self.assertEqual(arguments[2], "--output") + + def test_a_skills_report_that_was_not_written_exits_2(self): + checkout = Checkout(self) + checkout.set(SKILLS, stdout="{}", code=1) + code, _, _ = checkout.run("--dry-run", "--network") + self.assertEqual(code, 2) + + +class SummaryLineTests(unittest.TestCase): + def test_the_line_names_only_nonzero_counts_and_the_command(self): + line = cd.summary_line({"pins_behind": 2, "stale_receipts": 1, "due_layers": 3, "reopen_triggers": 0}) + self.assertEqual(line, f"stack currency: 2 pins behind, 1 stale receipt, 3 layers due; details: {COMMAND}") + self.assertEqual(cd.summary_line(dict.fromkeys(cd.DUE_KEYS, 0)), "stack currency: nothing due") + + def test_the_line_stays_within_160_characters_and_keeps_the_command(self): + line = cd.summary_line(dict.fromkeys(cd.DUE_KEYS, 10 ** 40)) + self.assertLessEqual(len(line), 160) + self.assertTrue(line.endswith(f"; details: {COMMAND}")) + + +class StateDirectoryTests(unittest.TestCase): + def test_xdg_state_home_is_used_only_when_absolute(self): + home = Path(tempfile.gettempdir()) / "synthetic-home" + state = Path(tempfile.gettempdir()) / "synthetic-state" + default = home / ".local/state/native-agent-stack" + cases = [({"XDG_STATE_HOME": str(state), "HOME": str(home)}, state / "native-agent-stack"), + ({"XDG_STATE_HOME": "", "HOME": str(home)}, default), + ({"XDG_STATE_HOME": "relative/state", "HOME": str(home)}, default), + ({"HOME": str(home)}, default)] + for environ, expected in cases: + with self.subTest(environ=environ): + self.assertEqual(cd.default_state_dir(environ), expected) + + +class ThisCheckoutTests(unittest.TestCase): + def test_the_real_checks_run_dry_and_write_nothing(self): + # The five-second budget is measured on the workstation by the acceptance command; this bound only + # catches a pathological slowdown on slower runners. + with tempfile.TemporaryDirectory() as temporary: + state = Path(temporary) / "state" + started = time.monotonic() + result = subprocess.run([sys.executable, str(ROOT / "scripts/currency_due.py"), "--dry-run", "--json", + "--state-dir", str(state)], capture_output=True, text=True, timeout=600, + cwd=ROOT, stdin=subprocess.DEVNULL, check=False) + elapsed = time.monotonic() - started + self.assertEqual(result.returncode, 0, result.stderr[-2000:]) + document = json.loads(result.stdout) + self.assertEqual(list(document), ["generated_at", "due", "summary_line", "details"]) + self.assertEqual(list(document["due"]), list(cd.DUE_KEYS)) + self.assertTrue(all(isinstance(value, int) and value >= 0 for value in document["due"].values())) + self.assertLessEqual(len(document["summary_line"]), 160) + self.assertFalse(state.exists()) + self.assertLess(elapsed, 120) + + +class UnitTemplateTests(unittest.TestCase): + def setUp(self): + self.service = (SYSTEMD_DIR / "stack-currency.service").read_text(encoding="utf-8").splitlines() + self.timer = (SYSTEMD_DIR / "stack-currency.timer").read_text(encoding="utf-8").splitlines() + + def test_the_service_is_a_guarded_oneshot_that_only_the_timer_starts(self): + for setting in ("Type=oneshot", "UMask=0077", "NoNewPrivileges=true", + "Environment=PYTHONDONTWRITEBYTECODE=1", + "ExecStart=/usr/bin/python3 @REPOSITORY@/scripts/currency_due.py"): + self.assertIn(setting, self.service) + self.assertNotIn("[Install]", self.service) + directives = [line for line in self.service if line and not line.startswith("#")] + self.assertFalse(any("--network" in line for line in directives)) + # The version probes of adoption_status.py --pinned-versions resolve through PATH (shutil.which), and the + # user manager's own PATH lacks the ecosystem bin directory. + search_path = next(line for line in directives if line.startswith("Environment=PATH=")) + self.assertIn("%h/.local/share/codex-ecosystem/bin", search_path) + + def test_the_timer_runs_daily_catches_up_and_spreads_its_start(self): + for setting in ("OnCalendar=daily", "Persistent=true", "RandomizedDelaySec=15m", + "Unit=stack-currency.service", "WantedBy=timers.target"): + self.assertIn(setting, self.timer) + + +if __name__ == "__main__": + unittest.main() From 5cddc995b2501118b6348e0049ed7d82b9b3e746 Mon Sep 17 00:00:00 2001 From: Scout Date: Tue, 29 Sep 2026 23:55:26 -0400 Subject: [PATCH 02/25] Session currency notice: startup rule, install paragraph and decision record docs/token-practice.md item 4: ordinary startup still runs no audits, model trials or network calls; one read-only SessionStart line of at most 160 characters from the timer's due-file is allowed, fail-open. adoption/lifecycle.md: render, verify and enable the stack-currency units (added after v2026.09.26.2). docs/decisions/2026-09-30-session-currency-notice.md: context, alternatives (UserPromptSubmit hook, SessionStart running the checks, weekly CI only, raw saturation count, dated-manifest pins, monotonic timer), the decision with the hook contract and gate for unit F2, overturn conditions and sources. The hook script and any AGENTS.md:28 amendment ship in the frozen units F2 and F1. Co-Authored-By: Claude Opus 5.5 --- adoption/lifecycle.md | 20 ++ .../2026-09-30-session-currency-notice.md | 181 ++++++++++++++++++ docs/token-practice.md | 9 +- 3 files changed, 208 insertions(+), 2 deletions(-) create mode 100644 docs/decisions/2026-09-30-session-currency-notice.md diff --git a/adoption/lifecycle.md b/adoption/lifecycle.md index 13488f00e..0884ab64c 100644 --- a/adoption/lifecycle.md +++ b/adoption/lifecycle.md @@ -225,6 +225,26 @@ retain that expected outcome explicitly. For final retirement use that unit. Keep its data directories. The shared MCPorter daemon is not an owned disposable service; do not stop it to clean up another component. +Added after `v2026.09.26.2`: a daily user timer keeps the next session's +currency notice current without a check at startup. +[`stack-currency.service`](templates/systemd/stack-currency.service) runs +`python3 scripts/currency_due.py`, which runs the receipt, pin and saturation +checks and writes `${XDG_STATE_HOME:-~/.local/state}/native-agent-stack/currency-due.json` +(mode 0600) only while something is due; a SessionStart hook prints its one +line ([decision](../docs/decisions/2026-09-30-session-currency-notice.md)). +Render the service from the live clone, verify both units, enable +[the timer](templates/systemd/stack-currency.timer) and read the current report; +the service file's header covers `PATH` on another host: + +```sh +sed 's#@REPOSITORY@#%h/code/native-agent-stack-live#g' adoption/templates/systemd/stack-currency.service > ~/.config/systemd/user/stack-currency.service +cp adoption/templates/systemd/stack-currency.timer ~/.config/systemd/user/ +systemd-analyze --user verify ~/.config/systemd/user/stack-currency.service ~/.config/systemd/user/stack-currency.timer +systemctl --user daemon-reload +systemctl --user enable --now stack-currency.timer +python3 scripts/currency_due.py --dry-run +``` + Foreground tools should close through their native exit path. For a child started by the qualification shell, record its PID, verify its identity, send SIGTERM if needed and `wait` for that same child. Record its exit status and diff --git a/docs/decisions/2026-09-30-session-currency-notice.md b/docs/decisions/2026-09-30-session-currency-notice.md new file mode 100644 index 000000000..b0f5f2182 --- /dev/null +++ b/docs/decisions/2026-09-30-session-currency-notice.md @@ -0,0 +1,181 @@ +# Decision: a daily timer writes the currency due-file, and one SessionStart line reads it (2026-09-30) + +**Decided by:** coordinator session `native-agent-stack-c5`, unit A2 of the 2026-09-30 wave (branch +`claude/sota-defaults-a2-20260930`, base `origin/main@e45328d3`). + +**Scope:** [`scripts/currency_due.py`](../../scripts/currency_due.py) and its tests +([`tests/test_currency_due.py`](../../tests/test_currency_due.py)), the drafted user units +[`adoption/templates/systemd/stack-currency.service`](../../adoption/templates/systemd/stack-currency.service) and +[`.timer`](../../adoption/templates/systemd/stack-currency.timer), their install paragraph in +[`adoption/lifecycle.md`](../../adoption/lifecycle.md#native-client-integration-and-process-lifecycle), and the +startup rule, item 4 of [`docs/token-practice.md`](../token-practice.md). + +**Not in this change.** The SessionStart hook script ships in unit F2, and any amendment of `AGENTS.md:28` in unit F1; +both are frozen Gate A units. This record gives F2 the file contract and the acceptance gate below. No unit was +installed and no host file was written: the timer and the service are drafted templates. + +## Context + +- The currency checks already exist and are read-only. `scripts/adoption_status.py --pinned-versions` compares the + installed tools with the platform pins, `scripts/receipt_staleness.py` finds host receipts that are old or at a + retired pin, and `scripts/saturation_ledger.py --report` lists the landscape layers due for a sweep and their + reopen triggers. Weekly workflows run the last two + ([`receipt-staleness.yml`](../../.github/workflows/receipt-staleness.yml), + [`saturation-tracking.yml`](../../.github/workflows/saturation-tracking.yml)) and publish an artifact and one + issue. The pinned-version probes need the host's own installed tools, so no workflow can run them for a + workstation. A session learns none of this unless someone runs the commands, which is instruction-only triage. +- A session may not run them at start. `AGENTS.md:28` says "Do not rerun the full audit or model trials at + startup", and item 4 of `docs/token-practice.md` said "Do not add hooks or schedulers, override providers, or + rerun model trials during ordinary startup". +- The coordinator's brief for this unit reports that on 2026-09-29 ten components behind upstream were found by + hand. The repository carries such hand-written notes + ([`docs/grand-catalog-handbook.md`](../grand-catalog-handbook.md) lines 1267, 1510, 1771 and 1892). The latest + dated SOTA-convergence manifest, `catalogs/sota-convergence/manifest-20260929.json`, marks 51 rows (33 distinct + component ids) `pin_behind_upstream: true`. That count was made for this record; the manifest has no summary + figure for it. +- The state on the workstation at base `e45328d3` on 2026-09-30, from a dry run of the new script (a local integration + check, not an upstream test): one component whose version probe did not report its pin (`codex`, pin 0.157.1), + 7 flagged receipt buckets (`pin_moved` 7, `no_bound_receipt` 4) and 12 layers with current reopen triggers (38 + `pin_moved` triggers). All 32 layers are not yet saturation candidates. Their last completed sweeps were on + 2026-09-26 (12 layers) and 2026-09-29 (20 layers). + +## Alternatives + +- **A UserPromptSubmit hook that prints the line.** Rejected. Claude Code adds that event's plain-text stdout as + context on every prompt, so the line would cost tokens on every turn instead of once per session. +- **A SessionStart hook that runs the checks.** Rejected. A dry run of the three checks took 1.56 s on the + workstation, about 30 times the 50 ms hook budget below, and they execute each pinned tool's version probe. Running them is the startup + audit that `AGENTS.md:28` rules out. SessionStart also fires on `resume`, `clear`, `compact` and `fork`, so the + cost would repeat within one session. +- **Weekly CI only.** Kept as the reviewed record, not replaced. The workflows cover what a host cannot, such as + catalog freshness from GitHub. They cannot reach a session, and they cannot see a host's installed versions. +- **The raw saturation count as `due_layers`.** Rejected. The report calls every layer that is not a saturation + candidate due, which is 32 of 32 today. The file would never be removed and the line would print in every + session. The sweep recipe's own cadence is "Sweep only the due layers, at most monthly ..., or sooner when a + reopen trigger fires", so `due_layers` follows the monthly cadence and `reopen_triggers` covers the "sooner". + `--sweep-cadence-days 0` restores the raw count. +- **Counting the `pin_behind_upstream` rows of the latest dated manifest in `pins_behind`.** Not adopted. The + manifest is a snapshot of its build date, so a pin bumped since then still reads as behind. The weekly + catalog-freshness rebuild is a workflow artifact, not a file on the host. Running + `tools/sota-convergence/github_freshness.py` over every catalog repository each day is out of proportion for a + notice. `--network` runs the bounded runtime-worker skill check instead. +- **A monotonic timer (`OnUnitActiveSec=1d`), as the host-request poll uses.** Rejected. That poll needs no + catch-up because every poll reads the full current state. A daily notice on a WSL distribution that is often + stopped does need one, and `Persistent=` applies only to `OnCalendar=` timers. +- **Removing the earlier due-file when a run fails.** Rejected, because it would hide due items exactly when the + checks break. A failed run leaves the state directory as it was, the unit shows in + `systemctl --user --failed`, and the hook's 48-hour age limit below keeps an old line from lingering. + +## Decision + +1. **The checks run in a daily timer, never at startup.** + [`stack-currency.timer`](../../adoption/templates/systemd/stack-currency.timer) sets `OnCalendar=daily`, + `Persistent=true` and `RandomizedDelaySec=15m`. + [`stack-currency.service`](../../adoption/templates/systemd/stack-currency.service) is a `Type=oneshot` unit + that runs `/usr/bin/python3 @REPOSITORY@/scripts/currency_due.py`. It sets `UMask=0077`, + `NoNewPrivileges=true`, `Nice=10`, the best-effort I/O class, `TimeoutStartSec=900`, + `PYTHONDONTWRITEBYTECODE=1` and an explicit `PATH` that holds the ecosystem bin directory. That `PATH` is + load-bearing. The workstation's user manager uses + `/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin:/usr/games:/usr/local/games:/snap/bin`, which finds + none of the six exec probes of the default profile. A dry run with that `PATH` read `pins_behind` 0 with 7 + components unchecked, against 1 and 1 with the unit's `PATH`. The install recipe is in `adoption/lifecycle.md`. +2. **The writer.** `scripts/currency_due.py` runs the three checks as subprocesses of the same checkout, with the + arguments the workflows use. As in `saturation-tracking.yml`, the receipt report reaches + `saturation_ledger.py --staleness` as a file. It derives four counts: + - `pins_behind`: the distinct `mismatched` ids across the selected profiles + (`scripts/adoption_status.py:1266-1270`). With `--network` it adds the skill pins in `skill-drift`, + `repository-drift` or `removed-at-head` and a drifted skills CLI pin + (`tools/adoption/runtime_skill_freshness.py:103,152`). A mismatch can also mean the installed version is ahead + of the pin; the detail entry names the component and the pin. + - `stale_receipts`: the report's `flagged` count (`scripts/receipt_staleness.py:162`). + - `due_layers`: the layers the saturation report marks due whose last completed sweep is at least + `--sweep-cadence-days` old (default 30), or that have no datable sweep. The default follows + `recipes/saturation-sweep.md:20-21` and matches `receipt_staleness.py`'s 30-day `DEFAULT_MAX_AGE_DAYS` + (`:60`). The raw count stays in the `coverage` detail as `due_layers_total`. + - `reopen_triggers`: the number of layers in `current_reopen_triggers` + (`scripts/saturation_ledger.py:999-1000`); the report's own summary counts layers too (`:1022`). + + The script writes `${XDG_STATE_HOME:-~/.local/state}/native-agent-stack/currency-due.json` only when a count is + nonzero, and removes the file otherwise. A relative `XDG_STATE_HOME` is ignored, as the XDG specification + requires. The write goes through a mode-0600, fsynced temporary file in the same directory and `os.replace`, + the pattern of `saturation_ledger.write_ledger` (`scripts/saturation_ledger.py:1159-1177`). The document's + keys are `generated_at`, `due`, `summary_line` (at most 160 characters, ending with + `python3 scripts/currency_due.py --dry-run`) and `details`. The script exits 0 whether or not anything is due, + and 2 on an internal error, which leaves the state directory as it was. It makes no network call unless + `--network` is given, and the unit does not pass it. It refuses a state directory inside the checkout. +3. **The startup rule.** Item 4 of `docs/token-practice.md` now allows exactly one read-only SessionStart line from + that file, printed fail-open; the checks never run at startup. `AGENTS.md:28` still reads "Do not rerun the full + audit or model trials at startup", which this design keeps. Any change to that wording is unit F1's. +4. **The contract for the hook in unit F2, which this change does not contain:** + - Read only that file and print its `summary_line` as plain stdout. Exit 0 in every case. + - Print nothing when the file is missing, unreadable or not a JSON object, when `summary_line` is not one line + of at most 160 characters, or when `generated_at` is more than 48 hours old, because a failing timer leaves + the last file in place. The file was 7,175 bytes on the workstation on 2026-09-30. + - Print only for the `startup` source, and for `clear` if the line is wanted after a context reset. Skip + `resume`, `compact` and `fork`: Claude Code saves injected text in the session transcript. + - The acceptance gate, which F2 measures and this change did not: `claude -p ok --output-format json` shows an + input plus cache-creation token delta of 0 without the file and at most 60 tokens with it, and the hook + finishes within 50 ms. +5. **Evidence on this branch.** None of it is an unchanged upstream test. + - `tests/test_currency_due.py`, integration checks with synthetic fixtures: fake checks that print the four + reports' shapes. They cover no file when nothing is due (an earlier file removed), an atomic 0600 write with a + line of at most 160 characters, a failed rename that keeps the earlier file, exit 2 and no write for malformed + output from each check and for a malformed ledger, dry runs that write nothing, the cadence boundary (30 days + counts, 29 does not), network checks off by default, and the two units' settings. Seven patched mutants of the + script (no removal, a direct non-atomic write, no cadence gate, no truncation, network ignored, adoption exit 2 + rejected, a relative `XDG_STATE_HOME` accepted) were each caught by the test named for them, in eight runs: + the direct write was run against both the failed-rename test and the file-mode test. One test runs this + checkout's real checks dry. + - A dry run on the workstation: exit 0 in 1.56 s, and 1.57 s under `/usr/bin/python3` 3.12.3 with the unit's + `PATH` and a cleared environment. + - `systemd-analyze --user verify` on both units rendered with the documented `sed` command: exit 0 with no + warnings. A copy with an unparseable calendar and a missing interpreter failed the same check, exit 1. + `systemd-analyze calendar daily` normalizes to `*-*-* 00:00:00`. These are synthetic checks; no unit was + loaded or started. + +## Overturn condition + +- F2's gate fails, meaning the line costs more than 60 tokens or the hook takes more than 50 ms: shorten the line, + or move the notice to a channel that does not enter the model's context. +- Claude Code gains a native session-start notice outside the model's context: use it instead of a hook. +- A month of daily runs leaves the file present on most days without a resulting action, or the weekly workflows + report a due item that the file misses: retune the counts or the cadence, which are the policy choices recorded + above. +- The catalog-freshness result becomes a file on the host that is refreshed at least weekly: add its + `pin_behind_upstream` rows to `pins_behind`. +- A check's JSON shape changes: update `aggregate()` in the script. Its tests quote the shapes it reads. + +## Sources + +Fetched on 2026-09-30. + +- systemd.timer(5), . `Persistent=`: + "the service unit is triggered immediately if it would have been triggered at least once during the time when + the timer was inactive. Such triggering is nonetheless subject to the delay imposed by RandomizedDelaySec=. This + is useful to catch up on missed runs of the service when the system was powered down. Note that this setting only + has an effect on timers configured with OnCalendar=." `RandomizedDelaySec=`: "Delay the timer by a randomly + selected, evenly distributed amount of time between 0 and the specified time value ... useful to stretch + dispatching of similarly configured timer events over a certain time interval, to prevent them from firing all + at the same time". The workstation's `man systemd.timer` (systemd 255.4-1ubuntu8.17) has the same `Persistent=` + text. +- systemd.time(7), : + `daily → *-*-* 00:00:00`. +- Claude Code hooks, : "For most events, Claude Code writes stdout to the + debug log and doesn't show it in the transcript. The exceptions are `UserPromptSubmit`, `UserPromptExpansion`, + `SessionStart`, and `PostModelSwitch`, where Claude Code adds plain-text stdout as context that Claude can see and + act on." The SessionStart matcher values are `startup`, `resume`, `clear`, `compact` and `fork`, and "Claude Code + saves the injected text in the session transcript". +- XDG Base Directory Specification, : "If + $XDG_STATE_HOME is either not set or empty, a default equal to $HOME/.local/state should be used", and "If an + implementation encounters a relative path in any of these variables it should consider the path invalid and + ignore it." +- Python, : "If successful, the renaming will be an atomic + operation (this is a POSIX requirement)", and it "may fail if src and dst are on different filesystems", which is + why the temporary file is created in the state directory. + : "The file is readable and writable only by + the creating user ID." +- This repository: `recipes/saturation-sweep.md:20-21,36-37`, `recipes/sota-convergence-practice.md:9`, + `scripts/receipt_staleness.py:60,156-164`, `scripts/adoption_status.py:1266-1270,1400`, + `scripts/saturation_ledger.py:966-1002,1022,1159-1177`, `tools/adoption/runtime_skill_freshness.py:103,152,154`, + and the unit conventions of `adoption/templates/systemd/credential-boot-receipt.service` (the `@REPOSITORY@` + render) and `host-requests-workstation.service:26-28` (an explicit `PATH`). diff --git a/docs/token-practice.md b/docs/token-practice.md index 92df67590..798ab0c42 100644 --- a/docs/token-practice.md +++ b/docs/token-practice.md @@ -72,8 +72,13 @@ hook acceptance is not authorization to enable capture on every runtime. original implementation before correctness decisions. Lossy retrieval and compression can omit necessary information. 4. Preserve native caching, compaction, tool discovery, accounts and model - behavior. Shared PATH is not host acceptance. Do not add hooks or schedulers, - override providers, or rerun model trials during ordinary startup. + behavior. Shared PATH is not host acceptance. Do not add hooks or + schedulers, override providers, run audits or network checks, or rerun + model trials during ordinary startup. One addition is allowed: a read-only + SessionStart hook that prints one line of at most 160 characters, the + `summary_line` of the due-file a daily user timer writes, and prints nothing + when that file is absent or unreadable (fail-open). The checks run in that + timer, never at startup ([session currency notice](decisions/2026-09-30-session-currency-notice.md)). 5. Count once at the proper boundary. Missing measurements are unknown. Never add cumulative snapshots, cache subsets, provider usage and artifact differences, or multiply a measured difference by repository count. From a731dc3b3419570fac8af545e54a06ce34361c6e Mon Sep 17 00:00:00 2001 From: Scout Date: Wed, 30 Sep 2026 03:57:55 -0400 Subject: [PATCH 03/25] Currency due-file: an incomplete skill check never removes the notice, and the notice command reproduces the run Repair of the cross-family review of 3d1cfada (three findings, each checked against the source before the change). scripts/currency_due.py - A --network skill check that answered incompletely (an error in its report, a skill left unfetched or in a state this script does not know, an unfetched skills CLI release) is unknown, and unknown is not nothing due: the run keeps the earlier due-file (no removal, no new file), exit 0, and writes what it found when a count is nonzero, with the gap in the coverage detail (skills_complete, skills_fetch_errors, skills_unresolved, the first five error strings). An invalid-pin is a fetched answer and counts in pins_behind as its own detail kind. Sources: runtime_skill_freshness.py:77-81,100-103,115,134. - The command that ends summary_line repeats the options that change what a run reports (--network, a non-default --sweep-cadence-days; the latter bounded to 36500), so running it reproduces the notice. The next-step line names adoption_status.py only for a pin mismatch, not for a skill pin. - pinned_versions must be a list of objects with a string id; a wrong type is CheckError (exit 2), not a TypeError. tests/test_currency_due.py: failing-first against the unchanged script (42 tests, 18 failures, 14 errors: the earlier file deleted, TypeError at currency_due.py:199, the command without --network); incomplete-check, invalid-pin, reproduced-command, ExecStart-flag, malformed-field and 2,277-case shape tests. adoption/templates/systemd/stack-currency.service: header comment only; the directives are unchanged. Co-Authored-By: Claude Sonnet 5.5 --- .../templates/systemd/stack-currency.service | 4 +- scripts/currency_due.py | 126 +++++-- tests/test_currency_due.py | 312 +++++++++++++++++- 3 files changed, 401 insertions(+), 41 deletions(-) diff --git a/adoption/templates/systemd/stack-currency.service b/adoption/templates/systemd/stack-currency.service index 306aa19f9..34ccc66ea 100644 --- a/adoption/templates/systemd/stack-currency.service +++ b/adoption/templates/systemd/stack-currency.service @@ -15,7 +15,9 @@ # receipt_staleness.py, adoption_status.py --pinned-versions and saturation_ledger.py --report, then writes # $XDG_STATE_HOME/native-agent-stack/currency-due.json (~/.local/state/... when XDG_STATE_HOME is unset; mode 0600) # when something is due and removes it otherwise. Its --network option (runtime-worker skill pins through gh api) -# stays off here. +# stays off here. A flag added to ExecStart changes the notice: the command that ends the due-file's summary_line +# repeats --network and a non-default --sweep-cadence-days, so it names the same flags as ExecStart (tests/ +# test_currency_due.py checks that), and a skill check that cannot fetch leaves an earlier due-file in place. # # PATH: adoption_status.py --pinned-versions finds each version probe with shutil.which, and the user manager's own # PATH lacks the ecosystem bin directory, so without this line every probe would be reported unchecked and diff --git a/scripts/currency_due.py b/scripts/currency_due.py index 65212b79d..d14cfac1d 100644 --- a/scripts/currency_due.py +++ b/scripts/currency_due.py @@ -7,7 +7,8 @@ - pins_behind: components whose platform pin's version probe did not observe the pinned version (scripts/adoption_status.py --pinned-versions --json, the "mismatched" ids of each selected profile, each id once); with --network also the runtime-worker skill pins that tools/adoption/runtime_skill_freshness.py reports - as drifted from upstream HEAD (skill-drift, repository-drift, removed-at-head) and a drifted skills CLI pin; + as drifted from upstream HEAD (skill-drift, repository-drift, removed-at-head) or as not matching their recorded + tree (invalid-pin, a fetched answer), and a drifted skills CLI pin; - stale_receipts: the component x platform buckets scripts/receipt_staleness.py --json flags ("flagged"); - due_layers: the layers scripts/saturation_ledger.py --report --json marks due (not a saturation candidate) whose last completed sweep is at least --sweep-cadence-days old, or that have none or an undatable one. The default, @@ -22,8 +23,17 @@ {"generated_at": "YYYY-MM-DDTHH:MM:SSZ", "due": {the four counts}, "summary_line": "at most 160 characters, ending with the command below", "details": [...]} -and otherwise removes that file. A SessionStart hook, a separate change, prints summary_line when the file exists -and nothing when it does not (docs/decisions/2026-09-30-session-currency-notice.md). +and otherwise removes that file, unless the run could not see everything it was asked to check. A skill check +(--network) that answered incompletely, meaning an error in its report, a skill left unfetched or in a state this +script does not know, or a skills CLI release that was not fetched, is unknown, and unknown is not "nothing due" +(the check's own report says "Incomplete fetches remain unknown"). Such a run writes the file when the counts it +did reach are nonzero, with the gap in the coverage entry of the details, and otherwise leaves the state +directory as it was: no removal and no new file, exit 0. + +The command that ends summary_line is "python3 scripts/currency_due.py --dry-run" plus the options that change what a +run reports, --network and a non-default --sweep-cadence-days, so that running it prints the details of the notice. +A SessionStart hook, a separate change, prints summary_line when the file exists and nothing when it does not +(docs/decisions/2026-09-30-session-currency-notice.md). python3 scripts/currency_due.py # write or remove the due-file; one line for the journal python3 scripts/currency_due.py --dry-run # the report as text; writes and removes nothing @@ -31,8 +41,8 @@ python3 scripts/currency_due.py --network # also compare runtime-worker skill pins through gh api No network call unless --network is given. It exits 0 whether or not anything is due, and 2 on an internal error: -a check that fails, times out or prints something other than its JSON report, an unreadable saturation ledger or -a failed write. An error leaves the state directory as it was. +a check that fails, times out or prints something other than its JSON report (a field of the wrong type included), +an unreadable saturation ledger or a failed write. An error leaves the state directory as it was. """ from __future__ import annotations @@ -61,15 +71,23 @@ # recipes/saturation-sweep.md: "Sweep only the due layers, at most monthly"; 30 days is also # scripts/receipt_staleness.py's DEFAULT_MAX_AGE_DAYS. DEFAULT_SWEEP_CADENCE_DAYS = 30 -# tools/adoption/runtime_skill_freshness.py main(): the states its own summary counts as drift. +# The notice repeats a non-default --sweep-cadence-days in its command; the bound (a hundred years) keeps that short. +MAX_SWEEP_CADENCE_DAYS = 36500 +# tools/adoption/runtime_skill_freshness.py compare_skill() (:77-81) gives each skill one state. main() (:152) counts +# the first three as drift. "invalid-pin" is a fetched answer: the manifest's tree is not the tree at the pinned ref. +# "unfetched" is no answer, and so is a state that is none of these, which a newer check may add. SKILL_DRIFT_STATES = ("skill-drift", "repository-drift", "removed-at-head") +SKILL_INVALID_STATE = "invalid-pin" +SKILL_CURRENT_STATE = "current" +ERROR_SAMPLES = 5 # of the skill check's own error strings that the details keep (value-free by design, :46-47) ISO_UTC = re.compile(r"[0-9]{4}-[0-9]{2}-[0-9]{2}T[0-9]{2}:[0-9]{2}:[0-9]{2}Z") # host_receipts.ISO_UTC_PATTERN LEDGER = "catalogs/saturation/ledger.json" # scripts/saturation_ledger.py LEDGER SKILLS_MANIFEST = "blueprints/runtime-workers/skills/manifest.json" # runtime_skill_freshness.py DEFAULT_MANIFEST # (script, exit codes that still carry its JSON report, timeout in seconds). adoption_status.py exits 2 whenever # prerequisites are missing and bounds each version probe at 30 s; runtime_skill_freshness.py exits 1 whenever its -# report is not ok (drift or fetch errors), and its gh api calls time out at 60 s each. +# report is not ok (errors, or a skill whose manifest tree does not match its pin, :115; drift alone exits 0), and +# its gh api calls time out at 60 s each. RECEIPTS = ("scripts/receipt_staleness.py", frozenset({0}), 120) LAYERS = ("scripts/saturation_ledger.py", frozenset({0}), 120) PINS = ("scripts/adoption_status.py", frozenset({0, 2}), 600) @@ -169,12 +187,23 @@ def collect(root: Path, now_text: str, network: bool) -> dict: return {"receipts": receipts, "layers": layers, "pins": pins, "skills": skills, "sweep_dates": sweep_dates(root)} -def summary_line(due: dict) -> str: - """The nonzero counts and the command that prints the details, in at most SUMMARY_LIMIT characters.""" +def details_command(network: bool, cadence_days: int) -> str: + """The command that prints the details of a run made with these options: DETAILS_COMMAND plus the options that + change what a run reports. --root, --state-dir, --now, --json and --dry-run do not belong in a notice.""" + options = ["--network"] if network else [] + if cadence_days != DEFAULT_SWEEP_CADENCE_DAYS: + options += ["--sweep-cadence-days", str(cadence_days)] + return " ".join([DETAILS_COMMAND, *options]) + + +def summary_line(due: dict, command: str = DETAILS_COMMAND, complete: bool = True) -> str: + """The nonzero counts and the command that prints the details, in at most SUMMARY_LIMIT characters. With no + count and a check that could not answer, the line says so rather than "nothing due".""" parts = [f"{due[key]} {LABELS[key][0] if due[key] == 1 else LABELS[key][1]}" for key in DUE_KEYS if due[key]] if not parts: - return "stack currency: nothing due" - prefix, suffix = "stack currency: ", f"; details: {DETAILS_COMMAND}" + return ("stack currency: nothing due" if complete else + "stack currency: nothing known due, skill check incomplete") + prefix, suffix = "stack currency: ", f"; details: {command}" counts, room = ", ".join(parts), SUMMARY_LIMIT - len(prefix) - len(suffix) if len(counts) > room: counts = counts[:room - 3] + "..." @@ -196,8 +225,11 @@ def aggregate(reports: dict, now: datetime, now_text: str, cadence_days: int) -> if not isinstance(profile, dict): raise CheckError(f"{name} report has a profile that is not an object") summary = field(profile, "pinned_versions_summary", dict, name) - versions = {item.get("id"): item.get("pinned_version") for item in profile.get("pinned_versions") or [] - if isinstance(item, dict)} + versions = {} + for item in field(profile, "pinned_versions", list, name): + if not isinstance(item, dict) or not isinstance(item.get("id"), str): + raise CheckError(f"{name} report has a pinned_versions entry that is not an object with a string 'id'") + versions[item["id"]] = item.get("pinned_version") for component in strings(summary.get("mismatched"), "mismatched", name): entry = mismatched.setdefault(component, {"kind": "pin_mismatch", "component_id": component, "pinned_version": versions.get(component), "profiles": []}) @@ -206,20 +238,35 @@ def aggregate(reports: dict, now: datetime, now_text: str, cadence_days: int) -> details += [mismatched[component] for component in sorted(mismatched)] pins_behind = len(mismatched) - skills, skills_errors = reports["skills"], None + # A skill entry or the CLI release that the check could not answer is unresolved, and an error in its report is an + # error; either makes the check incomplete, which keeps an earlier due-file (main) and is never "nothing due". + skills, skills_errors, skills_unresolved = reports["skills"], None, None if skills is not None: name = Path(SKILLS[0]).name - for entry in field(skills, "skills", list, name): - if isinstance(entry, dict) and entry.get("state") in SKILL_DRIFT_STATES: - details.append({"kind": "skill_drift", "skill": entry.get("name"), "source": entry.get("source"), - "state": entry.get("state"), "pinned_ref": entry.get("pinned_ref"), - "head_ref": entry.get("head_ref")}) + entries = field(skills, "skills", list, name) + errors = strings(skills.get("errors"), "errors", name) + skills_errors, skills_unresolved = len(errors), 0 + for entry in entries: + if not isinstance(entry, dict): + raise CheckError(f"{name} report has a skill that is not an object") + state = entry.get("state") + if state == SKILL_CURRENT_STATE: + continue + if state in SKILL_DRIFT_STATES or state == SKILL_INVALID_STATE: + details.append({"kind": "skill_drift" if state in SKILL_DRIFT_STATES else "skill_pin_invalid", + "skill": entry.get("name"), "source": entry.get("source"), "state": state, + "pinned_ref": entry.get("pinned_ref"), "head_ref": entry.get("head_ref")}) pins_behind += 1 + else: + skills_unresolved += 1 cli = skills.get("cli") if isinstance(skills.get("cli"), dict) else {} if cli.get("drift") is True: details.append({"kind": "skills_cli_drift", "pinned": cli.get("pinned"), "latest": cli.get("latest")}) pins_behind += 1 - skills_errors = len(field(skills, "errors", list, name)) + elif cli.get("drift") is not False: + skills_unresolved += 1 + details += [{"kind": "skills_probe_error", "error": error[:200]} for error in errors[:ERROR_SAMPLES]] + skills_complete = None if skills is None else not (skills_errors or skills_unresolved) name = Path(RECEIPTS[0]).name receipts = reports["receipts"] @@ -261,10 +308,17 @@ def aggregate(reports: dict, now: datetime, now_text: str, cadence_days: int) -> details.append({"kind": "coverage", "pins_unchecked": len(unchecked), "due_layers_total": due_total, "sweep_cadence_days": cadence_days, "network": skills is not None, - "skills_fetch_errors": skills_errors}) + "skills_complete": skills_complete, "skills_fetch_errors": skills_errors, + "skills_unresolved": skills_unresolved}) due = {"pins_behind": pins_behind, "stale_receipts": stale_receipts, "due_layers": due_layers, "reopen_triggers": reopen_triggers} - return {"generated_at": now_text, "due": due, "summary_line": summary_line(due), "details": details} + line = summary_line(due, details_command(skills is not None, cadence_days), skills_complete is not False) + return {"generated_at": now_text, "due": due, "summary_line": line, "details": details} + + +def incomplete(document: dict) -> bool: + """True when the run could not see everything it was asked to check; the coverage entry is always the last.""" + return document["details"][-1].get("skills_complete") is False def write_due_file(directory: Path, document: dict) -> Path: @@ -302,8 +356,10 @@ def render_text(document: dict) -> str: if kind == "pin_mismatch": lines.append(f" pin: {item['component_id']} did not report its pin {item['pinned_version']} " f"({', '.join(str(profile) for profile in item['profiles'])})") - elif kind == "skill_drift": + elif kind in ("skill_drift", "skill_pin_invalid"): lines.append(f" skill pin: {item['skill']} ({item['source']}): {item['state']}") + elif kind == "skills_probe_error": + lines.append(f" skill check error: {item['error']}") elif kind == "skills_cli_drift": lines.append(f" skills CLI pin: {item['pinned']}, latest release {item['latest']}") elif kind == "stale_receipt": @@ -317,16 +373,19 @@ def render_text(document: dict) -> str: names = sorted({str(trigger.get("trigger")) for trigger in item["triggers"] if isinstance(trigger, dict)}) lines.append(f" reopen trigger: {item['layer']}: {len(item['triggers'])} ({', '.join(names)})") elif kind == "coverage": - network = "off" if not item["network"] else ( - "on" + (f", {item['skills_fetch_errors']} fetch error(s) left unknown" - if item["skills_fetch_errors"] else "")) + network = ("off" if not item["network"] else "on" if item["skills_complete"] else + f"on, incomplete ({item['skills_fetch_errors']} error(s), {item['skills_unresolved']} " + f"unresolved): left unknown, never counted as nothing due") lines.append(f"coverage: {item['pins_unchecked']} pinned component(s) unchecked on this host; " f"{item['due_layers_total']} layer(s) not yet saturation candidates, due " f"{item['sweep_cadence_days']} days after their last sweep; network checks {network}") due = document["due"] + kinds = {item["kind"] for item in document["details"]} actions = [] - if due["pins_behind"]: + if "pin_mismatch" in kinds: actions.append("pins: python3 scripts/adoption_status.py --pinned-versions") + if kinds & {"skill_drift", "skill_pin_invalid", "skills_cli_drift"}: + actions.append("skill pins: blueprints/runtime-workers/skills/README.md") if due["stale_receipts"]: actions.append("receipts: python3 scripts/receipt_staleness.py and adoption/update.md") if due["due_layers"] or due["reopen_triggers"]: @@ -345,7 +404,8 @@ def build_parser() -> argparse.ArgumentParser: parser.add_argument("--dry-run", action="store_true", help="print the report; write and remove nothing") parser.add_argument("--json", action="store_true", help="print the due-file document instead of text") parser.add_argument("--network", action="store_true", - help="also run tools/adoption/runtime_skill_freshness.py (gh api calls; off by default)") + help="also run tools/adoption/runtime_skill_freshness.py (gh api calls; off by default); " + "an incomplete answer never removes an earlier due-file") parser.add_argument("--now", help="evaluate at this UTC time (YYYY-MM-DDTHH:MM:SSZ); default: the clock") parser.add_argument("--sweep-cadence-days", type=int, default=DEFAULT_SWEEP_CADENCE_DAYS, help=f"count a due layer once its last sweep is this many days old " @@ -360,8 +420,8 @@ def main(argv: list[str] | None = None) -> int: state = (args.state_dir if args.state_dir is not None else default_state_dir()).expanduser().resolve() if state == root or root in state.parents: parser.error(f"--state-dir must be outside the checkout ({root}); the due-file never goes into it") - if args.sweep_cadence_days < 0: - parser.error("--sweep-cadence-days must be zero or more") + if not 0 <= args.sweep_cadence_days <= MAX_SWEEP_CADENCE_DAYS: + parser.error(f"--sweep-cadence-days must be between 0 and {MAX_SWEEP_CADENCE_DAYS}") if args.now is None: now = datetime.now(timezone.utc).replace(microsecond=0) else: @@ -377,7 +437,11 @@ def main(argv: list[str] | None = None) -> int: action = "dry run" if not args.dry_run: if any(document["due"].values()): - action = f"wrote {write_due_file(state, document)}" + action = f"wrote {write_due_file(state, document)}" + ( + "; the skill check was incomplete" if incomplete(document) else "") + elif incomplete(document): + # Unknown is not "nothing due": leave the state directory as it was. + action = f"kept {state / DUE_FILE}" if (state / DUE_FILE).exists() else "no due-file" else: action = f"removed {state / DUE_FILE}" if remove_due_file(state) else "no due-file" except (CheckError, OSError) as error: diff --git a/tests/test_currency_due.py b/tests/test_currency_due.py index 88abeb1e0..a6fe92881 100644 --- a/tests/test_currency_due.py +++ b/tests/test_currency_due.py @@ -11,12 +11,14 @@ import contextlib import io import json +import shlex import stat import subprocess import sys import tempfile import time import unittest +from datetime import datetime, timezone from pathlib import Path from unittest import mock @@ -25,6 +27,7 @@ ROOT = Path(__file__).resolve().parents[1] SYSTEMD_DIR = ROOT / "adoption/templates/systemd" NOW = "2026-09-30T12:00:00Z" +NOW_DATETIME = datetime(2026, 9, 30, 12, 0, 0, tzinfo=timezone.utc) COMMAND = "python3 scripts/currency_due.py --dry-run" STALENESS = "scripts/receipt_staleness.py" @@ -117,15 +120,22 @@ def saturation_report(layers=(), triggers=None): "current_reopen_triggers": {key: value for key, value in triggers.items() if value}, "layers": rows} -def skills_report(states=(), cli_drift=False, errors=()): - """runtime_skill_freshness.py --output (build_report(), runtime_skill_freshness.py:106-116).""" +def skills_report(states=(), cli_drift=False, errors=(), cli_unknown=False): + """runtime_skill_freshness.py --output (build_report(), runtime_skill_freshness.py:106-116). ``cli_unknown`` is the + release fetch that failed: cli.latest and cli.drift stay None (build_report() starts them so, :100). The + ``manifest_tree_matches_pin`` value follows compare_skill() (:77-79): None when nothing was fetched, False for + an invalid pin.""" skills = [{"name": f"skill-{index}", "source": "example/skills", "path": f"skills/skill-{index}", "status": "selected", "pinned_ref": "a" * 40, "head_ref": "b" * 40, "pinned_tree": "c" * 40, - "head_tree": "d" * 40, "manifest_tree_matches_pin": True, "state": state, "native_check_ref": "a" * 40, - "native_check_advances_commit_pin": False} for index, state in enumerate(states)] + "head_tree": "d" * 40, + "manifest_tree_matches_pin": None if state == "unfetched" else state != "invalid-pin", + "state": state, "native_check_ref": "a" * 40, "native_check_advances_commit_pin": False} + for index, state in enumerate(states)] + cli = ({"pinned": "1.5.0", "latest": None, "drift": None} if cli_unknown else + {"pinned": "1.5.0", "latest": "v1.6.0" if cli_drift else "v1.5.0", "drift": cli_drift}) return {"schema_version": 1, "kind": "runtime_skill_freshness_report", "checked_at": "2026-09-30T12:00:00+00:00", "report_only": True, - "cli": {"pinned": "1.5.0", "latest": "v1.6.0" if cli_drift else "v1.5.0", "drift": cli_drift}, + "cli": cli, "native_check": {"executed": False, "source": None, "pinned_version": "1.5.0", "pinned_ref": "a" * 40, "reason": "synthetic"}, "skills": skills, "errors": list(errors), "ok": not errors} @@ -137,6 +147,19 @@ def ledger(*sweeps): "sweeps": [{"sweep_id": sweep_id, "date": day, "status": status} for sweep_id, day, status in sweeps]} +def json_paths(node, prefix=()): + """The path (a tuple of keys and indexes) to every value nested inside ``node``, not ``node`` itself.""" + if isinstance(node, dict): + children = list(node.items()) + elif isinstance(node, list): + children = list(enumerate(node)) + else: + return + for key, child in children: + yield (*prefix, key) + yield from json_paths(child, (*prefix, key)) + + class Checkout: """A temporary checkout whose checks are fakes, with a state directory beside it (outside the checkout). By default nothing is due.""" @@ -266,6 +289,26 @@ def test_a_malformed_ledger_exits_2(self): self.assertEqual(code, 2) self.assertFalse(checkout.state.exists()) + def test_a_malformed_pinned_versions_field_exits_2_and_writes_nothing(self): + # adoption_status.py:1258-1263,1331 always sets pinned_versions, a list of {"id": str, ...}, beside the + # summary. `True or []` used to reach a loop and raise TypeError, which is exit 1 with a traceback, not 2. + def report_with(value): + report = pinned_report(mismatched=("codex",)) + report["profiles"][0]["pinned_versions"] = value + return report + + malformed = {"a boolean": True, "a number": 1, "a string": "codex", "an object": {"id": "codex"}, + "a list of numbers": [1], "an entry whose id is a list": [{"id": ["codex"]}], + "an entry without an id": [{"pinned_version": "1.0.0"}], "null": None} + for label, value in malformed.items(): + with self.subTest(pinned_versions=label): + checkout = Checkout(self) + checkout.set(PINNED, report_with(value)) + code, _, stderr = checkout.run() + self.assertEqual(code, 2, stderr) + self.assertIn("adoption_status.py", stderr) + self.assertFalse(checkout.state.exists()) + def test_a_failed_run_leaves_an_earlier_file_byte_identical(self): checkout = Checkout(self) checkout.state.mkdir(mode=0o700) @@ -405,6 +448,47 @@ def test_every_check_gets_its_documented_arguments_and_the_same_clock(self): self.assertNotIn(checkout.state, handed.parents) +class AggregateShapeTests(unittest.TestCase): + """aggregate() reads nested fields of four JSON reports. Whatever type one of them has, it either ignores the field + or raises CheckError (exit 2); any other exception would be exit 1 with a traceback.""" + + REPLACEMENTS = (None, True, 0, -1, 1.5, "x", [], {}, [None], [[]], {"a": None}) + + @staticmethod + def reports() -> dict: + trigger = {"trigger": "pin_moved", "ref": "receipt_staleness:linux-wsl2-x86_64/codex"} + return {"receipts": staleness_report(("codex", ["pin_moved"])), + "pins": pinned_report(mismatched=("codex",)), + "layers": saturation_report([("foundation/workers", True, "sweep-a")], + {"foundation/workers": [trigger]}), + "skills": skills_report(("skill-drift", "invalid-pin", "unfetched", "current"), cli_drift=True, + errors=["gh api failed (exit 1): repos/example/skills/commits/HEAD"]), + "sweep_dates": {"sweep-a": "2026-08-01"}} + + def test_a_wrong_typed_field_is_ignored_or_a_check_error(self): + # The unmodified reports: codex's pin, a drifted skill, an invalid skill pin and the skills CLI make four. + document = cd.aggregate(self.reports(), NOW_DATETIME, NOW, 30) + self.assertEqual(document["due"], {"pins_behind": 4, "stale_receipts": 1, "due_layers": 1, + "reopen_triggers": 1}) + exercised = 0 + for name in ("receipts", "pins", "layers", "skills"): + for path in json_paths(self.reports()[name]): + for value in self.REPLACEMENTS: + reports = self.reports() + parent = reports[name] + for key in path[:-1]: + parent = parent[key] + parent[path[-1]] = value + with self.subTest(report=name, path=path, value=value): + try: + document = cd.aggregate(reports, NOW_DATETIME, NOW, 30) + except cd.CheckError: + continue + cd.render_text(document) + exercised += 1 + self.assertGreater(exercised, 500) + + class NetworkTests(unittest.TestCase): def test_network_checks_are_off_by_default(self): checkout = Checkout(self) @@ -414,9 +498,11 @@ def test_network_checks_are_off_by_default(self): self.assertIsNone(checkout.recorded(SKILLS)) self.assertEqual(json.loads(stdout)["due"]["pins_behind"], 0) - def test_network_adds_drifted_skill_pins_and_keeps_fetch_errors_unknown(self): + def test_network_adds_drifted_and_invalid_skill_pins_and_keeps_fetch_errors_unknown(self): # runtime_skill_freshness.py:152 counts skill-drift, repository-drift and removed-at-head as drift, and - # :103 sets cli.drift; it exits 1 whenever its report is not ok (:154), which is not a failure here. + # :103 sets cli.drift; it exits 1 whenever its report is not ok (:154), which is not a failure here. An + # invalid-pin is a fetched answer (:78-79: the manifest's tree is not the tree at the pinned ref), so it + # counts too; an unfetched skill and a failed fetch stay unknown. checkout = Checkout(self) states = ("current", "skill-drift", "repository-drift", "removed-at-head", "unfetched", "invalid-pin") checkout.set(SKILLS, stdout='{"ok": false}', code=1, output_file=json.dumps( @@ -424,10 +510,14 @@ def test_network_adds_drifted_skill_pins_and_keeps_fetch_errors_unknown(self): code, stdout, stderr = checkout.run("--dry-run", "--json", "--network") self.assertEqual(code, 0, stderr) document = json.loads(stdout) - self.assertEqual(document["due"]["pins_behind"], 4) + self.assertEqual(document["due"]["pins_behind"], 5) self.assertEqual(sorted(item["state"] for item in document["details"] if item["kind"] == "skill_drift"), ["removed-at-head", "repository-drift", "skill-drift"]) - self.assertEqual(document["details"][-1]["skills_fetch_errors"], 1) + self.assertEqual([item["state"] for item in document["details"] if item["kind"] == "skill_pin_invalid"], + ["invalid-pin"]) + coverage = document["details"][-1] + self.assertEqual((coverage["skills_fetch_errors"], coverage["skills_unresolved"], coverage["skills_complete"]), + (1, 1, False)) arguments = checkout.recorded(SKILLS) self.assertEqual(arguments[:2], ["--manifest", str(checkout.root / "blueprints/runtime-workers/skills/manifest.json")]) @@ -440,6 +530,186 @@ def test_a_skills_report_that_was_not_written_exits_2(self): self.assertEqual(code, 2) +class IncompleteSkillCheckTests(unittest.TestCase): + """A skill check that could not answer is unknown, and unknown is not "nothing due" (the check's own report says + "Incomplete fetches remain unknown", runtime_skill_freshness.py:134). It must never remove the notice an earlier + run wrote. The timer does not pass --network; these are the runs that do.""" + + FETCH_ERROR = "gh api failed (exit 1): repos/example/skills/commits/HEAD" + EARLIER = b'{"earlier": true}\n' + + def incomplete_reports(self) -> dict: + return { + "a failed fetch": skills_report(("unfetched",), errors=[self.FETCH_ERROR]), + "an unfetched skill that names no error": skills_report(("current", "unfetched")), + "a state this script does not know": skills_report(("current", "state-of-a-newer-check")), + "a failed release fetch": skills_report(("current",), cli_unknown=True, errors=[self.FETCH_ERROR]), + "an unknown release that names no error": skills_report(("current",), cli_unknown=True), + "an error beside skills that were all fetched": skills_report( + ("current",), errors=["cli pin: version None is not a release version"]), + } + + def skills_checkout(self, report: dict) -> Checkout: + checkout = Checkout(self) + checkout.set(SKILLS, stdout='{"ok": false}', code=1, output_file=json.dumps(report)) + return checkout + + def test_an_incomplete_check_keeps_the_earlier_due_file_byte_identical(self): + for label, report in self.incomplete_reports().items(): + with self.subTest(case=label): + checkout = self.skills_checkout(report) + checkout.state.mkdir(mode=0o700) + checkout.due_file.write_bytes(self.EARLIER) + code, stdout, stderr = checkout.run("--network") + self.assertEqual(code, 0, stderr) + self.assertEqual(checkout.due_file.read_bytes(), self.EARLIER) + self.assertEqual([path.name for path in checkout.state.iterdir()], ["currency-due.json"]) + self.assertIn("nothing known due", stdout) + self.assertIn("kept", stdout) + + def test_an_incomplete_check_creates_no_due_file(self): + for label, report in self.incomplete_reports().items(): + with self.subTest(case=label): + checkout = self.skills_checkout(report) + code, stdout, stderr = checkout.run("--network") + self.assertEqual(code, 0, stderr) + self.assertFalse(checkout.state.exists()) + self.assertIn("nothing known due", stdout) + + def test_a_complete_check_with_nothing_due_still_removes_the_earlier_file(self): + # The control: it is the incompleteness, not --network, that keeps the file. + checkout = self.skills_checkout(skills_report(("current", "current"))) + checkout.state.mkdir(mode=0o700) + checkout.due_file.write_bytes(self.EARLIER) + code, stdout, stderr = checkout.run("--network") + self.assertEqual(code, 0, stderr) + self.assertFalse(checkout.due_file.exists()) + self.assertIn("stack currency: nothing due", stdout) + + def test_an_incomplete_check_still_writes_what_the_other_checks_found(self): + checkout = self.skills_checkout(skills_report(("unfetched",), errors=[self.FETCH_ERROR])) + checkout.something_due() + checkout.state.mkdir(mode=0o700) + checkout.due_file.write_bytes(self.EARLIER) + code, stdout, stderr = checkout.run("--network") + self.assertEqual(code, 0, stderr) + document = json.loads(checkout.due_file.read_text(encoding="utf-8")) + self.assertEqual(document["due"], {"pins_behind": 1, "stale_receipts": 2, "due_layers": 0, + "reopen_triggers": 1}) + coverage = document["details"][-1] + self.assertEqual((coverage["skills_complete"], coverage["skills_fetch_errors"], coverage["skills_unresolved"]), + (False, 1, 1)) + self.assertEqual([item["error"] for item in document["details"] if item["kind"] == "skills_probe_error"], + [self.FETCH_ERROR]) + self.assertIn("incomplete", stdout) + + def test_an_invalid_pin_is_a_finding_that_replaces_the_earlier_file(self): + # runtime_skill_freshness.py:78-79: the pinned ref was fetched and the manifest's tree is not its tree. + checkout = self.skills_checkout(skills_report(("current", "invalid-pin"))) + checkout.state.mkdir(mode=0o700) + checkout.due_file.write_bytes(self.EARLIER) + code, _, stderr = checkout.run("--network") + self.assertEqual(code, 0, stderr) + document = json.loads(checkout.due_file.read_text(encoding="utf-8")) + self.assertEqual(document["due"]["pins_behind"], 1) + invalid = [item for item in document["details"] if item["kind"] == "skill_pin_invalid"] + self.assertEqual([(item["skill"], item["state"]) for item in invalid], [("skill-1", "invalid-pin")]) + self.assertIs(document["details"][-1]["skills_complete"], True) + self.assertEqual(document["summary_line"], f"stack currency: 1 pin behind; details: {COMMAND} --network") + + def test_a_dry_run_headline_says_nothing_known_due_and_lists_the_error(self): + checkout = self.skills_checkout(skills_report(("unfetched",), errors=[self.FETCH_ERROR])) + code, text, stderr = checkout.run("--dry-run", "--network") + self.assertEqual(code, 0, stderr) + lines = text.splitlines() + self.assertEqual(lines[0], "stack currency: nothing known due, skill check incomplete") + self.assertTrue(any(self.FETCH_ERROR in line for line in lines)) + self.assertTrue(any(line.startswith("coverage:") and "incomplete" in line for line in lines)) + self.assertFalse(checkout.state.exists()) + + +class DetailsCommandTests(unittest.TestCase): + """The command that ends the notice must print the details of the run that wrote it, so it repeats the options + that change what a run reports: --network and a non-default --sweep-cadence-days.""" + + def notice(self, checkout: Checkout, *options: str) -> dict: + code, _, stderr = checkout.run(*options) + self.assertEqual(code, 0, stderr) + return json.loads(checkout.due_file.read_text(encoding="utf-8")) + + def reproduced(self, checkout: Checkout, document: dict) -> dict: + """Run the notice's own command in this checkout (in process) and return the document it prints.""" + summary = document["summary_line"] + self.assertIn("; details: ", summary) + command = shlex.split(summary.split("; details: ", 1)[1]) + self.assertEqual(command[:3], ["python3", "scripts/currency_due.py", "--dry-run"], summary) + code, stdout, stderr = checkout.run(*command[2:], "--json") + self.assertEqual(code, 0, stderr) + return json.loads(stdout) + + def test_skill_drift_alone_is_reproduced_by_the_command_in_the_notice(self): + checkout = Checkout(self) + checkout.set(SKILLS, stdout="{}", output_file=json.dumps(skills_report(("skill-drift",)))) + document = self.notice(checkout, "--network") + self.assertEqual(document["due"], {"pins_behind": 1, "stale_receipts": 0, "due_layers": 0, + "reopen_triggers": 0}) + self.assertEqual(document["summary_line"], f"stack currency: 1 pin behind; details: {COMMAND} --network") + again = self.reproduced(checkout, document) + self.assertEqual((again["due"], again["summary_line"]), (document["due"], document["summary_line"])) + # Without the option the same command sees nothing, which is why the notice has to name it. + self.assertEqual(json.loads(checkout.run("--dry-run", "--json")[1])["due"]["pins_behind"], 0) + + def test_a_non_default_cadence_is_reproduced_by_the_command_in_the_notice(self): + checkout = Checkout(self) + # Swept the day before NOW: not yet due at the default 30 days, due at 0. + checkout.set(SATURATION, saturation_report([("foundation/workers", True, "sweep-a")])) + document = self.notice(checkout, "--sweep-cadence-days", "0") + self.assertEqual(document["due"]["due_layers"], 1) + self.assertEqual(document["summary_line"], + f"stack currency: 1 layer due; details: {COMMAND} --sweep-cadence-days 0") + again = self.reproduced(checkout, document) + self.assertEqual((again["due"], again["summary_line"]), (document["due"], document["summary_line"])) + self.assertEqual(json.loads(checkout.run("--dry-run", "--json")[1])["due"]["due_layers"], 0) + + def test_both_options_are_named_and_the_line_keeps_its_limit(self): + checkout = Checkout(self) + checkout.something_due() + checkout.set(SKILLS, stdout="{}", output_file=json.dumps(skills_report(("skill-drift",)))) + document = self.notice(checkout, "--network", "--sweep-cadence-days", "7") + self.assertTrue(document["summary_line"].endswith(f"; details: {COMMAND} --network --sweep-cadence-days 7"), + document["summary_line"]) + self.assertLessEqual(len(document["summary_line"]), 160) + self.assertEqual(self.reproduced(checkout, document)["due"], document["due"]) + + def test_options_that_do_not_change_the_counts_are_not_repeated(self): + checkout = Checkout(self) + checkout.something_due() + # checkout.run adds --root, --state-dir and --now; --json and --dry-run are output modes. + document = self.notice(checkout, "--json") + self.assertTrue(document["summary_line"].endswith(f"; details: {COMMAND}"), document["summary_line"]) + + def test_the_next_step_points_at_what_is_due(self): + # A skill pin that drifted is not something scripts/adoption_status.py --pinned-versions can report. + checkout = Checkout(self) + checkout.set(SKILLS, stdout="{}", output_file=json.dumps(skills_report(("skill-drift",)))) + _, text, _ = checkout.run("--dry-run", "--network") + self.assertEqual(text.splitlines()[-1], "next: skill pins: blueprints/runtime-workers/skills/README.md") + checkout = Checkout(self) + checkout.set(PINNED, pinned_report(mismatched=("codex",))) + _, text, _ = checkout.run("--dry-run") + self.assertEqual(text.splitlines()[-1], "next: pins: python3 scripts/adoption_status.py --pinned-versions") + + def test_the_cadence_option_is_bounded_so_the_command_stays_short(self): + checkout = Checkout(self) + for value in ("-1", "36501"): + with self.subTest(value=value): + with contextlib.redirect_stderr(io.StringIO()), self.assertRaises(SystemExit) as raised: + cd.main(["--root", str(checkout.root), "--state-dir", str(checkout.state), "--now", NOW, + "--sweep-cadence-days", value]) + self.assertEqual(raised.exception.code, 2) + self.assertEqual(checkout.run("--dry-run", "--sweep-cadence-days", "36500")[0], 0) + + class SummaryLineTests(unittest.TestCase): def test_the_line_names_only_nonzero_counts_and_the_command(self): line = cd.summary_line({"pins_behind": 2, "stale_receipts": 1, "due_layers": 3, "reopen_triggers": 0}) @@ -451,6 +721,18 @@ def test_the_line_stays_within_160_characters_and_keeps_the_command(self): self.assertLessEqual(len(line), 160) self.assertTrue(line.endswith(f"; details: {COMMAND}")) + def test_the_line_ends_with_the_command_it_is_given_within_160_characters(self): + command = f"{COMMAND} --network --sweep-cadence-days 36500" + line = cd.summary_line(dict.fromkeys(cd.DUE_KEYS, 10 ** 40), command) + self.assertLessEqual(len(line), 160) + self.assertTrue(line.endswith(f"; details: {command}")) + due = dict.fromkeys(cd.DUE_KEYS, 0) | {"pins_behind": 1} + self.assertEqual(cd.summary_line(due, command), f"stack currency: 1 pin behind; details: {command}") + + def test_an_incomplete_check_with_nothing_found_is_not_reported_as_nothing_due(self): + line = cd.summary_line(dict.fromkeys(cd.DUE_KEYS, 0), complete=False) + self.assertEqual(line, "stack currency: nothing known due, skill check incomplete") + class StateDirectoryTests(unittest.TestCase): def test_xdg_state_home_is_used_only_when_absolute(self): @@ -505,6 +787,18 @@ def test_the_service_is_a_guarded_oneshot_that_only_the_timer_starts(self): search_path = next(line for line in directives if line.startswith("Environment=PATH=")) self.assertIn("%h/.local/share/codex-ecosystem/bin", search_path) + def test_the_command_in_the_notice_names_exactly_the_flags_the_service_passes(self): + # ExecStart is "