diff --git a/agent/background_review.py b/agent/background_review.py index cfccfb323da7a..dbb52dd4a0a89 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -22,6 +22,7 @@ import json import logging import os +import uuid from typing import Any, Dict, List, Optional from agent.thread_scoped_output import thread_scoped_silence @@ -886,6 +887,10 @@ def _unregister_review_agent(agent_ref) -> None: # rebuild path, but these pins guarantee parity even # if a future code path bypasses the cache. review_agent.session_start = agent.session_start + # Cache/transcript attribution is not resource ownership. Pin a + # private task namespace before borrowing the parent's session id; + # both tool dispatch and close() must use this same namespace. + review_agent._resource_owner_task_id = str(uuid.uuid4()) review_agent.session_id = agent.session_id # The fork shares the parent's live session_id (pinned above for # prefix-cache parity). It is single-lifecycle and calls close() @@ -973,6 +978,7 @@ def _unregister_review_agent(agent_ref) -> None: else messages_snapshot ) review_agent.run_conversation( + task_id=review_agent._resource_owner_task_id, user_message=( prompt + "\n\nYou can only call memory and skill " diff --git a/hermes_cli/goals.py b/hermes_cli/goals.py index 891c3a66bb796..0de538e77189a 100644 --- a/hermes_cli/goals.py +++ b/hermes_cli/goals.py @@ -29,7 +29,6 @@ from __future__ import annotations -import hashlib import json import logging import os @@ -445,8 +444,8 @@ class GoalGate: attempts: int = 0 last_exit_code: Optional[int] = None last_output_tail: str = "" - # Workspace fingerprint at the time of the last FAILED run — used to skip - # re-running an identical gate when nothing changed since it failed. + # Legacy serialized field, retained for compatibility only. Never reuse + # a gate result based on a workspace fingerprint. last_failed_fingerprint: str = "" def to_dict(self) -> Dict[str, Any]: @@ -467,36 +466,6 @@ def from_dict(cls, data: Optional[Dict[str, Any]]) -> "GoalGate": ) -def workspace_fingerprint(cwd: Optional[str] = None) -> str: - """Cheap workspace change fingerprint for unchanged-gate skip. - - Uses ``git status --porcelain`` + ``git rev-parse HEAD`` when inside a git - repo (covers tracked edits, stages, and commits). Outside git, returns - an empty string — an empty fingerprint never matches, so gates simply - always re-run (safe fallback, no behavior regression for non-repo work). - """ - workdir = cwd or os.getcwd() - try: - head = subprocess.run( - ["git", "rev-parse", "HEAD"], - capture_output=True, text=True, encoding="utf-8", errors="replace", - timeout=10, cwd=workdir, - ) - if head.returncode != 0: - return "" - status = subprocess.run( - ["git", "status", "--porcelain"], - capture_output=True, text=True, encoding="utf-8", errors="replace", - timeout=30, cwd=workdir, - ) - if status.returncode != 0: - return "" - blob = head.stdout.strip() + "\n" + status.stdout - return hashlib.sha256(blob.encode("utf-8", "replace")).hexdigest() - except Exception: - return "" - - def run_gate(gate: GoalGate, *, cwd: Optional[str] = None) -> Tuple[bool, int, str]: """Run one gate command. Returns ``(passed, exit_code, output_tail)``. @@ -1496,26 +1465,16 @@ def _check_gates(self) -> Optional[Dict[str, Any]]: either a continuation carrying the gate's output (attempts left) or an auto-pause (retries exhausted). - An unchanged workspace since the last failure of the same gate is - NOT re-run — the recorded failure is replayed and the attempt count - advances, so a stalled agent can't spin re-running an identical red - suite (mirrors Prime-Agent's unchanged-gate rule). + Always execute fresh evidence. Git status/HEAD cannot identify edits + to an already-dirty file, external state, or another worktree targeted + by the command. Retry and turn budgets still bound repeated failures. """ state = self._state if state is None or not state.gates: return None - fingerprint = workspace_fingerprint() for gate in state.gates: - unchanged = ( - bool(fingerprint) - and gate.last_exit_code not in (None, 0) - and gate.last_failed_fingerprint == fingerprint - ) - if unchanged: - passed, exit_code, tail = False, int(gate.last_exit_code or -1), gate.last_output_tail - else: - passed, exit_code, tail = run_gate(gate) + passed, exit_code, tail = run_gate(gate) gate.last_exit_code = exit_code gate.last_output_tail = tail if passed: @@ -1524,8 +1483,7 @@ def _check_gates(self) -> Optional[Dict[str, Any]]: continue gate.attempts += 1 - gate.last_failed_fingerprint = fingerprint - skipped_note = " (workspace unchanged since last failure — not re-run)" if unchanged else "" + gate.last_failed_fingerprint = "" if gate.attempts > gate.max_retries: state.status = "paused" @@ -1564,7 +1522,7 @@ def _check_gates(self) -> Optional[Dict[str, Any]]: "reason": f"gate failed (exit {exit_code}): $ {gate.command}", "message": ( f"✗ Quality gate failed ({state.turns_used}/{state.max_turns} turns, " - f"attempt {gate.attempts}/{gate.max_retries}){skipped_note}: $ {gate.command}" + f"attempt {gate.attempts}/{gate.max_retries}): $ {gate.command}" ), } @@ -2136,7 +2094,6 @@ def _log(msg: str) -> None: "parse_contract", "draft_contract", "run_gate", - "workspace_fingerprint", "CONTINUATION_PROMPT_TEMPLATE", "CONTINUATION_PROMPT_WITH_SUBGOALS_TEMPLATE", "CONTINUATION_PROMPT_WITH_CONTRACT_TEMPLATE", diff --git a/run_agent.py b/run_agent.py index 12e7da647ebbd..10b5fe4950166 100644 --- a/run_agent.py +++ b/run_agent.py @@ -4312,7 +4312,13 @@ def close(self) -> None: Safe to call multiple times (idempotent). Each cleanup step is independently guarded so a failure in one does not prevent the rest. """ - task_id = getattr(self, "session_id", None) or "" + # Review forks borrow session_id for prompt-cache attribution, never + # ownership of the parent's processes/environments/browser/CUA state. + task_id = ( + getattr(self, "_resource_owner_task_id", None) + or getattr(self, "session_id", None) + or "" + ) # 1. Kill background processes for this task try: diff --git a/tests/hermes_cli/test_goal_gate_fresh_evidence.py b/tests/hermes_cli/test_goal_gate_fresh_evidence.py new file mode 100644 index 0000000000000..64523824c6558 --- /dev/null +++ b/tests/hermes_cli/test_goal_gate_fresh_evidence.py @@ -0,0 +1,60 @@ +"""Native goal gates must execute fresh evidence, not replay git-status receipts.""" + +import subprocess +import sys + +import pytest + +from hermes_cli.goals import GoalManager + + +def git(path, *args): + return subprocess.run( + ["git", "-C", str(path), *args], check=True, capture_output=True, text=True, + ).stdout + + +@pytest.mark.parametrize("other_worktree", [False, True]) +def test_gate_reruns_after_repeated_dirty_edits(tmp_path, monkeypatch, other_worktree): + repo = tmp_path / "repo" + repo.mkdir() + git(repo, "init") + gate_script = repo / "gate.py" + gate_script.write_text( + "from pathlib import Path\n" + "value = Path(__file__).with_name('value').read_text().strip()\n" + "print('actual gate value=' + value)\n" + "raise SystemExit(int(value))\n" + ) + (repo / "value").write_text("0") + git(repo, "add", "gate.py", "value") + git(repo, "-c", "user.name=Fixture", "-c", "user.email=fixture@example.invalid", + "commit", "-m", "fixture") + target = repo + if other_worktree: + target = tmp_path / "target" + git(repo, "worktree", "add", "--detach", str(target), "HEAD") + monkeypatch.chdir(repo) + mgr = GoalManager(session_id="gate-fresh-evidence") + mgr.set("fixture goal") + mgr.add_gate(f'"{sys.executable}" "{target / "gate.py"}"') + + (target / "value").write_text("1") + first_status = git(repo, "status", "--porcelain") + assert mgr._check_gates()["verdict"] == "gate_failed" + gate = mgr.state.gates[0] + assert gate.last_exit_code == 1 + assert "actual gate value=1" in gate.last_output_tail + + # Same dirty-path listing and HEAD, different bytes (or different worktree). + (target / "value").write_text("2") + assert git(repo, "status", "--porcelain") == first_status + assert mgr._check_gates()["verdict"] == "gate_failed" + assert gate.last_exit_code == 2, "replayed stale failure instead of executing gate" + assert "actual gate value=2" in gate.last_output_tail + + (target / "value").write_text("0") + assert mgr._check_gates() is None + assert gate.last_exit_code == 0 + assert "actual gate value=0" in gate.last_output_tail + assert gate.attempts == 0 diff --git a/tests/hermes_cli/test_goal_gates.py b/tests/hermes_cli/test_goal_gates.py index 4dfc79478784e..3ef3b7567693b 100644 --- a/tests/hermes_cli/test_goal_gates.py +++ b/tests/hermes_cli/test_goal_gates.py @@ -161,8 +161,7 @@ def test_status_line_mentions_gates(): def test_failing_gate_short_circuits_judge(): mgr = _mgr_with_goal("gate-fail-sid") mgr.add_gate("exit 5") - with patch("hermes_cli.goals.judge_goal") as mock_judge, \ - patch("hermes_cli.goals.workspace_fingerprint", return_value=""): + with patch("hermes_cli.goals.judge_goal") as mock_judge: decision = mgr.evaluate_after_turn("I think it's done!") mock_judge.assert_not_called() assert decision["verdict"] == "gate_failed" @@ -190,8 +189,7 @@ def test_gate_retry_exhaustion_pauses_goal(): mgr = _mgr_with_goal("gate-exhaust-sid") mgr.add_gate("exit 1") mgr.state.gates[0].max_retries = 2 - with patch("hermes_cli.goals.judge_goal") as mock_judge, \ - patch("hermes_cli.goals.workspace_fingerprint", return_value=""): + with patch("hermes_cli.goals.judge_goal") as mock_judge: d1 = mgr.evaluate_after_turn("attempt one") d2 = mgr.evaluate_after_turn("attempt two") d3 = mgr.evaluate_after_turn("attempt three") @@ -204,38 +202,38 @@ def test_gate_retry_exhaustion_pauses_goal(): assert "gate" in (mgr.state.paused_reason or "") -def test_unchanged_workspace_skips_rerun(): +def test_unchanged_workspace_reruns_gate_with_fresh_diagnostics(): mgr = _mgr_with_goal("gate-unchanged-sid") mgr.add_gate("exit 1") - with patch("hermes_cli.goals.workspace_fingerprint", return_value="fp-1"), \ - patch("hermes_cli.goals.judge_goal"): + with patch("hermes_cli.goals.judge_goal"): mgr.evaluate_after_turn("turn 1") - # Second turn, same fingerprint — run_gate must NOT run again. - with patch("hermes_cli.goals.run_gate") as mock_run: + with patch("hermes_cli.goals.run_gate", return_value=(False, 2, "fresh failure")) as mock_run: d2 = mgr.evaluate_after_turn("turn 2") - mock_run.assert_not_called() + mock_run.assert_called_once() assert d2["verdict"] == "gate_failed" - assert "unchanged" in d2["message"] + assert "fresh failure" in d2["continuation_prompt"] + assert mgr.state.gates[0].last_exit_code == 2 -def test_changed_workspace_reruns_gate(): - mgr = _mgr_with_goal("gate-changed-sid") - mgr.add_gate("exit 1") - with patch("hermes_cli.goals.judge_goal"): - with patch("hermes_cli.goals.workspace_fingerprint", return_value="fp-1"): - mgr.evaluate_after_turn("turn 1") - with patch("hermes_cli.goals.workspace_fingerprint", return_value="fp-2"), \ - patch("hermes_cli.goals.run_gate", return_value=(False, 1, "still red")) as mock_run: - mgr.evaluate_after_turn("turn 2") - mock_run.assert_called_once() +def test_legacy_failure_fingerprint_cannot_skip_passing_rerun(): + mgr = _mgr_with_goal("gate-legacy-sid") + mgr.add_gate("true") + gate = mgr.state.gates[0] + gate.last_exit_code = 1 + gate.last_output_tail = "old failure" + gate.last_failed_fingerprint = "legacy-fingerprint" + save_goal(mgr.session_id, mgr.state) + reloaded = GoalManager(session_id=mgr.session_id) + assert reloaded._check_gates() is None + assert reloaded.state.gates[0].last_exit_code == 0 + assert reloaded.state.gates[0].last_failed_fingerprint == "" def test_gate_continuation_respects_turn_budget(): mgr = GoalManager(session_id="gate-budget-sid", default_max_turns=1) mgr.set("budget goal") mgr.add_gate("exit 1") - with patch("hermes_cli.goals.judge_goal"), \ - patch("hermes_cli.goals.workspace_fingerprint", return_value=""): + with patch("hermes_cli.goals.judge_goal"): decision = mgr.evaluate_after_turn("only turn") assert decision["status"] == "paused" assert decision["should_continue"] is False diff --git a/tests/run_agent/test_review_resource_ownership.py b/tests/run_agent/test_review_resource_ownership.py new file mode 100644 index 0000000000000..fb5cbfa84640b --- /dev/null +++ b/tests/run_agent/test_review_resource_ownership.py @@ -0,0 +1,78 @@ +"""Review teardown must not own the live parent's tool resources.""" + +from types import SimpleNamespace + +import pytest + +import run_agent +from agent import background_review +from run_agent import AIAgent +from tests.run_agent.test_background_review import _bare_agent + + +@pytest.mark.parametrize("crash", [False, True]) +def test_real_review_lifecycle_closes_only_review_resources(monkeypatch, crash): + parent = _bare_agent() + resources = { + kind: {parent.session_id: SimpleNamespace(alive=True)} + for kind in ("job", "environment", "browser", "cua") + } + parent_resources = {kind: rows[parent.session_id] for kind, rows in resources.items()} + captured = {} + + def release(kind, task_id): + resource = resources[kind].pop(task_id, None) + if resource: + resource.alive = False + + # Stateful resource fixtures: no actual process killing or backend calls. + from tools.process_registry import process_registry + import tools.computer_use as cua + monkeypatch.setattr(process_registry, "kill_all", lambda task_id: release("job", task_id)) + monkeypatch.setattr(run_agent, "cleanup_vm", lambda task_id: release("environment", task_id)) + monkeypatch.setattr(run_agent, "cleanup_browser", lambda task_id: release("browser", task_id)) + monkeypatch.setattr(cua, "release_computer_use_session", lambda task_id: release("cua", task_id)) + monkeypatch.setattr("hermes_cli.mem_trim.trim_memory", lambda **kw: None) + monkeypatch.setattr(background_review, "_resolve_review_runtime", lambda agent: {"routed": False}) + monkeypatch.setattr("model_tools.get_tool_definitions", lambda **kw: []) + + class Review(AIAgent): + def __init__(self, **kwargs): + # Isolate provider initialization; retain the REAL lifecycle/close. + self.session_id = "review-constructor-session" + self._session_messages = [] + self.client = None + self._session_db = None + captured["review"] = self + + def run_conversation(self, **kwargs): + captured["ran"] = True + captured["task_id"] = kwargs.get("task_id") or "review-generated-turn" + captured["prompt"] = self._cached_system_prompt + captured["session_id"] = self.session_id + captured["resources"] = {} + for kind, rows in resources.items(): + resource = SimpleNamespace(alive=True) + rows[captured["task_id"]] = resource + captured["resources"][kind] = resource + if crash: + raise RuntimeError("fixture review failure") + + def shutdown_memory_provider(self): + pass + + monkeypatch.setattr(run_agent, "AIAgent", Review) + # Exercise the real review worker, including its success/exception finally. + background_review._run_review_in_thread(parent, [], "fixture review") + + assert captured.get("ran"), "review did not reach its execution seam" + assert captured["prompt"] == parent._cached_system_prompt + assert captured["session_id"] == parent.session_id # cache attribution unchanged + assert all(resource.alive for resource in parent_resources.values()), "review closed parent resources" + assert captured["task_id"] != parent.session_id + assert all(not resource.alive for resource in captured["resources"].values()), "review leaked its resources" + assert parent._active_children == [] + assert parent._background_review_agent is None + # Repeated close remains confined to this review's namespace. + captured["review"].close() + assert all(resource.alive for resource in parent_resources.values()) diff --git a/tests/tools/test_snapshot_authority_isolation.py b/tests/tools/test_snapshot_authority_isolation.py new file mode 100644 index 0000000000000..816305698a0e4 --- /dev/null +++ b/tests/tools/test_snapshot_authority_isolation.py @@ -0,0 +1,88 @@ +"""Invocation identity/authority must win over current and legacy snapshots.""" + +from concurrent.futures import ThreadPoolExecutor + +from pathlib import Path +import json +import shlex +import sys +import threading + +import pytest + +from agent.delegation_context import delegated_child_context, DELEGATED_CHILD_ENV_MARKER +from gateway.session_context import scoped_current_session_id +from tools.environments.local import LocalEnvironment + + +@pytest.fixture +def shell(tmp_path, monkeypatch): + # Never source or rewrite a live snapshot, nor read the user's shell rc. + monkeypatch.setattr(LocalEnvironment, "get_temp_dir", lambda self: str(tmp_path)) + env = LocalEnvironment(cwd=str(tmp_path), timeout=30) + Path(env._snapshot_path).write_text("export ORDINARY_SHELL_STATE=preserved\n") + env._snapshot_ready = True + try: + yield env + finally: + env.cleanup() + + +def run_as(shell, child, sid, barrier=None): + with delegated_child_context(sid) if child else scoped_current_session_id(sid): + if barrier: + barrier.wait(timeout=10) + # Actual subprocess guard, not a locally reimplemented permission rule. + code = ( + "import os,json; from argparse import Namespace; " + "from hermes_cli.kanban import _is_delegated_child_cli_mutation as denied; " + "print(json.dumps({'marker':os.getenv('HERMES_DELEGATED_CHILD_CONTEXT')," + "'sid':os.getenv('HERMES_SESSION_ID')," + "'state':os.getenv('ORDINARY_SHELL_STATE')," + "'denied':denied(Namespace(kanban_action='comment'))}))" + ) + result = shell.execute(f"{shlex.quote(sys.executable)} -c {shlex.quote(code)}") + assert result["returncode"] == 0, result + return json.loads(result["output"].strip()) + + +def assert_identity(result, child, sid): + assert result["marker"] == ("1" if child else None), result + assert result["denied"] is child, result + assert result["sid"] == sid, result + assert result["state"] == "preserved", result + + +@pytest.mark.parametrize("order", [(True, False), (False, True)]) +def test_sequential_authority_is_invocation_local(shell, order): + for index, child in enumerate(order): + sid = f"session-{index}" + assert_identity(run_as(shell, child, sid), child, sid) + snapshot = Path(shell._snapshot_path).read_text() + assert DELEGATED_CHILD_ENV_MARKER not in snapshot + assert "HERMES_SESSION_ID" not in snapshot + + +@pytest.mark.parametrize("child", [False, True]) +def test_legacy_snapshot_cannot_override_incoming_authority(shell, child): + with open(shell._snapshot_path, "a") as stream: + stream.write( + f'export HERMES_DELEGATED_CHILD_CONTEXT={"" if child else "1"}\n' + 'export HERMES_SESSION_ID=stale-session\n' + ) + assert_identity(run_as(shell, child, "current-session"), child, "current-session") + + +def test_concurrent_child_parent_share_snapshot_not_authority(shell): + barrier = threading.Barrier(2) + with ThreadPoolExecutor(max_workers=2) as pool: + parent = pool.submit(run_as, shell, False, "parent", barrier) + child = pool.submit(run_as, shell, True, "child", barrier) + assert_identity(parent.result(timeout=30), False, "parent") + assert_identity(child.result(timeout=30), True, "child") + + +def test_inherited_child_lineage_remains_denied(shell, monkeypatch): + # A real delegated subprocess inherits the marker rather than ContextVars. + monkeypatch.setenv(DELEGATED_CHILD_ENV_MARKER, "1") + assert_identity(run_as(shell, False, "inherited-child"), True, "inherited-child") diff --git a/tools/environments/base.py b/tools/environments/base.py index dbdb874240198..3590d11ca5fc3 100644 --- a/tools/environments/base.py +++ b/tools/environments/base.py @@ -529,11 +529,19 @@ def _cwd_marker(session_id: str) -> str: # as the Python-side contract for the exclusion set; the dump path unsets by # name/prefix instead of grepping declare lines (see below / issue #71296). _SNAPSHOT_EXCLUDED_ENV_REGEX = ( - "^declare -x (HERMES_SESSION_|HERMES_UI_SESSION_ID|HERMES_CRON_AUTO_DELIVER_|HERMES_CRON_SESSION)" + "^declare -x (HERMES_SESSION_|HERMES_UI_SESSION_ID|HERMES_CRON_AUTO_DELIVER_|HERMES_CRON_SESSION|HERMES_DELEGATED_CHILD_CONTEXT|HERMES_KANBAN_)" ) _SHELL_ENV_NAME_RE = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$") +def _snapshot_invocation_env_names() -> tuple[str, ...]: + """Canonical invocation identity/authority, never persistent shell state.""" + from agent.delegation_context import DELEGATED_CHILD_ENV_MARKER, KANBAN_ENV_KEYS + from gateway.session_context import _VAR_MAP + + return (*_VAR_MAP, DELEGATED_CHILD_ENV_MARKER, *KANBAN_ENV_KEYS, "AI_AGENT", "HERMES_AGENT") + + def _export_dump_excluding_session_vars( tmp_path: str, excluded_names: Iterable[str] = (), @@ -564,7 +572,7 @@ def _export_dump_excluding_session_vars( # Quote caller-provided names so malformed configuration can never become # shell syntax. Valid environment names remain unquoted by shlex.quote(). safe_names = { - name for name in excluded_names + name for name in (*_snapshot_invocation_env_names(), *excluded_names) if isinstance(name, str) and name } extra_unset = " ".join(shlex.quote(name) for name in sorted(safe_names)) @@ -572,7 +580,7 @@ def _export_dump_excluding_session_vars( extra_unset = f" {extra_unset}" return ( "{ ( " - "unset ${!HERMES_SESSION_*} ${!HERMES_CRON_AUTO_DELIVER_*} " + "unset ${!HERMES_SESSION_*} ${!HERMES_CRON_AUTO_DELIVER_*} ${!HERMES_KANBAN_*} " # AI_AGENT / HERMES_AGENT are per-command attribution markers # (re-exported by every _wrap_command with outer-harness-preserving # ${VAR:-default} semantics). Persisting them into the snapshot @@ -874,7 +882,11 @@ def _wrap_command(self, command: str, cwd: str) -> str: # Values stay in environment memory and never enter the shell command # string, so secrets are not exposed through process arguments/logs. saved_names: list[tuple[str, str, str]] = [] - for name in passthrough_names: + # The incoming process env has already been resolved by the canonical + # session/delegation bridge. A legacy snapshot must not override it, + # including when an authority marker was deliberately absent. Keep + # this per-command, not on the shared Environment or in os.environ. + for name in sorted(set(passthrough_names) | set(_snapshot_invocation_env_names())): marker = f"_HERMES_RUNTIME_PASSTHROUGH_{name}" present = f"{marker}_PRESENT" value = f"{marker}_VALUE" diff --git a/website/docs/user-guide/features/goals.md b/website/docs/user-guide/features/goals.md index b4a9f31585840..e51a70590e152 100644 --- a/website/docs/user-guide/features/goals.md +++ b/website/docs/user-guide/features/goals.md @@ -141,7 +141,7 @@ How it works, each turn: 1. **Gates run before the judge.** If any gate fails, the judge is *not called* — a red gate is deterministic evidence the goal isn't done. The gate's exit code and output tail (last ~3 KB) become the continuation prompt, so the agent iterates against the actual failure instead of a vibe. 2. **All gates pass → normal judging.** The LLM judge then decides done/continue/wait exactly as before. -3. **Unchanged workspace → no re-run.** If a gate failed and nothing changed in the workspace since (tracked via a git fingerprint of HEAD + working-tree status), the gate is not re-run — the recorded failure is replayed and the attempt count advances. A stuck agent can't burn wall-clock re-running an identical red suite. Outside a git repo, gates simply always re-run. +3. **Fresh evidence on every check.** Gates always re-run, even when git status and HEAD are unchanged: an already-dirty file, an external dependency, or a different worktree targeted by the command may have changed. Recorded failures are diagnostic history, never a substitute for execution. Retry limits, turn budgets, and per-command timeouts still bound repeated failures. 4. **Retries are bounded.** Each gate defaults to 3 retries and a 5-minute timeout. When a gate exhausts its retries the goal auto-pauses (like the turn budget) with a message telling you to fix it manually, remove the gate, or `/goal resume`. Gates persist with the goal in `SessionDB.state_meta` (they survive `/resume` and context compression), and gate management (`/goal gate …`) is safe mid-run on the gateway — gates only run at turn boundary.