Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -81,6 +81,8 @@

### Fixed

- **Agent updates no longer report a false failure when the gateway restart is transient — and never mask a real failure.** During an agent update, the launchd/gateway restart subprocess can briefly exit non-zero while the old process is being replaced; that was surfaced as a whole-update failure. The updater now confirms the outcome by the actual gateway PID for the exact profile being updated (retrying the restart once), so a genuine process handoff is reported as success — while a real restart failure, an unknown/ambient PID, a wrong-profile confirmation, or a compatibility-wrapped status helper that can't confirm the specific PID all fail closed (never masked as success). Restart targeting and the PID check are pinned to the update's profile (root aliases normalized), and malformed profile names are rejected before launch. Thanks @franksong2702. (#6054, #6045)

- **Gateway provider errors are reported accurately, and your prompt survives a failed turn.** On gateway-backed sessions, a terminal provider error (bad model id, exhausted credentials, etc.) mid-turn is now classified and shown as the real error instead of a generic empty turn, the error card is bound to the correct session, and the in-flight user prompt plus any partial output is persisted so it survives a reload — even when you'd just re-sent an identical prompt (e.g. "continue"). Thanks @rodboev. (#5969, #5940)

- **"Fork from here" works during an active response.** The fork button was silently dead while the agent was generating — clicking it did nothing. You can now fork from any past (already-committed) message while a response streams; only the currently-streaming message and the just-sent (not-yet-committed) prompt are blocked, with a clear toast. Thanks @kertisalex. (#5994, #5993)
Expand Down
150 changes: 150 additions & 0 deletions api/agent_health.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,7 @@
from __future__ import annotations

import importlib
import inspect
import json
import os
import threading
Expand Down Expand Up @@ -251,6 +252,155 @@ def _gateway_running_pid(gateway_status: Any, pid_path: Path | None) -> int | No
return get_running_pid()


def _callable_code_declares_parameter(get_running_pid: Any, name: str) -> bool:
"""Return True when a Python callable's own code declares *name*."""
target = getattr(get_running_pid, "__func__", get_running_pid)
code = getattr(target, "__code__", None)
if code is None:
return False
declared_count = int(getattr(code, "co_argcount", 0)) + int(
getattr(code, "co_kwonlyargcount", 0)
)
return name in code.co_varnames[:declared_count]


def _declared_pid_path_parameter(
get_running_pid: Any,
) -> tuple[inspect.Parameter, int, list[inspect.Parameter]] | None:
try:
signature = inspect.signature(get_running_pid, follow_wrapped=False)
except (TypeError, ValueError):
return None
params = list(signature.parameters.values())
for idx, param in enumerate(params):
if param.kind not in (
inspect.Parameter.POSITIONAL_ONLY,
inspect.Parameter.POSITIONAL_OR_KEYWORD,
inspect.Parameter.KEYWORD_ONLY,
):
continue
name = str(param.name or "").lower()
if (
"path" in name
and "pid" in name
and _callable_code_declares_parameter(get_running_pid, param.name)
):
positional_index = sum(
1
for earlier in params[:idx]
if earlier.kind
in (
inspect.Parameter.POSITIONAL_ONLY,
inspect.Parameter.POSITIONAL_OR_KEYWORD,
)
)
return param, positional_index, params
return None


def _positional_only_pid_args(
params: list[inspect.Parameter],
positional_index: int,
pid_path: Path,
) -> list[Any] | None:
positional_params = [
param
for param in params
if param.kind
in (
inspect.Parameter.POSITIONAL_ONLY,
inspect.Parameter.POSITIONAL_OR_KEYWORD,
)
]
if positional_index >= len(positional_params):
return None
args: list[Any] = []
for param in positional_params[:positional_index]:
if param.default is inspect.Parameter.empty:
return None
args.append(param.default)
args.append(pid_path)
return args


def _accepts_cleanup_stale_keyword(params: list[inspect.Parameter]) -> bool:
return any(
param.name == "cleanup_stale"
and param.kind
in (
inspect.Parameter.KEYWORD_ONLY,
inspect.Parameter.POSITIONAL_OR_KEYWORD,
)
for param in params
)


def _gateway_running_pid_strict_path(gateway_status: Any, pid_path: Path) -> int | None:
"""Read a PID from an explicit path only; never fall back to ambient state."""
get_running_pid = gateway_status.get_running_pid
pid_path_match = _declared_pid_path_parameter(get_running_pid)
if pid_path_match is None:
return None
pid_path_param, positional_index, params = pid_path_match

if pid_path_param.kind is inspect.Parameter.KEYWORD_ONLY:
kwargs = {pid_path_param.name: pid_path}
try:
return get_running_pid(**kwargs, cleanup_stale=False)
except TypeError:
try:
return get_running_pid(**kwargs)
except TypeError:
return None

if pid_path_param.kind is inspect.Parameter.POSITIONAL_OR_KEYWORD:
kwargs = {pid_path_param.name: pid_path}
try:
return get_running_pid(**kwargs, cleanup_stale=False)
except TypeError:
try:
return get_running_pid(**kwargs)
except TypeError:
return None

if pid_path_param.kind is inspect.Parameter.POSITIONAL_ONLY:
args = _positional_only_pid_args(params, positional_index, pid_path)
if args is None:
return None
if _accepts_cleanup_stale_keyword(params):
try:
return get_running_pid(*args, cleanup_stale=False)
except TypeError:
pass
try:
return get_running_pid(*args)
except TypeError:
return None
return None


def get_active_profile_gateway_running_pid(profile: str | None = None) -> int | None:
"""Return the confirmed active-profile PID without state fallbacks.

Update recovery needs process identity, not just a recent ``running`` state:
an old gateway can remain alive when the restart CLI fails before executing.
Use the same active-profile home targeted by the restart helper rather than
the default-root health path. Keep this fail-closed when status is unavailable.
"""
try:
from api.profiles import get_active_hermes_home, get_hermes_home_for_profile

gateway_status = _gateway_status_module()
if profile is None:
gateway_home = get_active_hermes_home()
else:
gateway_home = get_hermes_home_for_profile(profile)
gateway_pid_path = Path(gateway_home) / _GATEWAY_PID_FILE
return _gateway_running_pid_strict_path(gateway_status, gateway_pid_path)
except Exception:
return None


def _runtime_detail_subset(runtime_status: dict[str, Any] | None) -> dict[str, Any]:
"""Return only non-sensitive runtime fields for the browser.

Expand Down
57 changes: 49 additions & 8 deletions api/gateway_restart.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,7 +10,13 @@
import threading
from pathlib import Path

from api.profiles import get_active_hermes_home
from api.profiles import (
_PROFILE_ID_RE,
_is_root_profile,
get_active_hermes_home,
get_active_profile_name,
get_hermes_home_for_profile,
)

logger = logging.getLogger(__name__)

Expand Down Expand Up @@ -46,8 +52,31 @@ def _release_lock() -> None:
pass


def _gateway_restart_profile_context(profile: str | None = None) -> tuple[Path, str | None]:
"""Return the HERMES_HOME and CLI profile arg for a gateway restart."""
if profile is None:
raw_profile = str(get_active_profile_name() or "default").strip()
active_home = Path(get_active_hermes_home())
else:
raw_profile = str(profile or "")
if not raw_profile or not _PROFILE_ID_RE.fullmatch(raw_profile):
raise ValueError(f"Invalid profile for gateway restart: {profile!r}")
active_home = Path(get_hermes_home_for_profile(raw_profile))

if (
raw_profile == "default"
and active_home.name == "default"
and active_home.parent.name == "profiles"
):
return active_home, None
if not raw_profile or not _PROFILE_ID_RE.fullmatch(raw_profile) or _is_root_profile(raw_profile):
return active_home, "default"
return active_home, raw_profile


def restart_active_profile_gateway(
*,
profile: str | None = None,
quick_timeout_seconds: float = 2.0,
background_wait_seconds: float = 240.0,
) -> dict:
Expand All @@ -66,18 +95,30 @@ def restart_active_profile_gateway(
}

try:
active_home = get_active_hermes_home()
active_home, cli_profile = _gateway_restart_profile_context(profile)
env = os.environ.copy()
env["HERMES_HOME"] = str(active_home)
hermes_cmd = _resolve_hermes_command()
cmd = [hermes_cmd]
if cli_profile is not None:
cmd.extend(["--profile", cli_profile])
cmd.extend(["gateway", "restart"])

logger.info(
"Restarting gateway service via CLI command: %s gateway restart (HERMES_HOME=%s)",
hermes_cmd,
active_home,
)
if cli_profile is None:
logger.info(
"Restarting gateway service via CLI command: %s gateway restart (HERMES_HOME=%s)",
hermes_cmd,
active_home,
)
else:
logger.info(
"Restarting gateway service via CLI command: %s --profile %s gateway restart (HERMES_HOME=%s)",
hermes_cmd,
cli_profile,
active_home,
)
proc = subprocess.Popen(
[hermes_cmd, "gateway", "restart"],
cmd,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
Expand Down
58 changes: 56 additions & 2 deletions api/updates.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,9 @@
from pathlib import Path
from urllib.parse import urlparse

from api.agent_health import get_active_profile_gateway_running_pid
from api.gateway_restart import restart_active_profile_gateway
from api.profiles import get_active_profile_name
from api.config import REPO_ROOT, STREAMS, STREAMS_LOCK

logger = logging.getLogger(__name__)
Expand All @@ -42,6 +44,7 @@
_check_in_progress = False
_apply_lock = threading.Lock() # prevents concurrent stash/pull/pop on same repo
CACHE_TTL = 1800 # 30 minutes
_AGENT_GATEWAY_RESTART_RETRY_DELAY_S = 1.0
_GIT_DIAGNOSTIC_MAX_CHARS = 300
_CREDENTIAL_IN_URL_RE = re.compile(r"([a-zA-Z][a-zA-Z0-9+.-]*://)([^/@\s'\"]+)@")
_GITHUB_TOKEN_RE = re.compile(r"\b(?:gh[pousr]_[A-Za-z0-9_]{20,}|github_pat_[A-Za-z0-9_]{20,})\b")
Expand Down Expand Up @@ -1814,11 +1817,62 @@ def _ensure_gateway_restart_for_agent_update() -> tuple[bool, dict]:
- ok is False when restart did not complete and callers must abort success.
- restart_payload contains helper status fields for response shaping.
"""
restart_result = restart_active_profile_gateway()
target_profile = str(get_active_profile_name() or "default").strip() or "default"
gateway_pid_before_restart = get_active_profile_gateway_running_pid(profile=target_profile)
restart_result = restart_active_profile_gateway(profile=target_profile)
status = str(restart_result.get("status") or "")
if status in {"completed", "in_progress"}:
return True, restart_result
return False, restart_result
if status != "failed":
return False, restart_result

# launchd can briefly fail to spawn the replacement gateway while it is
# rotating the supervised process (#6045). Retry exactly once after a
# bounded delay so an already-applied Agent update is not reported as a
# complete failure because of that transient process handoff.
time.sleep(_AGENT_GATEWAY_RESTART_RETRY_DELAY_S)
retry_result = restart_active_profile_gateway(profile=target_profile)
retry_status = str(retry_result.get("status") or "")
if retry_status in {"completed", "in_progress"}:
return True, {
**retry_result,
"retry_attempted": True,
"initial_failure": restart_result.get("message"),
}
if retry_status != "failed":
return False, {
**retry_result,
"retry_attempted": True,
"initial_failure": restart_result.get("message"),
}

# A restart command can still exit non-zero after launchd has recovered the
# service. Only accept that recovery when the confirmed local PID changed;
# a merely-alive old gateway has not loaded the updated Agent checkout.
time.sleep(_AGENT_GATEWAY_RESTART_RETRY_DELAY_S)
gateway_pid_after_retry = get_active_profile_gateway_running_pid(profile=target_profile)
if (
gateway_pid_before_restart is not None
and gateway_pid_after_retry is not None
and gateway_pid_after_retry != gateway_pid_before_restart
):
return True, {
"status": "completed",
"message": "Gateway service recovered after a transient restart failure",
"retry_attempted": True,
"process_replaced": True,
"initial_failure": restart_result.get("message"),
"retry_failure": retry_result.get("message"),
}

initial_message = str(restart_result.get("message") or "Restart failed")
retry_message = str(retry_result.get("message") or "retry did not complete")
return False, {
**retry_result,
"message": f"{initial_message}; recovery retry did not complete: {retry_message}",
"retry_attempted": True,
"initial_failure": restart_result.get("message"),
}


def _agent_gateway_restart_failure_message(target: str, restart_result: dict) -> str:
Expand Down
Loading
Loading