diff --git a/agent/redact.py b/agent/redact.py index 9e6800847b7b1..d6e82803dd3a3 100644 --- a/agent/redact.py +++ b/agent/redact.py @@ -67,7 +67,10 @@ r"pypi-[A-Za-z0-9_-]{10,}", # PyPI API token r"dop_v1_[A-Za-z0-9]{10,}", # DigitalOcean PAT r"doo_v1_[A-Za-z0-9]{10,}", # DigitalOcean OAuth - r"am_[A-Za-z0-9_-]{10,}", # AgentMail API key + # AgentMail API key: ``am_`` / ``am_org_`` + an opaque alphanumeric body. The body has no ``_``/``-``, + # which is what separates it from ``am_example_identifier_123`` (#10983); public docs pin only the + # prefix, so the charset stays broad and the length floor does the discriminating. + r"am_(?:org_)?[A-Za-z0-9]{20,}", r"sk_[A-Za-z0-9_]{10,}", # ElevenLabs TTS key (sk_ underscore, not sk- dash) r"tvly-[A-Za-z0-9]{10,}", # Tavily search API key r"exa_[A-Za-z0-9]{10,}", # Exa search API key diff --git a/apps/desktop/src/api/messaging.ts b/apps/desktop/src/api/messaging.ts index b49bdc93b8288..e55137df11bf5 100644 --- a/apps/desktop/src/api/messaging.ts +++ b/apps/desktop/src/api/messaging.ts @@ -19,12 +19,20 @@ export function getMessagingPlatforms(profile?: null | string): Promise { - return hermesApi<{ ok: boolean; platform: string }>({ +): Promise { + return hermesApi({ ...profileScoped(profile), path: `/api/messaging/platforms/${encodeURIComponent(platformId)}`, method: 'PUT', diff --git a/apps/desktop/src/app/messaging/index.tsx b/apps/desktop/src/app/messaging/index.tsx index 1611dcf3c2089..b159de123ee4c 100644 --- a/apps/desktop/src/app/messaging/index.tsx +++ b/apps/desktop/src/app/messaging/index.tsx @@ -171,6 +171,18 @@ export function MessagingView({ setStatusbarItemGroup: _setStatusbarItemGroup, . [m, scopeProfile] ) + // A multiplexed named profile is re-served from its new config at once (`hot_served`): no restart + // banner; re-read status once the adapter had a moment to connect. Anything else keeps the restart + // action on the success toast. + const settleAfterUpdate = useCallback( + (hotServed: boolean | undefined) => { + if (hotServed) { + window.setTimeout(() => void refreshPlatforms(true), 4000) + } + }, + [refreshPlatforms] + ) + // Pairing has its own signal. platforms.changed tracks connect/disconnect // health via gateway_state.json, which a new pairing request never moves — // riding it would leave a pending row invisible until something unrelated @@ -297,7 +309,7 @@ export function MessagingView({ setStatusbarItemGroup: _setStatusbarItemGroup, . setSaving(`enabled:${platform.id}`) try { - await updateMessagingPlatform(platform.id, { enabled }, scopeProfile) + const result = await updateMessagingPlatform(platform.id, { enabled }, scopeProfile) setPlatforms( current => current?.map(row => @@ -310,11 +322,12 @@ export function MessagingView({ setStatusbarItemGroup: _setStatusbarItemGroup, . : row ) ?? current ) + settleAfterUpdate(result.hot_served) notify({ kind: 'success', title: enabled ? m.platformEnabled(platform.name) : m.platformDisabled(platform.name), - message: m.restartToApply, - action: restartGatewayAction + message: result.hot_served ? m.appliedLive : m.restartToApply, + action: result.hot_served ? undefined : restartGatewayAction }) } catch (err) { notifyError(err, m.failedUpdate(platform.name)) @@ -333,14 +346,15 @@ export function MessagingView({ setStatusbarItemGroup: _setStatusbarItemGroup, . setSaving(`env:${platform.id}`) try { - await updateMessagingPlatform(platform.id, { env }, scopeProfile) + const result = await updateMessagingPlatform(platform.id, { env }, scopeProfile) setEdits(current => ({ ...current, [platform.id]: {} })) await refreshPlatforms() + settleAfterUpdate(result.hot_served) notify({ kind: 'success', title: m.setupSaved(platform.name), - message: m.restartToReconnect, - action: restartGatewayAction + message: result.hot_served ? m.connectingLive : m.restartToReconnect, + action: result.hot_served ? undefined : restartGatewayAction }) } catch (err) { notifyError(err, m.failedSave(platform.name)) @@ -353,7 +367,7 @@ export function MessagingView({ setStatusbarItemGroup: _setStatusbarItemGroup, . setSaving(`clear:${key}`) try { - await updateMessagingPlatform(platform.id, { clear_env: [key] }, scopeProfile) + const result = await updateMessagingPlatform(platform.id, { clear_env: [key] }, scopeProfile) setEdits(current => ({ ...current, [platform.id]: { @@ -362,6 +376,7 @@ export function MessagingView({ setStatusbarItemGroup: _setStatusbarItemGroup, . } })) await refreshPlatforms() + settleAfterUpdate(result.hot_served) notify({ kind: 'success', title: m.keyCleared(key), message: m.setupUpdated(platform.name) }) } catch (err) { notifyError(err, m.failedClear(key)) diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index a7ab7c54ce5ff..6b74ba1e6954e 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -1411,6 +1411,8 @@ export const ar = defineLocale({ restartToApply: 'أعد التشغيل لتطبيق التغييرات.', setupSaved: name => `تم حفظ إعداد ${name}`, restartToReconnect: 'أعد التشغيل لإعادة الاتصال.', + appliedLive: 'تم التطبيق على البوابة قيد التشغيل.', + connectingLive: 'البوابة قيد التشغيل تتصل باستخدام بيانات الاعتماد الجديدة.', keyCleared: key => `تم مسح ${key}`, setupUpdated: name => `تم تحديث إعداد ${name}`, failedUpdate: name => `فشل تحديث ${name}`, diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index acc480147c14a..7441cf3048a39 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -1962,6 +1962,8 @@ export const en: Translations = { restartToApply: 'This change takes effect after a gateway restart.', setupSaved: name => `${name} setup saved`, restartToReconnect: 'New credentials take effect after a gateway restart.', + appliedLive: 'Applied to the running gateway.', + connectingLive: 'The running gateway is connecting with the new credentials.', keyCleared: key => `${key} cleared`, setupUpdated: name => `${name} setup was updated.`, failedUpdate: name => `Failed to update ${name}`, diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 0174930026275..e70e742953a59 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -1721,6 +1721,8 @@ export const ja = defineLocale({ restartToApply: 'この変更はゲートウェイの再起動後に有効になります。', setupSaved: name => `${name} の設定を保存しました`, restartToReconnect: '新しい認証情報はゲートウェイの再起動後に有効になります。', + appliedLive: '実行中のゲートウェイに適用されました。', + connectingLive: '実行中のゲートウェイが新しい認証情報で接続しています。', keyCleared: key => `${key} をクリアしました`, setupUpdated: name => `${name} の設定が更新されました。`, failedUpdate: name => `${name} の更新に失敗しました`, diff --git a/apps/desktop/src/i18n/ru.ts b/apps/desktop/src/i18n/ru.ts index e21ba2de636a3..66c8587207f2d 100644 --- a/apps/desktop/src/i18n/ru.ts +++ b/apps/desktop/src/i18n/ru.ts @@ -2036,6 +2036,8 @@ export const ru = defineLocale({ restartToApply: 'Это изменение вступит в силу после перезапуска шлюза.', setupSaved: name => `Настройка ${name} сохранена`, restartToReconnect: 'Новые учётные данные вступят в силу после перезапуска шлюза.', + appliedLive: 'Применено к работающему шлюзу.', + connectingLive: 'Работающий шлюз подключается с новыми учётными данными.', keyCleared: key => `${key} очищено`, setupUpdated: name => `Настройка ${name} обновлена.`, failedUpdate: name => `Не удалось обновить ${name}`, diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 9d9b3f68695fe..ec987882c102b 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -1743,6 +1743,8 @@ export interface Translations { restartToApply: string setupSaved: (name: string) => string restartToReconnect: string + appliedLive: string + connectingLive: string keyCleared: (key: string) => string setupUpdated: (name: string) => string failedUpdate: (name: string) => string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 71ef5e4c1211a..a47b37a923390 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -1665,6 +1665,8 @@ export const zhHant = defineLocale({ restartToApply: '此變更將在閘道重新啟動後生效。', setupSaved: name => `${name} 設定已儲存`, restartToReconnect: '新憑證將在閘道重新啟動後生效。', + appliedLive: '已套用到執行中的閘道。', + connectingLive: '執行中的閘道正在使用新憑證連線。', keyCleared: key => `${key} 已清除`, setupUpdated: name => `${name} 設定已更新。`, failedUpdate: name => `更新 ${name} 失敗`, diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 3ad3c6501462d..901a8e6c7e92f 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -2130,6 +2130,8 @@ export const zh: Translations = { restartToApply: '此更改将在网关重启后生效。', setupSaved: name => `${name} 设置已保存`, restartToReconnect: '新凭据将在网关重启后生效。', + appliedLive: '已应用到正在运行的网关。', + connectingLive: '正在运行的网关正在使用新凭据连接。', keyCleared: key => `${key} 已清除`, setupUpdated: name => `${name} 设置已更新。`, failedUpdate: name => `更新 ${name} 失败`, diff --git a/cron/AGENTS.md b/cron/AGENTS.md index 78a13580bf278..a5326aaade16e 100644 --- a/cron/AGENTS.md +++ b/cron/AGENTS.md @@ -39,8 +39,11 @@ zero outside a kanban task (footprint ladder rung 3). notify-*, dispatch, daemon, gc`. Argparse alias dispatch must accept both `list` and `ls` (root). - **Toolset:** `tools/kanban_tools.py` — `kanban_show, kanban_complete, kanban_request_review, kanban_request_changes, kanban_block, kanban_heartbeat, kanban_comment, kanban_create, kanban_link, - kanban_attach, kanban_attach_url, kanban_attachments`; profiles enabling `kanban` outside a - dispatched task also get `kanban_list` and `kanban_unblock` for board routing. + kanban_attach, kanban_attach_url, kanban_attachments`; platforms whose saved selection enables + `kanban` (`hermes tools enable kanban --platform

`; default-off, in `CONFIGURABLE_TOOLSETS`) get + the full set plus `kanban_list`/`kanban_unblock` for board routing. The check_fn reads the schema + build's own selection (`tools/kanban_toolset_context.py`), never the legacy top-level `toolsets` + key alone. - **Dispatcher:** long-lived loop (default 60s) that reclaims stale claims, promotes ready tasks, atomically claims, and spawns assigned profiles. Runs **inside the gateway** by default (`kanban.dispatch_in_gateway: true`). Standalone: `plugins/kanban/systemd/hermes-kanban-dispatcher.service`. diff --git a/cron/scheduler_provider.py b/cron/scheduler_provider.py index e008e3223da38..43e40f7ee2e71 100644 --- a/cron/scheduler_provider.py +++ b/cron/scheduler_provider.py @@ -66,6 +66,14 @@ def _existing_profile_homes(profile_homes: list) -> list: profile's home untouched, which is the correct invariant: a home that does not exist cannot hold jobs to fire. """ + if callable(profile_homes): + # Live enumerator (multiplex gateway): a profile created after startup is ticked without a + # restart; a raising enumerator keeps this cycle at zero homes rather than killing the ticker. + try: + profile_homes = list(profile_homes()) + except Exception: + logger.warning("cron profile enumeration failed; skipping this cycle", exc_info=True) + return [] return [entry for entry in profile_homes if Path(_profile_entry(entry)[1]).is_dir()] @@ -393,7 +401,7 @@ def start( # jobs actually fire instead of languishing in a store no ticker owns (#69377). Without this, only # the process-global HERMES_HOME (the default profile) is ticked. Heartbeats and recovery are also # scoped per profile so `hermes cron status` reflects liveness for every profile independently. - if profile_homes: + if profile_homes is not None and (callable(profile_homes) or profile_homes): self._start_multiplex( stop_event, profile_homes=profile_homes, adapters=adapters, loop=loop, interval=interval, can_dispatch=can_dispatch, profile_adapters=profile_adapters, @@ -462,10 +470,12 @@ def _start_multiplex( ) from cron.jobs import clear_ticker_error, record_ticker_error, record_ticker_heartbeat + initial_homes = _existing_profile_homes(profile_homes) logger.info( - "Multiplex cron scheduler started for %d profile(s): %s", - len(profile_homes), - [p[0] if isinstance(p, tuple) else p for p in profile_homes], + "Multiplex cron scheduler started for %d profile(s): %s%s", + len(initial_homes), + [p[0] if isinstance(p, tuple) else p for p in initial_homes], + " (re-enumerated every cycle)" if callable(profile_homes) else "", ) def tick_adapters_for(profile_name): @@ -482,7 +492,7 @@ def tick_adapters_for(profile_name): # Recovery + heartbeat per profile; one broken store must not abort startup for the others. # A profile may have been deleted since this snapshot was taken; never recreate a deleted home's # cron workspace via the heartbeat below (#47368). - for entry in _existing_profile_homes(profile_homes): + for entry in initial_homes: _, home = _profile_entry(entry) try: with _profile_cron_scope(home): diff --git a/gateway/config.py b/gateway/config.py index de05751a4dd75..6b445caa550df 100644 --- a/gateway/config.py +++ b/gateway/config.py @@ -8,7 +8,7 @@ import os from pathlib import Path from dataclasses import asdict, dataclass, field, fields, is_dataclass -from typing import Dict, List, Optional, Any, Callable +from typing import Dict, List, Optional, Any, Callable, Mapping from enum import Enum from hermes_cli.config import get_hermes_home @@ -73,7 +73,7 @@ def _normalize_multiplex_profile_allowlist(value: Any) -> Optional[List[str]]: return normalized -def _env_multiplex_profiles_override() -> "bool | None": +def _env_multiplex_profiles_override(environ: Optional[Mapping[str, str]] = None) -> "bool | None": """GATEWAY_MULTIPLEX_PROFILES operator override: True/False for a recognized token. ``None`` when unset, blank, or unrecognized so the caller keeps the config.yaml @@ -81,7 +81,7 @@ def _env_multiplex_profiles_override() -> "bool | None": a provisioned-but-unpopulated Fly secret arrives as ``""`` and must NOT shadow a config.yaml opt-in. """ - raw = os.getenv("GATEWAY_MULTIPLEX_PROFILES") + raw = (environ or os.environ).get("GATEWAY_MULTIPLEX_PROFILES") if not (raw or "").strip(): return None parsed = _bool_token(raw) diff --git a/gateway/config_env.py b/gateway/config_env.py index a1e80ec0be27e..44cad1fb4c8e8 100644 --- a/gateway/config_env.py +++ b/gateway/config_env.py @@ -21,6 +21,7 @@ PlatformConfig, _getenv_str, _has_usable_api_server_key, + platform_binds_port, ) from utils import is_truthy_value @@ -168,6 +169,15 @@ def _env_reply_mode(config: GatewayConfig, platform: Platform, env: str) -> None config.platforms.setdefault(platform, PlatformConfig()).reply_to_mode = mode +def _loading_secondary_under_multiplexer() -> bool: + """True while a multiplexer loads a NON-default profile's config (``_profile_runtime_scope`` sets the + home override; the runner sets the multiplex flag). Same signal ``gateway.config`` uses for scoped reads.""" + from agent.secret_scope import is_multiplex_active + from hermes_constants import get_hermes_home_override, profile_name_for_home + override = get_hermes_home_override() + return bool(override) and is_multiplex_active() and profile_name_for_home(override) != "default" + + def _enable_from_env( config: GatewayConfig, platform: Platform, *, pop_marker: bool = False, warn: bool = True ) -> PlatformConfig: @@ -184,7 +194,13 @@ def _enable_from_env( explicit = extra.pop("_enabled_explicit", False) if pop_marker else extra.get("_enabled_explicit", False) if platform_config.enabled: return platform_config - if not explicit: + if not explicit and not ( + platform_binds_port(platform.value, extra) and _loading_secondary_under_multiplexer() + ): + # A secondary's port-binding credential (the docs require API_SERVER_KEY in its .env for + # /p// auth) must not turn into listener intent: the default profile owns the one + # shared listener and ``_load_secondary_profile_config`` skips the WHOLE profile for it (#100397). + # The credential itself still lands in ``extra`` for the shared adapter to authenticate with. platform_config.enabled = True elif warn: _warn_explicit_disable_beats_env(platform) diff --git a/gateway/control_socket.py b/gateway/control_socket.py index bac58ce6b2187..b45608a9dd0fa 100644 --- a/gateway/control_socket.py +++ b/gateway/control_socket.py @@ -340,3 +340,11 @@ def pause_gateway_for_update(home: Path, *, timeout: float = _DEFAULT_CLIENT_TIM Step 2 of the socket migration (#92091). """ return query_gateway_control(home, "pause-for-update", timeout=timeout) + + +def rescan_gateway_profiles(home: Path, *, timeout: float = 8.0) -> Optional[dict[str, Any]]: + """Ask the multiplexer serving ``home`` to reconcile ``profiles/`` now (hot-serve a created profile, + unroute a deleted one). Returns its ``{"served_profiles", "added", "removed", ...}`` answer, or None + when no gateway answers / the gateway predates the verb — callers then rely on the periodic rescan + (or the restart reminder).""" + return query_gateway_control(home, "rescan-profiles", timeout=timeout) diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 8bd6ed54782b0..3e278204945b2 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -1092,6 +1092,8 @@ class APIServerAdapter(OpenAICompatRoutesMixin, BasePlatformAdapter): # Stateless request/response (``send()`` is a stub): async-delivery tools must not promise # delivery here, and a resumed turn completes the work rather than asking. supports_async_delivery: bool = False + # ``/p//v1/...`` on the shared listener (``_make_profile_prefix_middleware``). + serves_profile_prefix: bool = True # Same statelessness applies to the startup auto-resume prompt: no client is waiting to answer "session # restored — what next?", so a resumed turn should complete the interrupted work rather than acknowledge # (#57056). diff --git a/gateway/platforms/base.py b/gateway/platforms/base.py index 1c6cabb26b99b..a71a915cc8384 100644 --- a/gateway/platforms/base.py +++ b/gateway/platforms/base.py @@ -1822,6 +1822,11 @@ def set_status_text(self, chat_id: str, text: Optional[str]) -> None: # answer, and an acknowledgement would silently abandon the task (#57056). Read generically via # ``getattr(adapter, "interactive_resume", True)`` — no per-platform branching at the call site. interactive_resume: bool = True + # Port-binding adapter that answers ``/p//...`` for every served profile on the default + # listener under ``gateway.multiplex_profiles``. Declared per adapter (not in a central list) so + # ``hermes gateway migrate`` can tell "URL changes" from "this profile would be skipped" as new + # HTTP-inbound adapters gain the prefix. + serves_profile_prefix: bool = False # Back-reference to the running ``GatewayRunner`` (set by gateway/run.py); ``build_source`` # resolves the inbound profile via ``runner._profile_name_for_source``. gateway_runner = None # type: ignore[assignment] diff --git a/gateway/platforms/bluebubbles.py b/gateway/platforms/bluebubbles.py index 47b83e1bf9cc3..add1e18fe207c 100644 --- a/gateway/platforms/bluebubbles.py +++ b/gateway/platforms/bluebubbles.py @@ -108,6 +108,8 @@ def _ok(): class BlueBubblesAdapter(BasePlatformAdapter): + # Answers /p//... on the default listener for a served secondary (shared_ingress). + serves_profile_prefix: bool = True platform = Platform.BLUEBUBBLES SUPPORTS_MESSAGE_EDITING = False MAX_MESSAGE_LENGTH = MAX_TEXT_LENGTH diff --git a/gateway/platforms/msgraph_webhook.py b/gateway/platforms/msgraph_webhook.py index 326aed18c6060..318330b7dac70 100644 --- a/gateway/platforms/msgraph_webhook.py +++ b/gateway/platforms/msgraph_webhook.py @@ -98,6 +98,8 @@ def _resolve(match: re.Match[str]) -> str: class MSGraphWebhookAdapter(BasePlatformAdapter): """Receive Microsoft Graph change notifications and surface them internally.""" + # Answers /p//... on the default listener for a served secondary (shared_ingress). + serves_profile_prefix: bool = True def __init__(self, config: PlatformConfig): super().__init__(config, Platform.MSGRAPH_WEBHOOK) diff --git a/gateway/platforms/webhook.py b/gateway/platforms/webhook.py index d2f57b4266f1a..5dccb37f8abd4 100644 --- a/gateway/platforms/webhook.py +++ b/gateway/platforms/webhook.py @@ -154,6 +154,8 @@ class WebhookAdapter(BasePlatformAdapter): # The startup auto-resume turn must instruct the model to FINISH the interrupted work instead of # emitting an interactive acknowledgement that abandons the task (#57056). interactive_resume: bool = False + # ``/p//webhooks/`` on the shared listener (``_resolve_request_profile``). + serves_profile_prefix: bool = True def __init__(self, config: PlatformConfig): super().__init__(config, Platform.WEBHOOK) diff --git a/gateway/platforms/whatsapp_cloud.py b/gateway/platforms/whatsapp_cloud.py index 8daf427cb51c7..24503d7cc30b6 100644 --- a/gateway/platforms/whatsapp_cloud.py +++ b/gateway/platforms/whatsapp_cloud.py @@ -155,6 +155,8 @@ def check_whatsapp_cloud_requirements() -> bool: class WhatsAppCloudAdapter(WhatsAppBehaviorMixin, BasePlatformAdapter): """Outbound: Graph ``///messages``; inbound: aiohttp webhook server. The mixin comes first so its ``format_message`` overrides the base one.""" + # Answers /p//... on the default listener for a served secondary (shared_ingress). + serves_profile_prefix: bool = True splits_long_messages = True # send() chunks via truncate_message() diff --git a/gateway/run.py b/gateway/run.py index 83ee1f39412e2..41e8d64ce9d81 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -2111,6 +2111,7 @@ def _load_bridge_config(config_path: Path) -> dict: from gateway.run_inbound import GatewayInboundMixin from gateway.run_goals import GatewayGoalsMixin from gateway.run_agent_cache import GatewayAgentCacheMixin +from gateway.run_profile_reconcile import GatewayProfileReconcileMixin from gateway.platforms.base import ( BasePlatformAdapter, _reply_anchor_for_event, @@ -3272,7 +3273,7 @@ class GatewayRunner( GatewayVoiceMixin, GatewayAdapterLifecycleMixin, GatewayTopicThreadsMixin, GatewayTurnMixin, GatewayShutdownMixin, GatewayBusySessionMixin, GatewayConfigLoadersMixin, GatewayStartupMixin, GatewaySessionWatchersMixin, GatewayNotificationsMixin, GatewayInboundMixin, GatewayGoalsMixin, - GatewayAgentCacheMixin): + GatewayAgentCacheMixin, GatewayProfileReconcileMixin): """Main gateway controller: manages adapter lifecycles, routes messages to/from the agent.""" # Class-level defaults so partial construction in tests doesn't blow up on attribute access. @@ -5050,8 +5051,24 @@ def _request() -> None: "pausing": accepted, "already_stopping": not accepted, "pid": os.getpid(), "drain_timeout": _drain} + def _rescan_profiles_handler() -> dict: + """``hermes profile create/delete`` asks the multiplexer to reconcile ``profiles/`` now + (the watcher also rescans periodically). Runs on the socket executor: marshal onto the loop + and wait briefly so the caller learns whether the profile is served.""" + if not getattr(runner.config, "multiplex_profiles", False): + return {"multiplex": False, "served_profiles": runner.served_profile_names()} + future = asyncio.run_coroutine_threadsafe( + runner.reconcile_served_profiles(reason="control-socket"), _main_loop) + try: + # Bounded: a token-less create reconciles in milliseconds; a credential-add whose adapter + # connect outlasts this keeps running and the caller sees ``pending`` (not an error). + return {"multiplex": True, **future.result(timeout=5.0)} + except concurrent.futures.TimeoutError: + return {"multiplex": True, "pending": True, "served_profiles": runner.served_profile_names()} + _control_server = GatewayControlServer( - verb_handlers={"pause-for-update": _pause_for_update_handler}) + verb_handlers={"pause-for-update": _pause_for_update_handler, + "rescan-profiles": _rescan_profiles_handler}) if not await _control_server.start(): _control_server = None else: @@ -5079,7 +5096,9 @@ def _start_gateway_start_cron_and_housekeeping(runner): try: profile_homes = _multiplex_profile_homes(runner.config) if profile_homes: - cron_start_kwargs["profile_homes"] = profile_homes + # Live enumerator: the ticker re-reads profiles/ every cycle so a profile created while + # the multiplexer runs gets its jobs fired without a restart (hot-serve). + cron_start_kwargs["profile_homes"] = lambda: _multiplex_profile_homes(runner.config) # Per-profile adapters so each profile's cron output goes via its own bot, not the default's. cron_start_kwargs["profile_adapters"] = getattr(runner, "_profile_adapters", None) # runner.adapters belongs to "default"; naming it keeps the ticker from routing a secondary's diff --git a/gateway/run_adapters.py b/gateway/run_adapters.py index 3ea22ad640118..4a661499a1fb5 100644 --- a/gateway/run_adapters.py +++ b/gateway/run_adapters.py @@ -878,6 +878,7 @@ def _record_served_profiles(self, active: str, profile_homes) -> None: from gateway.status import write_runtime_status from gateway.pairing import PairingStore served = [active] + sorted(name for name, _home in profile_homes if name != active) + self._note_served_profiles(profile_homes) for name in served: if name and name not in self.pairing_stores: self.pairing_stores[name] = ( @@ -974,6 +975,11 @@ async def _start_one_profile_adapters( for platform, platform_config in profile_cfg.platforms.items(): if not platform_config.enabled: continue + # Runtime re-scan of a served profile (config/.env changed): only platforms that are not + # already live or queued for reconnect are built — never a second poller on the same bot. + if platform in profile_map or platform in ( + (getattr(self, "_profile_failed_platforms", None) or {}).get(profile_name) or {}): + continue # No credential in THIS profile's scope: an adapter would fan inbound across every such profile. if multiplex and not _platform_has_bot_credential(platform, platform_config): logger.info( diff --git a/gateway/run_profile_reconcile.py b/gateway/run_profile_reconcile.py new file mode 100644 index 0000000000000..94edddab3f0f4 --- /dev/null +++ b/gateway/run_profile_reconcile.py @@ -0,0 +1,229 @@ +"""Hot-serve for ``gateway.multiplex_profiles``: keep the served-profile set in step with ``profiles/`` +while the multiplexer runs, instead of snapshotting it once at boot. + +Three things were start-time snapshots: the secondary adapter set (``_start_secondary_profile_adapters``), +the ``served_profiles`` record in ``gateway_state.json`` (``_record_served_profiles``) and the cron +ticker's ``profile_homes`` list. Everything else (``/p//`` prefixes, profile-route eligibility, +handoff/kanban watchers, shared ingress) already reads ``profiles_to_serve()`` / ``_profile_adapters`` +live, so reconciling those three is enough for a profile created after boot to be served. + +``reconcile_served_profiles`` runs on the loop under one lock, triggered by the ``rescan-profiles`` control +verb (``hermes_cli/profiles.py`` create/delete fire it through the control socket) and by the supervised +``_profile_reconcile_watcher`` every ``_PROFILE_RESCAN_INTERVAL_SECS`` as the safety net. A served profile whose ``config.yaml``/``.env`` +changed since its adapters were last built is re-scanned too: creators make the profile first and add the +bot token afterwards, and without this an adapter-less profile would stay adapter-less forever. +""" +from __future__ import annotations + +import asyncio +import logging +import os +from pathlib import Path +from typing import Any, Dict, Optional + +from gateway.run_shutdown import _log_suppressed + +logger = logging.getLogger(__name__) + +_PROFILE_RESCAN_INTERVAL_SECS = 30.0 +_PROFILE_SIGNATURE_FILES = ("config.yaml", ".env") + + +def profile_serve_signature(home: "Path") -> tuple: + """Cheap change detector for a served profile's credentials/config: (mtime_ns, size) per file.""" + sig = [] + for name in _PROFILE_SIGNATURE_FILES: + try: + st = os.stat(Path(home) / name) + sig.append((st.st_mtime_ns, st.st_size)) + except OSError: + sig.append(None) + return tuple(sig) + + +class GatewayProfileReconcileMixin: + """Runtime reconciliation of the multiplexed served-profile set (hot add / unroute / credential-add).""" + + _served_profile_homes: Optional[Dict[str, "Path"]] = None + _served_profile_signatures: Optional[Dict[str, tuple]] = None + _profile_reconcile_lock: Optional[asyncio.Lock] = None + + # ── state helpers ───────────────────────────────────────────────────────────────────────────── + + def _reconcile_lock(self) -> asyncio.Lock: + if self._profile_reconcile_lock is None: + self._profile_reconcile_lock = asyncio.Lock() + return self._profile_reconcile_lock + + def served_profile_names(self) -> list: + """Profiles this multiplexer currently serves (active first), from the live bookkeeping.""" + homes = self._served_profile_homes or {} + active = getattr(self, "_primary_profile_name", None) or "default" + return ([active] if active in homes or not homes else []) + sorted(n for n in homes if n != active) + + def _note_served_profiles(self, profile_homes) -> None: + """Called by ``_record_served_profiles``: remember the served set and each home's signature.""" + homes = {str(name): Path(home) for name, home in profile_homes} + self._served_profile_homes = homes + sigs = self._served_profile_signatures if isinstance(self._served_profile_signatures, dict) else {} + self._served_profile_signatures = {name: sigs.get(name) or profile_serve_signature(home) + for name, home in homes.items()} + + # ── watcher ─────────────────────────────────────────────────────────────────────────────────── + + async def _profile_reconcile_watcher(self, interval: float = _PROFILE_RESCAN_INTERVAL_SECS) -> None: + """Supervised safety net: rescan ``profiles/`` every ``interval`` seconds (creators signal the + control socket for an immediate rescan). Returns at once, never respawned, when multiplexing is off.""" + if not self._multiplex_on(): + return + while self._running: + await asyncio.sleep(interval) + if not self._running: + return + try: + await self.reconcile_served_profiles(reason="watcher") + except asyncio.CancelledError: + raise + except Exception: + logger.warning("Served-profile reconcile failed; retrying next cycle", exc_info=True) + + # ── reconcile ───────────────────────────────────────────────────────────────────────────────── + + async def reconcile_served_profiles(self, *, reason: str = "request") -> Dict[str, Any]: + """Diff ``profiles/`` against the served set: start adapters for new profiles, tear down and + unroute deleted ones, (re)build adapters for served profiles whose config/.env changed. Other + profiles' adapters are never touched. Returns ``{"added", "removed", "rescanned", "served_profiles"}``.""" + from gateway.run import MultiplexConfigError, _multiplex_profile_homes + result: Dict[str, Any] = {"added": [], "removed": [], "rescanned": [], "reason": reason} + if not self._multiplex_on(): + return {**result, "multiplex": False, "served_profiles": self.served_profile_names()} + if not self._running or self._served_profile_homes is None: + # Startup enumerates profiles/ itself; a rescan before it finishes has nothing to diff against. + return {**result, "pending": True, "served_profiles": self.served_profile_names()} + async with self._reconcile_lock(): + active = getattr(self, "_primary_profile_name", None) or "default" + current = {str(name): Path(home) for name, home in _multiplex_profile_homes(self.config)} + known = dict(self._served_profile_homes or {}) + sigs = self._served_profile_signatures or {} + added = [n for n in current if n not in known and n != active] + removed = [n for n in known if n not in current and n != active] + changed = [n for n in current if n in known and n != active and n not in added + and profile_serve_signature(current[n]) != sigs.get(n)] + if not (added or removed or changed): + return {**result, "served_profiles": self.served_profile_names()} + for name in removed: + await self._unserve_profile(name, known[name]) + result["removed"].append(name) + for name in changed: + await self._reset_profile_adapters_for_rescan(name) + claimed = self._live_resource_claims(active) + for name in added + changed: + try: + connected = await self._start_one_profile_adapters(name, current[name], claimed) + except MultiplexConfigError as exc: + # Boot refuses to run with such a profile; at runtime we park just this profile. + logger.error("[MULTIPLEX] Profile '%s' not served: %s", name, exc) + connected = 0 + except Exception: + logger.error("[MULTIPLEX] Failed to start adapters for profile '%s'", name, exc_info=True) + connected = 0 + sigs[name] = profile_serve_signature(current[name]) + if name in added: + logger.info("[MULTIPLEX] Now serving profile '%s' (%s adapter(s) connected; %s)", name, connected, reason) + result["added"].append(name) + else: + logger.info("[MULTIPLEX] Re-scanned profile '%s' after config/.env change (%s adapter(s) connected)", name, connected) + result["rescanned"].append(name) + self._served_profile_signatures = sigs + # A profile deleted while an adapter above was still connecting must not be recorded back + # (the deleter's signal timed out against this lock and rmtree already ran). + live_now = {str(name) for name, _home in _multiplex_profile_homes(self.config)} + for name in [n for n in current if n not in live_now and n != active]: + await self._unserve_profile(name, current.pop(name)) + result["removed"].append(name) + added = [n for n in added if n != name] + self._record_served_profiles(active, list(current.items())) + if added or changed: + await self._after_profiles_added([(n, current[n]) for n in added + changed]) + result["served_profiles"] = self.served_profile_names() + return result + + async def _reset_profile_adapters_for_rescan(self, profile_name: str) -> None: + """Tear down a changed profile's old adapters before rebuilding from its new config.""" + pending = (getattr(self, "_profile_failed_platforms", None) or {}).pop(profile_name, None) or {} + tasks = [task for task in pending.values() if isinstance(task, asyncio.Task) and not task.done()] + for task in tasks: + task.cancel() + if tasks: + await asyncio.wait(tasks, timeout=self._adapter_disconnect_timeout_secs()) + adapters = (getattr(self, "_profile_adapters", None) or {}).pop(profile_name, None) or {} + for platform, adapter in adapters.items(): + await self._bounded_adapter_teardown(adapter, platform, profile=profile_name) + + def _live_resource_claims(self, active: str) -> Dict[tuple, str]: + """Startup's ``claimed`` map rebuilt from what is live now: primary claims plus every connected + secondary's credential/listener, so a hot-added profile reusing a token is parked, never a + second poller.""" + claimed = self._primary_resource_claims(active) + for profile_name, adapters in (getattr(self, "_profile_adapters", None) or {}).items(): + for platform, adapter in list(adapters.items()): + for claim in (self._adapter_credential_claim(platform, adapter), + self._adapter_listener_claim(platform, adapter)): + if claim is not None: + claimed[claim] = profile_name + return claimed + + async def _after_profiles_added(self, profile_homes) -> None: + """Per-profile startup side effects for hot-added profiles: log routing + scoped MCP discovery.""" + from gateway.run import _enable_multiplex_log_routing, _profile_runtime_scope + from contextvars import copy_context + with _log_suppressed(logging.DEBUG, "log routing refresh failed", exc_info=True): + _enable_multiplex_log_routing(self.config) + loop = asyncio.get_running_loop() + for profile_name, profile_home in profile_homes: + try: + from tools.mcp_tool_discovery import discover_mcp_tools + with _profile_runtime_scope(Path(profile_home)): + await loop.run_in_executor(None, copy_context().run, discover_mcp_tools) + except Exception: + logger.warning("MCP tool discovery failed for profile '%s'", profile_name, exc_info=True) + + async def _unserve_profile(self, name: str, home: "Path") -> None: + """Stop and unroute one deleted profile: cancel its reconnects, tear down its adapters, drop its + bookkeeping and release this process's handles into its home so the deleter's rmtree succeeds.""" + from gateway.run import _write_runtime_status_quiet + from hermes_constants import hermes_home_key + pending = (getattr(self, "_profile_failed_platforms", None) or {}).pop(name, None) or {} + tasks = [t for t in pending.values() if isinstance(t, asyncio.Task) and not t.done()] + for task in tasks: + task.cancel() + if tasks: + await asyncio.wait(tasks, timeout=self._adapter_disconnect_timeout_secs()) + adapters = (getattr(self, "_profile_adapters", None) or {}).pop(name, None) or {} + for platform, adapter in list(adapters.items()): + await self._bounded_adapter_teardown(adapter, platform, profile=name) + with _log_suppressed(logging.DEBUG, "MCP scope cleanup failed for deleted profile", exc_info=True): + from tools.mcp_tool_lifecycle import shutdown_mcp_servers + await asyncio.to_thread(shutdown_mcp_servers, scope=hermes_home_key(home)) + # Its ``:`` runtime entries describe a profile that no longer exists. + _write_runtime_status_quiet(drop_profile_platforms=name) + for attr in ("pairing_stores", "_busy_text_modes_by_profile", "_busy_input_modes_by_profile"): + store = getattr(self, attr, None) + if isinstance(store, dict): + store.pop(name, None) + if isinstance(self._served_profile_homes, dict): + self._served_profile_homes.pop(name, None) + if isinstance(self._served_profile_signatures, dict): + self._served_profile_signatures.pop(name, None) + prefix = f"agent:{name}:" + cache = getattr(self, "_agent_cache", None) + for key in [k for k in list(cache or {}) if str(k).startswith(prefix)]: + with _log_suppressed(logging.DEBUG, "agent eviction failed for %s", key, exc_info=True): + self._evict_cached_agent(key) + with _log_suppressed(logging.DEBUG, "profile handle release failed", exc_info=True): + from hermes_state_registry import close_all_under + close_all_under(home) + with _log_suppressed(logging.DEBUG, "memory-store release failed", exc_info=True): + from plugins.memory.holographic.store import MemoryStore + MemoryStore.release_all_under(home) + logger.info("[MULTIPLEX] Profile '%s' deleted — %d adapter(s) stopped and unrouted", name, len(adapters)) diff --git a/gateway/run_startup.py b/gateway/run_startup.py index 49213fbc20def..dd3206d695877 100644 --- a/gateway/run_startup.py +++ b/gateway/run_startup.py @@ -1274,7 +1274,9 @@ async def _start_finish_wiring(self, connected_count: int) -> None: "_session_housekeeping_watcher", "_model_catalog_refresh_watcher", "_session_stall_watcher", "_kanban_notifier_watcher", "_kanban_dispatcher_watcher", ) - _POST_RECONNECT_WATCHERS = ("_handoff_watcher", "_async_delegation_watcher", "_loop_wakeup_watcher") + _POST_RECONNECT_WATCHERS = ( + "_handoff_watcher", "_async_delegation_watcher", "_loop_wakeup_watcher", "_profile_reconcile_watcher", + ) def _start_spawn_background_watchers(self) -> None: """Spawn the long-lived supervised background watchers.""" diff --git a/gateway/run_turn.py b/gateway/run_turn.py index 143ad2283df21..ead97ba506848 100644 --- a/gateway/run_turn.py +++ b/gateway/run_turn.py @@ -2293,6 +2293,7 @@ async def _execute_mcp_reload(self, event: MessageEvent) -> str: _servers, _lock, _mcp_tool_server_names_by_scope, _server_public_names, _server_visible_in_scope, ) + from tools.mcp_tool_scope import _key_name from tools.mcp_tool_agent import reprobe_tool_availability from tools.registry import registry @@ -2301,12 +2302,12 @@ async def _execute_mcp_reload(self, event: MessageEvent) -> str: def _scoped_server_names() -> set: with _lock: names = { - _server_public_names.get(name, name) for name in _servers + _server_public_names.get(name, _key_name(name)) for name in _servers if _server_visible_in_scope(name, reload_scope) } if reload_scope is not None: names.update( - _server_public_names.get(owner, owner) + _server_public_names.get(owner, _key_name(owner)) for owner in _mcp_tool_server_names_by_scope.get(reload_scope, {}).values() ) return names @@ -2325,7 +2326,9 @@ def _scoped_server_names() -> set: provenance = _mcp_tool_server_names_by_scope.get(reload_scope, {}) new_tools = [ n for n in new_tools - if _server_public_names.get(provenance.get(n), provenance.get(n)) in connected_servers + if _server_public_names.get( + provenance.get(n), _key_name(provenance.get(n)) if provenance.get(n) is not None else n + ) in connected_servers ] # (label, i18n key, names); i18n lines list reconnected first, the injected note added first. changes = ( diff --git a/gateway/status.py b/gateway/status.py index b356a938b69a7..6774e1922629c 100644 --- a/gateway/status.py +++ b/gateway/status.py @@ -380,6 +380,18 @@ def _profile_name_for_home(profile_home: Path) -> Optional[str]: return profile_home.name if profile_home.parent.name == "profiles" else None +def profile_flag_value(command: str) -> Optional[str]: + """The ``-p``/``--profile`` argument of a command line, or None. Token equality is the only safe + profile match: a substring test lets ``-p ops`` claim (and ``gateway stop`` SIGTERM) ``-p ops-2``.""" + tokens = command.split() + for i, tok in enumerate(tokens): + if tok.startswith("--profile="): + return tok.partition("=")[2] + if tok in ("-p", "--profile") and i + 1 < len(tokens): + return tokens[i + 1] + return None + + def _command_line_belongs_to_profile(command: str, profile_home: Path) -> bool: """True when a gateway command line belongs to ``profile_home`` (mirrors ``hermes_cli.gateway._matches_current_profile``): a stale state file can record a PID recycled @@ -389,10 +401,7 @@ def _command_line_belongs_to_profile(command: str, profile_home: Path) -> bool: profile_name = _profile_name_for_home(profile_home) home_lc = str(profile_home).lower().replace("\\", "/") if profile_name is not None and profile_name != "default": - profile_lc = profile_name.lower() - return any(needle in command_lc for needle in ( - f"--profile {profile_lc}", f"-p {profile_lc}", f"hermes_home={home_lc}" - )) + return profile_flag_value(command_lc) == profile_name.lower() or f"hermes_home={home_lc}" in command_lc # Default profile: accept unless argv names another profile or a conflicting explicit # HERMES_HOME= (its absence is not disqualifying -- HERMES_HOME usually arrives via the env). if "--profile " in command_lc or " -p " in command_lc: @@ -441,13 +450,15 @@ def _get_code_identity_fields() -> dict[str, Any]: return {} -def _pid_record_belongs_to_current_profile(record: Optional[dict[str, Any]]) -> bool: +def _pid_record_belongs_to_current_profile( + record: Optional[dict[str, Any]], *, expected_home: Optional[Path] = None +) -> bool: """True when the record's ``hermes_home`` matches the current process (legacy records: True); another HERMES_HOME's record must be ignored or the default gateway assumes its identity.""" if not isinstance(record, dict): return False record_home = record.get("hermes_home") - return not record_home or _same_hermes_home(record_home, _get_process_hermes_home()) + return not record_home or _same_hermes_home(record_home, expected_home or _get_process_hermes_home()) def _build_runtime_status_record() -> dict[str, Any]: @@ -795,20 +806,24 @@ def write_runtime_status( active_agents: Any = _UNSET, platform: Any = _UNSET, platform_state: Any = _UNSET, error_code: Any = _UNSET, error_message: Any = _UNSET, needs_attention: Any = _UNSET, retrying_since: Any = _UNSET, served_profiles: Any = _UNSET, session_store: Any = _UNSET, - clear_profile_platforms: bool = False, + ingress_url: Any = _UNSET, clear_profile_platforms: bool = False, + drop_profile_platforms: Optional[str] = None, ) -> None: - """Persist gateway runtime health information for diagnostics/status.""" + """Persist gateway runtime health information for diagnostics/status. ``drop_profile_platforms`` + removes one deleted profile's ``:`` entries (hot unroute).""" path = _get_runtime_status_path() payload = _read_json_file(path) or _build_runtime_status_record() previous_payload = copy.deepcopy(payload) current_record = _build_pid_record() payload.setdefault("platforms", {}) - if clear_profile_platforms: + if clear_profile_platforms or drop_profile_platforms: # Secondary-profile entries are keyed ``:``. A fresh process must not # inherit them or /api/status stays degraded until every old adapter re-emits. platforms = payload["platforms"] if isinstance(payload["platforms"], dict) else {} + drop_prefix = f"{drop_profile_platforms}:" if drop_profile_platforms else None payload["platforms"] = { - k: v for k, v in platforms.items() if not isinstance(k, str) or ":" not in k + k: v for k, v in platforms.items() + if not isinstance(k, str) or ":" not in k or (drop_prefix is not None and not k.startswith(drop_prefix)) } # Re-stamp identity + code fields on every write: the file can outlive its creator and the # top-level record must describe the CURRENT writer. @@ -833,6 +848,8 @@ def write_runtime_status( ("needs_attention", needs_attention, bool), # ISO start of the current retry episode; None clears it. ("retrying_since", retrying_since, None), + # Shared-listener secondaries: the /p// callback URL the vendor console must target. + ("ingress_url", ingress_url, None), )) # Per-entry writer provenance: top-level pid/start_time only identify the most recent # writer; /api/status tells "live" from "preserved" by exact (pid, start_time) equality. @@ -1442,7 +1459,8 @@ def planned_stop_marker_targets_self() -> bool: def get_running_pid( - pid_path: Optional[Path] = None, *, cleanup_stale: bool = True + pid_path: Optional[Path] = None, *, cleanup_stale: bool = True, + expected_home: Optional[Path] = None, ) -> Optional[int]: """PID of a running gateway (lock + PID file verified against the live process), or None.""" resolved_pid_path = pid_path or _get_pid_path() @@ -1453,12 +1471,12 @@ def get_running_pid( ) for record in records: pid = _live_pid_from_record(record) - if pid is None or not _pid_record_belongs_to_current_profile(record): + if pid is None or not _pid_record_belongs_to_current_profile(record, expected_home=expected_home): continue - if _record_matches_live_gateway_pid(record, pid): + if _record_matches_live_gateway_pid(record, pid, expected_home=expected_home): return pid _cleanup_invalid_pid_path(resolved_pid_path, cleanup_stale=cleanup_stale) - return get_runtime_status_running_pid() if pid_path is None else None + return get_runtime_status_running_pid(expected_home=expected_home) if pid_path is None else None # Lock inactive: the runtime-status fallback runs BEFORE cleanup here. runtime_pid = get_runtime_status_running_pid() if pid_path is None else None if runtime_pid is None: diff --git a/hermes_cli/AGENTS.md b/hermes_cli/AGENTS.md index 7b7f7b1cc2215..b9896c080e546 100644 --- a/hermes_cli/AGENTS.md +++ b/hermes_cli/AGENTS.md @@ -128,5 +128,48 @@ matchers; parser-derived flag sets; never blanket-exclude gateway ancestors, #87 `_apply_profile_override()` in `hermes_cli/main.py` sets `HERMES_HOME` before any module import, so every `get_hermes_home()` scopes to the active profile (rules in root). Profiles are independent -islands by design — no live config inheritance; `--clone` copies at creation. Multiplex -(`gateway.multiplex_profiles`) secret-scope rules: `gateway/AGENTS.md`. +islands by design — no live config inheritance; `--clone` copies at creation, minus messaging +channels (`profile_channels.py` derives the token/allowlist/platform-section key set from the adapter +registry + `gateway/config_env._ENV_STEPS`, never a hand list; `--clone-channels` opts in). Multiplex +(`gateway.multiplex_profiles`) secret-scope rules: `gateway/AGENTS.md`. The served set is +`profiles.py::profiles_to_serve(multiplex=True)` = default + every live (non-tombstoned) dir under +`profiles/` — there is no allowlist (`gateway.multiplex_profile_allowlist` was retired in config v43). +Enumeration is a pure read: never `mkdir` a profile home from a served path (`SessionDB`, logging, +cron all go through `mkdir_under_hermes_home` / `_ensure_cron_dir`, which refuse a deleted or +missing named profile, #94590). Process-global per-profile slots (MCP discovery in `mcp_startup.py`, +tool registry overlays) key on `hermes_constants.hermes_home_key()`, never a single flag. +Migration from per-profile gateways: `hermes_cli/gateway_migrate.py` (`hermes gateway migrate +--multiplex|--standalone`, table-driven `_PREFLIGHT_CHECKS`, manifest `/gateway_migration.json`); +`update_cmd_fleet._verify_fleet_after_update` calls `maybe_auto_migrate_after_update` on the success +path only. Blockers reuse `GatewayRunner._adapter_credential_fingerprint` and `platform_binds_port`; +"has a `/p//` ingress" is the adapter class attribute `serves_profile_prefix` — set it on a +new HTTP-inbound adapter when it answers the prefix, never extend a list here. + +## Nous free tier (`hermes_cli/anon_auth.py`) + +Sign-in completion is one function, `settle_after_upgrade`, called by every caller that persists an +account over a free-tier identity (CLI `upgrade_guest`, the desktop poller): it moves a config on the +welcome route to the account's host and the tier's recommended default +(`models.recommended_nous_default_model`, shared with `GET /api/model/recommended-default`). + +The shared flow, states, and copy live in `anon_sign_in.py`; CLI rendering lives in +`anon_sign_in_cli.py`. `anon_auth.py` keeps identity, promotion polling, and settlement, and +re-exports the existing sign-in API. The flow resolves identity and persistence collaborators +through `anon_auth` at call time to preserve module-attribute monkeypatch seams. + +The sign-in itself is one composition: `anon_auth.run_sign_in()` yields `SignInState`s (`Code`, +`Waiting`, `Completed`, `Declined`, `Superseded`, `TimedOut`, `Retired`, `Failed`, +`AlreadySignedIn`, `Unavailable`). It reads the current state itself, holds one absolute deadline +across both waits, persists only after a completed promotion **and** a token grant, runs +`settle_after_upgrade` exactly once per completion, and never lets a persist or settle failure +escape as an exception — it becomes `Failed`. Every state carries its own `.copy` (the chat form, +which never contains a raw exception, a URL or a `hermes` verb) and `.copy_terminal`, so no caller +maps a reason to a string. `cancelled()` stops an attempt; `cancel_wins_after_promotion` decides +what happens when the server had already completed the transfer — the desktop keeps `True` (a +DELETE means "not on this machine"), the gateway passes `False` (a supersede must not discard a +transfer the user actually approved). `scope` is entered only around the precondition and persist +blocks, never across a `yield` or a network wait, because `run_in_executor` does not carry +contextvars. `upgrade_guest` (`hermes auth upgrade`), the CLI `/login` handler and the desktop +promotion poller are renderers over it; a surface that needs the cancel check and the save to be +atomic passes `persist_guard`. The desktop's plain "connect another Nous account" device-code login +is a separate path (`_nous_plain_poller`) and must stay one. diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 28d4f37c9dcda..56d9563eefd27 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1948,6 +1948,19 @@ def _aux(timeout, *, reasoning_effort=True, **extra): # (primary copy: state.db gateway_routing table). True for external tooling and downgrade # safety; False stops producing the file. "write_sessions_json": True, + # One gateway for every profile on this host: the DEFAULT profile's gateway also connects + # each named profile's bots (their own .env / config.yaml, per-profile secret scope) and + # stamps the profile into session keys. Flip with `hermes gateway migrate --multiplex` + # (records a rollback manifest; `--standalone` undoes it) or `hermes config set + # gateway.multiplex_profiles true` + `hermes gateway restart`. GATEWAY_MULTIPLEX_PROFILES + # in the environment overrides. Two profiles configuring the same bot token cannot be + # served together — the duplicate adapter is parked; `hermes profile create --clone` + # therefore leaves messaging channels behind unless --clone-channels is passed. + "multiplex_profiles": False, + # Route inbound chats of the default profile's bots to another profile + # (gateway/profile_routing.py): [{profile, platform, chat_id|user_id|guild_id|...}]. + # Most-specific match wins; only read by the multiplexing default gateway. + "profile_routes": [], # Scale-to-zero idle TIMEOUT only. When an instance is opted in via the NAS "Labs" toggle # (HERMES_SCALE_TO_ZERO env stamp) AND messaging is relay-only/absent AND a wakeUrl is # registered, the relay transport goes dormant so the platform (e.g. Fly autostop) can diff --git a/hermes_cli/cron.py b/hermes_cli/cron.py index 478b0b9bf2797..08c177b7a8351 100644 --- a/hermes_cli/cron.py +++ b/hermes_cli/cron.py @@ -390,7 +390,7 @@ def _warn(headline: str) -> None: def cron_status(): """Show cron execution status.""" from cron.jobs import list_jobs - from hermes_cli.gateway import find_gateway_pids + from hermes_cli.gateway import find_gateway_pids, named_profile_served_by_running_multiplexer print() provider = _active_cron_provider_name() @@ -401,6 +401,12 @@ def cron_status(): "not the in-process ticker.", Colors.GREEN)) print(color(" (No ticker heartbeat is expected for an external provider; " "due jobs are delivered by an authenticated webhook.)", Colors.DIM)) + elif not find_gateway_pids() and named_profile_served_by_running_multiplexer(): + # Satellite profile: the default multiplexer's ticker fires this store (same answer as + # `_builtin_gateway_liveness`, which `cron list` uses -- the two must not disagree). + print(color("✓ Gateway is running via the default-profile multiplexer — it ticks this profile's jobs.", + Colors.GREEN)) + print(color(" Ticker health is reported by `hermes cron status` on the default profile.", Colors.DIM)) else: pids = find_gateway_pids() gateway_alive_via_lock = False diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py index 8d0f20ea25009..1f4cbe7268924 100644 --- a/hermes_cli/gateway.py +++ b/hermes_cli/gateway.py @@ -571,7 +571,9 @@ def _scan_gateway_pids( pids: list[int] = [] # Strict matcher shared with gateway.status: requires a real ``gateway run`` argv, so # ``gateway status``/``dashboard`` siblings and ``python -m tui_gateway`` don't match. - from gateway.status import looks_like_gateway_command_line, looks_like_gateway_runtime_command_line + from gateway.status import ( + looks_like_gateway_command_line, looks_like_gateway_runtime_command_line, profile_flag_value, + ) current_home = str(get_hermes_home().resolve()) # Forward slashes on both sides of the HERMES_HOME= match (mirrors gateway.status). current_home_lc = current_home.lower().replace("\\", "/") @@ -582,9 +584,9 @@ def _scan_gateway_pids( def _matches_current_profile(command: str) -> bool: command_lc = command.lower().replace("\\", "/") if current_profile_name: + # Token equality, not substring: `-p ops` must not claim (or SIGTERM) an `-p ops-2` gateway. return ( - f"--profile {current_profile_name_lc}" in command_lc - or f"-p {current_profile_name_lc}" in command_lc + profile_flag_value(command_lc) == current_profile_name_lc or f"hermes_home={current_home_lc}" in command_lc ) @@ -1962,6 +1964,19 @@ def _profile_suffix() -> str: return _profile_name_from_home(home, default) or hashlib.sha256(str(home).encode()).hexdigest()[:8] +def _current_profile_name() -> str: + """Return the profile name represented by this process's ``HERMES_HOME``. + + This facade seam is used by pooled dashboard requests when resolving an unscoped + request to the backend's profile. + """ + try: + from hermes_cli.profiles import get_active_profile_name + return get_active_profile_name() or "default" + except Exception: + return _profile_suffix() or "default" + + def _profile_arg(hermes_home: str | None = None, default_root: str | Path | None = None) -> str: """``--profile `` for ``/profiles/``, else "". *hermes_home*/*default_root* let a sudo/root process generate a unit for another user (the defaults would refer to root).""" @@ -4291,13 +4306,16 @@ def named_profile_served_by_running_multiplexer(profile_name: str | None = None) return False try: - from gateway.status import _pid_exists, _pid_from_record, _read_pid_record - rec = _read_pid_record(default_root / "gateway.pid") - if not rec: - return False - pid = _pid_from_record(rec) - if not pid or not _pid_exists(pid): + from hermes_cli.gateway_multiplex_served import live_default_gateway_pid, recorded_served_profiles + if live_default_gateway_pid() is None: return False + from hermes_cli.profiles import normalize_profile_name + # The live gateway's own record wins: the CLI process cannot see an env-only opt-in on the + # default profile (`hermes -p X` loads X's .env) and a config edit after start is not live yet. + # Only a record without the key (pre-multiplex writer) falls through to config derivation. + recorded = recorded_served_profiles(default_root) + if recorded is not None: + return normalize_profile_name(suffix) in {normalize_profile_name(p) for p in recorded} from gateway.config import _env_multiplex_profiles_override cfg_path = default_root / "config.yaml" @@ -4321,7 +4339,6 @@ def named_profile_served_by_running_multiplexer(profile_name: str | None = None) else: raw_allowlist = gateway_cfg.get("multiplex_profile_allowlist") from gateway.config import _normalize_multiplex_profile_allowlist - from hermes_cli.profiles import normalize_profile_name profile_allowlist = _normalize_multiplex_profile_allowlist(raw_allowlist) return profile_allowlist is None or normalize_profile_name(suffix) in profile_allowlist except Exception: @@ -4329,17 +4346,21 @@ def named_profile_served_by_running_multiplexer(profile_name: str | None = None) return False -def _guard_named_profile_under_multiplexer(force: bool = False) -> None: - """Refuse a named-profile gateway when a multiplexing default gateway already serves it (a second one - would double-bind its platforms: two pollers on one token, port fights). ``--force`` overrides.""" +def _named_profile_refused_under_multiplexer(force: bool = False) -> bool: + """Print the served-profile refusal and return True when a named-profile gateway must not start: + a multiplexing default gateway already serves it (a second one would double-bind its platforms: two + pollers on one token, port fights). ``--force`` overrides. Shared by ``run`` and the service verbs + (``start``/``install``/``restart``): a refusal only inside ``gateway run`` leaves the service manager + to discover it — systemd parks the unit on exit 78 while the CLI prints "started"; launchd + (KeepAlive, no exit-status gating) respawns it every ThrottleInterval forever.""" if force: - return + return False try: suffix = _profile_suffix() except Exception: - return + return False if not named_profile_served_by_running_multiplexer(): - return + return False print_error( f"The default gateway is running as a profile multiplexer and already " @@ -4357,6 +4378,13 @@ def _guard_named_profile_under_multiplexer(force: bool = False) -> None: print() print(" Pass --force to start a separate profile gateway anyway (not") print(" recommended while the multiplexer is running).") + return True + + +def _guard_named_profile_under_multiplexer(force: bool = False) -> None: + """Exit-78 form of ``_named_profile_refused_under_multiplexer`` for the CLI entry points.""" + if not _named_profile_refused_under_multiplexer(force=force): + return # EX_CONFIG, not 1: the refusal is decided purely by config, so it is permanent. The systemd unit # (Restart=always, StartLimitIntervalSec=0) relies on RestartPreventExitStatus=78 as its only # backstop — exit 1 turned a correct refusal into an unbounded restart loop; s6 maps 78 to @@ -5988,6 +6016,8 @@ def _cmd_install(args): managed_error("install gateway service") return force = getattr(args, "force", False) + # `--force` doubles as the reinstall flag here; a served profile's unit would only ever exit 78. + _guard_named_profile_under_multiplexer(force=force) system = getattr(args, "system", False) run_as_user = getattr(args, "run_as_user", None) if is_termux(): @@ -6032,9 +6062,7 @@ def _cmd_start(args): clear_drain_request(home=get_hermes_home()) system = getattr(args, "system", False) start_all = getattr(args, "all", False) - from gateway.drain_control import clear_drain_request - - clear_drain_request(home=get_hermes_home()) + _guard_named_profile_under_multiplexer(force=getattr(args, "force", False)) if not start_all and _dispatch_via_service_manager_if_s6("start"): return if start_all: @@ -6107,6 +6135,8 @@ def _cmd_restart(args): _refuse_from_inside_gateway("restart", "restart loops") system = getattr(args, "system", False) restart_all = getattr(args, "all", False) + force = getattr(args, "force", False) + _guard_named_profile_under_multiplexer(force=force) if restart_all and _dispatch_all_via_service_manager_if_s6("restart"): return if not restart_all and _dispatch_via_service_manager_if_s6("restart"): @@ -6150,7 +6180,7 @@ def _cmd_restart(args): print("✓ Stopped gateway for this profile") _wait_for_gateway_exit(timeout=10.0, force_after=5.0) print("Starting gateway...") - run_gateway(verbose=0) + run_gateway(verbose=0, force=force) # ``hermes gateway status`` hints for a manually-run / stopped gateway, keyed by host kind. diff --git a/hermes_cli/gateway_migrate.py b/hermes_cli/gateway_migrate.py new file mode 100644 index 0000000000000..d0009e4fb2059 --- /dev/null +++ b/hermes_cli/gateway_migrate.py @@ -0,0 +1,718 @@ +"""``hermes gateway migrate --multiplex`` / ``--standalone``: move a per-profile-gateway install onto one +multiplexed default gateway (and back), with a table-driven preflight. + +Standalone per-profile gateways stay supported; this is a migration path, not a removal. The +preflight reuses the gateway's own conflict logic (``GatewayRunner._adapter_credential_fingerprint``, +``platform_binds_port``, the adapters' ``serves_profile_prefix`` declaration) so its verdict matches +what the multiplexer would do at startup. ``hermes update`` calls :func:`maybe_auto_migrate_after_update`. +""" + +from __future__ import annotations + +import contextlib +import json +import logging +import os +import sys +import time +from dataclasses import dataclass, field +from pathlib import Path +from types import SimpleNamespace +from typing import Callable, Iterator, Optional + +logger = logging.getLogger(__name__) + +MANIFEST_NAME = "gateway_migration.json" +MIGRATE_COMMAND = "hermes gateway migrate --multiplex" +_SERVED_WAIT_SECONDS = 90.0 + + +# --------------------------------------------------------------------------- data + + +@dataclass +class ProfileGateway: + """One profile's standalone gateway footprint: live PID and/or installed service.""" + name: str + home: Path + pid: Optional[int] = None + service: Optional[tuple[str, bool]] = None # ("systemd", system) | ("launchd", False) + + @property + def is_default(self) -> bool: + return self.name == "default" + + @property + def has_gateway(self) -> bool: + return self.pid is not None or self.service is not None + + def service_label(self) -> str: + if self.service is None: + return "none" + kind, system = self.service + return f"{kind} ({'system' if system else 'user'})" if kind == "systemd" else kind + + def to_dict(self) -> dict: + return { + "profile": self.name, "home": str(self.home), "pid": self.pid, + "service": None if self.service is None else {"kind": self.service[0], "system": self.service[1]}, + } + + +@dataclass +class MigrationPlan: + default_home: Path + profiles: list[ProfileGateway] + multiplex_flag_on: bool + live_served: Optional[list[str]] # served_profiles the live default gateway recorded, if any + blockers: list[str] = field(default_factory=list) + notices: list[str] = field(default_factory=list) + + @property + def secondaries(self) -> list[ProfileGateway]: + return [p for p in self.profiles if not p.is_default] + + @property + def default(self) -> ProfileGateway: + return next(p for p in self.profiles if p.is_default) + + @property + def already_multiplexed(self) -> bool: + # The flag can be left behind by a partially applied migration. Standalone secondary + # gateways still have to be stopped/uninstalled before this fleet is complete. + if self.standalone_secondaries: + return False + return self.multiplex_flag_on or bool(self.live_served and len(self.live_served) > 1) + + @property + def standalone_secondaries(self) -> list[ProfileGateway]: + return [p for p in self.secondaries if p.has_gateway] + + @property + def blocked(self) -> bool: + return bool(self.blockers) + + def target_service_kind(self) -> Optional[tuple[str, bool]]: + """Service manager the default gateway should end up on: its own, else the one the + secondaries used (so a systemd-managed fleet stays systemd-managed).""" + if self.default.service is not None: + return self.default.service + return next((p.service for p in self.secondaries if p.service is not None), None) + + def to_dict(self) -> dict: + return { + "default_home": str(self.default_home), + "profiles": [p.to_dict() for p in self.profiles], + "multiplex_flag_on": self.multiplex_flag_on, + "live_served": self.live_served, + "already_multiplexed": self.already_multiplexed, + "blockers": list(self.blockers), + "notices": list(self.notices), + "eligible": self.eligible_for_migration(), + "command": MIGRATE_COMMAND, + } + + def eligible_for_migration(self) -> bool: + """>= 2 profiles, at least one secondary with its own gateway, multiplex off, no blockers. + This is the AUTO-migration (``hermes update``) bar; the explicit command also proceeds with + zero standalone secondaries (see :func:`cmd_migrate`).""" + return ( + len(self.profiles) >= 2 and bool(self.standalone_secondaries) + and not self.already_multiplexed and not self.blocked + ) + + +# --------------------------------------------------------------------------- home / env plumbing + + +@contextlib.contextmanager +def _home_env(home: Path) -> Iterator[None]: + """Run service-manager helpers as if ``home`` were the active HERMES_HOME. Both the contextvar + override (``get_hermes_home``) and ``os.environ`` (``gateway.status`` identity files, unit + generation) are switched, then restored.""" + from hermes_constants import reset_hermes_home_override, set_hermes_home_override + import hermes_constants + previous = os.environ.get("HERMES_HOME") + token = set_hermes_home_override(str(home)) + os.environ["HERMES_HOME"] = str(home) + hermes_constants._default_hermes_root_memo = None + try: + yield + finally: + reset_hermes_home_override(token) + if previous is None: + os.environ.pop("HERMES_HOME", None) + else: + os.environ["HERMES_HOME"] = previous + hermes_constants._default_hermes_root_memo = None + + +def _default_home() -> Path: + from hermes_constants import get_default_hermes_root + return get_default_hermes_root() + + +def _profile_homes(profile_allowlist: Optional[list[str]] = None) -> list[tuple[str, Path]]: + from hermes_cli.profiles import profiles_to_serve + return list(profiles_to_serve(multiplex=True, profile_allowlist=profile_allowlist)) + + +def _live_gateway_pid(home: Path) -> Optional[int]: + """PID of a standalone gateway owned by ``home`` (pid file, then runtime status), else None.""" + from gateway.status import get_running_pid, get_runtime_status_running_pid, read_runtime_status + with contextlib.suppress(Exception): + pid = get_running_pid(home / "gateway.pid", cleanup_stale=False) + if pid is not None: + return pid + with contextlib.suppress(Exception): + return get_runtime_status_running_pid(read_runtime_status(home / "gateway_state.json"), expected_home=home) + return None + + +def _installed_service(home: Path) -> Optional[tuple[str, bool]]: + """Installed service kind for ``home``'s gateway (unit / plist on disk), else None.""" + from hermes_cli import gateway as gw + with _home_env(home): + if gw.supports_systemd_services(): + for system in (False, True): + if gw.get_systemd_unit_path(system=system).exists(): + return ("systemd", system) + if gw.is_macos() and gw.get_launchd_plist_path().exists(): + return ("launchd", False) + return None + + +def _service_op(kind: str, system: bool, verb: str, home: Path) -> None: + """``stop`` / ``uninstall`` / ``start`` / ``restart`` / ``install`` on ``home``'s service.""" + from hermes_cli import gateway as gw + with _home_env(home): + if verb == "install": + if kind == "launchd": + gw.launchd_install() + else: + gw.systemd_install(system=system, non_interactive=True) + return + gw._service_call(kind, verb, system) + + +def _stop_gateway_process(home: Path) -> None: + from hermes_cli.profiles import _stop_gateway_process + _stop_gateway_process(home) + + +def _spawn_detached_gateway(home: Path) -> bool: + from hermes_cli import gateway as gw + with _home_env(home): + return gw._spawn_detached_gateway() + + +def _read_multiplex_flag(default_home: Path) -> bool: + from gateway.config import _env_multiplex_profiles_override + from agent.secret_scope import load_env_file + scoped_env = dict(os.environ) + scoped_env.update(load_env_file(default_home / ".env")) + env = _env_multiplex_profiles_override(scoped_env) + if env is not None: + return env + cfg_path = default_home / "config.yaml" + if not cfg_path.exists(): + return False + from hermes_cli.config import read_user_config_raw + cfg = read_user_config_raw(cfg_path) or {} + gateway_section = cfg.get("gateway") if isinstance(cfg.get("gateway"), dict) else {} + return bool(cfg.get("multiplex_profiles") or gateway_section.get("multiplex_profiles")) + + +def _write_multiplex_flag(default_home: Path, value: bool) -> None: + """Set ``gateway.multiplex_profiles`` in the DEFAULT profile's config.yaml through the config API + (same read-guard + nested-set + atomic write ``hermes config set`` uses; no raw YAML edits).""" + from hermes_cli.config import _set_nested, _write_user_config, require_readable_config_before_write + cfg_path = default_home / "config.yaml" + user_config = require_readable_config_before_write(cfg_path) + # A stale top-level alias would shadow the nested key the docs describe. + user_config.pop("multiplex_profiles", None) + _set_nested(user_config, "gateway.multiplex_profiles", value) + _write_user_config(cfg_path, user_config) + + +# --------------------------------------------------------------------------- preflight checks + + +def _profile_gateway_config(home: Path): + """This profile's ``GatewayConfig`` read exactly the way the multiplexer reads it: under the + profile's own secret scope with multiplexing active, so a missing token stays missing instead of + borrowing the CLI process's ``os.environ`` (which holds the launch profile's ``.env``).""" + from gateway.config import load_gateway_config + from gateway.run import _profile_runtime_scope + with _profile_runtime_scope(home): + return load_gateway_config() + + +@contextlib.contextmanager +def _multiplex_read_mode() -> Iterator[None]: + from agent.secret_scope import is_multiplex_active, set_multiplex_active + previous = is_multiplex_active() + set_multiplex_active(True) + try: + yield + finally: + set_multiplex_active(previous) + + +def _credential_probe(platform_config) -> SimpleNamespace: + """Config-shaped stand-in for ``GatewayRunner._adapter_credential_fingerprint`` (which probes + adapter attributes): token/api_key plus the id-style credentials adapters expose from ``extra``.""" + extra = getattr(platform_config, "extra", None) or {} + return SimpleNamespace( + token=getattr(platform_config, "token", None) or getattr(platform_config, "api_key", None), + _app_id=extra.get("app_id"), _client_id=extra.get("client_id"), _bot_id=extra.get("bot_id"), + _project_secret=extra.get("project_secret"), config=platform_config, + ) + + +def _credential_claims(config) -> dict[tuple, str]: + """``(platform, fingerprint)`` for every enabled platform with a discoverable credential.""" + from gateway.run import GatewayRunner + claims: dict[tuple, str] = {} + for platform, platform_config in config.platforms.items(): + if not platform_config.enabled: + continue + fp = GatewayRunner._adapter_credential_fingerprint(_credential_probe(platform_config)) + if fp is not None: + claims[(platform.value, fp)] = platform.value + return claims + + +def _check_duplicate_credentials(plan: MigrationPlan, configs: dict[str, object]) -> None: + """BLOCKER: the same bot credential configured on two profiles — the multiplexer would park + the duplicate adapter, so one profile's bot would go silent after migration.""" + owners: dict[tuple, str] = {} + for profile in plan.profiles: # default first: it wins the claim, like at multiplexer startup + cfg = configs.get(profile.name) + if cfg is None: + continue + for claim, platform_value in _credential_claims(cfg).items(): + owner = owners.setdefault(claim, profile.name) + if owner == profile.name: + continue + plan.blockers.append( + f"Profiles '{owner}' and '{profile.name}' both configure {platform_value} with the same " + f"credential: the bot can only belong to one profile; remove the token from " + f"'{profile.name}' or keep it in {owner} and route {profile.name}'s chats with " + f"profile_routes (gateway.profile_routes in {owner}'s config.yaml)." + ) + + +def platform_serves_profile_prefix(platform_value: str) -> bool: + """True when the adapter for ``platform_value`` declares ``serves_profile_prefix`` (it answers + ``/p//...`` on the default listener). Read from the adapter CLASS — builtin table or the + plugin registry entry — never from a hand-kept list, so new ingress adapters count automatically.""" + from gateway.platforms.base import BasePlatformAdapter + + def _declares(cls) -> bool: + return isinstance(cls, type) and issubclass(cls, BasePlatformAdapter) and bool( + getattr(cls, "serves_profile_prefix", False)) + + with contextlib.suppress(Exception): + from gateway.config import Platform + from gateway.run import _BUILTIN_ADAPTERS, _builtin_adapter_import + spec = _BUILTIN_ADAPTERS.get(Platform(platform_value)) + if spec is not None: + adapter_cls, _ok = _builtin_adapter_import(spec[0], spec[1], spec[2]) + return _declares(adapter_cls) + with contextlib.suppress(Exception): + # Plugin-shipped adapters (sms, line, teams, feishu, wecom, ...) only exist in the registry + # after discovery; a bare CLI process has not run it yet. + from hermes_cli.plugins import discover_plugins + discover_plugins() # idempotent + from gateway.platform_registry import platform_registry + entry = platform_registry.get(platform_value) + if entry is not None: + factory = entry.adapter_factory + if _declares(factory): + return True + # Lambda factories: the adapter class lives in the factory's module. + import importlib + module = importlib.import_module(factory.__module__) + return any(_declares(getattr(module, name)) for name in dir(module)) + return False + + +def _listener_url(default_cfg, platform_value: str, profile: str) -> str: + from gateway.config import Platform + extra = {} + with contextlib.suppress(Exception): + extra = (default_cfg.platforms.get(Platform(platform_value)) or SimpleNamespace(extra={})).extra or {} + defaults = {"api_server": ("127.0.0.1", 8642), "webhook": ("0.0.0.0", 8644)} + host, port = defaults.get(platform_value, ("", "")) + host = extra.get("host") or host + port = extra.get("port") or port + tail = {"api_server": "/v1/...", "webhook": "/webhooks/"}.get(platform_value, "/...") + return f"http://{host}:{port}/p/{profile}{tail}" + + +def _check_secondary_port_binders(plan: MigrationPlan, configs: dict[str, object]) -> None: + """BLOCKER when a secondary enables a port-binding platform with no ``/p//`` ingress + (the multiplexer skips the whole profile); NOTICE (URL changes) when the ingress exists.""" + from gateway.config import platform_binds_port + default_cfg = configs.get("default") + for profile in plan.secondaries: + cfg = configs.get(profile.name) + if cfg is None: + continue + for platform, platform_config in cfg.platforms.items(): + if not platform_config.enabled or not platform_binds_port(platform.value, platform_config.extra): + continue + plan.blockers.append( + f"Profile '{profile.name}' enables {platform.value}, which still binds its own port; " + f"the multiplexer would skip the whole profile even though its adapter may declare a " + f"/p/{profile.name}/ ingress. Disable it there (platforms.{platform.value}.enabled: false) " + f"or keep '{profile.name}' on a standalone gateway (hermes -p {profile.name} gateway start --force)." + ) + + +_PREFLIGHT_CHECKS: tuple[Callable[[MigrationPlan, dict[str, object]], None], ...] = ( + _check_duplicate_credentials, + _check_secondary_port_binders, +) + + +def _load_profile_configs(plan: MigrationPlan) -> dict[str, object]: + configs: dict[str, object] = {} + with _multiplex_read_mode(): + for profile in plan.profiles: + try: + configs[profile.name] = _profile_gateway_config(profile.home) + except Exception as exc: # unreadable config is itself a blocker, not a crash + plan.blockers.append(f"Profile '{profile.name}': could not load its gateway config ({exc}).") + return configs + + +def build_migration_plan() -> MigrationPlan: + """Enumerate profiles + their gateway footprint, then run every preflight check.""" + from hermes_cli.gateway_multiplex_served import recorded_served_profiles + default_home = _default_home() + allowlist = None + with contextlib.suppress(Exception): + allowlist = getattr(_profile_gateway_config(default_home), "multiplex_profile_allowlist", None) + profiles = [ + ProfileGateway(name=name, home=home, pid=_live_gateway_pid(home), service=_installed_service(home)) + for name, home in _profile_homes(allowlist) + ] + plan = MigrationPlan( + default_home=default_home, profiles=profiles, + multiplex_flag_on=_read_multiplex_flag(default_home), + live_served=recorded_served_profiles(default_home), + ) + if len(plan.profiles) < 2: + plan.notices.append("Only one profile exists: nothing to multiplex.") + return plan + configs = _load_profile_configs(plan) + for check in _PREFLIGHT_CHECKS: + check(plan, configs) + plan.notices.append( + "Profiles created after the migration are served by the running multiplexer as soon as " + "they exist (it rescans profiles/ on create/delete and every 30s)." + ) + return plan + + +# --------------------------------------------------------------------------- printing + + +def _print(lines: list[str]) -> None: + for line in lines: + print(line) + + +def format_plan(plan: MigrationPlan, *, dry_run: bool) -> list[str]: + head = "Migration plan (dry run — nothing changed)" if dry_run else "Migration plan" + lines = [head, f" default home: {plan.default_home}", "", " profile gateway pid service"] + for p in plan.profiles: + lines.append(f" {p.name:<12} {str(p.pid or '-'):<13} {p.service_label()}") + lines.append("") + if plan.already_multiplexed: + lines.append(" ✓ The default gateway is already multiplexing" + + (f" (serving {', '.join(plan.live_served)})" if plan.live_served else " (flag on)") + ".") + return lines + steps = [] + for p in plan.standalone_secondaries: + what = " + ".join(x for x in (f"stop pid {p.pid}" if p.pid else "", f"uninstall {p.service_label()}" if p.service else "") if x) + steps.append(f" - {p.name}: {what}") + if len(plan.profiles) < 2: # the notice already says "only one profile exists" + return lines + _plan_tail(plan) + if not steps: + lines.append(" No secondary profile runs its own gateway; the only step is turning the flag on:") + else: + lines += [" Steps:", *steps] + lines.append(f" - default: set gateway.multiplex_profiles: true in {plan.default_home / 'config.yaml'}") + target = plan.target_service_kind() + lines.append(f" - default: {'restart' if plan.default.has_gateway else 'start'} the gateway" + + (f" via {target[0]}" if target else " (detached)") + f", verify it serves {len(plan.profiles)} profiles") + lines.append(f" - record the previous state in {plan.default_home / MANIFEST_NAME} (rollback: hermes gateway migrate --standalone)") + return lines + _plan_tail(plan) + + +def _plan_tail(plan: MigrationPlan) -> list[str]: + lines: list[str] = [] + if plan.blockers: + lines += ["", " ✗ Blockers (fix these first, nothing will be changed):"] + lines += [f" • {b}" for b in plan.blockers] + if plan.notices: + lines += ["", " Notices:"] + lines += [f" • {n}" for n in plan.notices] + return lines + + +def format_update_warning(plan: MigrationPlan) -> list[str]: + return [ + "⚠ Your profiles each run their own gateway. A single multiplexed gateway is the recommended", + " setup, but this install cannot be migrated automatically yet:", + *[f" • {b}" for b in plan.blockers], + f" After fixing the above, run: {MIGRATE_COMMAND}", + " (`hermes update` will migrate automatically once nothing blocks it.)", + ] + + +# --------------------------------------------------------------------------- apply / rollback + + +def _manifest_path(default_home: Path) -> Path: + return default_home / MANIFEST_NAME + + +def _read_manifest(default_home: Path) -> Optional[dict]: + path = _manifest_path(default_home) + if not path.exists(): + return None + try: + data = json.loads(path.read_text(encoding="utf-8")) + except (OSError, ValueError): + return None + return data if isinstance(data, dict) else None + + +def _write_manifest(default_home: Path, data: dict) -> None: + _manifest_path(default_home).write_text(json.dumps(data, indent=2), encoding="utf-8") + + +def _wait_for_served(default_home: Path, expected: set[str], timeout: float) -> Optional[list[str]]: + """Poll the default's ``gateway_state.json`` until ``served_profiles`` covers ``expected``.""" + from hermes_cli.gateway_multiplex_served import recorded_served_profiles + deadline = time.monotonic() + timeout + served: Optional[list[str]] = None + while time.monotonic() < deadline: + with _home_env(default_home): + served = recorded_served_profiles(default_home) + if served is not None and expected <= set(served): + return served + time.sleep(0.5) + return served + + +def _restart_default(plan_default: ProfileGateway, target: Optional[tuple[str, bool]], default_home: Path) -> str: + """Bring the default gateway up on the new flag value; returns a one-line description.""" + if plan_default.service is not None: + kind, system = plan_default.service + _service_op(kind, system, "restart", default_home) + return f"restarted the default gateway via {kind}" + if target is not None: + kind, system = target + _service_op(kind, system, "install", default_home) + _service_op(kind, system, "start", default_home) + return f"installed and started the default gateway via {kind}" + verb = "restarted" if plan_default.pid is not None else "started" + if plan_default.pid is not None: + _stop_gateway_process(default_home) + if not _spawn_detached_gateway(default_home): + raise RuntimeError("could not spawn the default gateway (detached)") + return f"{verb} the default gateway (detached; no service manager was in use)" + + +def _restore_default_gateway(default_home: Path, default_rec: dict) -> None: + """Restore exactly the default gateway footprint recorded before migration.""" + recorded_service = default_rec.get("service") + desired_service = ( + (recorded_service["kind"], bool(recorded_service.get("system"))) + if isinstance(recorded_service, dict) and recorded_service.get("kind") else None + ) + desired_pid = default_rec.get("pid") + current_pid = _live_gateway_pid(default_home) + current_service = _installed_service(default_home) + + def stop_current() -> None: + nonlocal current_pid, current_service + if current_pid is not None: + _stop_gateway_process(default_home) + current_pid = None + if current_service is not None: + kind, system = current_service + _service_op(kind, system, "stop", default_home) + _service_op(kind, system, "uninstall", default_home) + current_service = None + + if desired_service is None and desired_pid is None: + # Migration may have installed the secondary service manager on the default home. + # A default that was absent before migration must be absent after rollback too. + stop_current() + return + + if desired_service is not None: + if current_service == desired_service: + _service_op(*desired_service, "restart", default_home) + return + stop_current() + _service_op(*desired_service, "install", default_home) + _service_op(*desired_service, "start", default_home) + return + + # The recorded default was a detached gateway, not a service-managed one. + stop_current() + if not _spawn_detached_gateway(default_home): + raise RuntimeError("could not restore the default gateway (detached)") + + +def apply_migration(plan: MigrationPlan, *, served_wait: float = _SERVED_WAIT_SECONDS) -> bool: + """Stop/uninstall every secondary gateway, flip the flag, bring up the multiplexer, verify. + Returns True when the multiplexer verifiably serves every profile.""" + if plan.blocked: + _print(["✗ Migration refused:", *[f" • {b}" for b in plan.blockers]]) + return False + if plan.already_multiplexed: + print("✓ Already multiplexed — nothing to do.") + return True + manifest = { + "version": 1, "migrated_at": time.strftime("%Y-%m-%dT%H:%M:%S%z"), + "flag_was": plan.multiplex_flag_on, + "default": plan.default.to_dict(), + "secondaries": [p.to_dict() for p in plan.standalone_secondaries], + } + # The complete rollback record must exist before the first stop/uninstall/kill/config mutation. + _write_manifest(plan.default_home, manifest) + for p in plan.standalone_secondaries: + if p.service is not None: + kind, system = p.service + _service_op(kind, system, "stop", p.home) + _service_op(kind, system, "uninstall", p.home) + print(f" ✓ {p.name}: stopped and removed its {p.service_label()} service") + if p.pid is not None: + _stop_gateway_process(p.home) + print(f" ✓ {p.name}: stopped standalone gateway (pid {p.pid})") + _write_multiplex_flag(plan.default_home, True) + print(f" ✓ default: gateway.multiplex_profiles: true ({plan.default_home / 'config.yaml'})") + print(f" ✓ {_restart_default(plan.default, plan.target_service_kind(), plan.default_home)}") + + expected = {p.name for p in plan.profiles} + served = _wait_for_served(plan.default_home, expected, served_wait) + if served is not None and expected <= set(served): + _print(["", f"✓ Migrated: the default gateway now serves {len(served)} profiles: {', '.join(served)}", + f" Rollback any time with: hermes gateway migrate --standalone", + *[f" • {n}" for n in plan.notices]]) + return True + missing = sorted(expected - set(served or [])) + _print(["", f"⚠ Migration applied, but the default gateway has not confirmed serving: {', '.join(missing)}", + " Check `hermes gateway status` and the gateway log; the flag and manifest are in place.", + " Rollback: hermes gateway migrate --standalone"]) + return False + + +def rollback_migration(default_home: Optional[Path] = None) -> bool: + """``--standalone``: flag off, reinstall/start the recorded per-profile gateways, restart default.""" + default_home = default_home or _default_home() + manifest = _read_manifest(default_home) + if manifest is None: + print(f"✗ No migration manifest at {_manifest_path(default_home)}; nothing to roll back.") + print(" To leave multiplex mode by hand: hermes config set gateway.multiplex_profiles false && hermes gateway restart") + return False + _write_multiplex_flag(default_home, bool(manifest.get("flag_was", False))) + print(" ✓ default: gateway.multiplex_profiles restored") + default_rec = manifest.get("default") or {} + _restore_default_gateway(default_home, default_rec) + if default_rec.get("service") or default_rec.get("pid"): + print(" ✓ default: restored the gateway recorded before migration") + else: + print(" ✓ default: restored its pre-migration stopped state") + ok = True + for rec in manifest.get("secondaries", []): + home = Path(rec["home"]) + name = rec["profile"] + try: + service = rec.get("service") + if service: + kind, system = service["kind"], bool(service.get("system")) + _service_op(kind, system, "install", home) + _service_op(kind, system, "start", home) + print(f" ✓ {name}: reinstalled and started its {kind} service") + elif rec.get("pid"): + if _spawn_detached_gateway(home): + print(f" ✓ {name}: started its standalone gateway (detached)") + else: + ok = False + print(f" ✗ {name}: could not start its standalone gateway") + except Exception as exc: + ok = False + print(f" ✗ {name}: {exc}") + if ok: + _manifest_path(default_home).unlink(missing_ok=True) + print("✓ Rolled back to per-profile gateways.") + else: + print(f"⚠ Rollback incomplete; manifest kept at {_manifest_path(default_home)}.") + return ok + + +# --------------------------------------------------------------------------- CLI + update hook + + +def _host_supports_migration() -> Optional[str]: + """Reason the host cannot be migrated by this command (s6 slots / Windows tasks), else None.""" + from hermes_cli import gateway as gw + if gw._running_under_s6(): + return "s6-supervised container: per-profile gateways are s6 slots; set gateway.multiplex_profiles on the default profile and restart the container instead." + if gw.is_windows(): + return "Windows Scheduled Tasks are not migrated automatically; set gateway.multiplex_profiles true, stop the per-profile tasks, and `hermes gateway restart`." + return None + + +def cmd_migrate(args) -> None: + """``hermes gateway migrate [--multiplex|--standalone] [--dry-run] [--yes]``.""" + if getattr(args, "standalone", False): + sys.exit(0 if rollback_migration() else 1) + reason = _host_supports_migration() + if reason: + print(f"✗ {reason}") + sys.exit(1) + plan = build_migration_plan() + dry_run = getattr(args, "dry_run", False) + _print(format_plan(plan, dry_run=dry_run)) + if dry_run: + return + if plan.already_multiplexed: + return + if plan.blocked or len(plan.profiles) < 2: + sys.exit(1 if plan.blocked else 0) + # Zero standalone secondaries is still a migration when the user asks for it explicitly: the flag + # goes on and the default gateway restarts (the update hook keeps treating that case as a no-op). + if not getattr(args, "yes", False) and sys.stdin.isatty(): + from hermes_cli.setup import prompt_yes_no + if not prompt_yes_no("Apply this migration now?", True): + print("Aborted; nothing changed.") + return + print() + sys.exit(0 if apply_migration(plan) else 1) + + +def maybe_auto_migrate_after_update() -> None: + """``hermes update`` hook: with >= 2 profiles, per-profile gateways present and multiplex off, + migrate automatically when unblocked (deterministic, never prompts) or print the blocker block.""" + if _host_supports_migration() is not None: + return + plan = build_migration_plan() + if plan.already_multiplexed or len(plan.profiles) < 2 or not plan.standalone_secondaries: + return + print() + if plan.blocked: + _print(format_update_warning(plan)) + return + print("→ Migrating per-profile gateways onto one multiplexed default gateway...") + _print(format_plan(plan, dry_run=False)) + apply_migration(plan) diff --git a/hermes_cli/gateway_multiplex_served.py b/hermes_cli/gateway_multiplex_served.py new file mode 100644 index 0000000000000..6e15e080506ff --- /dev/null +++ b/hermes_cli/gateway_multiplex_served.py @@ -0,0 +1,114 @@ +"""Which profiles does the LIVE default multiplexer serve? One answer for every CLI/dashboard surface. + +``gateway/run_adapters.py::_record_served_profiles`` writes ``served_profiles`` into the default +home's ``gateway_state.json`` at startup. That record is the truth about the running process; the +default ``config.yaml`` plus ``GATEWAY_MULTIPLEX_PROFILES`` as seen by the *CLI* process is only a +guess (``hermes -p coder ...`` loads coder's ``.env``, so an env-only opt-in on the default profile +is invisible to it, and an allowlist edited after start flips the guess before the restart). +""" + +from __future__ import annotations + +import logging +from pathlib import Path +from typing import Optional + +logger = logging.getLogger(__name__) + + +def live_default_gateway_pid() -> Optional[int]: + """PID of the default gateway only when its lock, identity, and live command all validate.""" + from hermes_constants import get_default_hermes_root + from gateway.status import get_running_pid + default_root = get_default_hermes_root() + try: + pid = get_running_pid( + default_root / "gateway.pid", cleanup_stale=False, expected_home=default_root) + if pid is not None: + return pid + # Pre-multiplex Hermes versions wrote only gateway.pid and had no lock file. Keep + # the historical probe during upgrade; current records use the strict path above. + from gateway.status import _pid_exists, _pid_from_record, _read_pid_record + if (default_root / "gateway.lock").exists(): + return None + record = _read_pid_record(default_root / "gateway.pid") + pid = _pid_from_record(record) if record else None + return pid if pid and _pid_exists(pid) else None + except Exception: + logger.debug("default gateway identity probe failed", exc_info=True) + return None + + +def recorded_served_profiles(default_root: Optional[Path] = None) -> Optional[list[str]]: + """``served_profiles`` the live default gateway recorded, or None when the key is absent (a record + from before the multiplexer recorded it, or a stopped/absent gateway). Callers fall back to config + derivation only on None: an empty list is an authoritative "serves nobody else".""" + from hermes_constants import get_default_hermes_root + from gateway.status import read_runtime_status + if live_default_gateway_pid() is None: + return None + runtime = read_runtime_status((default_root or get_default_hermes_root()) / "gateway_state.json") + served = (runtime or {}).get("served_profiles") + return [str(p) for p in served] if isinstance(served, list) else None + + +def multiplexer_served_secondaries() -> list[str]: + """Named profiles the live default multiplexer serves (excludes ``default``); empty when none.""" + return [p for p in (recorded_served_profiles() or []) if p and p != "default"] + + +def served_profile_ingress_urls(profile: Optional[str] = None) -> dict[str, dict[str, str]]: + """``{profile: {platform: url}}`` for every secondary inbound-port platform the live multiplexer + serves on its shared listener (``:`` entries carrying ``ingress_url``). This is + what the user pastes into the vendor console (Twilio, LINE, Teams, ...). ``profile`` narrows the map.""" + from hermes_constants import get_default_hermes_root + from gateway.status import read_runtime_status, shared_listener_mirror_platforms + if live_default_gateway_pid() is None: + return {} + runtime = read_runtime_status(get_default_hermes_root() / "gateway_state.json") or {} + platforms = runtime.get("platforms") + if not isinstance(platforms, dict): + return {} + urls: dict[str, dict[str, str]] = {} + for key, entry in platforms.items(): + if not (isinstance(key, str) and ":" in key and isinstance(entry, dict)): + continue + url = entry.get("ingress_url") + if not url or entry.get("state") in ("fatal", "disconnected", "stopped"): + continue + name, platform = key.split(":", 1) + if profile and name != profile: + continue + urls.setdefault(name, {})[platform] = str(url) + # api_server/webhook are the default's adapters mirrored at /p// (no entry of their own). + served = [str(p) for p in (runtime.get("served_profiles") or []) if p and p != "default"] + for name in served if not profile else [p for p in served if p == profile]: + for platform, entry in shared_listener_mirror_platforms(runtime, name).items(): + if entry.get("ingress_url"): + urls.setdefault(name, {})[platform] = str(entry["ingress_url"]) + return urls + + +def format_ingress_url_lines(urls: dict[str, str], indent: str = " ") -> list[str]: + """One ``: `` line per platform, sorted.""" + return [f"{indent}{platform}: {url}" for platform, url in sorted(urls.items())] + + +def notify_multiplexer_profiles_changed(profile_name: str, *, timeout: float = 8.0) -> Optional[list[str]]: + """Tell the live default multiplexer that ``profiles/`` changed (``profile_name`` was created or + deleted) so it hot-serves / unroutes it now instead of at its next periodic rescan. Returns the + served-profile list the gateway answered with, or None when no multiplexer answered (no live default + gateway, single-profile gateway, or a gateway predating the verb). Never raises.""" + try: + from hermes_constants import get_default_hermes_root + from gateway.control_socket import rescan_gateway_profiles + if live_default_gateway_pid() is None: + return None + answer = rescan_gateway_profiles(get_default_hermes_root(), timeout=timeout) + except Exception: + logger.debug("multiplexer rescan notification failed for %r", profile_name, exc_info=True) + return None + if not isinstance(answer, dict) or answer.get("multiplex") is False: + return None + served = answer.get("served_profiles") + return [str(p) for p in served] if isinstance(served, list) else None diff --git a/hermes_cli/kanban_db.py b/hermes_cli/kanban_db.py index 51dd9523cc3c6..2edce44cb5c18 100644 --- a/hermes_cli/kanban_db.py +++ b/hermes_cli/kanban_db.py @@ -1310,8 +1310,10 @@ def _normalize_task_skills(skills: Optional[Iterable[str]]) -> Optional[list[str noun = "is a toolset name" if len(toolset_typos) == 1 else "are toolset names" raise ValueError( f"{quoted} {noun}, not skill name(s). " - "Put toolsets in the assignee profile's `toolsets:` config " - "instead of per-task skills. Skills are named skill bundles " + "Enable toolsets for the assignee's platform with " + "`hermes tools enable kanban --platform ` or add " + "`kanban` to the appropriate `platform_toolsets.` entry " + "instead of putting toolsets in per-task skills. Skills are named skill bundles " "(e.g. `blogwatcher`, `github-code-review`); toolsets are runtime " "capabilities (e.g. `web`, `browser`, `terminal`)." ) diff --git a/hermes_cli/profile_channels.py b/hermes_cli/profile_channels.py new file mode 100644 index 0000000000000..a86a91cc7505b --- /dev/null +++ b/hermes_cli/profile_channels.py @@ -0,0 +1,389 @@ +"""Messaging-channel settings a profile clone must NOT inherit. + +A ``--clone``d profile that keeps the source's bot tokens, allowlists and platform state makes two +gateways fight over one bot (standalone) or blocks ``hermes gateway migrate --multiplex`` with a +duplicate-credential finding per platform. The key set is DERIVED from the platform adapters — the +``Platform`` enum + plugin registry (``required_env``, allowlist/allow-all/home-channel env names), +the gateway env-override table (``gateway.config_env._ENV_STEPS`` / ``_ENV_ENABLE_CREDENTIALS``) and +the ``_`` env prefix every adapter's keys share — so a new adapter is covered without a +hand-written list. Model/provider keys, tool keys, memory and general config are never touched. +""" + +from __future__ import annotations + +import contextlib +import logging +import re +from functools import partial +from pathlib import Path +from typing import Dict, Iterable, List, Optional, Set, Tuple + +logger = logging.getLogger(__name__) + +# Platforms whose env names do not share the ``_`` prefix of their config id. The +# dashboard Channels page uses the same table to decide which Keys-page fields a card owns. +_PLATFORM_ENV_PREFIX_ALIASES: dict[str, tuple[str, ...]] = { + "email": ("EMAIL_",), + "homeassistant": ("HASS_",), + "qqbot": ("QQ_", "QQBOT_"), + "sms": ("TWILIO_",), + "wecom": ("WECOM_BOT_", "WECOM_SECRET"), + "wecom_callback": ("WECOM_CALLBACK_",), +} + +# Multiplexer-owner settings: a clone of the default that inherits them and is then started +# standalone tries to be a second multiplexer for every profile on the host. +_GATEWAY_OWNER_KEYS = ("multiplex_profiles", "profile_routes") + +_ENV_LINE_RE = re.compile(r"^\s*(?:export\s+)?([A-Za-z_][A-Za-z0-9_]*)\s*=") + + +def platform_env_prefixes(platform_id: str) -> tuple[str, ...]: + """Env-var prefixes owned by one messaging platform.""" + return _PLATFORM_ENV_PREFIX_ALIASES.get(platform_id, (platform_id.upper().replace("-", "_") + "_",)) + + +def platform_ids() -> List[str]: + """Every messaging platform id: built-in ``Platform`` members plus registered plugin adapters.""" + from gateway.config import Platform + ids = {m.value for m in Platform.__members__.values() if m.value != "local"} + with contextlib.suppress(Exception): + from hermes_cli.plugins import discover_plugins + discover_plugins() # idempotent + from gateway.platform_registry import platform_registry + ids.update(entry.name for entry in platform_registry.all_entries()) + return sorted(ids) + + +def _cred_row_envs(row) -> Set[str]: + """Every env name a ``gateway.config_env._Cred`` row reads.""" + names: Set[str] = set() + + def _flatten(spec) -> None: + if isinstance(spec, str): + names.add(spec) + elif isinstance(spec, (tuple, list)): + for item in spec: + _flatten(item) + + _flatten(row.creds) + if row.token: + names.add(row.token) + for key_env in (*row.fixed, *row.optional, *row.optional_stripped): + _flatten(key_env[1]) + if row.warn_missing: + names.add(row.warn_missing[0]) + if row.home: + names.update({row.home, f"{row.home}_NAME", f"{row.home}_THREAD_ID"}) + return names + + +def declared_channel_env_keys() -> Dict[str, str]: + """``{ENV_KEY: platform_id}`` for every env name an adapter declares outright (registry entry + fields, the gateway env-override table). Prefix matching covers the rest.""" + keys: Dict[str, str] = {} + with contextlib.suppress(Exception): + from hermes_cli.plugins import discover_plugins + discover_plugins() + from gateway.platform_registry import platform_registry + for entry in platform_registry.all_entries(): + for name in (*entry.required_env, entry.allowed_users_env, entry.allow_all_env, entry.cron_deliver_env_var): + if name: + keys[name] = entry.name + with contextlib.suppress(Exception): + from gateway import config_env + for platform, names in config_env._ENV_ENABLE_CREDENTIALS.items(): + keys.update(dict.fromkeys(names, platform.value)) + for step in config_env._ENV_STEPS: + if isinstance(step, config_env._Cred): + keys.update(dict.fromkeys(_cred_row_envs(step), step.platform.value)) + elif isinstance(step, partial): + platform = step.keywords.get("platform") + for kw in ("env", "env_base"): + if step.keywords.get(kw) and platform is not None: + keys[step.keywords[kw]] = platform.value + return keys + + +_CREDENTIAL_SUFFIXES = ( + "_TOKEN", "_SECRET", "_KEY", "_PASSWORD", "_APP_ID", "_CLIENT_ID", "_BOT_ID", "_ACCOUNT_SID", + "_SERVICE_ACCOUNT_JSON", "_PROJECT_ID", "_ACCOUNT", +) + + +def credential_env_keys() -> Dict[str, str]: + """``{ENV_KEY: platform_id}`` for the keys that make an adapter CONNECT AS a bot (token / app id / + client id / secret — the shape ``GatewayRunner._adapter_credential_fingerprint`` hashes). Enable + flags, URLs and hosts are excluded: two profiles pointing at one Mattermost server collide only + when they also share the token.""" + keys: Dict[str, str] = {} + with contextlib.suppress(Exception): + from hermes_cli.plugins import discover_plugins + discover_plugins() + from gateway.platform_registry import platform_registry + for entry in platform_registry.all_entries(): + keys.update(dict.fromkeys(entry.required_env, entry.name)) + with contextlib.suppress(Exception): + from gateway import config_env + for platform, names in config_env._ENV_ENABLE_CREDENTIALS.items(): + keys.update(dict.fromkeys(names, platform.value)) + for step in config_env._ENV_STEPS: + if isinstance(step, config_env._Cred): + creds: Set[str] = set() + for group in step.creds: + creds.update((group,) if isinstance(group, str) else group) + if step.token: + creds.add(step.token) + keys.update(dict.fromkeys(creds, step.platform.value)) + return {key: pid for key, pid in keys.items() if key.endswith(_CREDENTIAL_SUFFIXES)} + + +class ChannelKeyIndex: + """Resolves an env key to the messaging platform that owns it (``None`` = not a channel key).""" + + def __init__(self) -> None: + self.platforms = platform_ids() + self.declared = declared_channel_env_keys() + self._prefixes: List[Tuple[str, str]] = sorted( + ((prefix, pid) for pid in self.platforms for prefix in platform_env_prefixes(pid)), + key=lambda item: -len(item[0]), # longest prefix wins: WECOM_CALLBACK_ before WECOM_ + ) + + def platform_for(self, key: str) -> Optional[str]: + if key in self.declared: + return self.declared[key] + return next((pid for prefix, pid in self._prefixes if key.startswith(prefix)), None) + + +def _env_key_of_line(line: str) -> Optional[str]: + match = _ENV_LINE_RE.match(line) + return match.group(1) if match else None + + +def strip_channel_env_file( + env_path: Path, index: Optional[ChannelKeyIndex] = None, *, preserve_platforms: Optional[Set[str]] = None, +) -> Dict[str, List[str]]: + """Drop every messaging-channel assignment from ``env_path`` in place; comments, blank lines and + every other key survive verbatim. Returns ``{platform: [keys removed]}``.""" + if not env_path.is_file(): + return {} + index = index or ChannelKeyIndex() + removed: Dict[str, List[str]] = {} + kept: List[str] = [] + text = env_path.read_text(encoding="utf-8-sig", errors="replace") + for line in text.splitlines(): + key = _env_key_of_line(line) + platform = index.platform_for(key) if key else None + if key is None or platform is None or platform in (preserve_platforms or set()): + kept.append(line) + else: + removed.setdefault(platform, []).append(key) + if removed: + env_path.write_text("\n".join(kept) + ("\n" if text.endswith("\n") or kept else ""), encoding="utf-8") + return removed + + +def _channel_config_paths(raw: dict, platforms: Iterable[str]) -> List[Tuple[str, ...]]: + """Dotted paths in a raw config.yaml mapping that hold platform identity: ``platforms``, every + top-level ``:`` block, ``gateway.platforms`` / ``gateway.``, and the + multiplexer-owner keys (both spellings the gateway loader accepts).""" + paths: List[Tuple[str, ...]] = [] + gateway: dict = raw["gateway"] if isinstance(raw.get("gateway"), dict) else {} + if "platforms" in raw: + paths.append(("platforms",)) + if "platforms" in gateway: + paths.append(("gateway", "platforms")) + for key in _GATEWAY_OWNER_KEYS: + if key in raw: + paths.append((key,)) + if key in gateway: + paths.append(("gateway", key)) + for pid in platforms: + if pid in raw: + paths.append((pid,)) + if pid in gateway: + paths.append(("gateway", pid)) + return paths + + +def strip_channel_config(config_path: Path, index: Optional[ChannelKeyIndex] = None) -> List[str]: + """Remove platform sections from a raw ``config.yaml`` in place. Returns the dotted paths removed.""" + if not config_path.is_file(): + return [] + from hermes_cli.config import read_user_config_raw + from utils import atomic_yaml_write + index = index or ChannelKeyIndex() + raw = read_user_config_raw(config_path) + paths = _channel_config_paths(raw, index.platforms) + if not paths: + return [] + for path in paths: + node = raw + for seg in path[:-1]: + node = node[seg] + node.pop(path[-1], None) + if isinstance(raw.get("gateway"), dict) and not raw["gateway"]: + raw.pop("gateway") + atomic_yaml_write(config_path, raw, sort_keys=False) + return [".".join(path) for path in paths] + + +def channel_state_entries(root: Path, index: Optional[ChannelKeyIndex] = None) -> List[Path]: + """Root entries of a profile that hold per-bot runtime identity: pairing approvals and the + WhatsApp device session (``platforms/`` + legacy dirs), the gateway's per-platform ledgers, + channel directories and every ``_*`` state file an adapter writes beside config.yaml.""" + if not root.is_dir(): + return [] + index = index or ChannelKeyIndex() + fixed = {"platforms", "pairing", "whatsapp", "gateway", "channel_directory.json", "channel_aliases.json"} + prefixes = tuple(f"{pid}_" for pid in index.platforms) + return sorted( + entry for entry in root.iterdir() + if entry.name in fixed or (entry.is_file() and entry.name.startswith(prefixes)) + ) + + +def strip_channel_settings(profile_dir: Path, *, include_state: bool) -> Dict[str, List[str]]: + """Strip channel credentials/identity from a freshly cloned profile. ``include_state`` also + drops the runtime state ``--clone-all`` copied. Returns ``{platform|"config"|"state": [what]}``.""" + import shutil + index = ChannelKeyIndex() + preserve = set() + if not _platform_enabled_in_config(profile_dir / "config.yaml", "homeassistant"): + # HASS_TOKEN/HASS_URL are also the Home Assistant tool credentials. Keep them when the + # messaging adapter is disabled; tools_config.py treats the configured token as opt-in. + preserve.add("homeassistant") + stripped: Dict[str, List[str]] = dict( + strip_channel_env_file(profile_dir / ".env", index, preserve_platforms=preserve) + ) + config_paths = strip_channel_config(profile_dir / "config.yaml", index) + if config_paths: + stripped["config"] = config_paths + if include_state: + dropped = [] + for entry in channel_state_entries(profile_dir, index): + shutil.rmtree(entry, ignore_errors=True) if entry.is_dir() else entry.unlink(missing_ok=True) + dropped.append(entry.name) + if dropped: + stripped["state"] = dropped + return stripped + + +def channel_platforms_configured(profile_dir: Path) -> List[str]: + """Platform ids with any channel setting in ``profile_dir`` (.env keys or config.yaml sections) — + what a channel-less clone of it leaves behind. Pure read.""" + index = ChannelKeyIndex() + found: Set[str] = set() + homeassistant_enabled = _platform_enabled_in_config(profile_dir / "config.yaml", "homeassistant") + env_path = profile_dir / ".env" + if env_path.is_file(): + for line in env_path.read_text(encoding="utf-8-sig", errors="replace").splitlines(): + key = _env_key_of_line(line) + platform = index.platform_for(key) if key else None + if platform and (platform != "homeassistant" or homeassistant_enabled): + found.add(platform) + config_path = profile_dir / "config.yaml" + if config_path.is_file(): + from hermes_cli.config import read_user_config_raw + raw = read_user_config_raw(config_path) + for path in _channel_config_paths(raw, index.platforms): + node = raw + for seg in path: + node = node[seg] + if path[-1] == "platforms" and isinstance(node, dict): + found.update( + str(k) for k in node + if str(k) != "homeassistant" or homeassistant_enabled + ) + elif path[-1] in index.platforms: + if path[-1] != "homeassistant" or homeassistant_enabled: + found.add(path[-1]) + return sorted(found) + + +def _env_values(env_path: Path, wanted: Dict[str, str]) -> Dict[str, str]: + values: Dict[str, str] = {} + if not env_path.is_file(): + return values + from dotenv import dotenv_values + with contextlib.suppress(Exception): + for key, value in (dotenv_values(env_path, encoding="utf-8-sig") or {}).items(): + if key in wanted and value and value.strip(): + values[key] = value.strip() + return values + + +def _config_platform_tokens(config_path: Path) -> Dict[str, str]: + """``{platform: token}`` from ``platforms.

.token|api_key`` (both nesting spellings).""" + tokens: Dict[str, str] = {} + if not config_path.is_file(): + return tokens + from hermes_cli.config import read_user_config_raw + raw = read_user_config_raw(config_path) + gateway: dict = raw["gateway"] if isinstance(raw.get("gateway"), dict) else {} + for section in (raw.get("platforms"), gateway.get("platforms")): + if not isinstance(section, dict): + continue + for pid, block in section.items(): + if isinstance(block, dict): + token = block.get("token") or block.get("api_key") + if isinstance(token, str) and token.strip(): + tokens[str(pid)] = token.strip() + return tokens + + +def _platform_enabled_in_config(config_path: Path, platform_id: str) -> bool: + """Whether a platform is explicitly enabled in either accepted config nesting.""" + if not config_path.is_file(): + return False + from hermes_cli.config import read_user_config_raw + raw = read_user_config_raw(config_path) + gateway = raw.get("gateway") if isinstance(raw.get("gateway"), dict) else {} + sections = (raw.get("platforms"), gateway.get("platforms"), raw.get(platform_id), gateway.get(platform_id)) + for position, section in enumerate(sections): + if isinstance(section, dict) and platform_id in section and isinstance(section[platform_id], dict): + return bool(section[platform_id].get("enabled")) + if position >= 2 and isinstance(section, dict): + return bool(section.get("enabled")) + return False + + +def shared_channel_credentials(profile_dir: Path, source_dir: Path) -> List[str]: + """Platforms whose CONNECTING credential (bot token / app id / account) in ``profile_dir`` is + byte-identical to ``source_dir``'s — the bots that will collide. Pure file reads: no secret + manager, no gateway config load, so ``hermes profile list`` can afford it per profile.""" + wanted = credential_env_keys() + mine = _env_values(profile_dir / ".env", wanted) + theirs = _env_values(source_dir / ".env", wanted) + shared = {wanted[key] for key in mine if theirs.get(key) == mine[key]} + if "homeassistant" in shared and not ( + _platform_enabled_in_config(profile_dir / "config.yaml", "homeassistant") + or _platform_enabled_in_config(source_dir / "config.yaml", "homeassistant") + ): + shared.remove("homeassistant") + mine_cfg = _config_platform_tokens(profile_dir / "config.yaml") + theirs_cfg = _config_platform_tokens(source_dir / "config.yaml") + shared.update(pid for pid, token in mine_cfg.items() if theirs_cfg.get(pid) == token) + return sorted(shared) + + +def shared_credential_warning(profile: str, platforms: List[str], source: str = "default") -> str: + return ( + f"⚠ Profile '{profile}' shares its {', '.join(platforms)} credential with {source}: the bot can " + f"only belong to one profile. Give '{profile}' its own bot (hermes -p {profile} setup, or the " + f"dashboard Messaging page) or remove the token from '{profile}'; a multiplexed gateway parks " + f"the duplicate and `hermes gateway migrate --multiplex` refuses until it is gone." + ) + + +def format_stripped_notice(profile: str, platforms: List[str], clone_flag: str = "--clone") -> List[str]: + """Lines printed after a channel-less clone so the user knows what was left behind and how to + configure the new profile's own bots.""" + if not platforms: + return [] + return [ + f"Messaging channels were NOT cloned ({', '.join(platforms)}): a copied bot token or allowlist " + "would make two gateways fight over one bot.", + f" Configure this profile's own bots: hermes -p {profile} setup (or the dashboard Messaging page)", + f" To copy the source's channels anyway: hermes profile create {profile} {clone_flag} --clone-channels", + ] diff --git a/hermes_cli/profile_cmd.py b/hermes_cli/profile_cmd.py index e9eb16180e676..76559af37684f 100644 --- a/hermes_cli/profile_cmd.py +++ b/hermes_cli/profile_cmd.py @@ -132,6 +132,29 @@ def _profile_list(args): dist = f"{p.distribution_name}@{p.distribution_version or '?'}"[:30] if p.distribution_name else "—" print(f"{marker}{name:<15} {model:<28} {gw:<12} {alias:<12} {dist}") print() + for line in _shared_credential_warnings(profiles): + print(line) + + +def _shared_credential_warnings(profiles) -> list: + """One warning per named profile whose bot credential is byte-identical to the default's + (typically an old ``--clone`` that copied .env): the collision that parks a multiplexed + adapter or makes two standalone gateways fight over one bot.""" + from hermes_cli.profile_channels import shared_channel_credentials, shared_credential_warning + default = next((p for p in profiles if p.is_default), None) + if default is None: + return [] + lines = [] + for p in profiles: + if p.is_default: + continue + try: + shared = shared_channel_credentials(p.path, default.path) + except Exception: + continue + if shared: + lines.append(shared_credential_warning(p.name, shared)) + return lines + ([""] if lines else []) def _profile_use(args): @@ -144,6 +167,33 @@ def _profile_use(args): _die(f"Error: {e}") +def _source_profile_dir(source_label: str) -> Path: + from hermes_cli.profiles import get_profile_dir + source_dir = get_profile_dir(source_label) + if not source_dir.is_dir(): + raise FileNotFoundError(source_dir) + return source_dir + + +def _print_channel_clone_notice(name: str, source_label: str, clone_channels: bool, clone_flag: str) -> None: + from hermes_cli.profile_channels import ( + channel_platforms_configured, format_stripped_notice, shared_channel_credentials, + shared_credential_warning, + ) + from hermes_cli.profiles import get_profile_dir + try: + source_dir = _source_profile_dir(source_label) + except FileNotFoundError: + return + if not clone_channels: + for line in format_stripped_notice(name, channel_platforms_configured(source_dir), clone_flag): + print(line) + return + shared = shared_channel_credentials(get_profile_dir(name), source_dir) + if shared: + print(shared_credential_warning(name, shared, source_label)) + + def _profile_create(args): from hermes_cli.profiles import ( _get_wrapper_dir, _is_wrapper_dir_in_path, check_alias_collision, create_profile, @@ -155,22 +205,25 @@ def _profile_create(args): no_alias = getattr(args, "no_alias", False) no_skills = getattr(args, "no_skills", False) clone_from = getattr(args, "clone_from", None) + clone_channels = getattr(args, "clone_channels", False) clone_config = clone or clone_from is not None cloned = clone_config or clone_all + source_label = clone_from or get_active_profile_name() try: profile_dir = create_profile( name=name, clone_from=clone_from, clone_all=clone_all, clone_config=clone_config, no_alias=no_alias, no_skills=no_skills, description=getattr(args, "description", None), + clone_channels=clone_channels, ) except (ValueError, FileExistsError, FileNotFoundError) as e: _die(f"Error: {e}") print(f"\nProfile '{name}' created at {profile_dir}") if cloned: - source_label = clone_from or get_active_profile_name() if clone_all: print(f"Full copy from {source_label} (excluding session history, backups, and snapshots).") else: print(f"Cloned config, .env, SOUL.md, and skills from {source_label}.") + _print_channel_clone_notice(name, source_label, clone_channels, "--clone-all" if clone_all else "--clone") # Auto-clone Honcho config for the new profile (only with clone operations) try: from plugins.memory.honcho.cli import clone_honcho_for_profile @@ -208,7 +261,16 @@ def _profile_create(args): print("\nNext steps:") print(f" {name} setup Configure API keys and model") print(f" {name} chat Start chatting") - print(f" {name} gateway start Start the messaging gateway") + from hermes_cli.gateway_multiplex_served import live_default_gateway_pid, recorded_served_profiles + from hermes_cli.profiles import normalize_profile_name + served = recorded_served_profiles() if live_default_gateway_pid() is not None else None + if served is not None and normalize_profile_name(name) in {normalize_profile_name(p) for p in served}: + print(" (served now by the running multiplexed gateway — add its bot token and it connects)") + elif served is not None: + # The multiplexer did not pick the profile up (older gateway or the signal failed): a restart serves it. + print(" hermes gateway restart Serve this profile from the running multiplexed gateway") + else: + print(f" {name} gateway start Start the messaging gateway") if clone or clone_all: print(f"\n Edit {profile_dir_display}/.env for different API keys") print(f" Edit {profile_dir_display}/SOUL.md for different personality") diff --git a/hermes_cli/profiles.py b/hermes_cli/profiles.py index e4a0d3b1a3ae8..67161309e0edb 100644 --- a/hermes_cli/profiles.py +++ b/hermes_cli/profiles.py @@ -748,6 +748,28 @@ def _resolve_clone_source(clone_from: Optional[str]) -> Path: return source_dir +def _refuse_clone_channels_from_live_multiplexer(source_dir: Path, source_label: str) -> None: + """Reject a cross-surface channel clone when the live multiplexer already serves its source.""" + from hermes_cli.gateway_multiplex_served import live_default_gateway_pid, recorded_served_profiles + from hermes_cli.profile_channels import channel_platforms_configured + if live_default_gateway_pid() is None: + return + served = recorded_served_profiles() + if served is None: + return + source_name = normalize_profile_name(source_label) + if source_name not in {normalize_profile_name(name) for name in served}: + return + platforms = channel_platforms_configured(source_dir) + if platforms: + raise ValueError( + f"--clone-channels would copy {', '.join(platforms)} from '{source_label}', which the running " + "multiplexed gateway already serves: the bot can only belong to one profile, so the copy would be " + "parked as a duplicate credential. Clone without --clone-channels and give the new profile its own bot " + "(hermes -p setup), or route its chats with gateway.profile_routes instead." + ) + + def _seed_file_if_missing(path: Path, text: str, mode: Optional[int] = None) -> None: """Best-effort: write *text* to *path* unless it already exists; never raises.""" if path.exists(): @@ -776,6 +798,14 @@ def _clone_all_into(source_dir: Path, profile_dir: Path, canon: str) -> None: """--clone-all: full copytree minus infrastructure/history, then strip runtime files and cloned single-use OAuth grants.""" shutil.copytree(source_dir, profile_dir, symlinks=True, ignore=_clone_all_copytree_ignore(source_dir)) + env_path = profile_dir / ".env" + if env_path.is_symlink(): + # copytree(..., symlinks=True) preserves a managed source link. Materialize the clone before + # channel stripping so write_text cannot follow the link back into the source profile. + data = env_path.read_bytes() + env_path.unlink() + env_path.write_bytes(data) + os.chmod(str(env_path), 0o600) for stale in _CLONE_ALL_STRIP: (profile_dir / stale).unlink(missing_ok=True) # auth.json / .anthropic_oauth.json copied verbatim fork single-use OAuth grants @@ -813,11 +843,16 @@ def _bootstrap_profile_dir(profile_dir: Path, source_dir: Optional[Path]) -> Non def create_profile( name: str, clone_from: Optional[str] = None, clone_all: bool = False, clone_config: bool = False, no_alias: bool = False, no_skills: bool = False, description: Optional[str] = None, + clone_channels: bool = False, ) -> Path: """Create a new profile directory and return its path. ``clone_from`` defaults to the active profile when cloning. ``clone_all`` copies all state; ``clone_config`` copies config.yaml/.env/SOUL.md, installed skills, and identity files. + Either clone strips the source's messaging channels — bot tokens, allowlists, platform + sections, pairing/session state — unless ``clone_channels`` opts in: a copied bot credential + makes two gateways fight over one bot (``hermes_cli.profile_channels``; callers list what + was left behind with ``channel_platforms_configured(source_dir)``). ``no_skills`` creates an empty profile and writes a marker so ``hermes update`` skips re-seeding its skills; it is mutually exclusive with the clone options, which copy skills.""" if no_skills and (clone_from is not None or clone_config or clone_all): @@ -841,10 +876,19 @@ def create_profile( source_dir = None if clone_from is not None or clone_all or clone_config: source_dir = _resolve_clone_source(clone_from) + if clone_channels: + _refuse_clone_channels_from_live_multiplexer( + source_dir, clone_from or get_active_profile_name() + ) if clone_all and source_dir: _clone_all_into(source_dir, profile_dir, canon) else: _bootstrap_profile_dir(profile_dir, source_dir) + if source_dir is not None and not clone_channels: + from hermes_cli.profile_channels import strip_channel_settings + stripped = strip_channel_settings(profile_dir, include_state=clone_all) + if stripped: + logger.info("profile %s: cloned without messaging channels %s", canon, stripped) # Seed an empty .env so the profile owns a credentials file from day one. Without it, # profile-scoped env writes (dashboard Channels/Keys pages, `hermes -p auth add`) @@ -881,9 +925,17 @@ def create_profile( # `hermes -p gateway start` supervises via `s6-svc -u` instead of a bare # process. No-op on host (systemd/launchd/windows unit generation handles lifecycle). _maybe_register_gateway_service(canon) + # A running multiplexer enumerates profiles/ at boot: ask it to serve this one now (it also + # rescans periodically, so a missed signal only delays serving). + _notify_multiplexer(canon) return profile_dir +def _notify_multiplexer(canon: str) -> None: + from hermes_cli.gateway_multiplex_served import notify_multiplexer_profiles_changed + notify_multiplexer_profiles_changed(canon) + + def seed_profile_skills(profile_dir: Path, quiet: bool = False) -> Optional[dict]: """Seed bundled skills into a profile via subprocess (sync_skills() caches HERMES_HOME at module level). Returns the sync result dict, or None on failure. ``--no-skills`` profiles @@ -1007,7 +1059,13 @@ def _profile_bound_backend_pids(canon: str, profile_dir: Path) -> list[int]: except Exception: current_user = None pids: list[int] = [] - for proc in psutil.process_iter(["pid", "name", "username", "cmdline"]): + try: + processes = list(psutil.process_iter(["pid", "name", "username", "cmdline"])) + except Exception: + # Process enumeration itself can be denied on macOS hardened hosts. The caller + # can still remove the profile after the best-effort backend cleanup. + return [] + for proc in processes: try: info = proc.info pid = info.get("pid") @@ -1166,6 +1224,9 @@ def delete_profile(name: str, yes: bool = False) -> Path: # Tombstone before rmtree so a stale serve/logging mkdir cannot relist this name live. mark_named_profile_deleted(profile_dir) + # The multiplexer sees the tombstone, stops this profile's adapters and releases its handles + # into the directory before we remove it. + _notify_multiplexer(canon) # Release this process's holographic memory-store connections into the profile. The # Desktop's main serve process opens memory_store.db for every profile and is diff --git a/hermes_cli/status.py b/hermes_cli/status.py index 8d1536b1869b1..07386ef37f94b 100644 --- a/hermes_cli/status.py +++ b/hermes_cli/status.py @@ -219,12 +219,21 @@ def _render_platforms(ctx): def _render_gateway(ctx): _section("Gateway Service") try: - from hermes_cli.gateway import get_gateway_runtime_snapshot, _format_gateway_pids + from hermes_cli.gateway import ( + get_gateway_runtime_snapshot, _format_gateway_pids, named_profile_served_by_running_multiplexer) + from hermes_cli.gateway_multiplex_served import multiplexer_served_secondaries snapshot = get_gateway_runtime_snapshot() + # A satellite profile has no gateway.pid of its own; the default multiplexer is its live process. + if not snapshot.running and named_profile_served_by_running_multiplexer(): + _kv_flag("Status:", True, "running (via the default-profile multiplexer)", "stopped") + _kv("Manage with:", "hermes gateway status # from the default profile") + return _kv_flag("Status:", snapshot.running, "running", "stopped") _kv("Manager:", snapshot.manager) if snapshot.gateway_pids: _kv("PID(s):", _format_gateway_pids(snapshot.gateway_pids)) + if snapshot.running and (served := multiplexer_served_secondaries()): + _kv("Serves:", ", ".join(served)) if snapshot.has_process_service_mismatch: _kv("Service:", "installed but not managing the current running gateway") elif _is_termux() and not snapshot.gateway_pids: diff --git a/hermes_cli/subcommands/gateway.py b/hermes_cli/subcommands/gateway.py index 54dfa62a4943f..e0f1906b53fd3 100644 --- a/hermes_cli/subcommands/gateway.py +++ b/hermes_cli/subcommands/gateway.py @@ -8,6 +8,13 @@ from hermes_cli.subcommands._shared import add_accept_hooks_flag +# `start`/`restart` on a named profile refuse while the default multiplexer serves it (a second gateway +# would double-bind its platforms); `gateway run` carries its own broader --force text. +_FORCE_SERVED_PROFILE_HELP = ( + "Start a separate gateway for this profile even when the default multiplexer already serves it " + "(not recommended: two pollers on one bot token, port conflicts)") + + def _flag(parser, *names, help, **kw): parser.add_argument(*names, action="store_true", help=help, **kw) @@ -66,6 +73,7 @@ def build_gateway_parser( _add_system_flag(gateway_start) _flag(gateway_start, "--all", help="Kill ALL stale gateway processes across all profiles before starting") + _flag(gateway_start, "--force", help=_FORCE_SERVED_PROFILE_HELP) _add_compat_platform_flag(gateway_start) gateway_stop = gateway_subparsers.add_parser("stop", help="Stop gateway service") @@ -79,6 +87,7 @@ def build_gateway_parser( _add_system_flag(gateway_restart) _flag(gateway_restart, "--all", help="Kill ALL gateway processes across all profiles before restarting") + _flag(gateway_restart, "--force", help=_FORCE_SERVED_PROFILE_HELP) _add_compat_platform_flag(gateway_restart) gateway_status = gateway_subparsers.add_parser("status", help="Show gateway status") @@ -90,7 +99,8 @@ def build_gateway_parser( gateway_install = gateway_subparsers.add_parser( "install", help="Install gateway as a systemd/launchd background service") - _flag(gateway_install, "--force", help="Force reinstall") + _flag(gateway_install, "--force", + help="Force reinstall, and install even when the default multiplexer already serves this profile") _flag(gateway_install, "--system", help="Install as a Linux system-level service (starts at boot)") gateway_install.add_argument("--run-as-user", dest="run_as_user", diff --git a/hermes_cli/subcommands/profile.py b/hermes_cli/subcommands/profile.py index 35fbc4b4ec83f..16c83b2de276e 100644 --- a/hermes_cli/subcommands/profile.py +++ b/hermes_cli/subcommands/profile.py @@ -19,13 +19,19 @@ def build_profile_parser(subparsers, *, cmd_profile: Callable) -> None: profile_create.add_argument("profile_name", help="Profile name (lowercase, alphanumeric)") profile_create.add_argument( "--clone", action="store_true", - help="Copy config.yaml, .env, SOUL.md, and skills from active profile") + help="Copy config.yaml, .env, SOUL.md, and skills from active profile " + "(messaging bot tokens/allowlists are left behind; see --clone-channels)") profile_create.add_argument( "--clone-all", action="store_true", - help="Full copy of active profile (all state, excluding per-profile history)") + help="Full copy of active profile (all state, excluding per-profile history and messaging channels)") profile_create.add_argument( "--clone-from", metavar="SOURCE", help="Source profile to clone from; implies --clone unless --clone-all is set") + profile_create.add_argument( + "--clone-channels", action="store_true", + help="Also copy the source's messaging channels (bot tokens, allowlists, platform sections). " + "Two profiles holding one bot token collide; refused when the source is served by a live " + "multiplexed gateway.") profile_create.add_argument( "--no-alias", action="store_true", help="Skip wrapper script creation") profile_create.add_argument( diff --git a/hermes_cli/tools_config.py b/hermes_cli/tools_config.py index a7ed3c2db5dff..5246e0e38a8cf 100644 --- a/hermes_cli/tools_config.py +++ b/hermes_cli/tools_config.py @@ -64,6 +64,7 @@ ("stt", "🎙️ Speech-to-Text", "voice transcription (gateway voice messages + voice mode)"), ("skills", "📚 Skills", "list, view, manage"), ("todo", "📋 Task Planning", "todo_list"), + ("kanban", "📌 Kanban", "opt-in task board tools for this platform"), ("memory", "💾 Memory", "persistent memory across sessions"), ("context_engine", "🧩 Context Engine", "runtime tools from the active context engine"), ("session_search", "🔎 Session Search", "search past conversations"), @@ -92,7 +93,7 @@ def gui_toolset_label(label: str) -> str: # OFF by default for new installs (still in _HERMES_CORE_TOOLS; the checklist won't pre-select them). x_search # auto-enables when xAI creds exist (mirrors HASS_TOKEN → homeassistant); its check_fn still gates the schema. -_DEFAULT_OFF_TOOLSETS = {"homeassistant", "spotify", "discord", "discord_admin", "video", "video_gen", "x_search", "a2a"} +_DEFAULT_OFF_TOOLSETS = {"homeassistant", "spotify", "discord", "discord_admin", "video", "video_gen", "x_search", "a2a", "kanban"} # Config-only capabilities: provider setup in `hermes tools` (TOOL_CATEGORIES) but not model toolsets — zero # schemas, own switch (``stt.enabled``), never in ``platform_toolsets`` or the per-platform checklist. @@ -166,7 +167,7 @@ def _get_plugin_toolset_keys() -> set: def _checklist_toolset_keys(platform: str) -> Set[str]: """Toolset keys the ``hermes tools`` checklist offers for ``platform`` (mirrors ``_prompt_toolset_checklist``); - read-time-resolved toolsets (``kanban``, recovered composites, MCP names) are NOT here.""" + read-time-resolved toolsets (recovered composites, MCP names) are NOT here.""" return { ts_key for ts_key, _, _ in _get_effective_configurable_toolsets() if _toolset_allowed_for_platform(ts_key, platform) and ts_key not in _CONFIG_ONLY_TOOLSETS} @@ -589,6 +590,11 @@ def _get_platform_tools(config: dict, platform: str, *, include_default_mcp_serv explicit_passthrough = {ts for ts in toolset_names if ts not in explicit_known_keys and ts not in platform_default_keys} enabled_toolsets |= _merge_mcp_servers(config, toolset_names, explicit_passthrough, include_default_mcp_servers) + # Legacy profile opt-in is a fallback only. A saved platform list (even + # empty) is authoritative, so a later disable cannot silently re-enable it. + if not explicitly_configured and "kanban" in (config.get("toolsets") or []): + enabled_toolsets.add("kanban") + # agent.disabled_toolsets is a global suppression list (#86661) and runs LAST so it overrides everything # above. It may arrive as a JSON-array string ("['memory']") from `hermes config set` or a JSON-mode editor. disabled_toolsets = (config.get("agent") or {}).get("disabled_toolsets") @@ -944,7 +950,7 @@ def _configure_list(to_configure: List[str], config: dict, *, selected: bool = T def _checklist_diff(new_enabled: Set[str], prev: Set[str], platform: str) -> tuple[Set[str], Set[str]]: - """``(added, removed)`` scoped to the checklist universe, so read-time toolsets (``kanban``) the user never + """``(added, removed)`` scoped to the checklist universe, so read-time toolsets (MCP names) the user never saw a checkbox for don't print as spurious removals.""" universe = _checklist_toolset_keys(platform) return (new_enabled - prev) & universe, (prev - new_enabled) & universe diff --git a/hermes_cli/update_cmd_fleet.py b/hermes_cli/update_cmd_fleet.py index 7b51c6af96763..12ea0894c48e1 100644 --- a/hermes_cli/update_cmd_fleet.py +++ b/hermes_cli/update_cmd_fleet.py @@ -1409,6 +1409,9 @@ def _verify_fleet_after_update(restart, *, _pre_update_plan, _windows_gateway_re # doesn't treat the fleet as healthy; leave the pending marker for catch-up. sys.exit(1) _clear_fleet_restart_pending_marker() + with _best_effort('Automatic gateway migration check failed: %s'): + from hermes_cli.gateway_migrate import maybe_auto_migrate_after_update + maybe_auto_migrate_after_update() def _restart_phase_failure_is_incomplete(surviving, pre_restart_pids) -> bool: diff --git a/hermes_cli/web_models.py b/hermes_cli/web_models.py index e6a0e54dc0b4c..90f430bcbdb26 100644 --- a/hermes_cli/web_models.py +++ b/hermes_cli/web_models.py @@ -401,6 +401,9 @@ class ProfileCreate(BaseModel): clone_from: Optional[str] = None clone_from_default: bool = False # legacy clients; new ones send clone_from explicitly clone_all: bool = False + # Opt-in: also copy the source's messaging channels (bot tokens, allowlists, platform sections). + # Default False — a copied bot credential makes two profiles collide over one bot. + clone_channels: bool = False no_skills: bool = False description: Optional[str] = None provider: Optional[str] = None diff --git a/hermes_cli/web_routers/messaging.py b/hermes_cli/web_routers/messaging.py index b0fcb43010700..30dd899d84bf0 100644 --- a/hermes_cli/web_routers/messaging.py +++ b/hermes_cli/web_routers/messaging.py @@ -870,7 +870,22 @@ def _apply(): "env_keys=%s cleared_keys=%s", platform_id, target_profile or "current", body.enabled, sorted(body.env), sorted(body.clear_env), ) - return {"ok": True, "platform": platform_id} + # A live multiplexer serving this named profile builds the adapter from the new token now + # (its periodic rescan would otherwise pick it up within a cycle); no gateway restart. + hot_served = await asyncio.to_thread(_notify_multiplexer_hot_serve, target_profile) + return {"ok": True, "platform": platform_id, "hot_served": hot_served} + + +def _notify_multiplexer_hot_serve(profile: Optional[str]) -> bool: + """True when a live multiplexer serves the written profile and was told to rebuild its adapters. + Unscoped (no ``?profile=``) means THIS process's profile: Desktop routes a pooled + ``hermes --profile X serve`` without the query (#109088), so X must resolve here too.""" + from hermes_cli.gateway import _current_profile_name, named_profile_served_by_running_multiplexer + from hermes_cli.gateway_multiplex_served import notify_multiplexer_profiles_changed + name = (profile or "").strip() or _current_profile_name() + if not name or name == "default" or not named_profile_served_by_running_multiplexer(name): + return False + return notify_multiplexer_profiles_changed(name) is not None @router.post("/api/messaging/platforms/{platform_id}/test") diff --git a/hermes_cli/web_routers/ops.py b/hermes_cli/web_routers/ops.py index 1bfe78043db41..c3fc4a15af472 100644 --- a/hermes_cli/web_routers/ops.py +++ b/hermes_cli/web_routers/ops.py @@ -250,6 +250,15 @@ async def set_webhook_enabled(name: str, body: WebhookEnabledToggle): @router.post("/api/gateway/start") async def start_gateway(profile: Optional[str] = None): + from hermes_cli.gateway import named_profile_served_by_running_multiplexer + # The spawned `hermes -p X gateway start` would refuse with exit 78 into an action log nobody reads; + # surface the same refusal here so the UI can point at the multiplexer instead of showing "started". + if profile and profile != "default" and await asyncio.to_thread(named_profile_served_by_running_multiplexer, profile): + raise HTTPException( + status_code=409, + detail=f"The default gateway already serves profile '{profile}' as a multiplexer; " + "restart it from the default profile instead of starting a separate gateway.", + ) with http_failure("Failed to spawn gateway start", 500, "Failed to start gateway"): proc = _spawn_hermes_action(_gateway_subcommand(profile, "start"), "gateway-start") return {"ok": True, "pid": proc.pid, "name": "gateway-start"} diff --git a/hermes_cli/web_routers/profiles.py b/hermes_cli/web_routers/profiles.py index 1f0da2fb174cc..d713a57e936ec 100644 --- a/hermes_cli/web_routers/profiles.py +++ b/hermes_cli/web_routers/profiles.py @@ -660,7 +660,8 @@ async def create_profile_endpoint(body: ProfileCreate): bad_request=(ValueError, FileExistsError, FileNotFoundError)): path = profiles_mod.create_profile( name=body.name, clone_from=clone_from, clone_all=body.clone_all, - clone_config=clone_config, no_skills=body.no_skills, description=body.description) + clone_config=clone_config, no_skills=body.no_skills, description=body.description, + clone_channels=body.clone_channels) # Match the CLI flow: fresh named profiles get the bundled skills (cloning already # copied the source's; no_skills wrote the opt-out marker so seeding no-ops) and a # ~/.local/bin wrapper when the alias is safe. diff --git a/hermes_cli/web_server_messaging.py b/hermes_cli/web_server_messaging.py index 6e88d7bbc8353..3aba26be484af 100644 --- a/hermes_cli/web_server_messaging.py +++ b/hermes_cli/web_server_messaging.py @@ -281,18 +281,11 @@ def _channel_managed_env_keys() -> frozenset[str]: "GATEWAY_ALLOW_ALL_USERS", "GATEWAY_PROXY_KEY", "GATEWAY_PROXY_URL"}) -_PLATFORM_ENV_PREFIX_ALIASES: dict[str, tuple[str, ...]] = { - "email": ("EMAIL_",), - "homeassistant": ("HASS_",), - "qqbot": ("QQ_", "QQBOT_"), - "sms": ("TWILIO_",), - "wecom": ("WECOM_BOT_", "WECOM_SECRET"), - "wecom_callback": ("WECOM_CALLBACK_",)} - - def _platform_env_prefixes(platform_id: str) -> tuple[str, ...]: - """Env-var prefixes owned by a messaging platform card.""" - return _PLATFORM_ENV_PREFIX_ALIASES.get(platform_id, (platform_id.upper().replace("-", "_") + "_",)) + """Env-var prefixes owned by a messaging platform card (shared with the profile-clone + channel stripper so a card and a clone agree on which keys belong to a platform).""" + from hermes_cli.profile_channels import platform_env_prefixes + return platform_env_prefixes(platform_id) def _discover_platform_env_vars(platform_id: str) -> tuple[str, ...]: diff --git a/model_tools.py b/model_tools.py index 9eeae9444e4bb..12f03bb55985f 100644 --- a/model_tools.py +++ b/model_tools.py @@ -453,8 +453,11 @@ def _compute_tool_definitions(enabled_toolsets: Optional[List[str]] = None, disa quiet_mode: bool = False, skip_tool_search_assembly: bool = False) -> List[Dict[str, Any]]: """Uncached implementation of :func:`get_tool_definitions`.""" tools_to_include = _select_tool_names(enabled_toolsets, disabled_toolsets, quiet_mode) - # Registry returns only tools whose check_fn passes. - filtered_tools = _apply_dynamic_schemas(registry.get_definitions(tools_to_include, quiet=quiet_mode)) + # Selection is per schema, not per process/profile. Kanban's local checks + # are uncached; the outer definitions cache already keys on this selection. + from tools.kanban_toolset_context import scoped_kanban_toolset_selection + with scoped_kanban_toolset_selection(enabled_toolsets): + filtered_tools = _apply_dynamic_schemas(registry.get_definitions(tools_to_include, quiet=quiet_mode)) global _last_resolved_tool_names _last_resolved_tool_names = [t["function"]["name"] for t in filtered_tools] diff --git a/plugins/platforms/feishu/adapter.py b/plugins/platforms/feishu/adapter.py index eb48876cafb30..c526cfa3d7278 100644 --- a/plugins/platforms/feishu/adapter.py +++ b/plugins/platforms/feishu/adapter.py @@ -1194,6 +1194,8 @@ def _sdk_build(request_cls: Any, **fields: Any) -> Any: class FeishuAdapter(BasePlatformAdapter): """Feishu/Lark bot adapter.""" + # Answers /p//... on the default listener for a served secondary (shared_ingress). + serves_profile_prefix: bool = True supports_code_blocks = True # Feishu renders fenced code blocks splits_long_messages = True # send() chunks via truncate_message(MAX_MESSAGE_LENGTH) diff --git a/plugins/platforms/line/adapter.py b/plugins/platforms/line/adapter.py index 6c0dfe08da69d..afca20c84853e 100644 --- a/plugins/platforms/line/adapter.py +++ b/plugins/platforms/line/adapter.py @@ -366,6 +366,8 @@ def _coerce(cast: Callable[[Any], Any], value: Any, default: Any) -> Any: class LineAdapter(BasePlatformAdapter): """LINE Messaging API gateway adapter (no message editing → REQUIRES_EDIT_FINALIZE stays False).""" + # Answers /p//... on the default listener for a served secondary (shared_ingress). + serves_profile_prefix: bool = True def __init__(self, config, **kwargs): super().__init__(config=config, platform=Platform("line")) diff --git a/plugins/platforms/sms/adapter.py b/plugins/platforms/sms/adapter.py index d17d4e11915c0..c36425866be43 100644 --- a/plugins/platforms/sms/adapter.py +++ b/plugins/platforms/sms/adapter.py @@ -83,6 +83,8 @@ def check_sms_requirements() -> bool: class SmsAdapter(BasePlatformAdapter): """Twilio SMS <-> Hermes: one session per inbound number; replies always from TWILIO_PHONE_NUMBER.""" + # Answers /p//... on the default listener for a served secondary (shared_ingress). + serves_profile_prefix: bool = True MAX_MESSAGE_LENGTH = MAX_SMS_LENGTH diff --git a/plugins/platforms/teams/adapter.py b/plugins/platforms/teams/adapter.py index 41b441df3012a..f649413f19871 100644 --- a/plugins/platforms/teams/adapter.py +++ b/plugins/platforms/teams/adapter.py @@ -332,6 +332,8 @@ def _approval_body(cmd: str, desc: str, *, always: bool = False) -> list: class TeamsAdapter(BasePlatformAdapter): """Microsoft Teams adapter using the microsoft-teams-apps SDK.""" + # Answers /p//... on the default listener for a served secondary (shared_ingress). + serves_profile_prefix: bool = True MAX_MESSAGE_LENGTH = 28000 # Teams text message limit (~28 KB) splits_long_messages = True # send() chunks via truncate_message() diff --git a/plugins/platforms/wecom/callback_adapter.py b/plugins/platforms/wecom/callback_adapter.py index 6fd03e05cdbdd..ae70ce93bf994 100644 --- a/plugins/platforms/wecom/callback_adapter.py +++ b/plugins/platforms/wecom/callback_adapter.py @@ -81,6 +81,8 @@ def _ack(): class WecomCallbackAdapter(BasePlatformAdapter): + # Answers /p//... on the default listener for a served secondary (shared_ingress). + serves_profile_prefix: bool = True def __init__(self, config: PlatformConfig): super().__init__(config, Platform.WECOM_CALLBACK) extra = config.extra or {} diff --git a/tests/agent/test_redact.py b/tests/agent/test_redact.py index 8d5a964e63634..2811a05d03512 100644 --- a/tests/agent/test_redact.py +++ b/tests/agent/test_redact.py @@ -59,6 +59,13 @@ def test_gitlab_prefix_requires_word_boundary_and_length(self): ]: assert redact_sensitive_text(benign) == benign + def test_agentmail_prefix_needs_an_opaque_key_body(self): + """The documented prefix alone is not enough to identify an AgentMail secret.""" + for benign in ["schema.am_example_identifier_123", "path/to/am_monthly_report.sql"]: + assert redact_sensitive_text(benign) == benign + for key in ("am_" + "0123456789abcdef" * 2, "am_" + "Ab9" * 8, "am_org_" + "Zq7k" * 6): + assert key[-12:] not in redact_sensitive_text(f"leaked {key} in output"), key + def test_slack_token(self): token = "xoxb-" + "0" * 12 + "-" + "a" * 14 result = redact_sensitive_text(token) diff --git a/tests/cron/test_scheduler_provider.py b/tests/cron/test_scheduler_provider.py index 172a1492d171c..cd1851a975eda 100644 --- a/tests/cron/test_scheduler_provider.py +++ b/tests/cron/test_scheduler_provider.py @@ -914,3 +914,42 @@ def _tick(*args, **kwargs): assert recovery_homes == [str(failing_home), str(healthy_home)] # The failing profile stays in rotation: its ledger may still hold jobs. assert set(tick_homes) == {str(failing_home), str(healthy_home)} + + +def test_multiplex_ticker_reenumerates_profiles_each_cycle(tmp_path): + """Hot-serve: with a callable ``profile_homes`` the ticker re-reads the served set every cycle, + so a profile created after the multiplexer started gets its jobs fired without a restart.""" + import threading + from unittest.mock import patch + from cron.scheduler_provider import InProcessCronScheduler + from hermes_constants import get_hermes_home + + alpha = tmp_path / "alpha" + gamma = tmp_path / "gamma" + (alpha / "cron").mkdir(parents=True) + homes = [("alpha", alpha)] + stop = threading.Event() + ticked: list[str] = [] + + def _tick(*args, **kwargs): + ticked.append(str(get_hermes_home())) + if len(ticked) == 1: # "hermes profile create gamma" happens between two cycles + (gamma / "cron").mkdir(parents=True) + homes.append(("gamma", gamma)) + if len(ticked) >= 4: + stop.set() + return 0 + + provider = InProcessCronScheduler() + with patch("cron.scheduler.tick", side_effect=_tick): + thread = threading.Thread( + target=provider.start, args=(stop,), + kwargs={"interval": 0, "profile_homes": lambda: list(homes)}, daemon=True, + ) + thread.start() + thread.join(timeout=5) + stop.set() + thread.join(timeout=5) + + assert not thread.is_alive() + assert str(gamma) in ticked, ticked diff --git a/tests/gateway/test_multiplex_hot_serve.py b/tests/gateway/test_multiplex_hot_serve.py new file mode 100644 index 0000000000000..fa447e734f2e0 --- /dev/null +++ b/tests/gateway/test_multiplex_hot_serve.py @@ -0,0 +1,164 @@ +"""Hot-serve invariants for ``gateway.multiplex_profiles`` (``gateway/run_profile_reconcile.py``). + +The multiplexer used to enumerate ``profiles/`` once at boot; these pin the runtime reconcile: a +profile created afterwards is served, a deleted one is torn down and unrouted, a served profile whose +config/.env changed (bot token added after create) gets its adapters, and none of it touches the other +profiles' live adapters. The cron ticker's live enumerator is covered in ``tests/cron``. +""" +import asyncio +import json +from pathlib import Path +from unittest.mock import MagicMock, patch + +import pytest + +from gateway.config import GatewayConfig, Platform +from gateway.run import GatewayRunner +from gateway.run_profile_reconcile import profile_serve_signature + + +class _Adapter: + platform = Platform.DISCORD + + def __init__(self, token): + self.token = token + self.disconnected = False + self.cancelled = False + + async def disconnect(self): + self.disconnected = True + + async def cancel_background_tasks(self): + self.cancelled = True + + +def _runner(tmp_path, monkeypatch): + home = tmp_path / ".hermes" + (home / "profiles").mkdir(parents=True) + monkeypatch.setenv("HERMES_HOME", str(home)) + monkeypatch.setattr(Path, "home", lambda: tmp_path) + runner = object.__new__(GatewayRunner) + runner.config = GatewayConfig(multiplex_profiles=True) + runner._running = True + runner._primary_profile_name = "default" + runner.adapters = {} + runner._profile_adapters = {} + runner._profile_failed_platforms = {} + runner._failed_platforms = {} + runner._agent_cache = {} + runner._agent_cache_lock = None + runner.pairing_store = MagicMock() + runner.pairing_stores = {} + runner._adapter_disconnect_timeout_secs = lambda: 0.5 + started = [] + + async def _start(profile_name, profile_home, claimed): + started.append(profile_name) + token = (profile_home / ".env").read_text() if (profile_home / ".env").exists() else "" + if "DISCORD_BOT_TOKEN" not in token: + return 0 + runner._profile_adapters.setdefault(profile_name, {})[Platform.DISCORD] = _Adapter(token) + return 1 + + runner._start_one_profile_adapters = _start + runner._adapter_credential_fingerprint = lambda adapter: getattr(adapter, "token", None) + runner._started = started + return runner, home + + +def _mkprofile(home, name, env=""): + d = home / "profiles" / name + d.mkdir(parents=True, exist_ok=True) + (d / "config.yaml").write_text("model: {default: m}\n") + (d / ".env").write_text(env) + return d + + +def _served_record(home): + return json.loads((home / "gateway_state.json").read_text()).get("served_profiles") + + +@pytest.mark.asyncio +async def test_created_then_credentialed_profile_is_served_without_restart(tmp_path, monkeypatch): + runner, home = _runner(tmp_path, monkeypatch) + alpha_dir = _mkprofile(home, "alpha", "DISCORD_BOT_TOKEN=alpha-token\n") + with patch("hermes_cli.profiles.get_active_profile_name", return_value="default"): + await runner._start_secondary_profile_adapters() + alpha_adapter = runner._profile_adapters["alpha"][Platform.DISCORD] + assert _served_record(home) == ["default", "alpha"] + + # 1. Created while running, no token yet: served (routes/prefixes/cron), zero adapters. + gamma_dir = _mkprofile(home, "gamma") + result = await runner.reconcile_served_profiles() + assert result["added"] == ["gamma"] + assert _served_record(home) == ["default", "alpha", "gamma"] + assert "gamma" in runner.pairing_stores + assert Platform.DISCORD not in runner._profile_adapters.get("gamma", {}) + + # 2. Token added afterwards: the rescan builds the adapter (never "adapter-less forever"). + (gamma_dir / ".env").write_text("DISCORD_BOT_TOKEN=gamma-token\n") + result = await runner.reconcile_served_profiles() + assert result["rescanned"] == ["gamma"] + assert runner._profile_adapters["gamma"][Platform.DISCORD].token.strip().endswith("gamma-token") + + # 3. A no-op rescan and the whole sequence never touched alpha's live adapter. + assert await runner.reconcile_served_profiles() == { + "added": [], "removed": [], "rescanned": [], "reason": "request", + "served_profiles": ["default", "alpha", "gamma"], + } + assert runner._profile_adapters["alpha"][Platform.DISCORD] is alpha_adapter + assert alpha_adapter.disconnected is False + assert runner._started.count("alpha") == 1 + assert profile_serve_signature(alpha_dir) == runner._served_profile_signatures["alpha"] + + +@pytest.mark.asyncio +async def test_deleted_profile_is_torn_down_and_unrouted_others_untouched(tmp_path, monkeypatch): + runner, home = _runner(tmp_path, monkeypatch) + _mkprofile(home, "alpha", "DISCORD_BOT_TOKEN=alpha-token\n") + gamma_dir = _mkprofile(home, "gamma", "DISCORD_BOT_TOKEN=gamma-token\n") + with patch("hermes_cli.profiles.get_active_profile_name", return_value="default"): + await runner._start_secondary_profile_adapters() + alpha_adapter = runner._profile_adapters["alpha"][Platform.DISCORD] + gamma_adapter = runner._profile_adapters["gamma"][Platform.DISCORD] + runner._agent_cache = {"agent:gamma:discord:dm:1": ("agent",), "agent:alpha:discord:dm:1": ("agent",)} + evicted = [] + runner._evict_cached_agent = evicted.append + reconnect = asyncio.get_running_loop().create_task(asyncio.sleep(3600)) + runner._profile_failed_platforms = {"gamma": {Platform.TELEGRAM: reconnect}} + + from hermes_constants import mark_named_profile_deleted + mark_named_profile_deleted(gamma_dir) # what ``delete_profile`` does before rmtree + result = await runner.reconcile_served_profiles() + + assert result["removed"] == ["gamma"] + assert gamma_adapter.disconnected is True and gamma_adapter.cancelled is True + assert "gamma" not in runner._profile_adapters + assert "gamma" not in runner.pairing_stores + assert reconnect.cancelled() + assert evicted == ["agent:gamma:discord:dm:1"] + assert _served_record(home) == ["default", "alpha"] + assert runner._profile_adapters["alpha"][Platform.DISCORD] is alpha_adapter + assert alpha_adapter.disconnected is False + + +@pytest.mark.asyncio +async def test_hot_added_profile_cannot_double_claim_a_live_secondary_token(tmp_path, monkeypatch): + """Boot's duplicate-credential guard sees every profile at once; a hot add must see the LIVE + secondaries' claims too, or the new profile starts a second poller on alpha's bot.""" + runner, home = _runner(tmp_path, monkeypatch) + _mkprofile(home, "alpha", "DISCORD_BOT_TOKEN=shared\n") + seen_claims = {} + + async def _start(profile_name, profile_home, claimed): + seen_claims[profile_name] = dict(claimed) + runner._profile_adapters.setdefault(profile_name, {})[Platform.DISCORD] = _Adapter("shared") + return 1 + + runner._start_one_profile_adapters = _start + with patch("hermes_cli.profiles.get_active_profile_name", return_value="default"): + await runner._start_secondary_profile_adapters() + _mkprofile(home, "dupe", "DISCORD_BOT_TOKEN=shared\n") + await runner.reconcile_served_profiles() + fp = GatewayRunner._adapter_credential_fingerprint(_Adapter("shared")) + assert seen_claims["dupe"].get((Platform.DISCORD, fp)) == "alpha" diff --git a/tests/gateway/test_multiplex_mcp_discovery.py b/tests/gateway/test_multiplex_mcp_discovery.py index a74332da1ee95..f9748ee0ad050 100644 --- a/tests/gateway/test_multiplex_mcp_discovery.py +++ b/tests/gateway/test_multiplex_mcp_discovery.py @@ -343,6 +343,53 @@ def test_scoped_teardown_restores_alias_from_surviving_profile( assert registry.get_toolset_alias_target("shared") == "mcp-shared" +@pytest.mark.asyncio +async def test_reload_mcp_projects_scoped_connection_keys_to_public_names( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """Tuple connection keys stay server names in the public reload summary.""" + from gateway.run import GatewayRunner + from tools import mcp_tool + from tools import mcp_tool_discovery as _mcp_discovery + from tools import mcp_tool_lifecycle as _mcp_lifecycle + + worker_home = tmp_path / "profiles" / "worker" + worker_home.mkdir(parents=True) + worker_scope = hermes_home_key(worker_home) + launch_scope = hermes_home_key(tmp_path / "default") + connection_key = (launch_scope, "shared") + + runner = GatewayRunner.__new__(GatewayRunner) + runner.config = GatewayConfig(multiplex_profiles=True) + runner._resolve_profile_home_for_source = MagicMock(return_value=worker_home) + runner._agent_cache = {} + runner._agent_cache_lock = None + runner._async_session_store = SimpleNamespace( + get_or_create_session=MagicMock(side_effect=RuntimeError("skip transcript")), + ) + + live_server = SimpleNamespace(session=object(), _config={}, _tools=[], tool_timeout=30, + initialize_result=None, _registered_tool_names=[]) + monkeypatch.setattr(mcp_tool, "_servers", {connection_key: live_server}) + monkeypatch.setattr(mcp_tool, "_server_scope_keys", {connection_key: launch_scope}) + monkeypatch.setattr(mcp_tool, "_server_tool_scopes", {connection_key: {worker_scope}}) + monkeypatch.setattr(mcp_tool, "_mcp_registry_scope", lambda: worker_scope) + monkeypatch.setattr(_mcp_lifecycle, "shutdown_mcp_servers", lambda **_kwargs: None) + monkeypatch.setattr(_mcp_discovery, "discover_mcp_tools", lambda: []) + + event = MessageEvent( + text="/reload-mcp", message_id="m1", + source=SessionSource( + platform=Platform.TELEGRAM, user_id="u1", chat_id="c1", + chat_type="dm", profile="worker", + ), + ) + + result = await runner._execute_mcp_reload(event) + + assert "shared" in result + + @pytest.mark.parametrize("worker_cfg", [ {"url": "https://worker.example/mcp"}, # different route {"url": "https://default.example/mcp", "headers": {"Authorization": "Bearer worker"}}, # same route, other credentials diff --git a/tests/gateway/test_multiplex_secondary_port_binding_env.py b/tests/gateway/test_multiplex_secondary_port_binding_env.py new file mode 100644 index 0000000000000..804797029122f --- /dev/null +++ b/tests/gateway/test_multiplex_secondary_port_binding_env.py @@ -0,0 +1,59 @@ +"""A secondary profile's port-binding credential wires the key without claiming the listener (#100397). + +The docs require ``API_SERVER_KEY`` in a secondary's ``.env`` for ``/p//`` auth, yet the env pass +turned it into ``api_server.enabled = True`` and ``_load_secondary_profile_config`` skipped the whole profile. +""" + +from __future__ import annotations + +import pytest + + +@pytest.fixture +def multiplex_root(tmp_path, monkeypatch): + root = tmp_path / "hermes" + (root / "profiles" / "coder").mkdir(parents=True) + (root / "config.yaml").write_text("model: {default: x}\n") + (root / "profiles" / "coder" / "config.yaml").write_text("model: {default: x}\n") + (root / "profiles" / "coder" / ".env").write_text("API_SERVER_KEY=abcdefghijklmnopqrstuvwxyz123456\n") + monkeypatch.setenv("HERMES_HOME", str(root)) + monkeypatch.delenv("API_SERVER_KEY", raising=False) + import hermes_constants + monkeypatch.setattr(hermes_constants, "_default_hermes_root_memo", None) + from agent import secret_scope + monkeypatch.setattr(secret_scope, "_MULTIPLEX_ACTIVE", True) + return root + + +def _api_server(home): + from gateway.config import Platform, load_gateway_config + from gateway.run import _profile_runtime_scope + with _profile_runtime_scope(home): + return load_gateway_config().platforms.get(Platform.API_SERVER) + + +def test_secondary_api_server_key_wires_key_but_does_not_enable_listener(multiplex_root): + pc = _api_server(multiplex_root / "profiles" / "coder") + assert pc is not None and pc.extra.get("key") + assert pc.enabled is False + + +def test_default_profile_api_server_key_still_enables_listener(multiplex_root): + (multiplex_root / ".env").write_text("API_SERVER_KEY=abcdefghijklmnopqrstuvwxyz123456\n") + pc = _api_server(multiplex_root) + assert pc is not None and pc.enabled is True and pc.extra.get("key") + + +@pytest.mark.parametrize( + ("cmdline", "expected"), + [ + ("/v/python -m hermes_cli.main -p ops-2 gateway run", False), + ("/v/python -m hermes_cli.main --profile ops2 gateway run", False), + ("/v/python -m hermes_cli.main -p ops gateway run", True), + ("/v/python -m hermes_cli.main --profile=ops gateway run", True), + ], +) +def test_profile_match_is_token_equality_not_substring(tmp_path, cmdline, expected): + """``-p ops`` must never claim (or let ``gateway stop`` SIGTERM) an ``-p ops-2`` gateway.""" + from gateway.status import _command_line_belongs_to_profile + assert _command_line_belongs_to_profile(cmdline, tmp_path / "profiles" / "ops") is expected diff --git a/tests/hermes_cli/test_config.py b/tests/hermes_cli/test_config.py index f58716552ff95..82a3065603241 100644 --- a/tests/hermes_cli/test_config.py +++ b/tests/hermes_cli/test_config.py @@ -1906,3 +1906,13 @@ def test_config_command_unset_exits_cleanly_on_broken_yaml(self, tmp_path, capsy assert excinfo.value.code == 1 assert "not valid YAML" in capsys.readouterr().err assert config_path.read_text(encoding="utf-8") == original + + +def test_gateway_multiplex_keys_are_recognized_config_keys(): + """``hermes config set gateway.multiplex_profiles true`` used to warn 'not a recognized config + key' although gateway/config.py reads it; the key (and profile_routes) live in DEFAULT_CONFIG.""" + from hermes_cli.config import _validate_config_key + from hermes_cli.config_defaults import DEFAULT_CONFIG + assert DEFAULT_CONFIG["gateway"]["multiplex_profiles"] is False + assert _validate_config_key("gateway.multiplex_profiles") == (True, None) + assert _validate_config_key("gateway.profile_routes") == (True, None) diff --git a/tests/hermes_cli/test_gateway_migrate_multiplex.py b/tests/hermes_cli/test_gateway_migrate_multiplex.py new file mode 100644 index 0000000000000..365ac6b6cbc1e --- /dev/null +++ b/tests/hermes_cli/test_gateway_migrate_multiplex.py @@ -0,0 +1,201 @@ +"""``hermes gateway migrate``: preflight verdicts, apply/rollback bookkeeping, and the update hook. + +Service layer is faked through the module's ``_installed_service`` / ``_service_op`` seams (the same +shape ``hermes gateway install`` tests use); the default gateway boot is faked by writing the +``served_profiles`` record the real multiplexer writes. Blockers reuse the gateway's own credential +fingerprint and port-binding predicates, so the tests assert verdict → effect, not internal lists. +""" + +from __future__ import annotations + +import json +import os +from pathlib import Path +from types import SimpleNamespace + +import pytest + +import hermes_constants +from hermes_cli import gateway_migrate as gm + + +@pytest.fixture +def fleet(tmp_path, monkeypatch): + """default + coder + ops; both secondaries run a 'live' standalone gateway with a systemd unit.""" + root = tmp_path / "hermes" + for sub in ("profiles/coder", "profiles/ops"): + (root / sub).mkdir(parents=True) + (root / "config.yaml").write_text("model:\n default: x\n", encoding="utf-8") + (root / ".env").write_text("TELEGRAM_BOT_TOKEN=111111:default-token\n", encoding="utf-8") + (root / "profiles/coder/.env").write_text("TELEGRAM_BOT_TOKEN=222222:coder-token\n", encoding="utf-8") + (root / "profiles/ops/.env").write_text("DISCORD_BOT_TOKEN=ops-discord-333333\n", encoding="utf-8") + monkeypatch.setenv("HERMES_HOME", str(root)) + monkeypatch.delenv("GATEWAY_MULTIPLEX_PROFILES", raising=False) + for name in ("TELEGRAM_BOT_TOKEN", "DISCORD_BOT_TOKEN", "API_SERVER_KEY", "WEBHOOK_ENABLED"): + monkeypatch.delenv(name, raising=False) + monkeypatch.setattr(hermes_constants, "_default_hermes_root_memo", None) + + state = SimpleNamespace( + services={"coder": ("systemd", False), "ops": ("systemd", False)}, + pids={"coder": 4101, "ops": 4102}, + ops=[], + ) + + def _name(home: Path) -> str: + return hermes_constants.profile_name_for_home(home) or "default" + + def _service_op(kind, system, verb, home): + name = _name(home) + state.ops.append((name, verb)) + if verb == "uninstall": + state.services.pop(name, None) + elif verb == "install": + state.services[name] = (kind, system) + elif verb in ("start", "restart") and name == "default": + # What the real multiplexer does at startup: record the served set in the default home. + (root / "gateway.pid").write_text(json.dumps({"pid": os.getpid(), "hermes_home": str(root)})) + (root / "gateway_state.json").write_text(json.dumps({ + "pid": os.getpid(), "hermes_home": str(root), "gateway_state": "running", + "served_profiles": ["default", "coder", "ops"], + })) + + monkeypatch.setattr(gm, "_installed_service", lambda home: state.services.get(_name(home))) + monkeypatch.setattr(gm, "_live_gateway_pid", lambda home: state.pids.get(_name(home))) + monkeypatch.setattr(gm, "_service_op", _service_op) + monkeypatch.setattr(gm, "_stop_gateway_process", lambda home: state.pids.pop(_name(home), None)) + monkeypatch.setattr(gm, "_host_supports_migration", lambda: None) + # The live identity probe is tested separately; this fixture models a verified running gateway. + from hermes_cli import gateway_multiplex_served + monkeypatch.setattr(gateway_multiplex_served, "live_default_gateway_pid", lambda: os.getpid()) + state.root = root + return state + + +def _config_flag(root: Path): + import yaml + raw = yaml.safe_load((root / "config.yaml").read_text(encoding="utf-8")) or {} + return (raw.get("gateway") or {}).get("multiplex_profiles") + + +def test_dry_run_and_blocked_preflight_change_nothing(fleet, capsys): + plan = gm.build_migration_plan() + assert not plan.blocked and plan.eligible_for_migration() + assert [p.name for p in plan.standalone_secondaries] == ["coder", "ops"] + + gm.cmd_migrate(SimpleNamespace(multiplex=True, standalone=False, dry_run=True, yes=True)) # returns; no exit + assert "dry run" in capsys.readouterr().out + # Blocked: coder reuses the default's Telegram token -> the gateway's own fingerprint says duplicate. + (fleet.root / "profiles/coder/.env").write_text("TELEGRAM_BOT_TOKEN=111111:default-token\n", encoding="utf-8") + blocked = gm.build_migration_plan() + assert blocked.blocked and "profile_routes" in blocked.blockers[0] and "'coder'" in blocked.blockers[0] + with pytest.raises(SystemExit) as exc: + gm.cmd_migrate(SimpleNamespace(multiplex=True, standalone=False, dry_run=False, yes=True)) + assert exc.value.code == 1 + assert fleet.ops == [] and fleet.services == {"coder": ("systemd", False), "ops": ("systemd", False)} + assert fleet.pids == {"coder": 4101, "ops": 4102} and _config_flag(fleet.root) is None + assert not (fleet.root / gm.MANIFEST_NAME).exists() + assert "nothing will be changed" in capsys.readouterr().out + + +def test_apply_records_manifest_flips_flag_and_rollback_restores(fleet, capsys): + plan = gm.build_migration_plan() + assert gm.apply_migration(plan, served_wait=5.0) is True + manifest = json.loads((fleet.root / gm.MANIFEST_NAME).read_text(encoding="utf-8")) + assert {s["profile"] for s in manifest["secondaries"]} == {"coder", "ops"} + assert all(s["service"] == {"kind": "systemd", "system": False} for s in manifest["secondaries"]) + assert _config_flag(fleet.root) is True + assert "coder" not in fleet.services and "ops" not in fleet.services and fleet.pids == {} + # The default is brought up on the SAME service manager the secondaries used. + assert fleet.services["default"] == ("systemd", False) + assert ("default", "install") in fleet.ops and ("default", "start") in fleet.ops + assert "serves 3 profiles" in capsys.readouterr().out + # Idempotent: a second run sees the live multiplexer and refuses cleanly. + again = gm.build_migration_plan() + assert again.already_multiplexed and gm.apply_migration(again) is True + + fleet.ops.clear() + assert gm.rollback_migration(fleet.root) is True + assert _config_flag(fleet.root) is False + # The default had no gateway before migration, so the service temporarily + # transferred from the secondaries must not survive rollback. + assert fleet.services == {"coder": ("systemd", False), "ops": ("systemd", False)} + assert [op for op in fleet.ops if op[0] != "default"] == [ + ("coder", "install"), ("coder", "start"), ("ops", "install"), ("ops", "start")] + assert not (fleet.root / gm.MANIFEST_NAME).exists() + + +def test_secondary_port_binder_is_blocked_before_migration(fleet): + """Every enabled secondary port binder is blocked until runtime consumes its prefixed config.""" + (fleet.root / "profiles/ops/.env").write_text( + "DISCORD_BOT_TOKEN=ops-discord-333333\nAPI_SERVER_KEY=ops-api-key-abcdef\n", encoding="utf-8") + (fleet.root / "profiles/ops/config.yaml").write_text( + "platforms:\n api_server:\n enabled: true\n extra:\n port: 9999\n", encoding="utf-8") + plan = gm.build_migration_plan() + assert plan.blocked and "api_server" in plan.blockers[0] and "/p/ops/" in plan.blockers[0] + + +def test_serves_profile_prefix_is_read_from_adapter_classes(): + from gateway.platforms.api_server import APIServerAdapter + from gateway.platforms.webhook import WebhookAdapter + assert APIServerAdapter.serves_profile_prefix and WebhookAdapter.serves_profile_prefix + assert gm.platform_serves_profile_prefix("api_server") is True + assert gm.platform_serves_profile_prefix("webhook") is True + # Every adapter that binds through shared_ingress.bind_listener is served at /p// + # for a secondary, so migrate must report it as a notice, never a blocker. + for platform in ("sms", "line", "teams", "bluebubbles", "whatsapp_cloud", "msgraph_webhook"): + assert gm.platform_serves_profile_prefix(platform) is True, platform + # An outbound-only adapter never declares it (and never needs to). + assert gm.platform_serves_profile_prefix("telegram") is False + + +def test_update_hook_migrates_when_unblocked_and_only_warns_when_blocked(fleet, capsys): + gm.maybe_auto_migrate_after_update() + out = capsys.readouterr().out + assert "Migrating per-profile gateways" in out and "serves 3 profiles" in out + assert _config_flag(fleet.root) is True and (fleet.root / gm.MANIFEST_NAME).exists() + + # Blocked fleet: warning block with the fix + one-liner; nothing changes. + for f in ("gateway.pid", "gateway_state.json", gm.MANIFEST_NAME): + (fleet.root / f).unlink() + (fleet.root / "config.yaml").write_text("model:\n default: x\n", encoding="utf-8") + fleet.services.update({"coder": ("systemd", False)}); fleet.services.pop("default", None) + fleet.pids.update({"coder": 4101}); fleet.ops.clear() + (fleet.root / "profiles/coder/.env").write_text("TELEGRAM_BOT_TOKEN=111111:default-token\n", encoding="utf-8") + gm.maybe_auto_migrate_after_update() + out = capsys.readouterr().out + assert gm.MIGRATE_COMMAND in out and "profile_routes" in out + assert fleet.ops == [] and _config_flag(fleet.root) is None + + +def test_update_hook_never_touches_single_profile_or_already_multiplexed(fleet, capsys): + fleet.services.clear(); fleet.pids.clear() # secondaries exist but run no gateway of their own + gm.maybe_auto_migrate_after_update() + assert capsys.readouterr().out == "" and _config_flag(fleet.root) is None + + +def test_explicit_migrate_with_no_standalone_secondaries_still_flips_flag_and_restarts_default(fleet, capsys, monkeypatch): + """The user typed --multiplex: 'nothing to migrate' + flag left off was a no-op the user did not ask + for. The update hook keeps its no-op (previous test); the explicit command proceeds.""" + fleet.services.clear(); fleet.pids.clear() + + def _detached(home): # no service manager anywhere -> detached start writes the served record + fleet.services["default-detached"] = True + (fleet.root / "gateway.pid").write_text(json.dumps({"pid": os.getpid(), "hermes_home": str(fleet.root)})) + (fleet.root / "gateway_state.json").write_text(json.dumps({ + "pid": os.getpid(), "hermes_home": str(fleet.root), "gateway_state": "running", + "served_profiles": ["default", "coder", "ops"]})) + return True + monkeypatch.setattr(gm, "_spawn_detached_gateway", _detached) + assert not gm.build_migration_plan().standalone_secondaries + with pytest.raises(SystemExit) as exc: + gm.cmd_migrate(SimpleNamespace(multiplex=True, standalone=False, dry_run=False, yes=True)) + assert exc.value.code == 0 + assert fleet.services.pop("default-detached") is True + assert _config_flag(fleet.root) is True + manifest = json.loads((fleet.root / gm.MANIFEST_NAME).read_text(encoding="utf-8")) + assert manifest["secondaries"] == [] and manifest["flag_was"] is False + assert ("default", "install") not in fleet.ops + out = capsys.readouterr().out + assert "serves 3 profiles" in out + # The same manifest rolls it back: flag restored, nothing to reinstall. + assert gm.rollback_migration(fleet.root) is True and _config_flag(fleet.root) is False diff --git a/tests/hermes_cli/test_gateway_multiplex_served_record.py b/tests/hermes_cli/test_gateway_multiplex_served_record.py new file mode 100644 index 0000000000000..715a27bb635ec --- /dev/null +++ b/tests/hermes_cli/test_gateway_multiplex_served_record.py @@ -0,0 +1,98 @@ +"""The live multiplexer's recorded ``served_profiles`` decides "served"; every start verb honours it. + +- Probe: ``named_profile_served_by_running_multiplexer`` reads the pid-verified default + ``gateway_state.json`` first (env-only opt-in on the default profile is invisible to ``hermes -p X``); + config/env derivation is only the fallback when the key is absent. +- ``gateway start`` / ``install`` / ``restart`` refuse (exit 78) like ``run`` does, so a served profile + never gets a permanently failing systemd unit or a launchd respawn loop; ``--force`` overrides. +- ``hermes status`` / ``cron status`` on a satellite say running-via-multiplexer; the default lists served. +""" + +from __future__ import annotations + +import argparse +import contextlib +import io +import json +import os + +import pytest + + +@pytest.fixture +def served_root(tmp_path, monkeypatch): + root = tmp_path / "hermes" + (root / "profiles" / "coder").mkdir(parents=True) + (root / "profiles" / "other").mkdir(parents=True) + (root / "config.yaml").write_text("model: {default: x}\n") # NO multiplex flag: env-only opt-in + (root / "gateway.pid").write_text(json.dumps({"pid": os.getpid(), "hermes_home": str(root)})) + (root / "gateway_state.json").write_text(json.dumps( + {"pid": os.getpid(), "hermes_home": str(root), "served_profiles": ["default", "coder"]})) + monkeypatch.setenv("HERMES_HOME", str(root / "profiles" / "coder")) + monkeypatch.delenv("GATEWAY_MULTIPLEX_PROFILES", raising=False) + import hermes_constants + monkeypatch.setattr(hermes_constants, "_default_hermes_root_memo", None) + from hermes_cli import gateway_multiplex_served + monkeypatch.setattr(gateway_multiplex_served, "live_default_gateway_pid", lambda: os.getpid()) + return root + + +def test_probe_trusts_live_record_over_cli_side_config(served_root): + from hermes_cli.gateway import named_profile_served_by_running_multiplexer + assert named_profile_served_by_running_multiplexer("coder") is True + assert named_profile_served_by_running_multiplexer("other") is False + # Config says multiplex on, but the running gateway did not pick up 'other': the record wins. + (served_root / "config.yaml").write_text("gateway: {multiplex_profiles: true}\n") + assert named_profile_served_by_running_multiplexer("other") is False + + +def test_probe_falls_back_to_config_only_without_recorded_key(served_root): + from hermes_cli.gateway import named_profile_served_by_running_multiplexer + (served_root / "gateway_state.json").write_text(json.dumps({"pid": os.getpid()})) + assert named_profile_served_by_running_multiplexer("coder") is False + (served_root / "config.yaml").write_text("gateway: {multiplex_profiles: true}\n") + assert named_profile_served_by_running_multiplexer("coder") is True + + +@pytest.mark.parametrize("verb", ["start", "install", "restart"]) +def test_service_verbs_refuse_served_profile_with_exit_78(served_root, monkeypatch, verb): + import hermes_cli.gateway as gw + calls: list = [] + monkeypatch.setattr(gw, "_service_backend", lambda: "systemd") + monkeypatch.setattr(gw, "_service_call", lambda backend, v, system: calls.append(v)) + monkeypatch.setattr(gw, "_install_systemd_from_cli", lambda *a, **k: calls.append("install")) + monkeypatch.setattr(gw, "_dispatch_via_service_manager_if_s6", lambda v: False) + monkeypatch.setattr(gw, "_installed_service_kind_for", lambda *a, **k: "systemd") + monkeypatch.setattr(gw, "is_managed", lambda: False) + monkeypatch.setattr(gw, "is_termux", lambda: False) + fn = getattr(gw, f"_cmd_{verb}") + ns = argparse.Namespace(system=False, all=False, force=False, run_as_user=None) + with contextlib.redirect_stdout(io.StringIO()), pytest.raises(SystemExit) as exc: + fn(ns) + assert exc.value.code == gw.GATEWAY_FATAL_CONFIG_EXIT_CODE and calls == [] + + ns.force = True + with contextlib.redirect_stdout(io.StringIO()): + fn(ns) + assert calls, f"--force must let `gateway {verb}` reach the service manager" + + +def test_status_surfaces_agree_for_a_satellite_profile(served_root, monkeypatch): + import hermes_cli.gateway as gw + import hermes_cli.status as st + import hermes_cli.cron as cr + monkeypatch.setattr(gw, "find_gateway_pids", lambda *a, **k: []) + monkeypatch.setattr(gw, "get_gateway_runtime_snapshot", + lambda system=False: gw.GatewayRuntimeSnapshot(manager="systemd (user)")) + monkeypatch.setattr(cr, "_active_cron_provider_name", lambda: "builtin") + monkeypatch.setattr(cr, "_print_active_jobs_summary", lambda jobs: None) + + buf = io.StringIO() + with contextlib.redirect_stdout(buf): + st._render_gateway(None) + assert "running" in buf.getvalue() and "multiplexer" in buf.getvalue() + + buf = io.StringIO() + with contextlib.redirect_stdout(buf): + cr.cron_status() + assert "NOT fire" not in buf.getvalue() and "multiplexer" in buf.getvalue() diff --git a/tests/hermes_cli/test_gateway_multiplex_status.py b/tests/hermes_cli/test_gateway_multiplex_status.py index 0c516db590918..4e379a2c09e85 100644 --- a/tests/hermes_cli/test_gateway_multiplex_status.py +++ b/tests/hermes_cli/test_gateway_multiplex_status.py @@ -19,6 +19,7 @@ def _fake_multiplexer(monkeypatch, tmp_path, *, multiplex: bool): import hermes_constants import gateway.status as status + from hermes_cli import gateway_multiplex_served (tmp_path / "profiles" / "beta").mkdir(parents=True) (tmp_path / "config.yaml").write_text( @@ -28,6 +29,9 @@ def _fake_multiplexer(monkeypatch, tmp_path, *, multiplex: bool): monkeypatch.setenv("HERMES_HOME", str(tmp_path / "profiles" / "beta")) monkeypatch.setattr(hermes_constants, "_default_hermes_root_memo", None) monkeypatch.setattr(status, "_pid_exists", lambda pid: True) + # The production probe validates lock, start-time, and command identity through + # gateway.status.get_running_pid; keep this unit test focused on multiplex routing. + monkeypatch.setattr(gateway_multiplex_served, "live_default_gateway_pid", lambda: os.getpid()) def _run_status(): diff --git a/tests/hermes_cli/test_kanban_db.py b/tests/hermes_cli/test_kanban_db.py index 3d8513ffc403a..66654e1ca4081 100644 --- a/tests/hermes_cli/test_kanban_db.py +++ b/tests/hermes_cli/test_kanban_db.py @@ -24,6 +24,14 @@ from hermes_cli import kanban_db_workspace as kbw +def test_normalize_task_skills_points_to_platform_toolset_configuration(): + with pytest.raises(ValueError) as exc_info: + kb._normalize_task_skills(["kanban"]) + message = str(exc_info.value) + assert "hermes tools enable kanban --platform " in message + assert "platform_toolsets." in message + + @pytest.fixture def kanban_home(tmp_path, monkeypatch): """Isolated HERMES_HOME with an empty kanban DB.""" diff --git a/tests/hermes_cli/test_kanban_worker_spawn_toolsets.py b/tests/hermes_cli/test_kanban_worker_spawn_toolsets.py index 3de7108811030..7537132618327 100644 --- a/tests/hermes_cli/test_kanban_worker_spawn_toolsets.py +++ b/tests/hermes_cli/test_kanban_worker_spawn_toolsets.py @@ -163,8 +163,14 @@ def test_resolve_worker_cli_toolsets_uses_profile_home_not_parent_config(monkeyp assert resolved is not None assert "terminal" in resolved assert "web" in resolved - # The dispatcher adds the task-scoped lifecycle surface later in - # model_tools; this resolver returns only the profile's CLI pin. + # Opt-in is no longer inferred for ordinary chats. The dispatcher-owned + # worker gets lifecycle tools at schema assembly, independently of the + # assignee's saved chat selection. + monkeypatch.setenv("HERMES_KANBAN_TASK", "t_spawn_tools") + from model_tools import get_tool_definitions + names = {t["function"]["name"] for t in get_tool_definitions(resolved, quiet_mode=True, skip_tool_search_assembly=True)} + assert "kanban_complete" in names + assert "kanban_list" not in names assert resolved != ["kanban"] @@ -194,5 +200,9 @@ def test_resolve_worker_cli_toolsets_expands_all_without_unrestricted_sentinel( assert "all" not in resolved assert "*" not in resolved assert "terminal" in resolved + # ``kanban`` is a configurable opt-in, not a native toolset recovered from + # the ``all`` composite. Dispatcher-owned lifecycle tools are appended by + # model_tools when the worker is assembled. + assert "kanban" not in resolved assert "desktop_ui" not in resolved assert "project" not in resolved diff --git a/tests/hermes_cli/test_profile_clone_channels.py b/tests/hermes_cli/test_profile_clone_channels.py new file mode 100644 index 0000000000000..f26a5e4d197fe --- /dev/null +++ b/tests/hermes_cli/test_profile_clone_channels.py @@ -0,0 +1,122 @@ +"""``hermes profile create --clone`` leaves messaging channels behind (``hermes_cli.profile_channels``). + +Invariant, not snapshot: the clone's credential fingerprint set — computed by the gateway's own +``_adapter_credential_fingerprint`` through the migrate preflight — is DISJOINT from the source's, +while provider/tool keys and general config survive; ``--clone-channels`` restores the copy. +""" + +from __future__ import annotations + +from pathlib import Path + +import pytest +import yaml + +import hermes_constants +from hermes_cli import gateway_migrate as gm +from hermes_cli.profile_channels import ( + channel_platforms_configured, shared_channel_credentials, strip_channel_env_file, +) +from hermes_cli.profiles import create_profile + +_SOURCE_ENV = ( + "OPENAI_API_KEY=sk-model-key\n" + "FIRECRAWL_API_KEY=fc-tool-key\n" + "# telegram\n" + "TELEGRAM_BOT_TOKEN=111111:default-telegram-token\n" + "TELEGRAM_ALLOWED_USERS=12345\n" + "TELEGRAM_GROUP_ALLOWED_CHATS=-100999\n" + "DISCORD_BOT_TOKEN=default-discord-token-abcdef\n" + "DISCORD_ALLOWED_USERS=777\n" + "WHATSAPP_ENABLED=true\n" + "API_SERVER_KEY=default-api-server-key-0123456789\n" +) +_SOURCE_CONFIG = { + "model": {"default": "gpt-5", "provider": "openai"}, + "memory": {"provider": "builtin"}, + "platforms": {"telegram": {"enabled": True, "token": "111111:default-telegram-token"}, + "discord": {"enabled": True}}, + "telegram": {"reactions": True, "allowed_chats": "-100999"}, + "discord": {"require_mention": False, "dm_role_auth_guild": "42"}, + "gateway": {"multiplex_profiles": True, "profile_routes": [{"profile": "x", "platform": "telegram"}], + "platform_connect_timeout": 45}, +} + + +@pytest.fixture +def home(tmp_path, monkeypatch): + root = tmp_path / ".hermes" + root.mkdir() + monkeypatch.setattr(Path, "home", lambda: tmp_path) + monkeypatch.setenv("HERMES_HOME", str(root)) + monkeypatch.setattr(hermes_constants, "_default_hermes_root_memo", None) + for name in ("TELEGRAM_BOT_TOKEN", "DISCORD_BOT_TOKEN", "API_SERVER_KEY", "WHATSAPP_ENABLED", + "GATEWAY_MULTIPLEX_PROFILES", "TELEGRAM_ALLOWED_USERS"): + monkeypatch.delenv(name, raising=False) + (root / ".env").write_text(_SOURCE_ENV, encoding="utf-8") + (root / "config.yaml").write_text(yaml.safe_dump(_SOURCE_CONFIG), encoding="utf-8") + (root / "SOUL.md").write_text("Be helpful.", encoding="utf-8") + monkeypatch.setattr(gm, "_installed_service", lambda home: None) + monkeypatch.setattr(gm, "_live_gateway_pid", lambda home: None) + return root + + +def _fingerprints(profile_home: Path) -> set: + """``(platform, fingerprint)`` claims exactly as the migrate preflight / multiplexer see them.""" + with gm._multiplex_read_mode(): + return set(gm._credential_claims(gm._profile_gateway_config(profile_home))) + + +def test_clone_strips_every_channel_credential_but_keeps_model_and_tool_keys(home): + source_claims = _fingerprints(home) + assert {p for p, _ in source_claims} >= {"telegram", "discord"} + + profile_dir = create_profile("bot2", clone_config=True, no_alias=True) + + assert _fingerprints(profile_dir).isdisjoint(source_claims) + assert shared_channel_credentials(profile_dir, home) == [] + env_text = (profile_dir / ".env").read_text(encoding="utf-8") + assert "OPENAI_API_KEY=sk-model-key" in env_text and "FIRECRAWL_API_KEY=fc-tool-key" in env_text + assert "TELEGRAM" not in env_text and "DISCORD" not in env_text + assert "WHATSAPP_ENABLED" not in env_text and "API_SERVER_KEY" not in env_text + cfg = yaml.safe_load((profile_dir / "config.yaml").read_text(encoding="utf-8")) + assert cfg["model"] == _SOURCE_CONFIG["model"] and cfg["memory"] == _SOURCE_CONFIG["memory"] + assert (profile_dir / "SOUL.md").read_text(encoding="utf-8") == "Be helpful." + for section in ("platforms", "telegram", "discord"): + assert section not in cfg + # The clone must not think it is the host's multiplexer, but unrelated gateway knobs survive. + assert "multiplex_profiles" not in cfg["gateway"] and "profile_routes" not in cfg["gateway"] + assert cfg["gateway"]["platform_connect_timeout"] == 45 + # The migrate preflight, which blocked with a duplicate finding per platform, is now clean. + plan = gm.build_migration_plan() + assert not plan.blocked, plan.blockers + + +def test_clone_channels_opt_in_keeps_the_source_channels(home): + profile_dir = create_profile("twin", clone_config=True, no_alias=True, clone_channels=True) + assert _fingerprints(profile_dir) == _fingerprints(home) + assert set(shared_channel_credentials(profile_dir, home)) >= {"telegram", "discord"} + assert set(channel_platforms_configured(profile_dir)) >= {"telegram", "discord", "whatsapp", "api_server"} + assert gm.build_migration_plan().blocked + + +def test_clone_all_drops_pairing_and_platform_state(home): + (home / "platforms" / "pairing").mkdir(parents=True) + (home / "platforms" / "pairing" / "telegram_approved.json").write_text("{}", encoding="utf-8") + (home / "discord_threads.json").write_text("{}", encoding="utf-8") + (home / "memories").mkdir() + (home / "memories" / "MEMORY.md").write_text("remember", encoding="utf-8") + + profile_dir = create_profile("full", clone_all=True, no_alias=True) + + assert not (profile_dir / "platforms").exists() and not (profile_dir / "discord_threads.json").exists() + assert (profile_dir / "memories" / "MEMORY.md").read_text(encoding="utf-8") == "remember" + assert _fingerprints(profile_dir) == set() + + +def test_strip_env_file_keeps_comments_and_unknown_keys_verbatim(tmp_path): + env = tmp_path / ".env" + env.write_text("# header\nexport OPENAI_API_KEY=abc\n\nTELEGRAM_BOT_TOKEN=1:x\nMY_CUSTOM_THING=1\n", encoding="utf-8") + removed = strip_channel_env_file(env) + assert removed == {"telegram": ["TELEGRAM_BOT_TOKEN"]} + assert env.read_text(encoding="utf-8") == "# header\nexport OPENAI_API_KEY=abc\n\nMY_CUSTOM_THING=1\n" diff --git a/tests/hermes_cli/test_tools_config.py b/tests/hermes_cli/test_tools_config.py index 5e65a97b2e4c0..9099bf1dd3183 100644 --- a/tests/hermes_cli/test_tools_config.py +++ b/tests/hermes_cli/test_tools_config.py @@ -848,38 +848,18 @@ def test_get_effective_configurable_toolsets_dedupes_bundled_plugins(): -# ── Checklist diff scope: non-configurable toolsets (kanban) must not be -# reported as added/removed by `hermes tools` ────────────────────────── - - - - -def test_kanban_not_reported_as_removed_in_diff(): - """Reproduces the false-signal bug: `hermes tools` printed ``- kanban`` - when saving a platform that resolves kanban as enabled, even though the - checklist never offered kanban as a toggle. - - The printed diff must be scoped to ``_checklist_toolset_keys`` so a tool - the user could not deselect is never reported as removed. The persisted - config still keeps kanban (verified separately by _save_platform_tools). - """ +# Kanban now participates in the checklist: an explicit deselection must be +# both visible in the diff and durable in the platform selection. +def test_kanban_checklist_reports_and_persists_explicit_removal(): config = {"platform_toolsets": {"telegram": ["kanban", "web", "terminal"]}} current = _get_platform_tools(config, "telegram", include_default_mcp_servers=False) - assert "kanban" in current # resolved as enabled at read time - - # The checklist can only return configurable keys it was shown; kanban - # is never one of them. universe = _checklist_toolset_keys("telegram") - new_enabled = {t for t in current if t != "kanban"} - - # Unscoped (old, buggy) diff would surface kanban. - assert (current - new_enabled) == {"kanban"} - # Scoped (fixed) diff drops it. - assert ((current - new_enabled) & universe) == set() - - - - + new_enabled = current - {"kanban"} + assert ((current - new_enabled) & universe) == {"kanban"} + with patch("hermes_cli.tools_config.save_config"): + _save_platform_tools(config, "telegram", new_enabled) + assert "kanban" not in _get_platform_tools(config, "telegram", include_default_mcp_servers=False) + assert {"web", "terminal"} <= set(config["platform_toolsets"]["telegram"]) def test_vision_picker_custom_endpoint(tmp_path, monkeypatch): @@ -1178,8 +1158,6 @@ def test_dispatcher_worker_recovers_kanban_and_injects_inter_agent_separately(mo monkeypatch.setattr("model_tools._is_delegated_child_context", lambda: False) enabled = _get_platform_tools({}, "cli", include_default_mcp_servers=False) - assert "kanban" in enabled - selected = _select_tool_names(sorted(enabled), [], quiet_mode=True) assert "kanban_complete" in selected assert "inter_agent" in selected diff --git a/tests/hermes_cli/test_web_server_messaging_profiles.py b/tests/hermes_cli/test_web_server_messaging_profiles.py index 193bb4b219a81..9e775f63d1e48 100644 --- a/tests/hermes_cli/test_web_server_messaging_profiles.py +++ b/tests/hermes_cli/test_web_server_messaging_profiles.py @@ -305,3 +305,37 @@ def test_scoped_enablement_uses_only_own_credentials(client, isolated_profiles, assert platform["enabled"] is enabled assert platform["configured"] is True assert "root-token" in (isolated_profiles["default"] / ".env").read_text(encoding="utf-8") + + +@pytest.mark.parametrize("topology", ["scoped_query", "pooled_unscoped"]) +def test_credential_write_hot_serves_a_multiplexed_profile(client, isolated_profiles, monkeypatch, topology): + """A token saved for a profile the live multiplexer serves is handed to the multiplexer right + away (``hot_served``), so the UI skips its restart banner. Both Desktop topologies: the dashboard's + ``?profile=`` and a pooled ``hermes --profile X serve`` that receives the PUT unscoped (#109088).""" + import hermes_cli.gateway as gateway_cli + import hermes_cli.gateway_multiplex_served as served_mod + notified = [] + monkeypatch.setattr(gateway_cli, "named_profile_served_by_running_multiplexer", lambda name=None: name == "worker_alpha") + monkeypatch.setattr(served_mod, "notify_multiplexer_profiles_changed", lambda name, **kw: notified.append(name) or ["default", name]) + if topology == "pooled_unscoped": + monkeypatch.setattr(gateway_cli, "_current_profile_name", lambda: "worker_alpha") + params = {} + else: + params = {"profile": "worker_alpha"} + resp = client.put("/api/messaging/platforms/telegram", params=params, + json={"enabled": True, "env": {"TELEGRAM_BOT_TOKEN": _VALID_WORKER_BOT_TOKEN}}) + assert resp.status_code == 200 + assert resp.json()["hot_served"] is True + assert notified == ["worker_alpha"] + + +def test_credential_write_on_default_profile_is_not_hot_served(client, isolated_profiles, monkeypatch): + """The default profile is the multiplexer itself (its own adapters are restart-managed): never + claim a hot serve for it.""" + import hermes_cli.gateway_multiplex_served as served_mod + monkeypatch.setattr(served_mod, "notify_multiplexer_profiles_changed", + lambda name, **kw: pytest.fail("default profile must not ping the multiplexer")) + resp = client.put("/api/messaging/platforms/telegram", + json={"enabled": True, "env": {"TELEGRAM_BOT_TOKEN": _VALID_WORKER_BOT_TOKEN}}) + assert resp.status_code == 200 + assert resp.json()["hot_served"] is False diff --git a/tests/test_toolsets.py b/tests/test_toolsets.py index daf06b469b10d..d716c696d6dc1 100644 --- a/tests/test_toolsets.py +++ b/tests/test_toolsets.py @@ -337,7 +337,7 @@ def counting_get_toolset(name, *, include_registry=True): f"got {get_toolset_calls['n']} calls" ) assert ( - "hermes-cli", True, registry_id, generation + "hermes-cli", True, registry_id, generation, registry.current_scope_key() ) in toolsets_mod._resolve_toolset_memo def test_generation_bump_invalidates_memo(self, monkeypatch): diff --git a/tests/test_tui_gateway_server.py b/tests/test_tui_gateway_server.py index 462c619eb6670..b0a67cbbef513 100644 --- a/tests/test_tui_gateway_server.py +++ b/tests/test_tui_gateway_server.py @@ -2663,14 +2663,18 @@ def test_load_enabled_toolsets_rejects_disabled_mcp_env(monkeypatch, capsys): config_mod, "load_config", lambda: {"platform_toolsets": {"cli": ["memory"]}} ) - # `kanban` is auto-recovered by the platform toolset resolver, while - # `project` is GUI-only and is folded in by _load_enabled_toolsets. + # Sorted: ["memory", "project"]. `kanban` is a configurable opt-in and is + # never recovered onto a saved list; `project` is GUI-only, folded in by + # _load_enabled_toolsets. Toolsets inside their first release + # (_RECENTLY_SHIPPED_TOOLSETS) are back-filled onto saved lists that never + # offered them — allow those too. from hermes_cli.tools_config import _RECENTLY_SHIPPED_TOOLSETS result = server._load_enabled_toolsets() assert result is not None - assert {"kanban", "memory", "project"} <= set(result) - assert set(result) - {"kanban", "memory", "project"} <= _RECENTLY_SHIPPED_TOOLSETS + assert {"memory", "project"} <= set(result) + assert "kanban" not in result + assert set(result) - {"memory", "project"} <= _RECENTLY_SHIPPED_TOOLSETS err = capsys.readouterr().err assert "ignoring disabled MCP servers" in err assert "mcp-off" in err @@ -2695,8 +2699,9 @@ def test_load_enabled_toolsets_falls_back_when_tui_env_invalid(monkeypatch, caps result = server._load_enabled_toolsets() assert result is not None - assert {"kanban", "memory", "project"} <= set(result) - assert set(result) - {"kanban", "memory", "project"} <= _RECENTLY_SHIPPED_TOOLSETS + assert {"memory", "project"} <= set(result) + assert "kanban" not in result + assert set(result) - {"memory", "project"} <= _RECENTLY_SHIPPED_TOOLSETS assert "using configured CLI toolsets" in capsys.readouterr().err diff --git a/tests/tools/test_kanban_toolset_opt_in.py b/tests/tools/test_kanban_toolset_opt_in.py new file mode 100644 index 0000000000000..591aa03072671 --- /dev/null +++ b/tests/tools/test_kanban_toolset_opt_in.py @@ -0,0 +1,134 @@ +"""Saved tool opt-ins must reach the real schema without leaking across platforms.""" +from __future__ import annotations + +import json +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +import pytest + + +def _names(selection, disabled=None): + from model_tools import get_tool_definitions + + return { + row["function"]["name"] + for row in get_tool_definitions( + selection, disabled_toolsets=disabled, quiet_mode=True, + skip_tool_search_assembly=True, + ) + if row["function"]["name"].startswith("kanban_") + } + + +@pytest.mark.parametrize("surface", ["cli", "http", "rpc"]) +def test_saved_opt_in_roundtrip_reaches_schema_and_board(surface, tmp_path, monkeypatch): + """Exercise the real config writer, availability gate, skills gate and handler.""" + monkeypatch.setattr(Path, "home", lambda: tmp_path) + monkeypatch.delenv("HERMES_KANBAN_TASK", raising=False) + monkeypatch.delenv("HERMES_KANBAN_BOARD", raising=False) + from hermes_cli.config import load_config, save_config + from hermes_cli.tools_config import _apply_toolset_change, _get_platform_tools + from tools.registry import registry + + save_config({"platform_toolsets": {"cli": ["file"], "telegram": ["file"]}}) + + def selected(platform="cli"): + return sorted(_get_platform_tools(load_config(), platform, include_default_mcp_servers=False)) + + before = _names(selected()) # Warm both cache layers before enabling. + assert not before + client = None + if surface == "http": + from fastapi import FastAPI + from fastapi.testclient import TestClient + from hermes_cli.web_routers.tools import router + + app = FastAPI() + app.include_router(router) + client = TestClient(app) + + def toggle(enabled): + action = "enable" if enabled else "disable" + if surface == "http": + response = client.put("/api/tools/toolsets/kanban", json={"enabled": enabled}) + assert response.status_code == 200, response.text + assert response.json()["enabled"] is enabled + elif surface == "rpc": + from tui_gateway.server import _methods + + response = _methods["tools.configure"]("kanban-opt-in", {"action": action, "names": ["kanban"]}) + assert "error" not in response, response + assert "kanban" in response["result"]["changed"] + else: + _apply_toolset_change(load_config(), "cli", ["kanban"], action) + + try: + toggle(True) + enabled_names = _names(selected()) + assert {"kanban_list", "kanban_create", "kanban_complete"} <= enabled_names + assert not _names(selected("telegram")), "CLI opt-in leaked to Telegram" + assert "file" in selected() + # A second profile in the same process must not borrow this grant or + # poison the first profile's cached schema on return. + from hermes_constants import set_hermes_home_override, reset_hermes_home_override + other_home = tmp_path / "profiles" / "observer" + token = set_hermes_home_override(other_home) + try: + save_config({"platform_toolsets": {"cli": ["file"]}}) + assert not _names(selected()) + finally: + reset_hermes_home_override(token) + assert _names(selected()) == enabled_names + from agent.skill_utils import _detect_kanban + assert _detect_kanban(), "Saved opt-in still hides the Kanban playbook" + result = json.loads(registry.dispatch("kanban_create", {"title": "opt-in roundtrip", "assignee": "default"})) + assert result.get("ok"), result + from hermes_cli.kanban_db_connect import connect_closing + from hermes_cli.kanban_db import get_task + with connect_closing() as conn: + assert get_task(conn, result["task_id"]).title == "opt-in roundtrip" + toggle(False) + assert not _names(selected()) + assert not _detect_kanban() + assert "file" in selected() + assert enabled_names, "Changing config must not mutate an already-built schema" + finally: + if client is not None: + client.close() + + +@pytest.mark.parametrize("legacy", [False, True]) +def test_selection_is_scoped_and_preserves_worker_and_deny_boundaries(legacy, tmp_path, monkeypatch): + monkeypatch.setattr(Path, "home", lambda: tmp_path) + monkeypatch.delenv("HERMES_KANBAN_TASK", raising=False) + monkeypatch.delenv("HERMES_KANBAN_BOARD", raising=False) + from hermes_cli.config import load_config, save_config + from hermes_cli.tools_config import _get_platform_tools + from agent.delegation_context import delegated_child_context + + save_config({"toolsets": ["kanban"] if legacy else [], "platform_toolsets": {"telegram": ["kanban"]}}) + # The same profile concurrently builds an explicitly opted-in schema and + # an all/default schema. A platform grant must not become a cached global grant. + with ThreadPoolExecutor(max_workers=2) as pool: + named, broad = list(pool.map(_names, [["kanban"], ["hermes-cli"]])) + assert "kanban_create" in named + assert bool(broad) is legacy + assert bool(_names(None)) is legacy + assert bool(_names(["all"])) is legacy + assert not _names(["kanban"], ["kanban"]) + assert not _names([]) + assert bool(_names(sorted(_get_platform_tools(load_config(), "cli")))) is legacy + # An explicitly saved (non-empty) selection is authoritative over the legacy key. + cfg = load_config() + cfg["platform_toolsets"]["cli"] = ["file"] + save_config(cfg) + assert not _names(sorted(_get_platform_tools(load_config(), "cli"))) + + monkeypatch.setenv("HERMES_KANBAN_TASK", "t_worker") + worker = _names(["file"]) + assert "kanban_complete" in worker + assert "kanban_list" not in worker + with delegated_child_context(): + assert not _names(["kanban"]) + assert "kanban_complete" in _names(["file"]) diff --git a/tests/tools/test_mcp_multiplex_connection_keys.py b/tests/tools/test_mcp_multiplex_connection_keys.py new file mode 100644 index 0000000000000..a524cce162434 --- /dev/null +++ b/tests/tools/test_mcp_multiplex_connection_keys.py @@ -0,0 +1,187 @@ +"""Two multiplexed profiles that name the same MCP server with different credentials are two +connections (#106005, #91654): the ledgers in ``tools.mcp_tool`` are keyed per owning profile +scope, and an owner's scoped reload re-registers the profiles that had adopted its connection.""" + +from __future__ import annotations + +from types import SimpleNamespace +from unittest.mock import patch + +import pytest + +from hermes_constants import hermes_home_key, reset_hermes_home_override, set_hermes_home_override + + +def _tool(): + return SimpleNamespace(name="t", description="d", inputSchema={"type": "object", "properties": {}}, + annotations=None) + + +def _server(name, cfg): + return SimpleNamespace(name=name, session=object(), _config=cfg, _tools=[_tool()], tool_timeout=30, + initialize_result=None, _registered_tool_names=[], _sampling=None) + + +@pytest.fixture +def two_profiles(tmp_path, monkeypatch): + """Multiplex on, clean MCP ledgers, a scope switcher for homes A and B; restores everything.""" + import tools.mcp_tool as core + from tools import mcp_tool_config as _config + from tools.registry import registry + + homes = {k: tmp_path / "profiles" / k for k in ("a", "b")} + for home in homes.values(): + home.mkdir(parents=True) + monkeypatch.setattr("agent.secret_scope.is_multiplex_active", lambda: True) + monkeypatch.setattr(core, "_ensure_mcp_sdk", lambda: True) + monkeypatch.setattr(_config, "_filter_suspicious_mcp_servers", lambda servers: servers) + ledgers = ("_servers", "_server_scope_keys", "_server_tool_scopes", "_server_connecting", + "_server_connect_errors", "_server_connect_retry_after", "_server_connect_failures", + "_server_error_counts", "_server_breaker_opened_at", "_lazy_server_configs", + "_mcp_tool_server_names", "_orphaned_adopters", "_parallel_safe_servers") + saved = {n: type(getattr(core, n))(getattr(core, n)) for n in ledgers} + for n in ledgers: + getattr(core, n).clear() + tokens = [] + + def enter(which): + tokens.append(set_hermes_home_override(homes[which])) + return hermes_home_key(homes[which]) + + yield enter + for tool_name in list(registry.get_tool_names_for_toolset("mcp-x")): + for home in homes.values(): + registry.deregister(tool_name, scope=hermes_home_key(home)) + for token in reversed(tokens): + reset_hermes_home_override(token) + for n in ledgers: + getattr(core, n).clear() + getattr(core, n).update(saved[n]) + + +def test_same_named_server_with_other_credentials_is_a_separate_connection(two_profiles): + import tools.mcp_tool as core + from tools import mcp_tool_discovery as disc, mcp_tool_handlers as handlers + from tools import mcp_tool_registration as reg + from tools.registry import registry + import toolsets + + cfg_a = {"url": "https://mcp.example/x", "headers": {"Authorization": "Bearer A"}} + cfg_b = {"url": "https://mcp.example/x", "headers": {"Authorization": "Bearer B"}} + + two_profiles("a") + srv_a = _server("x", cfg_a) + disc._adopt_server("x", srv_a) + srv_a._registered_tool_names = reg._register_server_tools("x", srv_a, cfg_a) + assert toolsets.resolve_toolset("mcp-x") == ["mcp__x__t"] + for _ in range(core._CIRCUIT_BREAKER_THRESHOLD): + core._bump_server_error("x") + disc._note_connect_failure("y", RuntimeError("boom")) + + two_profiles("b") + # B's own view: no tools yet, its memo is not A's, and A's connection is not "connected" for B. + assert registry.get_tool_names_for_toolset("mcp-x") == [] + assert toolsets.resolve_toolset("mcp-x") == [] + assert disc.get_mcp_status({"x": cfg_b})[0]["status"] == "configured" + # B's differently-authenticated 'x' is a connect candidate, not shadowed by A's ledger entries. + assert "x" in disc._select_new_servers({"x": cfg_b}) + assert not disc._connect_cooldown_active("y") + assert handlers._check_circuit_breaker("x") is None + + +def test_owner_reload_reregisters_profiles_that_adopted_its_connection(two_profiles): + import tools.mcp_tool as core + from tools import mcp_tool_discovery as disc, mcp_tool_lifecycle as lifecycle + from tools import mcp_tool_registration as reg + from tools.registry import registry + + cfg = {"url": "https://mcp.example/x", "headers": {"Authorization": "Bearer shared"}} + scope_a = two_profiles("a") + srv_a = _server("x", cfg) + disc._adopt_server("x", srv_a) + srv_a._registered_tool_names = reg._register_server_tools("x", srv_a, cfg) + + two_profiles("b") + assert reg.register_connected_into_current_scope({"x": cfg}) == 1 + assert registry.get_tool_names_for_toolset("mcp-x") == ["mcp__x__t"] + + # Owner A: scoped shutdown (no MCP loop here, so emulate the task teardown), then rediscovery. + two_profiles("a") + with patch.object(lifecycle._loop, "_stop_mcp_loop", lambda **_kw: False): + lifecycle.shutdown_mcp_servers(scope=scope_a) + for tool_name in list(srv_a._registered_tool_names): + reg._deregister_mcp_tool_all_scopes(srv_a, tool_name) + with core._lock: + for key in [k for k, v in core._servers.items() if v is srv_a]: + core._servers.pop(key) + core._server_scope_keys.pop(key, None) + core._server_tool_scopes.pop(key, None) + + def fake_pass(new_servers): + for name, config in new_servers.items(): + srv = _server(name, config) + disc._adopt_server(name, srv) + srv._registered_tool_names = reg._register_server_tools(name, srv, config) + + with patch.object(disc, "_run_discovery_pass", fake_pass), \ + patch.object(disc._loop, "_ensure_mcp_loop", lambda: None), \ + patch("tools.mcp_tool_config._load_mcp_config", lambda: {"x": cfg}): + disc.register_mcp_servers({"x": cfg}) + + # B never reloaded, yet has its tools back on the owner's new identical connection. + two_profiles("b") + assert registry.get_tool_names_for_toolset("mcp-x") == ["mcp__x__t"] + assert disc.get_mcp_status({"x": cfg})[0]["status"] == "connected" + + +def test_deregister_preserves_provenance_for_surviving_same_named_connection(two_profiles): + import tools.mcp_tool as core + from tools import mcp_tool_discovery as disc + from tools import mcp_tool_registration as reg + from tools.registry import registry + + cfg_a = {"url": "https://mcp.example/x", "headers": {"Authorization": "Bearer A"}} + cfg_b = {"url": "https://mcp.example/x", "headers": {"Authorization": "Bearer B"}} + + scope_a = two_profiles("a") + srv_a = _server("x", cfg_a) + disc._adopt_server("x", srv_a) + srv_a._registered_tool_names = reg._register_server_tools("x", srv_a, cfg_a) + + scope_b = two_profiles("b") + srv_b = _server("x", cfg_b) + disc._adopt_server("x", srv_b) + srv_b._registered_tool_names = reg._register_server_tools("x", srv_b, cfg_b) + + two_profiles("a") + reg._deregister_mcp_tool_all_scopes(srv_a, "mcp__x__t") + + assert registry.snapshot_registration("mcp__x__t", scope=scope_a) is None + assert registry.snapshot_registration("mcp__x__t", scope=scope_b) is not None + assert disc.has_registered_mcp_tools() is True + assert disc.get_registered_mcp_server_names() == {"x"} + + +def test_parallel_policy_is_scoped_to_same_named_connections(two_profiles): + from tools import mcp_tool_discovery as disc + from tools import mcp_tool_registration as reg + + cfg_a = {"url": "https://mcp.example/x", "supports_parallel_tool_calls": False} + cfg_b = {"url": "https://mcp.example/x", "supports_parallel_tool_calls": True} + + two_profiles("a") + srv_a = _server("x", cfg_a) + disc._adopt_server("x", srv_a) + srv_a._registered_tool_names = reg._register_server_tools("x", srv_a, cfg_a) + disc._select_new_servers({"x": cfg_a}) + + two_profiles("b") + srv_b = _server("x", cfg_b) + disc._adopt_server("x", srv_b) + srv_b._registered_tool_names = reg._register_server_tools("x", srv_b, cfg_b) + disc._select_new_servers({"x": cfg_b}) + + two_profiles("a") + assert disc.is_mcp_tool_parallel_safe("mcp__x__t") is False + two_profiles("b") + assert disc.is_mcp_tool_parallel_safe("mcp__x__t") is True diff --git a/tests/tools/test_mcp_tool.py b/tests/tools/test_mcp_tool.py index dcdd9a8c5fd87..591b3191876db 100644 --- a/tests/tools/test_mcp_tool.py +++ b/tests/tools/test_mcp_tool.py @@ -1570,6 +1570,52 @@ def test_secret_source_injected_vars_are_passed(self, monkeypatch): assert result["NOTION_TOKEN"] == "from-op" assert "UNTRACKED_SECRET_KEY" not in result + def test_scoped_external_secret_is_passed_and_shapes_connection_identity(self, monkeypatch, tmp_path): + """External-secret provenance is captured per profile home, not by ambient name metadata.""" + from agent import secret_scope + from hermes_cli import env_loader + from hermes_constants import reset_hermes_home_override, set_hermes_home_override + from tools.mcp_tool_config import _build_safe_env, _connection_config + from tools.mcp_tool_registration import _connection_identity + + monkeypatch.setitem(env_loader._SECRET_SOURCES, "MCP_TOKEN", "bitwarden") + monkeypatch.setenv("MCP_TOKEN", "launch-profile-token") + secret_scope.set_multiplex_active(True) + home_a = tmp_path / "profile-a" + home_b = tmp_path / "profile-b" + home_a.mkdir() + home_b.mkdir() + monkeypatch.setitem(env_loader._SECRET_SOURCE_VALUES_BY_HOME, str(home_a.resolve()), + {"MCP_TOKEN": "profile-a-external"}) + monkeypatch.setitem(env_loader._SECRET_SOURCE_VALUES_BY_HOME, str(home_b.resolve()), {}) + + home_token_a = set_hermes_home_override(home_a) + token_a = secret_scope.set_secret_scope({"MCP_TOKEN": "profile-a-external"}) + try: + config_a = _connection_config({"command": "mcp-server"}) + child_env_a = _build_safe_env( + None, external_env=config_a["_hermes_external_secret_env"] + ) + finally: + secret_scope.reset_secret_scope(token_a) + reset_hermes_home_override(home_token_a) + + home_token_b = set_hermes_home_override(home_b) + token_b = secret_scope.set_secret_scope({"MCP_TOKEN": "profile-b-dotenv"}) + try: + config_b = _connection_config({"command": "mcp-server"}) + child_env_b = _build_safe_env( + None, external_env=config_b["_hermes_external_secret_env"] + ) + finally: + secret_scope.reset_secret_scope(token_b) + reset_hermes_home_override(home_token_b) + secret_scope.set_multiplex_active(False) + + assert child_env_a["MCP_TOKEN"] == "profile-a-external" + assert "MCP_TOKEN" not in child_env_b + assert _connection_identity(config_a) != _connection_identity(config_b) + def test_windows_location_vars_passed_without_secrets(self): """Windows launcher tools need location vars, but secrets stay filtered.""" from tools.mcp_tool_config import _build_safe_env diff --git a/tools/kanban_tools.py b/tools/kanban_tools.py index 2c174fb935185..d75ed69f7dbef 100644 --- a/tools/kanban_tools.py +++ b/tools/kanban_tools.py @@ -19,7 +19,7 @@ from agent.redact import redact_sensitive_text from hermes_cli.goals import judge_goal -from tools.registry import registry, tool_error +from tools.registry import no_cache_check_fn, registry, tool_error from hermes_cli.config import cfg_get, load_config from tools.kanban_tools_schemas import ( KANBAN_ATTACH_SCHEMA, @@ -294,9 +294,28 @@ def _project_remote_worker_state(payload: dict, *, current_run_id: str | None) - # --- Gating --- def _profile_has_kanban_toolset() -> bool: - # load_config() is mtime-cached and check_fn results are TTL-cached (~30s). + from tools.kanban_toolset_context import kanban_toolset_requested + + requested = kanban_toolset_requested() + if requested: + return True try: - return "kanban" in load_config().get("toolsets", []) + config = load_config() + # Preserve the legacy profile-wide opt-in for callers using bundles. + if "kanban" in (config.get("toolsets") or []): + return True + if requested is not None: + # Never borrow another platform's opt-in during schema assembly. + return False + # Offer-time skill discovery has no platform selection. A saved opt-in + # makes the playbook relevant; actual schemas still use the scope above. + from hermes_cli.tools_config import _get_platform_tools + + platforms = config.get("platform_toolsets") or {} + return any( + "kanban" in _get_platform_tools(config, platform, include_default_mcp_servers=False) + for platform, names in platforms.items() if isinstance(names, list) + ) except Exception: return False @@ -330,11 +349,13 @@ def _visible(*, to_env_worker: bool) -> bool: return _profile_has_kanban_toolset() +@no_cache_check_fn def _check_kanban_mode() -> bool: """Lifecycle tools: dispatcher workers + profiles with the ``kanban`` toolset.""" return _visible(to_env_worker=True) +@no_cache_check_fn def _check_kanban_orchestrator_mode() -> bool: """Board-routing tools (kanban_list, kanban_unblock): hidden from task workers.""" return _visible(to_env_worker=False) diff --git a/tools/kanban_toolset_context.py b/tools/kanban_toolset_context.py new file mode 100644 index 0000000000000..da932744911af --- /dev/null +++ b/tools/kanban_toolset_context.py @@ -0,0 +1,28 @@ +"""Explicit Kanban selection during one model-schema build. + +The registry's ordinary availability cache is profile-wide, while a gateway +can build schemas for several platforms in the same profile concurrently. +Carry only the explicit selection through a ContextVar, never process env. +""" +from __future__ import annotations + +from contextlib import contextmanager +from contextvars import ContextVar +from typing import Iterable, Iterator, Optional + +_requested: ContextVar[Optional[bool]] = ContextVar("kanban_toolset_requested", default=None) + + +def kanban_toolset_requested() -> Optional[bool]: + """None outside schema assembly; otherwise whether Kanban was named explicitly.""" + return _requested.get() + + +@contextmanager +def scoped_kanban_toolset_selection(toolsets: Optional[Iterable[str]]) -> Iterator[None]: + """An all/default selection is not an explicit workflow opt-in.""" + token = _requested.set("kanban" in (toolsets or ())) + try: + yield + finally: + _requested.reset(token) diff --git a/tools/mcp_tool.py b/tools/mcp_tool.py index 3a7ee17ec8324..7504b734e3245 100644 --- a/tools/mcp_tool.py +++ b/tools/mcp_tool.py @@ -309,7 +309,7 @@ class MCPServerTask(MCPServerRunMixin, MCPServerTransportMixin, MCPServerHealthM keepalive/liveness live in the three mixins.""" __slots__ = ( - "name", "_registry_key", "session", "tool_timeout", "_task", "_ready", "_shutdown_event", "_reconnect_event", + "name", "session", "tool_timeout", "_task", "_ready", "_shutdown_event", "_reconnect_event", "_tools", "_error", "_config", "_sampling", "_elicitation", "_registered_tool_names", "_auth_type", "_refresh_lock", "_rpc_lock", "_pending_refresh_tasks", "_pending_call_context", "_lifecycle_started_at", "_last_tool_call_at", "_idle_timeout_seconds", "_max_lifetime_seconds", @@ -320,7 +320,6 @@ class MCPServerTask(MCPServerRunMixin, MCPServerTransportMixin, MCPServerHealthM def __init__(self, name: str): self.name = name - self._registry_key = name self.session: Optional[Any] = None self.tool_timeout: float = _DEFAULT_TOOL_TIMEOUT self._task: Optional[asyncio.Task] = None @@ -397,25 +396,30 @@ def __init__(self, name: str): # ---- Module-level state (every mutation under ``_lock``) ---- +# +# Every ledger below is keyed by the CONNECTION KEY from ``tools.mcp_tool_scope``: the bare +# server name outside a multiplexer, ``(owner_scope, name)`` under one. Two profiles naming the +# same server with their own credentials are two connections; a name-keyed ledger let the first +# profile's connection shadow the second's (never connected, silently tool-less — #106005). -_servers: Dict[str, MCPServerTask] = {} -# Internal connection key -> configured server name. A profile-owned OAuth/mTLS session can share -# a configured name with a peer while retaining a distinct transport and credential scope. -_server_public_names: Dict[str, str] = {} +_servers: Dict[Any, MCPServerTask] = {} # Profile registry scope per live connection (None outside multiplex) so a multiplexed # /reload-mcp tears down only its own profile's servers. -_server_scope_keys: Dict[str, Optional[str]] = {} +_server_scope_keys: Dict[Any, Optional[str]] = {} # Registry scopes that have adopted a live server connection. The owning scope above remains # authoritative for connection teardown; this set preserves visibility for shared connections. -_server_tool_scopes: Dict[str, set] = {} -_server_connecting: set[str] = set() -_server_connect_errors: Dict[str, str] = {} +_server_tool_scopes: Dict[Any, set] = {} +_server_connecting: set = set() +_server_connect_errors: Dict[Any, str] = {} +# adopter scope -> server names whose shared connection an owner's scoped shutdown tore down; +# drained by the next discovery pass so the adopter is re-registered (see mcp_tool_lifecycle). +_orphaned_adopters: Dict[str, set] = {} # Lazy startup: servers registered from the schema cache without connecting; popped on # first real connection. -# Keyed by server name; entries are popped once a real connection is established on first use. See #56832. -_lazy_server_configs: Dict[str, dict] = {} -_lazy_server_fingerprints: Dict[str, str] = {} -_lazy_server_tool_names: Dict[str, List[str]] = {} +# Keyed by connection key; entries are popped once a real connection is established on first use. See #56832. +_lazy_server_configs: Dict[Any, dict] = {} +_lazy_server_fingerprints: Dict[Any, str] = {} +_lazy_server_tool_names: Dict[Any, List[str]] = {} # Task-local claim around ``_connect_server``: discovery retains a recoverable parked task # while standalone probes never publish failed servers into module-global ownership. _connect_server_claim: contextvars.ContextVar[Optional[Callable[[MCPServerTask], None]]] = ( @@ -436,8 +440,8 @@ def __init__(self, name: str): # ``retry_after`` deadline with exponential backoff. ``register_mcp_servers`` skips a server whose cooldown # has not elapsed, so a chronically failing server is retried on a backoff schedule instead of on every # worker session -- isolating it from the rest of the bridge. A successful connection clears the state. -_server_connect_retry_after: Dict[str, float] = {} # name -> monotonic deadline -_server_connect_failures: Dict[str, int] = {} # name -> consecutive failures +_server_connect_retry_after: Dict[Any, float] = {} # connection key -> monotonic deadline +_server_connect_failures: Dict[Any, int] = {} # connection key -> consecutive failures _CONNECT_RETRY_BASE_BACKOFF_SEC, _CONNECT_RETRY_MAX_BACKOFF_SEC = 30.0, 600.0 # Per-server circuit breaker: closed -> open (calls short-circuit until the cooldown) -> @@ -450,8 +454,8 @@ def __init__(self, name: str): # ``_server_breaker_opened_at`` records the monotonic timestamp when the breaker most recently transitioned # into the open state. Use the ``_bump_server_error`` / ``_reset_server_error`` helpers to mutate this state # — they keep the count and timestamp in sync. -_server_error_counts: Dict[str, int] = {} -_server_breaker_opened_at: Dict[str, float] = {} +_server_error_counts: Dict[Any, int] = {} +_server_breaker_opened_at: Dict[Any, float] = {} _CIRCUIT_BREAKER_THRESHOLD, _CIRCUIT_BREAKER_COOLDOWN_SEC = 3, 60.0 # Trust-tier gating (``trust: full | untrusted``): on an untrusted server every write-capable @@ -460,34 +464,42 @@ def __init__(self, name: str): # already warned about, never widen access. Missing trust = full; unrecognized = untrusted (a # typo must never disable the gate). Classified at CALL time from DISCOVERY data: no schema # mutation, prompt cache intact. -_server_trust_levels: Dict[str, str] = {} -_tool_read_only_hints: Dict[str, Dict[str, bool]] = {} +_server_trust_levels: Dict[Any, str] = {} +_tool_read_only_hints: Dict[Any, Dict[str, bool]] = {} _TRUST_FULL, _TRUST_UNTRUSTED = "full", "untrusted" def _bump_server_error(server_name: str) -> None: - """Count a failure; at the threshold (re)stamp the breaker-open time.""" - n = _server_error_counts.get(server_name, 0) + 1 - _server_error_counts[server_name] = n - if n >= _CIRCUIT_BREAKER_THRESHOLD: - _server_breaker_opened_at[server_name] = time.monotonic() + """Count a failure; at the threshold (re)stamp the breaker-open time. Keyed by the calling + scope's connection so one profile's failing server never opens another profile's breaker.""" + from tools.mcp_tool_scope import _resolve_server_key + with _lock: + key = _resolve_server_key(server_name, lock_held=True) + n = _server_error_counts.get(key, 0) + 1 + _server_error_counts[key] = n + if n >= _CIRCUIT_BREAKER_THRESHOLD: + _server_breaker_opened_at[key] = time.monotonic() def _reset_server_error(server_name: str) -> None: """Close the breaker on any unambiguous success signal.""" - _server_error_counts[server_name] = 0 - _server_breaker_opened_at.pop(server_name, None) + from tools.mcp_tool_scope import _resolve_server_key + with _lock: + key = _resolve_server_key(server_name, lock_held=True) + _server_error_counts[key] = 0 + _server_breaker_opened_at.pop(key, None) -# Raw server names opted into parallel tool calls (``foo-bar``/``foo_bar`` sanitize alike but -# must not share policy). +# Connection keys opted into parallel tool calls. Outside multiplexing these remain bare raw +# names; under multiplexing they include the owning profile scope. _parallel_safe_servers: set = set() # registry tool name -> raw server name (the generated name is lossy; never re-parse it). _mcp_tool_server_names: Dict[str, str] = {} -# Profile overlays need independent provenance for identical public tool names. The legacy global -# map remains for unscoped callers and older plugins. -_mcp_tool_server_names_by_scope: Dict[str, Dict[str, str]] = {} +# Connection-key metadata is kept separately from the public registry names. Scoped keys must +# never leak into tool errors, reload summaries, or profile-local discovery results. +_server_public_names: Dict[Any, str] = {} +_mcp_tool_server_names_by_scope: Dict[Optional[str], Dict[str, Any]] = {} # Dedicated event loop in a background daemon thread; _lock guards the loop handles, _servers, # the status maps and the PID ledgers. @@ -629,20 +641,22 @@ def _mcp_registry_scope() -> Optional[str]: return registry.current_scope_key() -def _server_registry_scope(name: str) -> Optional[str]: - """Scope owning *name*'s tools: the one captured at adoption (teardown runs on the MCP - loop without the discovering profile's context), else the current one.""" - if name in _server_scope_keys: - return _server_scope_keys[name] - return _mcp_registry_scope() +def _server_registry_scope(key) -> Optional[str]: + """Scope owning the connection under *key*'s tools: the one captured at adoption (teardown + runs on the MCP loop without the discovering profile's context), else the current one.""" + if key in _server_scope_keys: + return _server_scope_keys[key] + from tools.mcp_tool_scope import _key_scope + return _key_scope(key) or _mcp_registry_scope() -def _server_visible_in_scope(name: str, scope: Optional[str]) -> bool: - """Whether a live server is visible from ``scope`` without changing its teardown owner.""" +def _server_visible_in_scope(key, scope: Optional[str]) -> bool: + """Whether the live connection under *key* is visible from ``scope`` without changing its + teardown owner.""" if scope is None: return True - return (_server_scope_keys.get(name) == scope - or scope in _server_tool_scopes.get(name, ())) + return (_server_scope_keys.get(key) == scope + or scope in _server_tool_scopes.get(key, ())) # Cross-process discovery guard: advisory file lock so gateway + CLI + TUI don't all discover. diff --git a/tools/mcp_tool_config.py b/tools/mcp_tool_config.py index 419385b727fe6..e194d5cc5415c 100644 --- a/tools/mcp_tool_config.py +++ b/tools/mcp_tool_config.py @@ -68,6 +68,46 @@ def _write_stderr_log_header(server_name: str) -> None: # ${VAR_NAME} interpolation; any non-} chars allowed so MY-VAR / my.var work. _ENV_VAR_PATTERN = re.compile(r"\$\{([^}]+)\}") +# Private connection metadata: unlike ``config.env``, these values are injected +# automatically by an external secret source and therefore must participate in +# scoped stdio connection reuse decisions. +_CONNECTION_EXTERNAL_ENV_KEY = "_hermes_external_secret_env" + + +def _external_secret_env() -> dict[str, str]: + """Return external-secret values from the active profile's immutable home snapshot.""" + try: + from agent.secret_scope import current_secret_scope, is_multiplex_active + from hermes_cli.env_loader import get_secret_source, get_secret_source_values + from hermes_constants import get_hermes_home + except Exception: # pragma: no cover - early bootstrap/import fallback + return {} + + home_values = get_secret_source_values(get_hermes_home()) + scope = current_secret_scope() + if scope is not None: + return { + key: value for key, value in home_values.items() + if isinstance(value, str) and scope.get(key) == value + } + if is_multiplex_active(): + return {} + if home_values: + return {key: value for key, value in home_values.items() if isinstance(value, str)} + return { + key: value for key, value in os.environ.items() + if get_secret_source(key) + } + + +def _connection_config(config: dict) -> dict: + """Capture automatic external-secret values on a stdio config before it enters a server task.""" + if "command" not in config or "url" in config or _CONNECTION_EXTERNAL_ENV_KEY in config: + return config + prepared = dict(config) + prepared[_CONNECTION_EXTERNAL_ENV_KEY] = _external_secret_env() + return prepared + def _workspace_folder() -> str: """Absolute workspace root for ``${workspaceFolder}``: the session's authoritative root @@ -91,18 +131,15 @@ def _workspace_basename() -> str: "workspaceFolderBasename": _workspace_basename, "pathSeparator": lambda: os.sep, "/": lambda: os.sep} -def _build_safe_env(user_env: Optional[dict]) -> dict: +def _build_safe_env(user_env: Optional[dict], *, external_env: Optional[dict] = None) -> dict: """Filtered env for stdio subprocesses so API keys/tokens don't leak: the safe baseline keys, ``XDG_*``, vars injected by an external secret source (users configured that backend precisely so subprocesses can consume them), plus the server config's own ``env``.""" - try: - from hermes_cli.env_loader import get_secret_source - except Exception: # pragma: no cover — early bootstrap/import fallback - get_secret_source = None env = { key: value for key, value in os.environ.items() if key in _SAFE_ENV_KEYS or key.upper() in _SAFE_ENV_KEYS_CASE_INSENSITIVE - or key.startswith("XDG_") or (get_secret_source is not None and get_secret_source(key))} + or key.startswith("XDG_")} + env.update(_external_secret_env() if external_env is None else external_env) for key in ("HERMES_KANBAN_DB", "HERMES_KANBAN_BOARD"): if key in os.environ: env[key] = os.environ[key] diff --git a/tools/mcp_tool_discovery.py b/tools/mcp_tool_discovery.py index b952a859541aa..2e7f904391382 100644 --- a/tools/mcp_tool_discovery.py +++ b/tools/mcp_tool_discovery.py @@ -16,27 +16,60 @@ from tools import mcp_tool_loop as _loop from tools import mcp_tool_registration as _registration from tools.mcp_tool_schema import MCP_TOOL_NAME_PREFIX +from tools.mcp_tool_scope import _key_name, _resolve_server_key, _server_key logger = logging.getLogger("tools.mcp_tool") +class _ScopedCandidateKey(str): + """Human-readable scoped key for candidate maps with lossless internal identity.""" + + def __new__(cls, public_name: str, ledger_key): + value = str.__new__( + cls, public_name if "::profile::" in public_name else f"{public_name}::profile::{ledger_key[0]}" + ) + value.public_name = public_name + value.ledger_key = ledger_key + return value + + +class _CandidateServerMap(dict): + """Candidate map with lossless scoped keys and public-name membership.""" + + def __contains__(self, key) -> bool: + if super().__contains__(key): + return True + return any(_candidate_public_name(candidate) == key for candidate in self) + + +def _candidate_public_name(key) -> str: + return getattr(key, "public_name", _key_name(key)) + + +def _candidate_ledger_key(key): + return getattr(key, "ledger_key", key) + + def _record_connect_failure(server_name: str) -> None: """Stamp a geometric, capped retry cooldown after a failed connect (under ``_lock``).""" - n = _core._server_connect_failures.get(server_name, 0) + 1 - _core._server_connect_failures[server_name] = n + key = _server_key(server_name) + n = _core._server_connect_failures.get(key, 0) + 1 + _core._server_connect_failures[key] = n backoff = min(_core._CONNECT_RETRY_BASE_BACKOFF_SEC * (2 ** (n - 1)), _core._CONNECT_RETRY_MAX_BACKOFF_SEC) - _core._server_connect_retry_after[server_name] = time.monotonic() + backoff + _core._server_connect_retry_after[key] = time.monotonic() + backoff def _clear_connect_failure(server_name: str) -> None: """Clear the connect-cooldown state after a successful connection.""" - _core._server_connect_failures.pop(server_name, None) - _core._server_connect_retry_after.pop(server_name, None) + key = _server_key(server_name) + _core._server_connect_failures.pop(key, None) + _core._server_connect_retry_after.pop(key, None) def _connect_cooldown_active(server_name: str) -> bool: - """True if ``server_name`` is still within its retry cooldown.""" - deadline = _core._server_connect_retry_after.get(server_name) + """True if ``server_name`` is still within its retry cooldown (this scope's connection: one + profile's failing ``x`` must not shadow another profile's healthy ``x``).""" + deadline = _core._server_connect_retry_after.get(_server_key(server_name)) return deadline is not None and time.monotonic() < deadline @@ -44,21 +77,17 @@ def _enabled(cfg: dict) -> bool: return _parse_boolish(cfg.get("enabled", True), default=True) -async def _connect_server(name: str, config: dict, *, registry_key: Optional[str] = None) -> _core.MCPServerTask: +async def _connect_server(name: str, config: dict) -> _core.MCPServerTask: """Create an MCPServerTask, start it, return once ready (tear down with ``server.shutdown()`` on the same loop). Raises on bad config, missing HTTP support or connect failure.""" server = _core.MCPServerTask(name) - # A profile-owned connection is constructed with its public OAuth identity, but all - # connection-local lifecycle state must already use the private registry key before - # the initial handshake starts. - server._registry_key = registry_key or name claim = _core._connect_server_claim.get() if claim is not None: claim(server) # The run task copies this context: don't retain the discovery closure for its life. claim_token = _core._connect_server_claim.set(None) if claim is not None else None try: - await server.start(config) + await server.start(_config._connection_config(config)) except asyncio.CancelledError: raise # start() already reaps server._task; shutdown() here could swallow the cancel except BaseException: @@ -115,8 +144,9 @@ def _note_connect_failure(name: str, exc: BaseException) -> str: """Record a failed connect (under ``_lock``): error text for status, cooldown stamp.""" message = _errors._format_connect_error(exc) with _core._lock: - _core._server_connecting.discard(name) - _core._server_connect_errors[name] = message + key = _server_key(name) + _core._server_connecting.discard(key) + _core._server_connect_errors[key] = message _record_connect_failure(name) return message @@ -124,18 +154,20 @@ def _note_connect_failure(name: str, exc: BaseException) -> str: def _note_connect_success(name: str) -> None: """Clear connecting/error/cooldown state after a successful connect (under ``_lock``).""" with _core._lock: - _core._server_connecting.discard(name) - _core._server_connect_errors.pop(name, None) + key = _server_key(name) + _core._server_connecting.discard(key) + _core._server_connect_errors.pop(key, None) _clear_connect_failure(name) def _adopt_server(name: str, server: _core.MCPServerTask) -> None: - """Publish *server* into ``_servers`` with its owning registry scope (under ``_lock``).""" + """Publish *server* into ``_servers`` under the connecting scope's key (under ``_lock``).""" with _core._lock: - server._registry_key = name - _core._servers[name] = server - _core._server_public_names[name] = server.name - _core._server_scope_keys[name] = _core._mcp_registry_scope() + public_name = _candidate_public_name(name) + key = _server_key(public_name) + _core._servers[key] = server + _core._server_scope_keys[key] = _core._mcp_registry_scope() + _core._server_public_names[key] = public_name def _ensure_lazy_server_connected(server_name: str) -> bool: @@ -146,15 +178,16 @@ def _ensure_lazy_server_connected(server_name: str) -> bool: See #50394. """ with _core._lock: - server = _core._servers.get(server_name) + key = _resolve_server_key(server_name, lock_held=True) + server = _core._servers.get(key) if server is not None and server.session is not None: return True - config = _core._lazy_server_configs.get(server_name) + config = _core._lazy_server_configs.get(key) if (not config or _connect_cooldown_active(server_name) - or server_name in _core._server_connecting): + or key in _core._server_connecting): return False - _core._server_connecting.add(server_name) - _core._server_connect_errors.pop(server_name, None) + _core._server_connecting.add(key) + _core._server_connect_errors.pop(key, None) logger.info("MCP server '%s': lazy start on first use", server_name) _loop._ensure_mcp_loop() connect_timeout = config.get("connect_timeout", _core._DEFAULT_CONNECT_TIMEOUT) @@ -166,16 +199,16 @@ def _ensure_lazy_server_connected(server_name: str) -> bool: return False _note_connect_success(server_name) with _core._lock: - _core._lazy_server_configs.pop(server_name, None) - stale_fingerprint = _core._lazy_server_fingerprints.pop(server_name, None) - cached_names = _core._lazy_server_tool_names.pop(server_name, None) or [] - server = _core._servers.get(server_name) + _core._lazy_server_configs.pop(key, None) + stale_fingerprint = _core._lazy_server_fingerprints.pop(key, None) + cached_names = _core._lazy_server_tool_names.pop(key, None) or [] + server = _core._servers.get(key) live_names = set(getattr(server, "_registered_tool_names", []) or []) # The cached manifest may advertise tools the live server no longer serves. phantom_names = [n for n in cached_names if n not in live_names] if phantom_names: for tool_name in phantom_names: - _registration._deregister_mcp_tool_all_scopes(server_name, tool_name) + _registration._deregister_mcp_tool_all_scopes(key, tool_name) logger.info("MCP server '%s': deregistered %d phantom cached tool(s) not served live (stale schema-cache " "fingerprint %s): %s", server_name, len(phantom_names), stale_fingerprint, ", ".join(phantom_names)) return server is not None and server.session is not None @@ -189,8 +222,9 @@ def _get_connected_server_for_call(server_name: str) -> Optional[_core.MCPServer AND the resource/prompt utility handlers all trigger the deferred spawn (#56832). """ with _core._lock: - server = _core._servers.get(server_name) - is_lazy = server_name in _core._lazy_server_configs + key = _resolve_server_key(server_name, lock_held=True) + server = _core._servers.get(key) + is_lazy = key in _core._lazy_server_configs if is_lazy and (server is None or server.session is None): _ensure_lazy_server_connected(server_name) elif server is not None and server.session is None and server._is_recycled_stdio(): @@ -198,19 +232,16 @@ def _get_connected_server_for_call(server_name: str) -> Optional[_core.MCPServer else: return server with _core._lock: - return _core._servers.get(server_name) + return _core._servers.get(key) async def _discover_and_register_server(name: str, config: dict) -> List[str]: """Connect one server, register its tools; return the registered names.""" # The claim fires inside _connect_server while this frame is suspended (list, not nonlocal). - public_name = _core._server_public_names.get(name, name) claimed: List[_core.MCPServerTask] = [] claim_token = _core._connect_server_claim.set(claimed.append) try: - connect = (_connect_server(public_name, config, registry_key=name) - if name != public_name else _connect_server(public_name, config)) - server = await asyncio.wait_for(connect, + server = await asyncio.wait_for(_connect_server(name, config), timeout=config.get("connect_timeout", _core._DEFAULT_CONNECT_TIMEOUT)) except BaseException: server = claimed[0] if claimed else None @@ -226,17 +257,11 @@ async def _discover_and_register_server(name: str, config: dict) -> List[str]: finally: _core._connect_server_claim.reset(claim_token) with _core._lock: - _core._server_connecting.discard(name) - _core._server_connect_errors.pop(name, None) + key = _server_key(name) + _core._server_connecting.discard(key) + _core._server_connect_errors.pop(key, None) _adopt_server(name, server) - if public_name == name: - # Preserve the established registration call contract for ordinary servers. The - # connection key is only an additional concern when profile-owned sessions share a - # configured public name. - registered_names = _registration._register_server_tools(public_name, server, config) - else: - registered_names = _registration._register_server_tools( - public_name, server, config, connection_name=name) + registered_names = _registration._register_server_tools(name, server, config) server._registered_tool_names = list(registered_names) logger.info("MCP server '%s' (%s): registered %d tool(s): %s", name, "HTTP" if "url" in config else "stdio", len(registered_names), ", ".join(registered_names)) @@ -248,84 +273,39 @@ def _select_new_servers(servers: Dict[str, dict]) -> Dict[str, dict]: refresh per-server bookkeeping. Known servers without a live session are parked or mid-reconnect with tools deregistered, so nothing else can nudge them: signal a reconnect.""" with _core._lock: - connecting = set(_core._server_connecting) current_scope = _core._mcp_registry_scope() + # This scope's own connections OR shared ones it adopted (``register_connected_into_current_scope`` + # ran first): a same-named server owned by ANOTHER profile with other credentials is not + # "connected" for us and must be a candidate, or this profile ends up silently tool-less. + keys = {k: _resolve_server_key(k, current_scope, current=False, lock_held=True) for k in servers} # Only attempt servers that aren't already connected (or currently connecting) and are enabled. # Checking ``_server_connecting`` prevents duplicate subprocess spawns when ``discover_mcp_tools()`` # is called from multiple entry-points before the first batch finishes (#58862). - new_servers = {} - new_server_public_names = {} - selected_connection_names = {} - for server_name, config in servers.items(): - connection_name = server_name - existing = _core._servers.get(server_name) - known_keys = (set(_core._server_public_names) | set(_core._server_scope_keys) - | set(_core._server_connecting) | set(_core._server_connect_errors) - | set(_core._server_connect_retry_after) - | set(_core._lazy_server_configs)) - owned_keys = { - key for key in known_keys - if _core._server_public_names.get(key, key) == server_name - } - current_owned_key = next( - (key for key in owned_keys if _core._server_scope_keys.get(key) == current_scope), - None, - ) - foreign_connection = (existing is not None or server_name in _core._lazy_server_configs - or server_name in connecting - or any(_core._server_scope_keys.get(key) != current_scope - for key in owned_keys)) - if (current_scope is not None and _registration._profile_owned_auth(config) - and current_owned_key is not None): - connection_name = current_owned_key - elif (foreign_connection and current_scope is not None - and _registration._profile_owned_auth(config)): - # OAuth tokens and mTLS clients are profile-owned even when route config matches. - # Give this profile a real connection key so it cannot be skipped by the global name. - connection_name = f"{server_name}::profile::{current_scope}" - selected_connection_names[server_name] = connection_name - if (connection_name not in _core._servers and connection_name not in connecting - and connection_name not in _core._lazy_server_configs - and _enabled(config) and not _connect_cooldown_active(connection_name)): - new_servers[connection_name] = config - new_server_public_names[connection_name] = server_name - stale_cached = [ - server for key, server in _core._servers.items() - if getattr(server, "session", None) is None - and _core._server_public_names.get(key, key) in servers - and (current_scope is None or _core._server_visible_in_scope(key, current_scope)) - ] - _core._server_connecting.update(new_servers) - for connection_name in new_servers: - _core._server_scope_keys[connection_name] = current_scope - _core._server_public_names[connection_name] = new_server_public_names[connection_name] - _core._server_connect_errors.pop(connection_name, None) - # Track the configured raw names for compatibility with the public policy ledger, and - # also track each matching internal connection key so profile-owned sessions retain - # independent policy when two profiles use the same configured name. + candidate_keys = { + k: _ScopedCandidateKey(k, keys[k]) if current_scope is not None else keys[k] + for k in servers + } + new_servers = _CandidateServerMap({ + candidate_keys[k]: v for k, v in servers.items() + if keys[k] not in _core._servers and keys[k] not in _core._server_connecting + and keys[k] not in _core._lazy_server_configs + and _enabled(v) and not _connect_cooldown_active(k)}) + stale_cached = [_core._servers[keys[k]] for k in servers + if keys[k] in _core._servers and getattr(_core._servers[keys[k]], "session", None) is None] + for candidate in new_servers: + key = _candidate_ledger_key(candidate) + _core._server_public_names[candidate] = _candidate_public_name(candidate) + _core._server_public_names[key] = _candidate_public_name(candidate) + _core._server_connecting.add(key) + _core._server_scope_keys[key] = current_scope + _core._server_connect_errors.pop(key, None) + # Track which servers opt-in to parallel tool calls (idempotent). for srv_name, srv_cfg in servers.items(): - matching_keys = { - key for key, public_name in _core._server_public_names.items() - if public_name == srv_name - } - matching_keys.update( - key for key in _core._servers - if key == srv_name or _core._server_public_names.get(key) == srv_name - ) - matching_keys.update( - key for key in new_servers - if _core._server_public_names.get(key, key) == srv_name - ) - current_connection_name = selected_connection_names.get(srv_name, srv_name) - policy_keys = matching_keys if current_scope is None else {current_connection_name} + key = keys[srv_name] if _parse_boolish(srv_cfg.get("supports_parallel_tool_calls", False), default=False): - if current_scope is None: - _core._parallel_safe_servers.add(srv_name) - _core._parallel_safe_servers.update(policy_keys) + _core._parallel_safe_servers.add(key) else: - if current_scope is None: - _core._parallel_safe_servers.discard(srv_name) - _core._parallel_safe_servers.difference_update(policy_keys) + _core._parallel_safe_servers.discard(key) for srv in stale_cached: _loop._signal_reconnect(srv) return new_servers @@ -344,24 +324,24 @@ def _register_lazy_from_cache(new_servers: Dict[str, dict]) -> Tuple[Dict[str, d from tools.mcp_schema_cache import config_fingerprint, get_cached_entry except Exception: # pragma: no cover - cache module missing return eager_servers, 0, 0 - for name, cfg in new_servers.items(): + for candidate, cfg in new_servers.items(): + key = _candidate_ledger_key(candidate) + name = _candidate_public_name(candidate) if not _resolve_server_lazy(name, cfg): continue - public_name = _core._server_public_names.get(name, name) - entry = get_cached_entry(public_name, config_fingerprint(cfg)) + entry = get_cached_entry(name, config_fingerprint(cfg)) if not entry: continue with _core._lock: - _core._server_connecting.discard(name) + _core._server_connecting.discard(key) try: - names = _registration._register_from_cache_sync( - name, cfg, entry, public_name=public_name) + names = _registration._register_from_cache_sync(name, cfg, entry) except Exception as exc: logger.warning("Failed lazy MCP registration for '%s': %s", name, exc) with _core._lock: - _core._server_connecting.add(name) + _core._server_connecting.add(key) continue - eager_servers.pop(name, None) + eager_servers.pop(candidate, None) lazy_registered += len(names) lazy_server_count += 1 return eager_servers, lazy_registered, lazy_server_count @@ -370,11 +350,13 @@ def _register_lazy_from_cache(new_servers: Dict[str, dict]) -> Tuple[Dict[str, d async def _discover_all(new_servers: Dict[str, dict]) -> None: """Connect every candidate concurrently; record per-server outcome.""" results = await asyncio.gather( - *(_discover_and_register_server(name, cfg) for name, cfg in new_servers.items()), + *(_discover_and_register_server(_candidate_public_name(candidate), cfg) + for candidate, cfg in new_servers.items()), return_exceptions=True) - for name, result in zip(new_servers, results): + for candidate, result in zip(new_servers, results): + name = _candidate_public_name(candidate) if isinstance(result, BaseException): - command = new_servers.get(name, {}).get("command") + command = new_servers.get(candidate, {}).get("command") message = _note_connect_failure(name, result) logger.warning("Failed to connect to MCP server '%s'%s: %s", name, f" (command={command})" if command else "", message) @@ -396,13 +378,16 @@ def _run_discovery_pass(new_servers: Dict[str, dict]) -> None: # Stranded _server_connecting entries would block future reconnects. how = "timed out" if isinstance(_e, TimeoutError) else "interrupted" with _core._lock: - stale = [n for n in new_servers if n in _core._server_connecting] + stale = [candidate for candidate in new_servers + if _candidate_ledger_key(candidate) in _core._server_connecting] if stale: logger.warning("MCP discovery %s while %d server(s) were still connecting; clearing stale " - "connecting set: %s", how, len(stale), ", ".join(stale)) - _core._server_connecting.difference_update(stale) - for _sn in stale: - _core._server_connect_errors.setdefault(_sn, f"Connection attempt {how} during discovery") + "connecting set: %s", how, len(stale), + ", ".join(_candidate_public_name(candidate) for candidate in stale)) + for candidate in stale: + key = _candidate_ledger_key(candidate) + _core._server_connecting.discard(key) + _core._server_connect_errors.setdefault(key, f"Connection attempt {how} during discovery") raise finally: if _was_interrupted: @@ -412,8 +397,11 @@ def _run_discovery_pass(new_servers: Dict[str, dict]) -> None: def _connected_summary(names, *, lazy_tools: int = 0, lazy_servers: int = 0) -> Tuple[int, int, int]: """(tool count, connected count, failed count) for candidate names, plus lazy servers.""" with _core._lock: - connected = [n for n in names if n in _core._servers and n not in _core._server_connect_errors] - tool_count = sum(len(getattr(_core._servers[n], "_registered_tool_names", [])) for n in connected) + keys = {n: _candidate_ledger_key(n) if _candidate_ledger_key(n) in _core._servers + else _server_key(_candidate_public_name(n)) for n in names} + connected = [n for n in names + if keys[n] in _core._servers and keys[n] not in _core._server_connect_errors] + tool_count = sum(len(getattr(_core._servers[keys[n]], "_registered_tool_names", [])) for n in connected) failed = len(names) - len(connected) return tool_count + lazy_tools, len(connected) + lazy_servers, failed @@ -435,6 +423,15 @@ def register_mcp_servers(servers: Dict[str, dict]) -> List[str]: logger.debug("MCP SDK not available -- skipping explicit MCP registration") return [] servers = _config._filter_suspicious_mcp_servers(servers) + try: + return _register_mcp_servers(servers) + finally: + # An owner's scoped reload orphaned adopters of its shared connections: now that this + # pass (its rediscovery) is done, give them their tools back under their own scope. + _lifecycle._reregister_orphaned_adopters() + + +def _register_mcp_servers(servers: Dict[str, dict]) -> List[str]: scoped_healed = _registration.register_connected_into_current_scope(servers) if not servers: logger.debug("No explicit MCP servers provided") @@ -509,9 +506,10 @@ def discover_mcp_tools(allowed_mcp_names: Optional[List[str]] = None) -> List[st cookie = _acquire_discovery_lock_with_retry() try: with _core._lock: - connecting = set(_core._server_connecting) + keys = {name: _resolve_server_key(name, lock_held=True) for name in servers} new_server_names = [name for name, cfg in servers.items() - if name not in _core._servers and name not in connecting and _enabled(cfg)] + if keys[name] not in _core._servers and keys[name] not in _core._server_connecting + and _enabled(cfg)] tool_names = register_mcp_servers(servers) if new_server_names: _log_summary(" MCP:", new_server_names) @@ -528,11 +526,14 @@ def is_mcp_tool_parallel_safe(tool_name: str) -> bool: return False with _core._lock: scope = _core._mcp_registry_scope() - scoped = getattr(_core, "_mcp_tool_server_names_by_scope", {}).get(scope, {}) - server_name = scoped.get(tool_name) - if server_name is None and scope is None: - server_name = _core._mcp_tool_server_names.get(tool_name) - return bool(server_name and server_name in _core._parallel_safe_servers) + scoped = _core._mcp_tool_server_names_by_scope.get(scope, {}) if scope is not None else {} + key = scoped.get(tool_name) + server_name = _core._mcp_tool_server_names.get(tool_name) + if key is None and not server_name: + return False + if key is None: + key = _resolve_server_key(server_name, lock_held=True) + return key in _core._parallel_safe_servers def get_mcp_status(configured: Optional[Dict[str, dict]] = None, *, include_runtime: bool = True) -> List[dict]: @@ -544,19 +545,15 @@ def get_mcp_status(configured: Optional[Dict[str, dict]] = None, *, include_runt return [] current_scope = _core._mcp_registry_scope() with _core._lock: - def visible(name: str) -> bool: + def visible(key) -> bool: # Runtime state belongs to the profile that adopted it; under a multiplexer only that # profile's view may show it, and ``include_runtime=False`` hides the launch profile's # servers from a status read scoped to a different profile. - return include_runtime and _core._server_visible_in_scope(name, current_scope) + return include_runtime and _core._server_visible_in_scope(key, current_scope) - active_servers = { - _core._server_public_names.get(n, n): s for n, s in _core._servers.items() if visible(n) - } - connecting = {_core._server_public_names.get(n, n) for n in _core._server_connecting if visible(n)} - connect_errors = { - _core._server_public_names.get(n, n): e for n, e in _core._server_connect_errors.items() if visible(n) - } + active_servers = {_key_name(k): s for k, s in _core._servers.items() if visible(k)} + connecting = {_key_name(k) for k in _core._server_connecting if visible(k)} + connect_errors = {_key_name(k): e for k, e in _core._server_connect_errors.items() if visible(k)} result: List[dict] = [] for name, cfg in configured.items(): @@ -617,16 +614,17 @@ def has_registered_mcp_tools() -> bool: """True if any MCP server has registered TOOLS (not merely connected), so the per-turn refresh hook stays idle for zero-tool servers.""" with _core._lock: - return bool(_core._mcp_tool_server_names or any( - getattr(_core, "_mcp_tool_server_names_by_scope", {}).values())) + return bool(_core._mcp_tool_server_names) def get_registered_mcp_server_names() -> set: """Server names that registered at least one tool (live, filtered — not config.yaml).""" with _core._lock: scope = _core._mcp_registry_scope() - if scope is None: - names = set(_core._mcp_tool_server_names.values()) - else: - names = set(getattr(_core, "_mcp_tool_server_names_by_scope", {}).get(scope, {}).values()) - return {_core._server_public_names.get(name, name) for name in names} + if scope is not None: + names = _core._mcp_tool_server_names_by_scope.get(scope, {}) + return { + _core._server_public_names.get(key, _key_name(key)) + for key in names.values() + } + return set(_core._mcp_tool_server_names.values()) diff --git a/tools/mcp_tool_handlers.py b/tools/mcp_tool_handlers.py index 5c4e2011643d9..83876ef861c63 100644 --- a/tools/mcp_tool_handlers.py +++ b/tools/mcp_tool_handlers.py @@ -35,47 +35,44 @@ "cleanly — do NOT retry this tool; ask the user to check the server's command and its stderr log.") -def _display_server_name(server_name: str, public_server_name: Optional[str] = None) -> str: - """Use the configured name in model-visible errors while state remains keyed by connection.""" - return public_server_name or server_name - - -def _trust_gate_check(server_name: str, tool_name: str, *, public_server_name: Optional[str] = None) -> Optional[str]: +def _trust_gate_check(server_name: str, tool_name: str) -> Optional[str]: """Approval gate for write-capable tools on ``trust: untrusted`` servers. None to proceed, else a ``tool_error``. Fail-closed: approval-system errors block.""" - display_name = _display_server_name(server_name, public_server_name) - if (_core._server_trust_levels.get(server_name, _core._TRUST_FULL) != _core._TRUST_UNTRUSTED - or _core._tool_read_only_hints.get(server_name, {}).get(tool_name) is True): + from tools.mcp_tool_scope import _resolve_server_key + key = _resolve_server_key(server_name) + if (_core._server_trust_levels.get(key, _core._TRUST_FULL) != _core._TRUST_UNTRUSTED + or _core._tool_read_only_hints.get(key, {}).get(tool_name) is True): return None try: # lazy: tools.approval routes the prompt to whichever surface owns the session from tools.approval_prompt import request_elicitation_consent answer = request_elicitation_consent( - f"MCP tool '{tool_name}' on UNTRUSTED server '{display_name}' wants to run. This tool is write-capable " + f"MCP tool '{tool_name}' on UNTRUSTED server '{server_name}' wants to run. This tool is write-capable " f"(no readOnlyHint=true annotation) and may modify external state.", - f"Server '{display_name}' is configured 'trust: untrusted'. " + f"Server '{server_name}' is configured 'trust: untrusted'. " f"Approve to run '{tool_name}' once, or deny to block it.", - surface=f"mcp-trust/{display_name}") + surface=f"mcp-trust/{server_name}") except Exception as exc: - logger.error("MCP trust gate: approval check failed for %s.%s: %s", display_name, tool_name, exc, exc_info=True) - return tool_error(f"MCP tool '{tool_name}' on untrusted server '{display_name}' was blocked: the approval " + logger.error("MCP trust gate: approval check failed for %s.%s: %s", server_name, tool_name, exc, exc_info=True) + return tool_error(f"MCP tool '{tool_name}' on untrusted server '{server_name}' was blocked: the approval " f"system was unavailable (fail-closed).") if answer == "accept": return None logger.info("MCP trust gate: user %s '%s' on untrusted server '%s'", - "cancelled" if answer == "cancel" else "denied", tool_name, display_name) + "cancelled" if answer == "cancel" else "denied", tool_name, server_name) return tool_error(f"The user did not approve running write-capable MCP tool '{tool_name}' on untrusted server " - f"'{display_name}'. The command was NOT run. Do not retry without explicit user direction.") + f"'{server_name}'. The command was NOT run. Do not retry without explicit user direction.") -def _check_circuit_breaker(server_name: str, *, public_server_name: Optional[str] = None) -> Optional[str]: +def _check_circuit_breaker(server_name: str) -> Optional[str]: """Open-breaker error, or None when calls may proceed. After the cooldown the breaker is half-open: the next call probes; success resets, failure re-bumps and re-arms the cooldown.""" - display_name = _display_server_name(server_name, public_server_name) - failures = _core._server_error_counts.get(server_name, 0) - age = time.monotonic() - _core._server_breaker_opened_at.get(server_name, 0.0) + from tools.mcp_tool_scope import _resolve_server_key + key = _resolve_server_key(server_name) + failures = _core._server_error_counts.get(key, 0) + age = time.monotonic() - _core._server_breaker_opened_at.get(key, 0.0) if failures < _core._CIRCUIT_BREAKER_THRESHOLD or age >= _core._CIRCUIT_BREAKER_COOLDOWN_SEC: return None - return tool_error(f"MCP server '{display_name}' is unreachable after {failures} consecutive failures. " + return tool_error(f"MCP server '{server_name}' is unreachable after {failures} consecutive failures. " f"Auto-retry available in ~{max(1, int(_core._CIRCUIT_BREAKER_COOLDOWN_SEC - age))}s. Do NOT retry " f"this tool yet — use alternative approaches or ask the user to check the MCP server.") @@ -85,7 +82,7 @@ def _acquire_call_server(server_name: str, tool_timeout: float, *, public_server reconnect may be completing, so wait briefly before a breaker strike; still down -> ask the server task to rebuild (probing a dead transport would re-arm the breaker forever).""" from tools import mcp_tool_discovery as _discovery # lazy: discovery -> registration -> handlers cycle - display_name = _display_server_name(server_name, public_server_name) + display_name = public_server_name or _core._server_public_names.get(server_name, server_name) not_connected = tool_error(f"MCP server '{display_name}' is not connected") server = _discovery._get_connected_server_for_call(server_name) wait = min(5.0, float(tool_timeout or 5.0)) @@ -125,8 +122,9 @@ def _mcp_loop_running() -> bool: def _lookup_reconnectable_server(server_name: str, require_loop: bool = False): """The registered server object when it can be signalled to reconnect, else None. With *require_loop*, also None unless the MCP loop is running (nothing to wait on).""" + from tools.mcp_tool_scope import _resolve_server_key with _core._lock: - srv = _core._servers.get(server_name) + srv = _core._servers.get(_resolve_server_key(server_name, lock_held=True)) ok = srv is not None and hasattr(srv, "_reconnect_event") and (_mcp_loop_running() or not require_loop) return srv if ok else None @@ -145,19 +143,17 @@ def _retry_once(server_name: str, retry_call, op_description: str, what: str): return result -def _handle_auth_error_and_retry(server_name: str, exc: BaseException, retry_call, op_description: str, - *, public_server_name: Optional[str] = None): +def _handle_auth_error_and_retry(server_name: str, exc: BaseException, retry_call, op_description: str): """OAuth recovery + one retry; None when *exc* is not an auth error. ``handle_401`` decides viability; if viable, signal a reconnect (fresh credentials), wait ready, retry once. Any failure returns the structured ``needs_reauth`` error so the model stops refreshing.""" if not _is_auth_error(exc): return None from tools.mcp_oauth_manager import get_manager - auth_server_name = public_server_name or server_name try: - recovered = _loop._run_on_mcp_loop(lambda: get_manager().handle_401(auth_server_name, None), timeout=10) + recovered = _loop._run_on_mcp_loop(lambda: get_manager().handle_401(server_name, None), timeout=10) except Exception as rec_exc: - logger.warning("MCP OAuth '%s': recovery attempt failed: %s", auth_server_name, rec_exc) + logger.warning("MCP OAuth '%s': recovery attempt failed: %s", server_name, rec_exc) recovered = False if recovered: srv = _lookup_reconnectable_server(server_name) @@ -169,8 +165,7 @@ def _handle_auth_error_and_retry(server_name: str, exc: BaseException, retry_cal result = _retry_once(server_name, retry_call, op_description, "auth recovery") if result is not None: return result - return _strike(server_name, _NEEDS_REAUTH_MSG.format(s=auth_server_name), needs_reauth=True, - server=auth_server_name) + return _strike(server_name, _NEEDS_REAUTH_MSG.format(s=server_name), needs_reauth=True, server=server_name) def _handle_session_expired_and_retry(server_name: str, exc: BaseException, retry_call, op_description: str): @@ -199,8 +194,7 @@ class _StdioChildExited(RuntimeError): """Stdio subprocess gone when (or while) a call ran. Deliberately NOT a TimeoutError.""" -def _handle_stdio_child_exited_and_retry(server_name: str, exc: Exception, retry_call, op_description: str, - *, public_server_name: Optional[str] = None): +def _handle_stdio_child_exited_and_retry(server_name: str, exc: Exception, retry_call, op_description: str): """Respawn a dead stdio child and retry once; None if not our error. Never spawns itself: it sets ``_reconnect_event`` and waits, so spawn frequency stays governed by ``run()``'s rapid-drop budget. Single-shot: a child that dies again reports and stops. @@ -210,32 +204,31 @@ def _handle_stdio_child_exited_and_retry(server_name: str, exc: Exception, retry session. Spawn frequency stays governed entirely by ``run()``'s rapid-drop budget, which parks a transport that keeps dropping without proving healthy (#62212). """ - display_name = _display_server_name(server_name, public_server_name) if not isinstance(exc, _StdioChildExited): return None reconnected = False srv = _lookup_reconnectable_server(server_name) if srv is not None: logger.info("MCP server '%s': %s found the stdio subprocess dead (%s); respawning and retrying once.", - display_name, op_description, exc) + server_name, op_description, exc) if _mcp_loop_running(): reconnected = _loop._signal_reconnect_and_wait( server_name, srv, op_description=op_description, timeout=_core._STDIO_RESPAWN_WAIT_SEC) else: # No MCP loop to wait on (non-async adapters, tests): still request the respawn. _loop._signal_reconnect(srv) if not reconnected: - return _strike(server_name, _STDIO_NO_RESPAWN_MSG.format(s=display_name, t=_core._STDIO_RESPAWN_WAIT_SEC)) + return _strike(server_name, _STDIO_NO_RESPAWN_MSG.format(s=server_name, t=_core._STDIO_RESPAWN_WAIT_SEC)) try: return _record_call_outcome(server_name, retry_call()) except _StdioChildExited as retry_exc: # Died again right after respawn: broken server; run()'s budget takes it to the park. logger.warning("MCP server '%s': %s stdio subprocess exited again right after respawn (%s); not retrying " - "further.", display_name, op_description, retry_exc) - return _strike(server_name, _STDIO_DIED_AGAIN_MSG.format(s=display_name)) + "further.", server_name, op_description, retry_exc) + return _strike(server_name, _STDIO_DIED_AGAIN_MSG.format(s=server_name)) except Exception as retry_exc: logger.warning("MCP %s/%s retry after stdio respawn failed: %s", server_name, op_description, retry_exc) return _strike(server_name, _sanitize_error( - f"MCP call failed after respawning the stdio subprocess for '{display_name}': " + f"MCP call failed after respawning the stdio subprocess for '{server_name}': " f"{type(retry_exc).__name__}: {_exc_str(retry_exc)}")) @@ -429,19 +422,18 @@ def _render_call_tool_result(result, server_name: str) -> str: return json.dumps({"result": text_result}, ensure_ascii=False) -def _make_tool_handler(server_name: str, tool_name: str, tool_timeout: float, - *, public_server_name: Optional[str] = None): +def _make_tool_handler(server_name: str, tool_name: str, tool_timeout: float, *, public_server_name: Optional[str] = None): """Sync registry handler (``handler(args_dict, **kwargs) -> str``) calling an MCP tool via the background loop.""" op = f"tools/call {tool_name}" def _handler(args: dict, **kwargs) -> str: # Security boundary: untrusted-server write tools need approval before ANY transport work (incl. lazy spawn). - display_name = _display_server_name(server_name, public_server_name) - error = (_trust_gate_check(server_name, tool_name, public_server_name=display_name) - or _check_circuit_breaker(server_name, public_server_name=display_name)) + error = _trust_gate_check(server_name, tool_name) or _check_circuit_breaker(server_name) if error is not None: return error - server, error = _acquire_call_server(server_name, tool_timeout, public_server_name=display_name) + server, error = _acquire_call_server( + server_name, tool_timeout, public_server_name=public_server_name, + ) if server is None: return error @@ -454,19 +446,14 @@ async def _call(): server._pending_call_context = None if getattr(server, "_mark_session_proven", None) is not None: # round-trip done: transport healthy server._mark_session_proven() - return _render_call_tool_result(result, display_name) + return _render_call_tool_result(result, server_name) def _on_failure(exc): _core._bump_server_error(server_name) logger.error("MCP tool %s/%s call failed: %s", server_name, tool_name, exc) - def _auth_recoverer(internal_name, exc, retry, description): - return _handle_auth_error_and_retry( - internal_name, exc, retry, description, public_server_name=public_server_name) return _dispatch( server_name, server, op, _call, tool_timeout, - (lambda internal, exc, retry, description: _handle_stdio_child_exited_and_retry( - internal, exc, retry, description, public_server_name=display_name), - _auth_recoverer, _handle_session_expired_and_retry), + (_handle_stdio_child_exited_and_retry, _handle_auth_error_and_retry, _handle_session_expired_and_retry), _on_failure, record_outcome=True) return _handler @@ -476,26 +463,22 @@ def _make_utility_handler(op: str, log_label: str, rpc, render, required: Option server_name)`` awaited under ``_rpc_lock``, ``render(result, server_name)`` -> JSON-able payload, ``required`` validated before any transport work.""" def _factory(server_name: str, tool_timeout: float, *, public_server_name: Optional[str] = None): - display_name = _display_server_name(server_name, public_server_name) - def _handler(args: dict, **kwargs) -> str: from tools import mcp_tool_discovery as _discovery # lazy: import cycle server = _discovery._get_connected_server_for_call(server_name) if not server or not server.session: + display_name = public_server_name or _core._server_public_names.get(server_name, server_name) return tool_error(f"MCP server '{display_name}' is not connected") if required and not args.get(required): return tool_error(f"Missing required parameter '{required}'") async def _call(): async with server._rpc_lock: - result = await rpc(server.session, args, display_name) - return json.dumps(render(result, display_name), ensure_ascii=False) - def _auth_recoverer(internal_name, exc, retry, description): - return _handle_auth_error_and_retry( - internal_name, exc, retry, description, public_server_name=public_server_name) + result = await rpc(server.session, args, server_name) + return json.dumps(render(result, server_name), ensure_ascii=False) return _dispatch( server_name, server, op, _call, tool_timeout, - (_auth_recoverer, _handle_session_expired_and_retry), + (_handle_auth_error_and_retry, _handle_session_expired_and_retry), lambda exc: logger.error("MCP %s/%s failed: %s", server_name, log_label, exc)) return _handler return _factory @@ -574,9 +557,12 @@ def _render_get_prompt(result, server_name: str) -> dict: def _make_check_fn(server_name: str): """Connection-alive check; lazy (schema-cache registered) servers count as available.""" + from tools.mcp_tool_scope import _resolve_server_key + def _check() -> bool: with _core._lock: - server = _core._servers.get(server_name) + key = _resolve_server_key(server_name, lock_held=True) + server = _core._servers.get(key) return ((server is not None and (server.session is not None or server._is_recycled_stdio())) - or server_name in _core._lazy_server_configs) + or key in _core._lazy_server_configs) return _check diff --git a/tools/mcp_tool_health.py b/tools/mcp_tool_health.py index c63cdb35d8995..fa39a6d0b8f06 100644 --- a/tools/mcp_tool_health.py +++ b/tools/mcp_tool_health.py @@ -127,7 +127,7 @@ def _deregister_owned(self, tool_names: Iterable[str]) -> None: from tools.registry import registry for tool_name in tool_names: if registry.get_toolset_for_tool(tool_name) == f"mcp-{self.name}": - _registration._deregister_mcp_tool_all_scopes(self._registry_key, tool_name) + _registration._deregister_mcp_tool_all_scopes(self, tool_name) async def _refresh_tools(self): """Re-fetch tools on ``tools/list_changed`` and update the registry. The lock serializes rapid-fire @@ -144,12 +144,7 @@ async def _refresh_tools(self): # Re-register; a raw name can become ambiguous after normalization without changing # its normalized name, so also drop old entries the final registration no longer owns. self._tools = new_mcp_tools - if self._registry_key == self.name: - registered_names = _registration._register_server_tools( - self.name, self, self._config) - else: - registered_names = _registration._register_server_tools( - self.name, self, self._config, connection_name=self._registry_key) + registered_names = _registration._register_server_tools(self.name, self, self._config) self._deregister_owned(old_tool_names - set(registered_names)) self._registered_tool_names = registered_names new_tool_names = set(registered_names) diff --git a/tools/mcp_tool_lifecycle.py b/tools/mcp_tool_lifecycle.py index e718ccefcca77..6503d5cbfad8b 100644 --- a/tools/mcp_tool_lifecycle.py +++ b/tools/mcp_tool_lifecycle.py @@ -84,16 +84,47 @@ def _filter_mcp_children(pids: set) -> set: return kept -def _clear_connect_cooldowns(names=None) -> None: +def _clear_connect_cooldowns(keys=None) -> None: """Drop connect-retry cooldowns: a restart must re-attempt every server immediately, not honour a stale per-server backoff. Caller holds ``_core._lock``.""" - if names is None: + if keys is None: _core._server_connect_retry_after.clear() _core._server_connect_failures.clear() else: - for name in names: - _core._server_connect_retry_after.pop(name, None) - _core._server_connect_failures.pop(name, None) + for key in keys: + _core._server_connect_retry_after.pop(key, None) + _core._server_connect_failures.pop(key, None) + + +def _reregister_orphaned_adopters() -> None: + """Re-run MCP registration for profiles whose ADOPTED shared connection an owner's + ``/reload-mcp`` just tore down. Their tools vanished with the owner's teardown and nothing + re-runs their discovery until THEY reload, so they sat tool-less behind a healthy-looking + status (#106005). Runs after the owner's rediscovery, under each adopter's own home + secret + scope (its ``${VAR}`` refs must resolve to ITS credentials): the adopter re-adopts the owner's + new identical connection or connects its own.""" + with _core._lock: + pending = dict(_core._orphaned_adopters) + _core._orphaned_adopters.clear() + if not pending: + return + from pathlib import Path + from agent.secret_scope import build_profile_secret_scope, reset_secret_scope, set_secret_scope + from hermes_constants import reset_hermes_home_override, set_hermes_home_override + from tools import mcp_tool_discovery as _discovery + from tools.mcp_tool_config import _load_mcp_config + for adopter, names in pending.items(): + home_token = set_hermes_home_override(adopter) + secret_token = set_secret_scope(build_profile_secret_scope(Path(adopter))) + try: + servers = {n: c for n, c in (_load_mcp_config() or {}).items() if n in names} + if servers: + _discovery.register_mcp_servers(servers) + except Exception: + logger.debug("MCP: re-registration for profile scope %s failed", adopter, exc_info=True) + finally: + reset_secret_scope(secret_token) + reset_hermes_home_override(home_token) def shutdown_mcp_servers(*, scope: Optional[str] = None): @@ -103,54 +134,80 @@ def shutdown_mcp_servers(*, scope: Optional[str] = None): (its ``/reload-mcp`` must not kill other profiles') and leaves the shared loop running if anything else is still connected.""" with _core._lock: - selected = [name for name in _core._servers if scope is None or _core._server_scope_keys.get(name) == scope] - servers_snapshot = [_core._servers[name] for name in selected] + selected = [key for key in _core._servers if scope is None or _core._server_scope_keys.get(key) == scope] + adopted = [] if scope is None else [ + key for key, scopes in _core._server_tool_scopes.items() + if scope in scopes and _core._server_scope_keys.get(key) != scope + ] + servers_snapshot = [_core._servers[key] for key in selected] selected_status = ( set(_core._servers) | set(_core._server_scope_keys) | set(_core._server_tool_scopes) | set(_core._server_connecting) | set(_core._server_connect_errors) - | set(_core._lazy_server_configs) | set(_core._lazy_server_tool_names) if scope is None else { - name for name in ( - set(_core._server_scope_keys) | set(_core._server_tool_scopes) - | set(_core._lazy_server_configs) | set(_core._lazy_server_tool_names) - ) if (_core._server_scope_keys.get(name) == scope - or scope in _core._server_tool_scopes.get(name, ())) + key for key, owner in _core._server_scope_keys.items() if owner == scope } ) - selected_names = set(selected) - adopted = { - name for name in selected_status - selected_names - if name in _core._servers and scope in _core._server_tool_scopes.get(name, ()) - } - - lazy_selected = [name for name in selected_status if name in _core._lazy_server_configs] - - def evict_selected_lazy(): - if not lazy_selected: - return - from tools import mcp_tool_registration as _registration - for name in lazy_selected: - _registration._evict_lazy_server(name, scope) + # Adopters of the connections being torn down lose their overlays with the tasks' own + # ``_deregister_tools``; remember them so the next discovery pass re-registers them + # (``_reregister_orphaned_adopters``). + if scope is not None: + from tools.mcp_tool_scope import _key_name + for key in selected: + for adopter in _core._server_tool_scopes.get(key, ()): + if adopter != scope: + _core._orphaned_adopters.setdefault(adopter, set()).add(_key_name(key)) + + # An adopted connection is not owned by the requesting profile, so reload only removes that + # profile's registry overlay and leaves the owner's transport alive. + if adopted: + from tools.mcp_tool_registration import _remove_server_scope + from tools.registry import registry + with _core._lock: + adopted_tools = { + tool_name + for tool_name, key in _core._mcp_tool_server_names_by_scope.get(scope, {}).items() + if key in adopted + } + for tool_name in adopted_tools: + registry.deregister(tool_name, scope=scope) + for key in adopted: + _remove_server_scope(key, scope) + with _core._lock: + scoped_names = _core._mcp_tool_server_names_by_scope.get(scope, {}) + for tool_name, mapped_key in list(scoped_names.items()): + if mapped_key in adopted: + scoped_names.pop(tool_name, None) + if not scoped_names: + _core._mcp_tool_server_names_by_scope.pop(scope, None) def clear_selected_status(): + from tools.registry import registry _core._server_connecting.difference_update(selected_status) - for name in selected_status: - if name in adopted: - continue - _core._server_connect_errors.pop(name, None) - _core._server_scope_keys.pop(name, None) - _core._server_public_names.pop(name, None) - _core._server_tool_scopes.pop(name, None) - - if adopted: - from tools import mcp_tool_registration as _registration - for name in adopted: - _registration._remove_server_scope(name, scope) - - # Lazy entries have no shutdown coroutine, but their scoped registry overlay is still - # live. Evict them before any server teardown clears the ownership maps they use. - evict_selected_lazy() + selected_tools = { + tool_name + for names in _core._mcp_tool_server_names_by_scope.values() + for tool_name, key in names.items() + if key in selected_status + } + for tool_name in selected_tools: + registry.deregister(tool_name, scope=scope) + for key in selected_status: + _core._server_connect_errors.pop(key, None) + _core._server_scope_keys.pop(key, None) + _core._server_tool_scopes.pop(key, None) + _core._lazy_server_configs.pop(key, None) + _core._lazy_server_fingerprints.pop(key, None) + _core._lazy_server_tool_names.pop(key, None) + _core._server_public_names.pop(key, None) + _core._parallel_safe_servers.discard(key) + for names in list(_core._mcp_tool_server_names_by_scope.values()): + for tool_name, key in list(names.items()): + if key in selected_status: + names.pop(tool_name, None) + for owner, names in list(_core._mcp_tool_server_names_by_scope.items()): + if not names: + _core._mcp_tool_server_names_by_scope.pop(owner, None) # Fast path: nothing to shut down. The connect-cooldown maps can still be populated here — a server that # failed to connect is never recorded in ``_servers`` (that is the very premise of the #50394 cooldown), @@ -163,10 +220,9 @@ async def _shutdown(): if isinstance(result, Exception): logger.debug("Error closing MCP server '%s': %s", server.name, result) with _core._lock: - for name in selected: - _core._servers.pop(name, None) - _core._server_scope_keys.pop(name, None) - _core._server_public_names.pop(name, None) + for key in selected: + _core._servers.pop(key, None) + _core._server_scope_keys.pop(key, None) clear_selected_status() _clear_connect_cooldowns(None if scope is None else selected_status) diff --git a/tools/mcp_tool_loop.py b/tools/mcp_tool_loop.py index 3337b278f5674..fa29d687d7101 100644 --- a/tools/mcp_tool_loop.py +++ b/tools/mcp_tool_loop.py @@ -194,8 +194,9 @@ def _signal_reconnect(server: Any) -> bool: def reconnect_mcp_server(server_name: str) -> bool: """Ask a currently-live MCP server to rebuild after external re-auth.""" + from tools.mcp_tool_scope import _resolve_server_key with _core._lock: - server = _core._servers.get(server_name) + server = _core._servers.get(_resolve_server_key(server_name, lock_held=True)) return server is not None and _signal_reconnect(server) diff --git a/tools/mcp_tool_registration.py b/tools/mcp_tool_registration.py index 9464ab705b757..de6872ffe7562 100644 --- a/tools/mcp_tool_registration.py +++ b/tools/mcp_tool_registration.py @@ -17,6 +17,7 @@ _make_list_resources_handler, _make_read_resource_handler) from tools.mcp_tool_schema import ( _UTILITY_CAPABILITY_ATTRS, _build_utility_schemas, _normalize_name_filter, matches_name_filter) +from tools.mcp_tool_scope import _key_name, _resolve_server_key, _server_key if TYPE_CHECKING: # pragma: no cover from tools.mcp_tool import MCPServerTask @@ -54,112 +55,131 @@ def _record_tool_trust_metadata(server_name: str, config: dict, tools: List[Any] """Capture per-server trust and per-tool readOnlyHint at discovery — the security boundary: the call-time gate classifies from data we control, never re-read server-supplied state.""" with _core._lock: - _core._server_trust_levels[server_name] = _normalize_server_trust((config or {}).get("trust")) - hints = _core._tool_read_only_hints.setdefault(server_name, {}) + key = _resolve_server_key(server_name, lock_held=True) + _core._server_trust_levels[key] = _normalize_server_trust((config or {}).get("trust")) + hints = _core._tool_read_only_hints.setdefault(key, {}) hints.update({t.name: _annotation_read_only_hint(t) for t in tools if getattr(t, "name", None)}) -def _track_mcp_tool_server(tool_name: str, server_name: str, *, scope: Optional[str] = None) -> None: - """Remember the exact raw MCP server that registered *tool_name*.""" +def _track_mcp_tool_server(tool_name: str, server_name: str, *, key=None, scope=None) -> None: + """Remember raw connection and public server provenance for *tool_name*.""" with _core._lock: - if scope is None: - _core._mcp_tool_server_names[tool_name] = server_name - else: - _core._mcp_tool_server_names_by_scope.setdefault(scope, {})[tool_name] = server_name + scope = _core._mcp_registry_scope() if scope is None else scope + if key is None: + key = server_name if server_name in _core._server_public_names else _server_key( + server_name, scope, current=False + ) + public_name = _core._server_public_names.get(key, _key_name(key)) + _core._mcp_tool_server_names[tool_name] = public_name + _core._server_public_names[key] = public_name + _core._mcp_tool_server_names_by_scope.setdefault(scope, {})[tool_name] = key -def _forget_mcp_tool_server(tool_name: str, *, scope: Optional[str] = None) -> None: - """Forget MCP server provenance for a deregistered tool.""" +def _forget_mcp_tool_server(tool_name: str, *, removed_key=None) -> None: + """Forget provenance only when no same-named live connection still owns the tool.""" + from tools.registry import registry with _core._lock: - if scope is None: - _core._mcp_tool_server_names.pop(tool_name, None) - else: - scoped = _core._mcp_tool_server_names_by_scope.get(scope) - if scoped is not None: - scoped.pop(tool_name, None) - if not scoped: - _core._mcp_tool_server_names_by_scope.pop(scope, None) + server_name = _core._mcp_tool_server_names.get(tool_name) + if server_name is None: + return + survivors = [ + (key, server) for key, server in _core._servers.items() + if key != removed_key and _key_name(key) == server_name + ] + for key, server in survivors: + scopes = set(_core._server_tool_scopes.get(key, ())) + scopes.add(_core._server_registry_scope(key)) + if any(registry.snapshot_registration(tool_name, scope=scope) is not None for scope in scopes): + return + _core._mcp_tool_server_names.pop(tool_name, None) + for scope, names in list(_core._mcp_tool_server_names_by_scope.items()): + if names.get(tool_name) == removed_key or (removed_key is None and tool_name in names): + names.pop(tool_name, None) + if not names: + _core._mcp_tool_server_names_by_scope.pop(scope, None) + if removed_key is not None and not any( + key == removed_key for names in _core._mcp_tool_server_names_by_scope.values() for key in names.values() + ): + _core._server_public_names.pop(removed_key, None) + + +def _server_key_for_task(server) -> object: + """Connection key of a live ``MCPServerTask`` (teardown runs on the MCP loop without the + discovering profile's context, so the key is found by identity, never re-derived).""" + with _core._lock: + for key, live in _core._servers.items(): + if live is server: + return key + return _server_key(server.name) -def _deregister_mcp_tool_all_scopes(server_name: str, tool_name: str) -> None: - """Deregister one server tool from every profile overlay that owns it.""" +def _deregister_mcp_tool_all_scopes(server, tool_name: str) -> None: + """Deregister one server tool from every profile overlay that owns it. *server* is the + live task or a connection key.""" from tools.registry import registry + key = server if isinstance(server, (str, tuple)) else _server_key_for_task(server) with _core._lock: - scopes = set(_core._server_tool_scopes.get(server_name, ())) + scopes = set(_core._server_tool_scopes.get(key, ())) if not scopes: - scopes = {_core._server_registry_scope(server_name)} + scopes = {_core._server_registry_scope(key)} for scope in scopes: registry.deregister(tool_name, scope=scope) - for scope in scopes: - _forget_mcp_tool_server(tool_name, scope=scope) - _forget_mcp_tool_server(tool_name) - _restore_server_toolset_alias(server_name) + _forget_mcp_tool_server(tool_name, removed_key=key) + _restore_server_toolset_alias(key) -def _restore_server_toolset_alias(server_name: str) -> None: - """Keep the process-global alias while any profile still owns this server's tools.""" +def _restore_server_toolset_alias(key) -> None: + """Keep the process-global alias while any profile still owns tools of a server with this + name — including another profile's same-named connection (the alias is per NAME; the + registry deregister dropped it after checking only one scope).""" from tools.registry import registry + server_name = _key_name(key) with _core._lock: - public_name = _core._server_public_names.get(server_name, server_name) - connection_keys = { - key for key, configured_name in _core._server_public_names.items() - if configured_name == public_name - } - connection_keys.add(server_name) - global_provenance = dict(_core._mcp_tool_server_names) - scoped_provenance = { - scope: dict(names) - for scope, names in getattr(_core, "_mcp_tool_server_names_by_scope", {}).items() - } + owned = [(server, set(_core._server_tool_scopes.get(k, ()))) + for k, server in _core._servers.items() if _key_name(k) == server_name] + scoped = [ + (key, scope) + for scope, names in _core._mcp_tool_server_names_by_scope.items() + for key in names.values() + if _core._server_public_names.get(key, _key_name(key)) == server_name + ] if any( - owner in connection_keys and registry.snapshot_registration(tool_name) is not None - for tool_name, owner in global_provenance.items() - ) or any( - owner in connection_keys and registry.snapshot_registration(tool_name, scope=scope) is not None - for scope, names in scoped_provenance.items() - for tool_name, owner in names.items() + registry.snapshot_registration(tool_name, scope=scope) is not None + for server, scopes in owned for scope in scopes + for tool_name in getattr(server, "_registered_tool_names", ()) ): - registry.register_toolset_alias(public_name, f"mcp-{public_name}") + registry.register_toolset_alias(server_name, f"mcp-{server_name}") + return + if scoped: + registry.register_toolset_alias(server_name, f"mcp-{server_name}") -def _remove_server_scope(server_name: str, scope: Optional[str]) -> None: +def _remove_server_scope(key, scope: str) -> None: """Remove one profile's MCP overlay for a shared live connection.""" from tools.registry import registry - with _core._lock: - public_name = _core._server_public_names.get(server_name, server_name) - scoped_provenance = _core._mcp_tool_server_names_by_scope.get(scope, {}) - tool_names = { - tool_name for tool_name, owner in scoped_provenance.items() - if owner == server_name - } - tool_names.update(_core._lazy_server_tool_names.get(server_name, ())) - server = _core._servers.get(server_name) - tool_names.update(getattr(server, "_registered_tool_names", ()) or ()) - for tool_name in tool_names: - if registry.snapshot_registration(tool_name, scope=scope) is None: - continue + server_name = _core._server_public_names.get(key, _key_name(key)) + for tool_name in registry.get_tool_names_for_toolset(f"mcp-{server_name}"): registry.deregister(tool_name, scope=scope) - _forget_mcp_tool_server(tool_name, scope=scope) with _core._lock: - scopes = set(_core._server_tool_scopes.get(server_name, ())) + scoped_names = _core._mcp_tool_server_names_by_scope.get(scope, {}) + removed_tools = [tool_name for tool_name, mapped_key in scoped_names.items() if mapped_key == key] + for tool_name in removed_tools: + scoped_names.pop(tool_name, None) + if not scoped_names: + _core._mcp_tool_server_names_by_scope.pop(scope, None) + with _core._lock: + scopes = set(_core._server_tool_scopes.get(key, ())) scopes.discard(scope) if scopes: - _core._server_tool_scopes[server_name] = scopes + _core._server_tool_scopes[key] = scopes else: - _core._server_tool_scopes.pop(server_name, None) - _restore_server_toolset_alias(server_name) - - -def _evict_lazy_server(server_name: str, scope: Optional[str]) -> None: - """Deregister a cached lazy overlay before dropping its ownership ledger.""" - _remove_server_scope(server_name, scope) - with _core._lock: - _core._lazy_server_configs.pop(server_name, None) - _core._lazy_server_fingerprints.pop(server_name, None) - _core._lazy_server_tool_names.pop(server_name, None) + _core._server_tool_scopes.pop(key, None) + for tool_name in removed_tools: + _forget_mcp_tool_server(tool_name, removed_key=key) + _restore_server_toolset_alias(key) def _select_utility_schemas(server_name: str, server: "MCPServerTask", config: dict) -> List[dict]: @@ -198,16 +218,16 @@ def _existing_tool_names() -> List[str]: from tools.registry import registry with _core._lock: + scoped_names = set(_core._mcp_tool_server_names_by_scope.get(scope, {})) server_names = [ - _core._server_public_names.get(name, name) for name in _core._servers - if _core._server_visible_in_scope(name, scope) + _core._server_public_names.get(key, _key_name(key)) for key in _core._servers + if _core._server_visible_in_scope(key, scope) ] server_names.extend( - _core._server_public_names.get(name, name) - for name in _core._lazy_server_tool_names - if name not in _core._servers + _core._server_public_names.get(key, _key_name(key)) for key in _core._lazy_server_tool_names + if key not in _core._servers and _core._server_visible_in_scope(key, scope) ) - return sorted({ + return sorted(scoped_names | { tool_name for server_name in server_names for tool_name in registry.get_tool_names_for_toolset(f"mcp-{server_name}") @@ -218,8 +238,8 @@ def _existing_tool_names() -> List[str]: names.extend(server._registered_tool_names if hasattr(server, "_registered_tool_names") else (_schema._convert_mcp_schema(server.name, t)["name"] for t in server._tools)) with _core._lock: - names.extend(n for sname, tool_names in _core._lazy_server_tool_names.items() - if sname not in _core._servers for n in tool_names) + names.extend(n for key, tool_names in _core._lazy_server_tool_names.items() + if key not in _core._servers for n in tool_names) return names @@ -265,7 +285,7 @@ def is_utility(self) -> bool: def _tool_candidates(name: str, tools: Iterable[Any], should_register: Callable[[str], bool], - tool_timeout, *, connection_name: Optional[str] = None) -> List[_Candidate]: + tool_timeout) -> List[_Candidate]: """Native tools (live SDK objects or cache stand-ins) -> candidates. The injection scan runs on BOTH paths: the cache file is user-writable JSON.""" out: List[_Candidate] = [] @@ -275,26 +295,19 @@ def _tool_candidates(name: str, tools: Iterable[Any], should_register: Callable[ continue _schema._scan_mcp_description(name, t.name, t.description or "") schema = _schema._convert_mcp_schema(name, t) - connection_key = connection_name or name - handler = (_handlers._make_tool_handler( - connection_key, t.name, tool_timeout, public_server_name=name) - if connection_key != name else _handlers._make_tool_handler( - connection_key, t.name, tool_timeout)) + handler = _handlers._make_tool_handler(name, t.name, tool_timeout) out.append(_Candidate(schema["name"], f"tool {t.name!r}", schema, handler)) return out -def _utility_candidates(name: str, entries: Iterable[Any], tool_timeout, *, connection_name: Optional[str] = None) -> List[_Candidate]: +def _utility_candidates(name: str, entries: Iterable[Any], tool_timeout) -> List[_Candidate]: """``{schema, handler_key}`` rows (live selection or cache) -> candidates; malformed rows dropped.""" out: List[_Candidate] = [] for raw in entries: schema, key = (raw.get("schema"), raw.get("handler_key")) if isinstance(raw, dict) else (None, None) if isinstance(schema, dict) and key in _UTILITY_HANDLER_FACTORIES and schema.get("name"): - connection_key = connection_name or name - factory = _UTILITY_HANDLER_FACTORIES[key] - handler = (factory(connection_key, tool_timeout, public_server_name=name) - if connection_key != name else factory(connection_key, tool_timeout)) - out.append(_Candidate(schema["name"], f"{_UTILITY_ORIGIN_PREFIX}{key!r}", schema, handler)) + out.append(_Candidate(schema["name"], f"{_UTILITY_ORIGIN_PREFIX}{key!r}", schema, + _UTILITY_HANDLER_FACTORIES[key](name, tool_timeout))) return out @@ -342,15 +355,17 @@ def _resolve_name_collisions(name: str, candidates: List[_Candidate]) -> List[_C def _register_candidates(name: str, candidates: List[_Candidate], *, check_fn: Callable, - scope: Callable[[], Optional[str]], lazy: bool, - server_key: Optional[str] = None) -> List[str]: + scope: Callable[[], Optional[str]], lazy: bool, key=None) -> List[str]: """Register candidates under toolset ``mcp-{name}``; returns the names that landed. The ownership pre-check is advisory (servers connect in parallel): ``registry.register()`` is - the atomic gate and its verdict is re-read after every call.""" + the atomic gate and its verdict is re-read after every call. *key* is the connection whose + ``_server_tool_scopes`` records the registering scope (default: this scope's own).""" from tools.registry import registry toolset_name = f"mcp-{name}" registered: List[str] = [] scope_value = scope() + if key is None: + key = _server_key(name, scope_value, current=False) for c in candidates: existing_toolset = registry.get_toolset_for_tool(c.registry_name) if existing_toolset and existing_toolset != toolset_name: # foreign owner: skip, preserve it @@ -369,10 +384,10 @@ def _register_candidates(name: str, candidates: List[_Candidate], *, check_fn: C name=c.registry_name, toolset=toolset_name, schema=c.schema, handler=c.handler, check_fn=check_fn, is_async=False, description=c.schema.get("description") or "", scope=scope_value) if registry.get_toolset_for_tool(c.registry_name) == toolset_name: - _track_mcp_tool_server(c.registry_name, server_key or name, scope=scope_value) + _track_mcp_tool_server(c.registry_name, name, key=key, scope=scope_value) if scope_value is not None: with _core._lock: - _core._server_tool_scopes.setdefault(server_key or name, set()).add(scope_value) + _core._server_tool_scopes.setdefault(key, set()).add(scope_value) registered.append(c.registry_name) elif not lazy: logger.error("MCP server '%s': registration of %s as '%s' was rejected by the registry; " @@ -412,24 +427,18 @@ def _write_schema_cache(name: str, server: "MCPServerTask", config: dict, should logger.debug("MCP schema cache write failed for '%s': %s", name, exc) -def _register_server_tools( - name: str, server: "MCPServerTask", config: dict, *, connection_name: Optional[str] = None -) -> List[str]: +def _register_server_tools(name: str, server: "MCPServerTask", config: dict) -> List[str]: """Register a connected server's tools plus utilities (initial discovery and list_changed refresh); returns the names. Toolset aliases derive from the live registry, not ``toolsets.TOOLSETS``; lossy normalization collisions (``read-file``/``read_file``) fail closed.""" - connection_name = connection_name or name should_register = _make_tool_filter(name, config) - _record_tool_trust_metadata(connection_name, config, server._tools) - candidates = _tool_candidates(name, server._tools, should_register, server.tool_timeout, - connection_name=connection_name) - candidates += _utility_candidates(name, _select_utility_schemas(name, server, config), server.tool_timeout, - connection_name=connection_name) + key = _server_key_for_task(server) + _record_tool_trust_metadata(name, config, server._tools) + candidates = _tool_candidates(name, server._tools, should_register, server.tool_timeout) + candidates += _utility_candidates(name, _select_utility_schemas(name, server, config), server.tool_timeout) registered = _register_candidates( name, _resolve_name_collisions(name, candidates), - check_fn=_make_check_fn(connection_name or name), - scope=lambda: _core._server_registry_scope(connection_name or name), lazy=False, - server_key=connection_name) + check_fn=_make_check_fn(name), scope=lambda: _core._server_registry_scope(key), lazy=False, key=key) if registered: _write_schema_cache(name, server, config, should_register) return registered @@ -447,6 +456,7 @@ def _connection_identity(config: dict) -> tuple: OAuth and mTLS inputs are included here as well, even though profile-owned credentials still require an owner-scope check below.""" from tools.mcp_schema_cache import config_fingerprint + from tools.mcp_tool_config import _CONNECTION_EXTERNAL_ENV_KEY, _external_secret_env def _frozen(value): return json.dumps(value or {}, sort_keys=True, default=str) @@ -456,8 +466,11 @@ def _frozen(value): for key in ("auth", "oauth", "client_cert", "client_key") if config.get(key) is not None } + external_env = config.get(_CONNECTION_EXTERNAL_ENV_KEY) + if external_env is None and "command" in config and "url" not in config: + external_env = _external_secret_env() return (config_fingerprint(config), _frozen(config.get("env")), _frozen(config.get("headers")), - _frozen(auth_inputs)) + _frozen(auth_inputs), _frozen(external_env)) def _profile_owned_auth(config: dict) -> bool: @@ -470,7 +483,7 @@ def _same_server_route(server: Any, config: dict) -> bool: return _connection_identity(getattr(server, "_config", {}) or {}) == _connection_identity(config) -def _connection_reusable_in_scope(name: str, server: Any, config: dict, scope: str) -> bool: +def _connection_reusable_in_scope(name: str, server: Any, config: dict, scope: str, *, owner_key=None) -> bool: """Allow a profile to adopt only a connection safe for its scope. Route/auth config equality is not enough for OAuth because the token store is under the @@ -479,7 +492,8 @@ def _connection_reusable_in_scope(name: str, server: Any, config: dict, scope: s """ if not _same_server_route(server, config): return False - return not (_profile_owned_auth(config) and _core._server_scope_keys.get(name) != scope) + owner_key = _server_key_for_task(server) if owner_key is None else owner_key + return not (_profile_owned_auth(config) and _core._server_scope_keys.get(owner_key) != scope) def register_connected_into_current_scope(servers: dict) -> int: @@ -505,67 +519,73 @@ def _register_connected_into_current_scope(servers: dict) -> int: if scope is None: return 0 + # Lazy schema-cache entries have no live task in ``_servers``, so they are not covered by + # the live-connection sweep below. Reconcile their stored fingerprint against the current + # profile config first; otherwise a removed/changed server leaves a callable stale overlay + # and its provenance makes the next discovery pass appear already satisfied. from tools.mcp_schema_cache import config_fingerprint + with _core._lock: + stale_lazy = [] + for key, fingerprint in _core._lazy_server_fingerprints.items(): + if _core._server_scope_keys.get(key) != scope: + continue + name = _core._server_public_names.get(key, _key_name(key)) + config = servers.get(name) + if config is None or not _server_enabled(config) or config_fingerprint(config) != fingerprint: + stale_lazy.append(key) + for key in stale_lazy: + with _core._lock: + cached_tool_names = list(_core._lazy_server_tool_names.get(key, ())) + for tool_name in cached_tool_names: + registry.deregister(tool_name, scope=scope) + _remove_server_scope(key, scope) + with _core._lock: + _core._lazy_server_configs.pop(key, None) + _core._lazy_server_fingerprints.pop(key, None) + _core._lazy_server_tool_names.pop(key, None) + _core._server_scope_keys.pop(key, None) + _core._server_public_names.pop(key, None) with _core._lock: stale = [] - for name, scopes in _core._server_tool_scopes.items(): + for key, scopes in _core._server_tool_scopes.items(): if scope not in scopes: continue - if name in _core._lazy_server_configs: - # A cached lazy overlay has no live task yet, but remains valid only while the - # current config matches the cached manifest. Changed or removed config must - # release the overlay so the next discovery can reselect the server. - public_name = _core._server_public_names.get(name, name) - config = servers.get(public_name) - fingerprint = config_fingerprint(config) if isinstance(config, dict) else None - if (config is None or not _server_enabled(config) - or fingerprint != _core._lazy_server_fingerprints.get(name)): - stale.append(name) - continue - server = _core._servers.get(name) - config = servers.get(_core._server_public_names.get(name, name)) + server = _core._servers.get(key) + config = servers.get(_core._server_public_names.get(key, _key_name(key))) if (config is None or not _server_enabled(config) or server is None or getattr(server, "session", None) is None - or not _connection_reusable_in_scope(name, server, config, scope)): - stale.append(name) - for name in stale: - _remove_server_scope(name, scope) - with _core._lock: - if name in _core._lazy_server_configs: - _core._lazy_server_configs.pop(name, None) - _core._lazy_server_fingerprints.pop(name, None) - _core._lazy_server_tool_names.pop(name, None) + or getattr(server, "session", None) is None or not _same_server_route(server, config)): + stale.append(key) + for key in stale: + _remove_server_scope(key, scope) registered_servers = 0 for name, config in servers.items(): if not _server_enabled(config): continue with _core._lock: - connection_name = next( - (key for key, public_name in _core._server_public_names.items() - if public_name == name and _core._server_visible_in_scope(key, scope)), - name if name in _core._servers else None, - ) - server = _core._servers.get(connection_name) if connection_name else None - if (server is None or getattr(server, "session", None) is None - or not _connection_reusable_in_scope(connection_name or name, server, config, scope)): + if _server_key(name, scope, current=False) in _core._servers: + continue # this profile has its own connection for the name + # Any other profile's live connection with the same route AND credentials is shareable. + shared = [(key, live) for key, live in _core._servers.items() + if _core._server_public_names.get(key, _key_name(key)) == name + and getattr(live, "session", None) is not None + and _connection_reusable_in_scope(name, live, config, scope, owner_key=key)] + if not shared: continue + key, server = shared[0] # Visibility for this profile: the owner keeps teardown, this scope sees the connection. with _core._lock: - _core._server_tool_scopes.setdefault(connection_name or name, set()).add(scope) + _core._server_tool_scopes.setdefault(key, set()).add(scope) if registry.get_tool_names_for_toolset(f"mcp-{name}"): continue - candidates = _tool_candidates( - name, server._tools, _make_tool_filter(name, config), server.tool_timeout, - connection_name=connection_name or name) + candidates = _tool_candidates(name, server._tools, _make_tool_filter(name, config), server.tool_timeout) candidates += _utility_candidates( - name, _select_utility_schemas(name, server, config), server.tool_timeout, - connection_name=connection_name or name) + name, _select_utility_schemas(name, server, config), server.tool_timeout) names = _register_candidates( name, _resolve_name_collisions(name, candidates), - check_fn=_make_check_fn(connection_name or name), scope=lambda: scope, lazy=False, - server_key=connection_name or name) + check_fn=_make_check_fn(name), scope=lambda: scope, lazy=False, key=key) if names: registered_servers += 1 with _core._lock: @@ -574,9 +594,7 @@ def _register_connected_into_current_scope(servers: dict) -> int: return registered_servers -def _register_from_cache_sync( - name: str, config: dict, entry: dict, *, public_name: Optional[str] = None -) -> List[str]: +def _register_from_cache_sync(name: str, config: dict, entry: dict) -> List[str]: """Lazy startup: register from a cached manifest with no child process (first real call goes through ``_ensure_lazy_server_connected``). Trust metadata is recorded first so the call-time gate is identical for live and cached registrations. @@ -585,21 +603,18 @@ def _register_from_cache_sync( call routes through ``_get_connected_server_for_call`` → ``_ensure_lazy_server_connected``. """ from tools.mcp_schema_cache import config_fingerprint, tools_from_cache_entry, utility_tools_from_cache_entry - public_name = public_name or _core._server_public_names.get(name, name) tool_timeout = _resolve_tool_timeout(config) cached_tools = _cached_tools(tools_from_cache_entry(entry)) _record_tool_trust_metadata(name, config, cached_tools) - candidates = _tool_candidates(public_name, cached_tools, _make_tool_filter(public_name, config), tool_timeout, - connection_name=name) - candidates += _utility_candidates(public_name, utility_tools_from_cache_entry(entry), tool_timeout, - connection_name=name) + candidates = _tool_candidates(name, cached_tools, _make_tool_filter(name, config), tool_timeout) + candidates += _utility_candidates(name, utility_tools_from_cache_entry(entry), tool_timeout) registered = _register_candidates( - public_name, candidates, check_fn=_make_check_fn(name), scope=_core._mcp_registry_scope, lazy=True, - server_key=name) + name, candidates, check_fn=_make_check_fn(name), scope=_core._mcp_registry_scope, lazy=True) if registered: with _core._lock: - _core._lazy_server_configs[name] = dict(config) - _core._lazy_server_fingerprints[name] = config_fingerprint(config) - _core._lazy_server_tool_names[name] = list(registered) + key = _server_key(name) + _core._lazy_server_configs[key] = dict(config) + _core._lazy_server_fingerprints[key] = config_fingerprint(config) + _core._lazy_server_tool_names[key] = list(registered) logger.info("MCP server '%s' (lazy): registered %d tool(s) from schema cache", name, len(registered)) return registered diff --git a/tools/mcp_tool_scope.py b/tools/mcp_tool_scope.py new file mode 100644 index 0000000000000..8b9cde4b3e263 --- /dev/null +++ b/tools/mcp_tool_scope.py @@ -0,0 +1,85 @@ +"""Connection-ledger keys for tools.mcp_tool under a profile multiplexer. + +Every ledger in ``tools.mcp_tool`` (``_servers``, connecting/error/cooldown maps, circuit +breaker, lazy configs, trust metadata) is keyed by a *connection key*: the bare server name +outside a multiplexer (single-profile processes are unchanged, byte for byte), and +``(owner_scope, name)`` under one. Two profiles that both configure ``github`` with their own +token are two connections; keying by name alone let the first profile's connection shadow the +second's forever — its ``register_mcp_servers`` saw the name as "already connected", adopted +nothing (different credentials) and left the profile silently tool-less (#106005, #91654). + +A profile may still *adopt* another profile's live connection when the route and credentials +match (``mcp_tool_registration._same_server_route``); ``_server_tool_scopes[key]`` records every +scope that has done so, and ``_resolve_server_key`` finds that shared connection for a caller +whose own scope has none. +""" + +from __future__ import annotations + +from typing import Optional, Tuple, Union + +from tools.mcp_tool_common import _core + +ServerKey = Union[str, Tuple[str, str]] + + +def _server_key(name: str, scope: Optional[str] = None, *, current: bool = True) -> ServerKey: + """Connection key for *name* owned by *scope* (the current registry scope when *current*). + ``None`` scope (no multiplexer) keeps the bare name.""" + if scope is None and current: + scope = _core._mcp_registry_scope() + return name if scope is None else (scope, name) + + +def _key_name(key: ServerKey) -> str: + return key[1] if isinstance(key, tuple) else key + + +def _key_scope(key: ServerKey) -> Optional[str]: + """Owning registry scope encoded in *key* (None for a bare, unscoped key).""" + return key[0] if isinstance(key, tuple) else None + + +def _key_visible_in_scope(key: ServerKey, scope: Optional[str]) -> bool: + """Whether the connection under *key* serves *scope*: owned by it or adopted into it. + Caller holds ``_core._lock`` or tolerates a racy read (status surfaces).""" + if scope is None: + return True + return _key_scope(key) == scope or scope in _core._server_tool_scopes.get(key, ()) + + +def _resolve_server_key( + name: str, scope: Optional[str] = None, *, current: bool = True, lock_held: bool = False +) -> ServerKey: + """The connection key a call to *name* from *scope* must use: the scope's own connection + (live, connecting or lazily registered) when it has one, else a shared connection it + adopted, else its own (not yet existing) key so bookkeeping lands under this scope. + + The connection ledgers are mutable from gateway and MCP-loop threads. Callers that already + hold the non-reentrant core lock pass ``lock_held=True``; all other callers get a short + lock-protected lookup. + """ + if scope is None and current: + scope = _core._mcp_registry_scope() + + def resolve_unlocked() -> ServerKey: + own = _server_key(name, scope, current=False) + # Older in-process callers may expose a human-readable scoped key. Prefer its + # explicit provenance metadata rather than parsing the key (server names may contain + # the delimiter), while keeping the tuple key for new connections. + if scope is not None: + for key in set(_core._servers) | set(_core._lazy_server_configs) | set(_core._server_scope_keys): + if (_core._server_scope_keys.get(key) == scope + and _core._server_public_names.get(key) == name): + return key + if scope is None or own in _core._servers or own in _core._lazy_server_configs: + return own + for key, scopes in tuple(_core._server_tool_scopes.items()): + if scope in scopes and _key_name(key) == name and key in _core._servers: + return key + return own + + if lock_held: + return resolve_unlocked() + with _core._lock: + return resolve_unlocked() diff --git a/tools/mcp_tool_server_run.py b/tools/mcp_tool_server_run.py index 30e61ab833ee6..2363aff4a1127 100644 --- a/tools/mcp_tool_server_run.py +++ b/tools/mcp_tool_server_run.py @@ -414,5 +414,5 @@ def _deregister_tools(self) -> None: """Drop this server's tools from the registry (idempotent); on shutdown AND budget exhaustion, so a dead server never leaves phantom tools in the prompt.""" for tool_name in list(getattr(self, "_registered_tool_names", [])): - _registration._deregister_mcp_tool_all_scopes(self._registry_key, tool_name) + _registration._deregister_mcp_tool_all_scopes(self, tool_name) self._registered_tool_names = [] diff --git a/tools/mcp_tool_transport.py b/tools/mcp_tool_transport.py index 2132c574d5b83..7995b6d858c05 100644 --- a/tools/mcp_tool_transport.py +++ b/tools/mcp_tool_transport.py @@ -117,7 +117,7 @@ async def _serve_session(self, session, connect_timeout: float, await self._discover_tools() self._ready.set() self._ever_connected = True - _core._reset_server_error(self._registry_key) + _core._reset_server_error(self.name) # Session is live again: clear any breaker state from a prior outage so the first call after # recovery isn't gated on a stale consecutive-failure count (#16788). # A completed handshake alone is NOT proof of health: a flapping transport can handshake fine and @@ -213,7 +213,13 @@ async def _run_stdio(self, config: dict): command = config.get("command") if not command: raise ValueError(f"MCP server '{self.name}' has no 'command' in config") - command, safe_env = _config._resolve_stdio_command(command, _config._build_safe_env(config.get("env"))) + command, safe_env = _config._resolve_stdio_command( + command, + _config._build_safe_env( + config.get("env"), + external_env=config.get(_config._CONNECTION_EXTERNAL_ENV_KEY), + ), + ) # OSV malware preflight, then the cached-npx swap (ordering enforced there). command, args = await _core._preflight_stdio_command(self.name, command, config.get("args", [])) server_params = _core.StdioServerParameters( @@ -449,15 +455,11 @@ def _register_discovered_tools_if_needed(self) -> None: if self._registered_tool_names: return with _core._lock: - owned = _core._servers.get(self._registry_key) is self + owned = [key for key, live in _core._servers.items() if live is self] if not owned and not self._ready.is_set(): return - if self._registry_key == self.name: - self._registered_tool_names = _registration._register_server_tools( - self.name, self, self._config) - else: - self._registered_tool_names = _registration._register_server_tools( - self.name, self, self._config, connection_name=self._registry_key) + self._registered_tool_names = _registration._register_server_tools(self.name, self, self._config) with _core._lock: # a retained initial-failure server that just published tools has recovered - if _core._servers.get(self._registry_key) is self: - _core._server_connect_errors.pop(self._registry_key, None) + for key in owned: + if _core._servers.get(key) is self: + _core._server_connect_errors.pop(key, None) diff --git a/toolsets.py b/toolsets.py index 41e0208a40fc3..896ca71c5f0bd 100644 --- a/toolsets.py +++ b/toolsets.py @@ -325,9 +325,11 @@ def bundle_non_core_tools(toolset_name: str) -> Set[str]: return to_remove - core -# Memo keyed on (name, include_registry, id(registry), registry generation); -# engages only at the public entry (visited is None). -_resolve_toolset_memo: Dict[Tuple[str, bool, int, int], List[str]] = {} +# Memo keyed on (name, include_registry, id(registry), registry generation, profile scope); +# engages only at the public entry (visited is None). The scope is part of the key because a +# multiplexed process resolves ``mcp-`` per profile overlay: without it profile B got +# profile A's tool names for a server B never connected (#106005). +_resolve_toolset_memo: Dict[Tuple[str, bool, int, int, str], List[str]] = {} def _plugin_platform_bundle(name: str) -> List[str]: @@ -361,7 +363,7 @@ def resolve_toolset(name: str, visited: Set[str] = None, *, include_registry: bo """ external_call = visited is None if external_call: - memo_key = (name, include_registry, *_registry_generation()) + memo_key = (name, include_registry, *_registry_generation(), _registry_call("current_scope_key", "")) cached = _resolve_toolset_memo.get(memo_key) if cached is not None: return list(cached) diff --git a/tui_gateway/methods_profiles.py b/tui_gateway/methods_profiles.py index a17065e4e20c4..f1cfec4f38dc4 100644 --- a/tui_gateway/methods_profiles.py +++ b/tui_gateway/methods_profiles.py @@ -350,6 +350,10 @@ def _mirror_launch_credentials(path, params: dict) -> dict: # .env: only over the seeded comment-only stub (never a clone's secrets). mirrored["env"] = _try(lambda: _mirror_secret(path, launch_home, ".env", lambda src, dst: ( _env_has_content(src) and not _try(lambda: _env_has_content(dst), False))), False) + if mirrored["env"] and not is_truthy_value(params.get("clone_channels", False)): + # Provider/tool keys are what "mirror credentials" means; the launch profile's bot tokens + # and allowlists would make the new bot collide with it over one Telegram/Discord bot. + _best_effort(lambda: _lazy("hermes_cli.profile_channels", "strip_channel_env_file")(path / ".env")) if not share_auth: # a copy forks token state: the first refresh in either store strands the other mirrored["auth"] = _try(lambda: _mirror_secret(path, launch_home, "auth.json", lambda src, dst: not dst.exists()), False) @@ -364,8 +368,9 @@ def _mirror_launch_credentials(path, params: dict) -> dict: @method("profiles.create") def _(rid, params: dict) -> dict: """Create a profile (ws twin of POST /api/profiles). Params: ``name``, ``description``, - ``clone_from`` (omitted = fresh + bundled skills), ``clone_all``, ``no_skills``, ``soul``, - ``model`` + ``provider``, ``share_auth``, ``mirror_credentials`` (default true: a bare + ``clone_from`` (omitted = fresh + bundled skills), ``clone_all``, ``clone_channels`` (opt-in: keep the + source's bot tokens/allowlists — default strips them so two profiles never hold one bot), ``no_skills``, ``soul``, + ``model`` + ``provider``, ``share_auth``, ``no_alias``, ``mirror_credentials`` (default true: a bare ``create_profile()`` seeds a comment-only .env and no auth.json = NO provider headless).""" name = str(params.get("name") or "").strip() if not name: @@ -378,7 +383,8 @@ def _(rid, params: dict) -> dict: name=name, clone_from=clone_from, clone_all=clone_all, clone_config=bool(clone_from) and not clone_all, no_skills=is_truthy_value(params.get("no_skills", False)), - description=str(params.get("description") or "").strip() or None) + description=str(params.get("description") or "").strip() or None, + clone_channels=is_truthy_value(params.get("clone_channels", False))) except (ValueError, FileExistsError, FileNotFoundError) as e: return _err(rid, 4062, str(e)) except Exception as e: diff --git a/web/src/lib/api.ts b/web/src/lib/api.ts index dee205d9d2074..2f01e27ecde03 100644 --- a/web/src/lib/api.ts +++ b/web/src/lib/api.ts @@ -875,8 +875,10 @@ export const api = { // Messaging platforms (gateway channels) getMessagingPlatforms: () => fetchJSON("/api/messaging/platforms"), + // `hot_served`: a live multiplexer serving the selected named profile rebuilt its adapters from the + // new credentials right away (no gateway restart needed). updateMessagingPlatform: (id: string, body: MessagingPlatformUpdate) => - fetchJSON<{ ok: boolean; platform: string }>( + fetchJSON<{ ok: boolean; platform: string; hot_served?: boolean }>( `/api/messaging/platforms/${encodeURIComponent(id)}`, { method: "PUT", diff --git a/web/src/pages/ChannelsPage.tsx b/web/src/pages/ChannelsPage.tsx index 08d781e306324..f7b70b9d74524 100644 --- a/web/src/pages/ChannelsPage.tsx +++ b/web/src/pages/ChannelsPage.tsx @@ -215,11 +215,17 @@ export default function ChannelsPage() { setSaving(true); try { const body: MessagingPlatformUpdate = { env, enabled: true }; - await api.updateMessagingPlatform(editing.id, body); - showToast(`${editing.name} saved`, "success"); + const result = await api.updateMessagingPlatform(editing.id, body); + showToast( + result.hot_served + ? `${editing.name} saved; the running gateway is connecting` + : `${editing.name} saved`, + "success", + ); setEditing(null); - setRestartNeeded(true); + if (!result.hot_served) setRestartNeeded(true); await load(); + if (result.hot_served) setTimeout(() => void load(), 4000); } catch (e) { showToast(`Failed to save: ${e}`, "error"); } finally { @@ -231,7 +237,7 @@ export default function ChannelsPage() { const next = !platform.enabled; setTogglingId(platform.id); try { - await api.updateMessagingPlatform(platform.id, { enabled: next }); + const result = await api.updateMessagingPlatform(platform.id, { enabled: next }); setPlatforms((prev) => prev.map((p) => p.id === platform.id @@ -239,7 +245,8 @@ export default function ChannelsPage() { : p, ), ); - setRestartNeeded(true); + if (result.hot_served) setTimeout(() => void load(), 4000); + else setRestartNeeded(true); } catch (e) { showToast(`Error: ${e}`, "error"); } finally { diff --git a/website/docs/reference/toolsets-reference.md b/website/docs/reference/toolsets-reference.md index c6925facb0098..eb6476e4f7324 100644 --- a/website/docs/reference/toolsets-reference.md +++ b/website/docs/reference/toolsets-reference.md @@ -69,7 +69,7 @@ Or in-session: | `context_engine` | (varies) | Runtime tools exposed by the active context-engine plugin (empty until a plugin populates it). | | `image_gen` | `image_generate` | Text-to-image generation via FAL.ai (with opt-in OpenAI / xAI backends). | | `video_gen` | `video_generate`, `xai_video_edit`, `xai_video_extend` | Text-to-video and image-to-video via plugin-registered backends (xAI Grok-Imagine, FAL.ai Veo 3.1 / Pixverse v6 / Kling O3). Pass `image_url` to animate an image; omit it for text-to-video. `xai_video_edit` / `xai_video_extend` are provider-specific edit/extend tools, gated on xAI Imagine credentials. | -| `kanban` | `kanban_attach`, `kanban_attach_url`, `kanban_attachments`, `kanban_block`, `kanban_comment`, `kanban_complete`, `kanban_create`, `kanban_heartbeat`, `kanban_link`, `kanban_list`, `kanban_request_changes`, `kanban_request_review`, `kanban_show`, `kanban_unblock` | Multi-agent coordination tools. Registered for dispatcher-spawned task workers (`HERMES_KANBAN_TASK`) and for profiles that explicitly list the `kanban` toolset by name (the `all`/`*` wildcard does **not** enable it). Workers mark tasks done, request first-class review, block, heartbeat, comment, and create/link follow-up tasks; orchestrator profiles additionally get board-routing tools like list/unblock. `delegate_task` children are not Kanban run owners: their schema strips/disables this toolset and runtime guards reject direct board mutations, even if parent `HERMES_KANBAN_*` env vars are present. | +| `kanban` | `kanban_attach`, `kanban_attach_url`, `kanban_attachments`, `kanban_block`, `kanban_comment`, `kanban_complete`, `kanban_create`, `kanban_heartbeat`, `kanban_link`, `kanban_list`, `kanban_request_changes`, `kanban_request_review`, `kanban_show`, `kanban_unblock` | Multi-agent coordination tools. Registered for dispatcher-spawned task workers (`HERMES_KANBAN_TASK`) and for platforms whose saved selection lists `kanban` (`hermes tools enable kanban --platform

`; the `all`/`*` wildcard does **not** enable it). Workers mark tasks done, request first-class review, block, heartbeat, comment, and create/link follow-up tasks; orchestrator profiles additionally get board-routing tools like list/unblock. `delegate_task` children are not Kanban run owners: their schema strips/disables this toolset and runtime guards reject direct board mutations, even if parent `HERMES_KANBAN_*` env vars are present. | | `memory` | `memory` | Persistent cross-session memory management. | | `desktop_ui` | `annotate_preview`, `close_preview`, `close_terminal`, `drive_preview`, `focus_pane`, `open_preview`, `react_to_message`, `read_preview`, `read_terminal`, `read_window_below`, `tour` | Affordances that act on the Hermes desktop app itself — read/close the embedded terminal pane, open, read, close, interact with, and annotate the in-app browser, identify the OS window behind the app, reveal a pane, react to a message, run a guided tour (highlight + narrate UI elements in the app or the preview pane). Enabled for sessions whose source is the desktop app, whichever backend it's connected to (local, SSH, URL, or Hermes Cloud). Never present on CLI, TUI, messaging, or cron sessions. | | `project` | `project_create`, `project_list`, `project_switch` | Create and switch desktop [Projects](../user-guide/cli.md) (named, multi-folder workspaces). GUI / desktop sessions only. | diff --git a/website/docs/user-guide/features/kanban.md b/website/docs/user-guide/features/kanban.md index acddf18bb9fb4..459f89f0ee850 100644 --- a/website/docs/user-guide/features/kanban.md +++ b/website/docs/user-guide/features/kanban.md @@ -341,6 +341,29 @@ parent, missing input, unmet capability) before unblocking, or raise `BLOCK_RECURRENCE_LIMIT` if the loop is expected. ::: +## Enabling tools for a chat profile + +The Desktop Kanban plugin displays the board; it does not grant the chat agent +permission to manage tasks. Enable the `kanban` toolset for the profile and +platform that should orchestrate work: + +```bash +hermes -p planner tools enable kanban # CLI / TUI / Desktop chats +hermes -p planner tools enable kanban --platform telegram # a gateway platform +``` + +Each platform has its own selection under `platform_toolsets.` in +`config.yaml`; the toolset is also a checkbox in `hermes tools` and the dashboard. +A gateway agent that has it can `kanban_create` from a chat and is auto-subscribed +to that task's completion/block notifications in the same thread. Start a new chat +after changing this setting; existing conversations retain their tool schemas and +prompt cache. `agent.disabled_toolsets` remains authoritative. Legacy top-level +`toolsets: [kanban]` is honoured as a fallback only when no platform selection was +saved; `all` alone is not a Kanban opt-in. + +Dispatcher-owned workers receive their task lifecycle tools automatically. +`delegate_task` children do not gain permission to mutate the board. + ## How workers interact with the board **Workers do not shell out to `hermes kanban`.** When the dispatcher spawns a worker it sets `HERMES_KANBAN_TASK=t_abcd` in the child's env, and that env var flips on a dedicated **kanban toolset** in the model's schema. The same toolset is also available to orchestrator profiles that enable `kanban` in their toolsets config. These tools read and mutate the board directly via the Python `kanban_db` layer, same as the CLI does. A running worker calls these like any other tool; it never sees or needs the `hermes kanban` CLI. diff --git a/website/docs/user-guide/multi-profile-gateways.md b/website/docs/user-guide/multi-profile-gateways.md index 9825f4d047840..40cc3e5881c38 100644 --- a/website/docs/user-guide/multi-profile-gateways.md +++ b/website/docs/user-guide/multi-profile-gateways.md @@ -115,24 +115,45 @@ moment the flag is off. #### 1. Secondary profiles must not start their own gateway -With a multiplexer running, a named-profile `hermes gateway start` / `run` is a -**hard error**, pointing you back at the multiplexer: +With a multiplexer running, a named-profile `hermes gateway run`, `start`, +`install` or `restart` is a **hard error** (exit code 78), pointing you back at +the multiplexer: ``` The default gateway is running as a profile multiplexer and already serves profile 'coder'. ... ``` +The refusal happens in the CLI before any service manager is touched, so a served +profile never ends up with a permanently failed systemd unit or a launchd respawn +loop. `hermes -p coder gateway stop` refuses the same way (exit 78) when coder has no +gateway of its own — there is nothing to stop but the multiplexer, which +`hermes gateway stop` on the default profile takes down for every served profile. +The dashboard and Desktop app follow the CLI: for a served profile the "Start" and +"Stop" gateway actions answer `409` with the same explanation (rendered as an inline +notice on the System page), and "Restart" restarts the multiplexer (the process that +actually serves the profile) instead of spawning a `-p coder gateway restart` that +could only fail. Because that restart reconnects every bot on the device, both apps +first ask *"Restart the shared gateway? All bots on this device reconnect: default, +coder, research"* (the list is the running gateway's `served_profiles`) and report +*"Shared gateway restarted (3 bots)"* when it completes. A standalone profile keeps +the plain restart. `/api/status?profile=coder` carries the same list as +`gateway_shared_with` (null for a standalone gateway). +"Served" is read from the running gateway's own record (`served_profiles` in the +default home's `gateway_state.json`), so it stays correct when the multiplexer was +enabled only through `GATEWAY_MULTIPLEX_PROFILES` in the default profile's +environment, or when profiles were added after the gateway started. The multiplexer is the single inbound process; a second profile gateway would -double-bind that profile's platforms. Pass `--force` only if you deliberately -want a separate process for that profile (not recommended while the multiplexer -is running). The cross-profile lifecycle wrapper script earlier on this page is -therefore **not** used in multiplex mode — you only manage the default gateway. +double-bind that profile's platforms. Pass `--force` (accepted by `run`, `start`, +`install` and `restart`) only if you deliberately want a separate process for that +profile (not recommended while the multiplexer is running). The cross-profile +lifecycle wrapper script earlier on this page is therefore **not** used in +multiplex mode — you only manage the default gateway. #### 2. HTTP-inbound platforms are reached via a `/p//` URL prefix -Webhook (and other HTTP-inbound) traffic for a secondary profile arrives on the -default listener under a profile prefix, **not** a second port: +HTTP-inbound traffic for a secondary profile arrives on the default profile's +**one** listener under a profile prefix, **not** a second port: ``` # default profile @@ -141,56 +162,73 @@ POST http://host:8644/webhooks/ POST http://host:8644/p/coder/webhooks/ ``` -An unknown or unconfigured profile in the prefix returns `404`. Because the one -shared listener already serves every profile this way, a **secondary profile -must not enable a port-binding platform itself** — doing so is a config error -that skips the entire secondary profile while the default and other healthy -profiles continue. The warning names the skipped profile and every conflicting -platform: - -``` -Skipping secondary profile 'coder' due to port-binding config error: Profile -'coder' enables port-binding platform(s) webhook, but gateway.multiplex_profiles -is on. ... Remove these platform entries from profile 'coder's config.yaml or -configure them only on the default profile. -``` - -Port-binding platforms covered by this rule: `webhook`, `api_server`, -`msgraph_webhook`, `feishu`, `wecom_callback`, `bluebubbles`, `sms`, -`whatsapp_cloud`, `line`, `teams`. Configure any of these **only on the default profile**; -every profile is reachable through its `/p//` prefix. +An unknown or unconfigured profile in the prefix returns `404`. The shared +listener is the default profile's `api_server` port (or its `webhook` port when +no API server is enabled); it serves three kinds of profile-prefixed paths: + +- **`api_server` and `webhook` are mirrored**, never duplicated. `/p/coder/v1/...` + and `/p/coder/webhooks/` are answered by the default profile's own + adapter under coder's scope. A secondary must therefore **not** enable + `api_server` or `webhook` itself (the dashboard refuses with `409`; an + `API_SERVER_KEY` or `WEBHOOK_ENABLED` in the secondary's `.env` wires the + credential without starting a listener). +- **Other port-binding platforms remain standalone-only.** A secondary that + configures an inbound platform without a tested `/p//` ingress is + rejected during multiplex startup; keep that profile on a standalone gateway + or disable the platform there. The adapter declaration alone does not create + shared-listener forwarding. Authentication follows the profile named in the URL. Unprefixed endpoints keep using the default listener's existing credentials. - `/p/coder/...` API-server requests must use `API_SERVER_KEY` from - `~/.hermes/profiles/coder/.env`; the default listener key is rejected. + `~/.hermes/profiles/coder/.env`; the default listener key is rejected. Under + the multiplexer that key only authenticates the prefix — it does not turn on a + second `api_server` listener in the secondary profile, so you do not need to + pin `platforms.api_server.enabled: false` in the secondary's `config.yaml`. - A webhook route that targets `coder` must declare `profile: coder` beside its existing route-specific `secret` in the default profile's `config.yaml`. That secret is then accepted only at `/p/coder/webhooks/` and is rejected on every other profile prefix. - Webhook routes without `profile` remain default-profile routes and are not reachable through a named profile prefix. +- Delivery follows the same binding. A `profile: coder` route's reply (or + `deliver_only` message) goes out through **coder's** adapter for the + `deliver` platform, falls back to **coder's** home channel when + `deliver_extra.chat_id` is unset, and a `github_comment` delivery runs `gh` + with `GH_TOKEN` / `GITHUB_TOKEN` from `profiles/coder/.env`. If coder has no + adapter for that platform the delivery fails (502) rather than posting as + another profile's bot; a default route likewise never borrows a platform that + is enabled only on a secondary profile. +- `/p/coder/api/platforms//events` callbacks are verified and + dispatched by coder's adapter; when coder has none the callback is a 503. -Keep port-binding platforms disabled in secondary profile configs. The shared -listener and its route definitions stay on the default profile; profile -binding controls which profile each authenticated webhook route may execute. Named API requests fail closed when the target profile has no -`API_SERVER_KEY`. +`API_SERVER_KEY`. Security configuration errors remain fatal: for example, an +`open` own-policy platform without `GATEWAY_ALLOW_ALL_USERS` or its +platform-specific allow-all opt-in still aborts gateway startup rather than +silently dropping the unsafe profile. + +#### Inbound-port platforms under the multiplexer -Only this shared-listener conflict degrades to a skipped profile. Security -configuration errors remain fatal: for example, an `open` own-policy platform -without `GATEWAY_ALLOW_ALL_USERS` or its platform-specific allow-all opt-in -still aborts gateway startup rather than silently dropping the unsafe profile. +Only `api_server` and `webhook` currently have a verified shared-listener +implementation. Other inbound-port adapters may declare future compatibility, +but until their forwarding and profile-scoped verification is implemented they +remain standalone-only and are blocked by the multiplexer preflight. #### 3. Per-credential platforms still need their own token per profile Polling/connection platforms (Telegram, Discord, Slack, Matrix, Signal, …) work fine multiplexed, but each profile that enables one must supply its **own** bot token — the same token cannot be polled by two profiles at once. If two profiles -configure the same `(platform, token)`, startup fails fast naming both profiles -(see [Token-conflict safety](#token-conflict-safety) — the rule is unchanged, -it's just enforced inside the one process now). +configure the same `(platform, token)`, the gateway logs an error naming both +profiles and parks the **duplicate** adapter (it shows as `fatal / +duplicate_credential` in runtime status) while the first claimant and every +other profile keep running — the gateway itself does not exit. The default +profile's adapters connect first and claim their credentials, so the parked +adapter is always the secondary's (see +[Token-conflict safety](#token-conflict-safety) — the rule is unchanged, it's +just enforced inside the one process now). #### 4. Session keys are namespaced by profile @@ -198,55 +236,169 @@ Each profile's sessions live under an `agent::…` namespace so two profiles on the same platform/chat never collide in the shared session store. The **default** profile keeps the historical `agent:main:…` namespace byte-for-byte, so existing default-profile sessions are unaffected — no -migration, no orphaned history. +migration, no orphaned history. Every gateway path that reads a key back — +delegation completions after a restart, shutdown notices, a per-user-thread +`/stop` of a sibling's run, `/undo`, QQ approval buttons — accepts the +`agent::…` shape too, so secondary profiles get the same behaviour +as the default one. + +Each profile's rows land in **its own** `state.db`: a named profile's under +`profiles//state.db`, the default profile's under the launch home — even +when the write happens inside another profile's routed turn or background tick. +The Desktop/TUI backend's own store is likewise pinned to the home it launched +under, and a Bot Chat's side agents (`prompt.background`) persist next to their +parent conversation. #### 5. One PID/lock and one status surface There is a single process-level PID and lock (the multiplexer, under the default -home). `hermes status` reports the multiplexer and the profiles it serves; -`hermes status -p ` slices to one profile. Each profile still writes its -own `runtime_status.json` under its own home, so existing per-profile readers -keep working. +home). `hermes status` on the default profile reports the multiplexer and lists +the profiles it serves (`Serves: coder, research`); `hermes -p coder status`, +`hermes -p coder gateway status` and `hermes -p coder cron status` all report +"running via the default-profile multiplexer" instead of "stopped", and the +dashboard's `/api/status?profile=coder` / Channels page report the multiplexer as +coder's running gateway (with coder's own adapters as its platforms). The single +`gateway_state.json` lives under the default home: secondary adapters appear +there as `:` entries beside `served_profiles`; nothing is +written under a secondary profile's home. #### What does **not** change Per-profile `.env` credential isolation is preserved and, if anything, stricter: a profile's keys are resolved from its own scope and are never unioned -into a shared environment (this also means subprocesses like MCP servers and -Kanban workers only ever see their own profile's secrets). Terminal settings +into a shared environment. Subprocesses like MCP servers and Kanban workers only +ever see their own profile's secrets — including credentials injected by an +external secret source (1Password, Bitwarden, …): a stdio MCP server started for +profile B receives B's value for such a name, or nothing if B has none, never the +default profile's. MCP servers are connected **per profile**: two profiles that +both name a server `github` with their own token get two connections and each +sees only its own tools; profiles whose `mcp_servers` entry is identical (same +route *and* credentials) share one connection, and an owner's `/reload-mcp` +re-registers the sharing profiles' tools without them reloading. Terminal settings (`terminal.backend`, `terminal.cwd`, `terminal.docker_volumes`, `terminal.docker_shared_container_key`, SSH targets, …) are likewise resolved per profile on every routed turn: a profile that omits a terminal key gets the documented default, never the launch profile's value, and a profile whose `config.yaml`/`.env` cannot be parsed has terminal execution refused rather than -run under another profile's sandbox policy. Kanban, -profile-scoped skills/memory/SOUL, and model routing all behave per-profile -exactly as they do with separate gateways. - -### Serving selected profiles - -By default, `gateway.multiplex_profiles: true` serves every valid named profile -on the host. To keep unrelated profiles installed without starting their -adapters or cron jobs, set `gateway.multiplex_profile_allowlist`: - -```yaml -gateway: - multiplex_profiles: true - multiplex_profile_allowlist: - - worker - - guest -``` - -The default profile is always served and does not need to be listed. An unset -allowlist preserves the historical serve-all behavior; an empty list serves -only the default profile. Names are normalized and deduplicated. Invalid list -entries or names that are not installed are skipped with a warning. A malformed -non-list value fails safely to default-only. - -The resulting served set also controls `/p//` API and webhook prefixes, -runtime status, profile-route eligibility, and which profiles the in-process -cron scheduler ticks. A named profile outside the allowlist may still run its -own standalone gateway. +run under another profile's sandbox policy. The media-delivery credential +guard (the denylist behind `MEDIA:` attachments — `.env`, `auth.json`, +`config.yaml`, `state.db`, session transcripts, OAuth token stores) covers every +profile under `profiles/`, so no profile's turn can attach another profile's +secrets or chat history to a reply. Authorization is per profile too: +`GATEWAY_ALLOW_ALL_USERS`, `GATEWAY_ALLOWED_USERS` and every platform allowlist +or allow-all opt-in are read from the owning profile's `.env` — the default +profile opting into open access never opens a secondary profile's bot, and a +secondary that opts in only in its own `.env` is honored. The same holds for +per-bot behaviour written in a profile's `config.yaml` (`require_mention`, +`mention_patterns`, `allow_bots`, `reactions`, `auto_thread`, `dm_policy`, +`ignored_channels`, Matrix `session_scope`, …): a secondary profile's YAML never +lands in the shared process environment, so it cannot become the default +profile's policy, and the default profile's YAML never governs a secondary +bot. The `terminal.env_passthrough` allowlist, the Yuanbao auto-designated +home channel, and the write guards protecting each profile's own `config.yaml` +are resolved per profile as well. Kanban, profile-scoped skills/memory/SOUL, and +model routing all behave per-profile exactly as they do with separate gateways. + +Outbound identity is per profile too. A turn running for profile `P` that calls +the `send_message` tool (send, react, media) posts through `P`'s own bot; +so do the "Gateway shutting down/restarted" and `/update` notices for `P`'s +sessions, `/loop` wakeups set from `P`'s chats, and the Discord +unauthorized-slash operator alert of `P`'s Discord bot (to `P`'s home +channel). If `P` has no connected bot for that platform the send fails with a +clear error — it never falls back to the default profile's bot. + +Tool and memory-provider credentials follow the same rule. Hosted OCR +(`FIRECRAWL_API_KEY`), Modal / Browser Use cloud gates, the mem0 OSS OpenAI +key, xAI video, and every memory-provider identity (`MEM0_USER_ID`, +`SUPERMEMORY_CONTAINER_TAG`, `RETAINDB_PROJECT`, `OPENVIKING_ACCOUNT/USER`, +`HINDSIGHT_BANK_ID`, `HERMES_HONCHO_HOST`) are read from the routed profile's +`.env`, so a secondary profile's memories land in **its** account/bank/project +(or the provider's per-profile default), never the default profile's. Custom +endpoints travel with their keys — `OPENAI_BASE_URL`, `XAI_BASE_URL`, +`NOUS_INFERENCE_BASE_URL`, `GATEWAY_PROXY_URL`, Firecrawl / Browserbase / +RetainDB / Supermemory / Honcho / Hindsight URLs — so a profile's key is never +sent to another profile's proxy or self-hosted server. `WEIXIN_HOME_CHANNEL`, +`HERMES_LANGUAGE` and `display.language`, and `hooks.outbound[].secret_env` are +likewise per profile, and end-of-session memory extraction for an evicted +secondary session runs under that profile's scope. + +Per-turn runtime settings follow the routed profile as well: `agent.max_turns`, +`fallback_providers`, `file_read_max_chars`, `tool_output.*`, `browser.*` +timeouts, `timezone` (including the `TZ` handed to `execute_code` sandboxes), +the media-delivery policy (`gateway.strict`, `media_delivery_allow_dirs`, +`trust_recent_files*`) and the Nous `auth.json` used for auxiliary calls are all +read from the profile serving the turn, never from the profile the gateway was +launched under. The same holds for per-profile state files (`processes.json`, +`checkpoints/`, sandbox snapshot stores, Feishu comment rules/pairing) and for +gateway hooks: each profile's `hooks/` directory is loaded on its own and fires +only for that profile's events. Shell hooks run with the routed profile's +`HERMES_HOME`, without the default profile's secrets in their environment, and +their stdin payload carries a `profile` field naming the profile that fired them. + +#### What is isolated per profile + +A quick reference for what a multiplexed turn resolves from **its own** +profile and never shares with the default or any sibling: + +| Concern | Resolved from | Behaviour when the profile lacks it | +|---|---|---| +| Provider keys, bot tokens, `${VAR}` refs in `config.yaml` | The profile's own `.env` (its secret scope) | Unresolved / no adapter — never the default profile's value | +| Authorization (`GATEWAY_ALLOW_ALL_USERS`, `GATEWAY_ALLOWED_USERS`, per-platform allowlists and allow-all opt-ins) | The owning profile's `.env` and `config.yaml` | Closed — a default-profile opt-in never opens a secondary's bot | +| HTTP endpoints (`/p//api/...`, `/p//webhooks/...`, platform event callbacks) | The named profile's `API_SERVER_KEY`, `profile:`-bound webhook routes, and its own adapter | `401`/`404`; delivery without an adapter is `502`/`503`, never another profile's bot | +| Inbound-port platforms (`/p//webhooks/twilio`, `/p//line/webhook`, `/p//api/messages`, …) | The named profile's own adapter and its secret (Twilio auth token, LINE channel secret, Teams app, BlueBubbles password, …); replies leave through that adapter | `401`/`403` on a wrong secret, `404` when the profile has no such adapter — never the default profile's adapter | +| Adapter settings (`*_REQUIRE_MENTION`, `*_REACTIONS`, `*_PROXY`, webhook host/port/URL, Matrix thread/session/E2EE policy, Discord backfill/attachment caps, Buzz reply mode, A2A agent card) | The owning profile's `.env` and `config.yaml` | The adapter's documented default — never the default profile's setting | +| `MEDIA:` attachment denylist | Every home under `profiles/` plus the default home, enumerated at check time | A turn can never attach another profile's `.env`, `auth.json`, `state.db`, sessions or token stores | +| stdio MCP child environment | Safe baseline + the profile's scoped values for secret-source names + the server's own `env:` | A name the profile lacks is absent from the child — no default-profile fallthrough | +| Outbound egress (`send_message`, shutdown/restart/`/update` notices, `/loop` wakeups, `profile:`-bound webhook delivery, `github_comment` tokens) | The profile's own connected adapter and `.env` | Clear failure; never posts through the default profile's bot | +| Session namespace | `agent::…` (default keeps `agent:main:…`) | Two profiles on the same chat never share history | +| Logs | `agent.log` / `errors.log` / `gateway.log` under the profile's own home | — | +| Terminal sandbox settings (`terminal.*`, SSH targets) | The profile's `config.yaml` | Documented default; unparsable config → execution refused | +| Working directory of a turn (unset `terminal.cwd`) | Same rule as a standalone gateway: `$HOME` for the local backend, sandbox default otherwise | Never the directory the multiplexer process was launched from | +| Command approvals (`command_allowlist`, "always" choices) | The profile's own `config.yaml` | A default-profile "always" never pre-approves a secondary's command; a secondary's choice is saved to its own config | +| Sandbox credential-file mounts (`terminal.credential_files`), `security.redact_secrets`, `browser.*` engine/headed flags, `lsp.*`, auxiliary-provider health marks, `logs/mcp-stderr.log` | The profile's own `config.yaml` / `.env` | Documented default — never the launch profile's cached value | +| Cloud-SDK credential clients (Bedrock boto3 clients + model discovery, Azure Entra credential), credential-fetched catalogs (DeepInfra, Copilot context limits, Nous reasoning caps, Ramp Router efforts, xAI / OpenRouter image models, custom-endpoint `/models`), Camofox VNC address, computer-use aux-vision routing, skill-sync push, remote-backend probe text, learned image token costs, `display.skin`, guest-mint back-off, banner skills, Yuanbao "active" adapter, Langfuse client | The profile's own `.env` / `config.yaml` / `/cache` | Documented default — never the launch profile's cached value or its credentials | +| Session-search knobs (`sessions.cjk_fts`, `sessions.search_slow_ms`) | The profile's `config.yaml` | Documented default — never the default profile's bridged value | +| Platform proxies (`TELEGRAM_PROXY`, `DISCORD_PROXY`, `HTTPS_PROXY`, …) | The profile's own `.env` | Direct connection — never the default profile's proxy | +| MCP discovery in the Desktop/dashboard backend | Once per served profile home | A profile selected after another has already built an agent still discovers its own `mcp_servers` | +| Dashboard actions (`hermes -p …` spawned by the Desktop/dashboard) | A scrubbed child env pinned to that profile's `HERMES_HOME` | The child loads its own `.env`; the dashboard profile's tokens and ports are not inherited | +| Cron `.env` tuning (`HERMES_CRON_TIMEOUT`, `HERMES_MODEL` fallback, `HERMES_CRON_MAX_PARALLEL`, prefill file), worker / Bot Chat child env | The profile's own `.env`; children never inherit the default profile's `.env` settings or bridged `TERMINAL_*` policy | Cron defaults / model refusal, exactly as a standalone `hermes -p gateway run` | +| Kanban workers and notifications for a profile's tasks | The assignee's `.env` + `config.yaml` (toolset pin, terminal backend, media policy, display language) | — | +| `/loop` ticks, `background_process_notifications` gate, `notice_delivery`, background-process checkpoint recovery | The owning profile's `state.db` / `config.yaml` / `processes.json` | — | + +What is **shared** by design: the process, its PID/lock and `gateway_state.json` +(default home), the one HTTP listener, and the `profile_routes` table (declared +on the default profile). + +### Which profiles are served + +`gateway.multiplex_profiles: true` serves the default profile plus **every** +live named profile under `profiles/` — there is no per-profile opt-out list. +(The former `gateway.multiplex_profile_allowlist` key is retired; a config +migration removes it from `config.yaml`, and a profile you do not want served is +archived or deleted instead — `hermes profile delete `, or move the +directory out of `profiles/`.) Deleted profiles leave a tombstone and are never +enumerated; a profile whose directory is gone is never recreated by a served +turn, the cron ticker or log routing. + +The served set controls `/p//` API and webhook prefixes, runtime +status, profile-route eligibility, and which profiles the in-process cron +scheduler ticks (the Desktop backend's ticker enumerates the same set and stands +down for any profile a running multiplexer or its own gateway already serves). A +multiplexer started as `hermes -p gateway run` always ticks its own +profile's cron store as well. + +The served set is **live**. A profile created while the multiplexer is running +(`hermes profile create`, the dashboard, Desktop or the TUI) is served at once: +the creator pings the multiplexer over its control socket, and the multiplexer +also rescans `profiles/` every 30 seconds as a safety net. The new profile's +adapters are built the moment its `config.yaml`/`.env` carries a bot token +(creators usually create first, then add the token), `served_profiles` in the +default profile's `gateway_state.json` is updated, and `hermes -p gateway +status` reports it as served — no restart, and the other profiles' adapters and +in-flight turns are untouched. Deleting a profile stops and unroutes its +adapters the same way. The one-credential-one-poller rule still applies: a +hot-added profile that reuses another profile's token is parked with a +`duplicate_credential` error, never started as a second poller. ### Routing shared-bot chats to profiles (`profile_routes`) @@ -293,6 +445,29 @@ no route stay on the default/active profile. The routed profile gets the full per-profile isolation described above (config, skills, memory, credentials, session namespace). Routing works on every platform adapter, not just Discord. +A route applies only to messages received by the **default profile's bot** +unless it names another bot with `bot_profile: `. Telegram DMs use the +same `chat_id` for every bot (the user's id), so without this a +`chat_id` route meant for the shared bot would also capture that user's DMs +with a secondary profile's dedicated bot. Messages arriving at a secondary +profile's own bot stay in that profile: + +```yaml + # Pin one user's DM with team_b's OWN bot to a third profile + - name: teamb-owner-dm + platform: telegram + bot_profile: team_b + chat_id: "72719239" + profile: ops-for-team-b +``` + +Authorization for a routed message is always decided by the **receiving bot's +profile** (its token and allowlist), including follow-ups sent while the agent +is busy and mid-turn checks such as `/topic` or `/stop`; the routed profile +itself needs no copy of the allowlist. A routed profile without a bot of its +own also receives background notifications (process completions, heartbeats, +async delegation results) through the shared bot after a gateway restart. + On WhatsApp and WhatsApp Cloud, a `chat_id` route matches across user-identity forms: a bare phone number (`15551234567`), a JID (`15551234567@s.whatsapp.net`), and a LID (`…@lid`) all refer to the same @@ -307,16 +482,18 @@ ids are unchanged. `profile_routes` requires `gateway.multiplex_profiles: true`; with multiplexing off the routes are ignored. If an explicit route matches but its -target profile is not installed or is outside `multiplex_profile_allowlist`, -the gateway rejects that ingress and logs the route and target. It does not run +target profile is not installed (or was deleted), the gateway rejects that ingress and logs the route and target. It does not run the default profile. Traffic that matches no route keeps the historical default-profile behavior. Cron jobs owned by a routed profile deliver through the shared bot too, but only to targets an enabled route with a `chat_id`/`thread_id` maps to that -profile — a routed profile's job targeting an unrouted chat (or a chat routed -to another profile) is never sent through the shared bot. Guild-only routes do -not qualify a cron target; add a `chat_id` route for the delivery channel. +profile (a `guild_id + chat_id` route qualifies its channel) — a routed +profile's job targeting an unrouted chat (or a chat routed to another profile) +is never sent through the shared bot. Guild-only routes do not qualify a cron +target; add a `chat_id` route for the delivery channel. The routed profile does +not need its own `platforms.` block for this: the shared bot's +authorization comes from the route, not from the satellite's config. ## Start, stop, or restart all gateways at once @@ -532,7 +709,10 @@ and reboots. Each profile must use unique bot tokens for each platform. If two profiles share a Telegram, Discord, Slack, WhatsApp, or Signal token, the second -gateway refuses to start with an error naming the conflicting profile. +gateway refuses to start with an error naming the conflicting profile. Under +[multiplexing](#alternative-one-gateway-for-all-profiles-multiplexing) the same +rule parks only the duplicate profile's adapter and the shared gateway keeps +running. To audit: @@ -541,6 +721,118 @@ grep -H 'TELEGRAM_BOT_TOKEN\|DISCORD_BOT_TOKEN' \ ~/.hermes/.env ~/.hermes/profiles/*/.env ``` +## Migrating from per-profile gateways + +If your profiles each run their own gateway today (one systemd unit or launchd +agent per profile), you can fold them into a single multiplexed default gateway +with one command — and roll back with another. Standalone per-profile gateways +remain fully supported; this is an optional migration, not a removal. + +```bash +hermes gateway migrate --multiplex --dry-run # print the plan and any blockers; changes nothing +hermes gateway migrate --multiplex # apply (asks for confirmation on a TTY; -y skips) +hermes gateway migrate --standalone # roll back to per-profile gateways +``` + +### What `hermes update` does + +After a successful update, when the install has two or more profiles, at least +one secondary profile runs its own gateway (a live process or an installed +service) and `gateway.multiplex_profiles` is off, `hermes update` runs the same +preflight: + +- **Nothing blocks it** → the migration runs automatically (the same code path + as `hermes gateway migrate --multiplex --yes`) and prints what it did. This + is deterministic and never prompts, so it also runs on headless/cron updates. +- **Something blocks it** → a warning block lists each blocker with its exact + fix and the one-liner to run later. Nothing is changed. + +Single-profile installs are never migrated (there is nothing to gain), and an +install that is already multiplexing is left alone. `hermes update` also does +nothing when no secondary profile runs its own gateway — it never flips modes +on an install where nothing was running. + +The explicit command is different: `hermes gateway migrate --multiplex` with +two or more profiles and **no** standalone secondary gateway still applies the +one remaining step — it sets `gateway.multiplex_profiles: true`, (re)starts the +default gateway and writes the same rollback manifest (with an empty +`secondaries` list), so `--standalone` undoes it. You asked for multiplex; you +get multiplex. + +:::tip Clones do not carry channels +`hermes profile create --clone` leaves the source's bot tokens and allowlists +behind (see [Profiles → messaging channels are never cloned](./profiles.md#messaging-channels-are-never-cloned---clone-channels-to-opt-in)), +so a fleet of clones no longer trips the duplicate-credential blocker below. +Older clones that still carry them are flagged by `hermes profile list`. +::: + +### What the migration does + +1. Stops each secondary profile's standalone gateway and uninstalls its + service (systemd user/system unit or launchd agent). What was removed is + recorded in `~/.hermes/gateway_migration.json` for rollback. +2. Sets `gateway.multiplex_profiles: true` in the **default** profile's + `config.yaml`. +3. Restarts the default gateway — or installs and starts it on the same service + manager the secondaries were using, so a systemd-managed fleet stays + systemd-managed. +4. Waits for the default gateway to record `served_profiles` covering every + profile, then prints a summary. + +### Blockers and fixes + +| Blocker | Why | Fix | +|---|---|---| +| Two profiles configure the same platform credential (e.g. the same `TELEGRAM_BOT_TOKEN`) | Under one process a bot token can only be polled once; the multiplexer would park the duplicate and that profile's bot would go silent | Remove the token from the second profile, or keep it in `default` and route that profile's chats with [`profile_routes`](#routing-shared-bot-chats-to-profiles-profile_routes) | +| A secondary profile enables a port-binding platform that has **no** verified `/p//` ingress on the default listener | The multiplexer skips that whole profile (see [rule 2](#2-http-inbound-platforms-are-reached-via-a-pprofile-url-prefix)) | Disable the platform in that profile (`platforms..enabled: false`), or keep the profile on a standalone gateway with `hermes -p gateway start --force` | + +The credential check reuses the gateway's own conflict detection, so its verdict +matches what the multiplexer does at startup. Which port-binding platforms have +a `/p//` ingress is read from the adapters themselves (each declares +`serves_profile_prefix`), so the preflight stays correct as new HTTP-inbound +adapters gain the prefix. + +### What changes for inbound-port profiles + +A secondary profile that used `api_server` or `webhook` on its own port is +**not** blocked — but its URL changes. The preflight prints the exact new URL, +for example: + +``` +Profile 'coder': api_server moves onto the default listener at +http://127.0.0.1:8642/p/coder/v1/... (its key/secret is unchanged; update +clients that call the old per-profile port). +``` + +The profile's own `API_SERVER_KEY` / webhook secret keeps authenticating the +prefixed URL; nothing else about the key changes. + +### Profiles created after the migration + +A profile created while the multiplexer runs is served without a restart (see +above). `hermes profile create` confirms this when the live multiplexer picked the +profile up; it prints the `hermes gateway restart` reminder only when it could not +reach the multiplexer (for example, a gateway started from an older build). + +### Rollback + +```bash +hermes gateway migrate --standalone +``` + +reads `gateway_migration.json`, sets `gateway.multiplex_profiles` back to its +previous value, restarts the default gateway, and reinstalls/starts every +recorded per-profile service. The manifest is removed once everything is back. +If no manifest exists (you enabled multiplexing by hand), leave multiplex mode +with `hermes config set gateway.multiplex_profiles false && hermes gateway restart` +and reinstall the per-profile services you want. + +Not covered automatically: s6-supervised containers (set the flag on the +default profile and restart the container) and Windows Scheduled Tasks (set the +flag, stop the per-profile tasks, `hermes gateway restart`). The dashboard's +System page offers the same migration as a button when the preflight finds an +eligible install. + ## Updating the code `hermes update` pulls the latest code once and syncs new bundled skills into @@ -551,6 +843,11 @@ hermes update hermes-gateways restart ``` +Running gateways are restarted by the update itself; on an install that still +runs one gateway per profile, the update then offers the +[migration to a single multiplexed gateway](#migrating-from-per-profile-gateways) +— automatically when nothing blocks it, otherwise as a warning with the fixes. + User-modified skills are never overwritten. ## Troubleshooting diff --git a/website/docs/user-guide/profiles.md b/website/docs/user-guide/profiles.md index ca3349defd7f2..0119babd0c485 100644 --- a/website/docs/user-guide/profiles.md +++ b/website/docs/user-guide/profiles.md @@ -80,6 +80,34 @@ hermes profile create work --clone-from coder hermes profile create work-backup --clone-from coder --clone-all ``` +### Messaging channels are never cloned (`--clone-channels` to opt in) + +Every clone — `--clone`, `--clone-from`, `--clone-all`, and the dashboard / Desktop / TUI +"clone from profile" option — copies the source **without its messaging channels**: bot tokens +and allowlists (`TELEGRAM_BOT_TOKEN`, `DISCORD_ALLOWED_USERS`, `WHATSAPP_ENABLED`, +`API_SERVER_KEY`, `WEBHOOK_SECRET`, …), the `platforms:` / `telegram:` / `discord:` sections of +`config.yaml`, `gateway.multiplex_profiles` / `profile_routes`, and (for `--clone-all`) the +pairing store, WhatsApp session and other per-bot state. Provider and tool API keys, the model +block, memory settings, skills and `SOUL.md` are copied as before. The command prints which +platforms were left behind. + +The reason is that a bot can only belong to one profile: two standalone gateways holding the +same token fight over its long-poll, and a [multiplexed gateway](./multi-profile-gateways.md) +parks the duplicate adapter (and `hermes gateway migrate --multiplex` refuses with one +duplicate-credential blocker per platform). Configure the new profile's own bots with +`hermes -p setup` or the dashboard Messaging page. + +```bash +hermes profile create twin --clone --clone-channels # keep the source's bots anyway +``` + +`--clone-channels` is refused when a running multiplexed gateway already serves the source +(the copy would be parked immediately) and otherwise prints a warning naming the platforms +now shared with the source. `hermes profile list` prints the same warning for any existing +profile whose bot credential is byte-identical to the default's, so older clones surface +before they bite. The key set is derived from the platform adapters themselves (registry +entries and the gateway's env table), so a newly added platform is covered automatically. + :::tip Honcho memory + profiles When Honcho is enabled, clone operations automatically create a dedicated AI peer for the new profile while sharing the same user workspace. Each profile builds its own observations and identity. See [Honcho -- Multi-agent / Profiles](./features/memory-providers.md#honcho) for details. :::