Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions acp_adapter/session.py
Original file line number Diff line number Diff line change
Expand Up @@ -649,6 +649,12 @@ def _make_agent(
kwargs.update(
{
"provider": runtime.get("provider"),
"requested_provider": (
runtime.get("requested_provider")
or requested_provider
or config_provider
or ""
),
"api_mode": api_mode or runtime.get("api_mode"),
"base_url": base_url or runtime.get("base_url"),
"api_key": runtime.get("api_key"),
Expand Down
6 changes: 5 additions & 1 deletion agent/agent_init.py
Original file line number Diff line number Diff line change
Expand Up @@ -1156,7 +1156,11 @@ def init_agent(
# every client construction path below (Anthropic native, OpenAI-wire,
# router-based implicit auth) can apply it consistently. Bedrock
# Claude uses its own timeout path and is not covered here.
_provider_timeout = get_provider_request_timeout(agent.provider, agent.model)
_provider_timeout = get_provider_request_timeout(
agent.provider,
agent.model,
requested_provider_id=getattr(agent, "requested_provider", None),
)

if agent.api_mode == "anthropic_messages":
from agent.anthropic_adapter import build_anthropic_client, resolve_anthropic_token
Expand Down
24 changes: 20 additions & 4 deletions agent/agent_runtime_helpers.py
Original file line number Diff line number Diff line change
Expand Up @@ -1493,7 +1493,11 @@ def try_recover_primary_transport(
agent._anthropic_base_url = rt["anthropic_base_url"]
agent._anthropic_client = build_anthropic_client(
rt["anthropic_api_key"], rt["anthropic_base_url"],
timeout=get_provider_request_timeout(agent.provider, agent.model),
timeout=get_provider_request_timeout(
agent.provider,
agent.model,
requested_provider_id=getattr(agent, "requested_provider", None),
),
)
agent._is_anthropic_oauth = rt["is_anthropic_oauth"]
agent.client = None
Expand Down Expand Up @@ -1781,7 +1785,11 @@ def restore_primary_runtime(agent) -> bool:
agent._anthropic_base_url = rt["anthropic_base_url"]
agent._anthropic_client = build_anthropic_client(
rt["anthropic_api_key"], rt["anthropic_base_url"],
timeout=get_provider_request_timeout(agent.provider, agent.model),
timeout=get_provider_request_timeout(
agent.provider,
agent.model,
requested_provider_id=getattr(agent, "requested_provider", None),
),
)
agent._is_anthropic_oauth = rt["is_anthropic_oauth"]
agent.client = None
Expand Down Expand Up @@ -3060,7 +3068,11 @@ def _restore_snapshot() -> None:
agent._anthropic_base_url = base_url or getattr(agent, "_anthropic_base_url", None)
agent._anthropic_client = build_anthropic_client(
effective_key, agent._anthropic_base_url,
timeout=get_provider_request_timeout(agent.provider, agent.model),
timeout=get_provider_request_timeout(
agent.provider,
agent.model,
requested_provider_id=getattr(agent, "requested_provider", None),
),
)
agent._is_anthropic_oauth = _is_oauth_token(effective_key) if (_is_native_anthropic and isinstance(effective_key, str)) else False
agent.client = None
Expand Down Expand Up @@ -3090,7 +3102,11 @@ def _restore_snapshot() -> None:
)
except Exception:
logger.debug("custom-provider TLS resolution skipped on switch_model", exc_info=True)
_sm_timeout = get_provider_request_timeout(agent.provider, agent.model)
_sm_timeout = get_provider_request_timeout(
agent.provider,
agent.model,
requested_provider_id=getattr(agent, "requested_provider", None),
)
if _sm_timeout is not None:
agent._client_kwargs["timeout"] = _sm_timeout
# Reapply provider-specific headers (e.g. OpenRouter HTTP-Referer,
Expand Down
8 changes: 8 additions & 0 deletions agent/background_review.py
Original file line number Diff line number Diff line change
Expand Up @@ -336,6 +336,7 @@ def _resolve_review_runtime(
parent_api_mode = "codex_responses"
parent = {
"provider": agent.provider,
"requested_provider": getattr(agent, "requested_provider", agent.provider),
"model": agent.model,
"api_key": parent_runtime.get("api_key") or None,
"base_url": parent_runtime.get("base_url") or None,
Expand Down Expand Up @@ -366,6 +367,7 @@ def _resolve_review_runtime(
)
return {
"provider": rp.get("provider") or task_provider,
"requested_provider": rp.get("requested_provider") or task_provider,
"model": rp.get("model") or task_model,
"api_key": rp.get("api_key"),
"base_url": rp.get("base_url"),
Expand Down Expand Up @@ -1223,6 +1225,12 @@ def build_cache_parity_fork(
quiet_mode=True,
platform=agent.platform,
provider=_rt.get("provider") or agent.provider,
requested_provider=(
_rt.get("requested_provider")
or _rt.get("provider")
or agent.provider
or ""
),
api_mode=_rt.get("api_mode"),
base_url=_rt.get("base_url") or None,
api_key=_rt.get("api_key") or None,
Expand Down
18 changes: 15 additions & 3 deletions agent/chat_completion_helpers.py
Original file line number Diff line number Diff line change
Expand Up @@ -824,7 +824,11 @@ def _derive_stream_stale_timeout(agent, api_kwargs: dict) -> float:
watchdog shares the exact same patience budget as the OpenAI/Anthropic
stale-stream detector below.
"""
_cfg_stale = get_provider_stale_timeout(agent.provider, agent.model)
_cfg_stale = get_provider_stale_timeout(
agent.provider,
agent.model,
requested_provider_id=getattr(agent, "requested_provider", None),
)
if _cfg_stale is not None:
_base = _cfg_stale
else:
Expand Down Expand Up @@ -3902,7 +3906,11 @@ def _call_chat_completions(stream_attempt_id: int):
import httpx as _httpx
# Per-provider / per-model request_timeout_seconds (from config.yaml)
# wins over the HERMES_API_TIMEOUT env default if the user set it.
_provider_timeout_cfg = get_provider_request_timeout(agent.provider, agent.model)
_provider_timeout_cfg = get_provider_request_timeout(
agent.provider,
agent.model,
requested_provider_id=getattr(agent, "requested_provider", None),
)
_base_timeout = (
_provider_timeout_cfg
if _provider_timeout_cfg is not None
Expand Down Expand Up @@ -5093,7 +5101,11 @@ def _call():
)

# Provider-configured stale timeout takes priority over env default.
_cfg_stale = get_provider_stale_timeout(agent.provider, agent.model)
_cfg_stale = get_provider_stale_timeout(
agent.provider,
agent.model,
requested_provider_id=getattr(agent, "requested_provider", None),
)
if _cfg_stale is not None:
_stream_stale_timeout_base = _cfg_stale
else:
Expand Down
3 changes: 3 additions & 0 deletions agent/curator.py
Original file line number Diff line number Diff line change
Expand Up @@ -1890,6 +1890,7 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
_base_url = None
_api_mode = None
_resolved_provider = None
_requested_provider = None
_credential_pool = None
_request_overrides: Dict[str, Any] = {}
_max_tokens = None
Expand All @@ -1912,6 +1913,7 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
_base_url = _rp.get("base_url")
_api_mode = _rp.get("api_mode")
_resolved_provider = _rp.get("provider") or _provider
_requested_provider = _rp.get("requested_provider") or _provider
_credential_pool = _rp.get("credential_pool")
_request_overrides = _merge_request_overrides(
_rp.get("request_overrides"),
Expand Down Expand Up @@ -1939,6 +1941,7 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
review_agent = AIAgent(
model=_model_name,
provider=_resolved_provider,
requested_provider=_requested_provider or _resolved_provider or "",
api_key=_api_key,
base_url=_base_url,
api_mode=_api_mode,
Expand Down
3 changes: 3 additions & 0 deletions gateway/platforms/api_server.py
Original file line number Diff line number Diff line change
Expand Up @@ -367,6 +367,7 @@ def _coerce_request_bool(value: Any, default: bool = False) -> bool:
"api_key",
"base_url",
"provider",
"requested_provider",
"api_mode",
"command",
"args",
Expand Down Expand Up @@ -478,6 +479,7 @@ def _resolve_request_runtime_agent_kwargs(provider: str, target_model: Optional[
"api_key": runtime.get("api_key"),
"base_url": runtime.get("base_url"),
"provider": runtime.get("provider"),
"requested_provider": runtime.get("requested_provider") or provider,
"api_mode": runtime.get("api_mode"),
"command": runtime.get("command"),
"args": list(runtime.get("args") or []),
Expand Down Expand Up @@ -3024,6 +3026,7 @@ def _resolve_provider_runtime(
_apply_runtime_agent_overrides(runtime_kwargs, provider_runtime)
elif effective_provider and effective_provider != current_provider:
runtime_kwargs["provider"] = effective_provider
runtime_kwargs["requested_provider"] = effective_provider
model = effective_model
# Per-route explicit transport secrets/base URLs win within the
# route contract after provider resolution.
Expand Down
5 changes: 5 additions & 0 deletions hermes_cli/cli_commands_mixin.py
Original file line number Diff line number Diff line change
Expand Up @@ -2305,6 +2305,11 @@ def run_background():
api_key=turn_route["runtime"].get("api_key"),
base_url=turn_route["runtime"].get("base_url"),
provider=turn_route["runtime"].get("provider"),
requested_provider=(
turn_route["runtime"].get("requested_provider")
or turn_route["runtime"].get("provider")
or ""
),
api_mode=turn_route["runtime"].get("api_mode"),
acp_command=turn_route["runtime"].get("command"),
acp_args=turn_route["runtime"].get("args"),
Expand Down
126 changes: 92 additions & 34 deletions hermes_cli/timeouts.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,62 +11,120 @@ def _coerce_timeout(raw: object) -> float | None:
return timeout


def get_provider_request_timeout(
provider_id: str, model: str | None = None
def _provider_timeout_candidates(
provider_id: str,
requested_provider_id: str | None,
) -> tuple[str, ...]:
"""Return config provider IDs from most to least route-specific."""
runtime_id = (provider_id or "").strip().lower()
requested_id = (requested_provider_id or "").strip().lower()
candidates: list[str] = []

# Named custom routes intentionally canonicalize to ``custom`` at runtime.
# Preserve the requested identity for config lookup, while retaining bare
# ``providers.custom`` as a backward-compatible fallback.
if runtime_id == "custom" and requested_id and requested_id != "custom":
candidates.append(requested_id)
if requested_id.startswith("custom:"):
candidates.append(requested_id.removeprefix("custom:"))
else:
candidates.append(f"custom:{requested_id}")

candidates.append(runtime_id)
return tuple(dict.fromkeys(candidate for candidate in candidates if candidate))


def _provider_config_for_candidate(
providers: dict[object, object], candidate: str
) -> object:
"""Resolve a provider block using runtime custom-provider aliases."""
provider_config = providers.get(candidate)
if provider_config is not None:
return provider_config

for key, value in providers.items():
route_ids: set[str] = set()
for raw_name in (key, value.get("name") if isinstance(value, dict) else None):
if raw_name is None:
continue
raw_id = str(raw_name).strip().lower()
normalized_id = raw_id.replace(" ", "-")
route_ids.update((raw_id, normalized_id, f"custom:{normalized_id}"))
if candidate in route_ids:
return value
return None


def _get_provider_timeout(
provider_id: str,
model: str | None,
*,
requested_provider_id: str | None,
model_field: str,
provider_field: str,
) -> float | None:
"""Return a configured provider request timeout in seconds, if any."""
if not provider_id:
return None

try:
from hermes_cli.config import load_config_readonly

config = load_config_readonly()
except Exception:
return None

providers = config.get("providers", {}) if isinstance(config, dict) else {}
provider_config = (
providers.get(provider_id, {}) if isinstance(providers, dict) else {}
)
if not isinstance(provider_config, dict):
if not isinstance(providers, dict):
return None

model_config = _get_model_config(provider_config, model)
if model_config is not None:
timeout = _coerce_timeout(model_config.get("timeout_seconds"))
for candidate in _provider_timeout_candidates(provider_id, requested_provider_id):
provider_config = _provider_config_for_candidate(providers, candidate)
if not isinstance(provider_config, dict):
continue

model_config = _get_model_config(provider_config, model)
if model_config is not None:
timeout = _coerce_timeout(model_config.get(model_field))
if timeout is not None:
return timeout

timeout = _coerce_timeout(provider_config.get(provider_field))
if timeout is not None:
return timeout

return _coerce_timeout(provider_config.get("request_timeout_seconds"))
return None


def get_provider_stale_timeout(
provider_id: str, model: str | None = None
def get_provider_request_timeout(
provider_id: str,
model: str | None = None,
*,
requested_provider_id: str | None = None,
) -> float | None:
"""Return a configured non-stream stale timeout in seconds, if any."""
if not provider_id:
return None

try:
from hermes_cli.config import load_config_readonly
config = load_config_readonly()
except Exception:
return None

providers = config.get("providers", {}) if isinstance(config, dict) else {}
provider_config = (
providers.get(provider_id, {}) if isinstance(providers, dict) else {}
"""Return a configured provider request timeout in seconds, if any."""
return _get_provider_timeout(
provider_id,
model,
requested_provider_id=requested_provider_id,
model_field="timeout_seconds",
provider_field="request_timeout_seconds",
)
if not isinstance(provider_config, dict):
return None

model_config = _get_model_config(provider_config, model)
if model_config is not None:
timeout = _coerce_timeout(model_config.get("stale_timeout_seconds"))
if timeout is not None:
return timeout

return _coerce_timeout(provider_config.get("stale_timeout_seconds"))
def get_provider_stale_timeout(
provider_id: str,
model: str | None = None,
*,
requested_provider_id: str | None = None,
) -> float | None:
"""Return a configured non-stream stale timeout in seconds, if any."""
return _get_provider_timeout(
provider_id,
model,
requested_provider_id=requested_provider_id,
model_field="stale_timeout_seconds",
provider_field="stale_timeout_seconds",
)


def _get_model_config(
Expand Down
5 changes: 5 additions & 0 deletions plugins/platforms/feishu/feishu_comment.py
Original file line number Diff line number Diff line change
Expand Up @@ -1076,6 +1076,11 @@ def _run_comment_agent(prompt: str, client: Any, session_key: str = "") -> str:
base_url=runtime_kwargs.get("base_url"),
api_key=runtime_kwargs.get("api_key"),
provider=runtime_kwargs.get("provider"),
requested_provider=(
runtime_kwargs.get("requested_provider")
or runtime_kwargs.get("provider")
or ""
),
api_mode=runtime_kwargs.get("api_mode"),
credential_pool=runtime_kwargs.get("credential_pool"),
quiet_mode=True,
Expand Down
Loading