diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 884866dc1173a..5ca7e462a7208 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -2400,6 +2400,18 @@ def extract_api_error_context(error: Exception) -> Dict[str, Any]: if value not in {None, ""}: context["reset_at"] = value break + # Numeric relative reset (e.g. Anthropic/codex usage_limit_reached body + # carries "resets_in_seconds": N rather than an absolute timestamp). + if "reset_at" not in context: + for key in ("resets_in_seconds", "reset_in_seconds"): + resets_in = payload.get(key) + if resets_in not in {None, ""}: + try: + context["reset_at"] = time.time() + float(resets_in) + except (TypeError, ValueError): + pass + else: + break retry_after = payload.get("retry_after") if retry_after not in {None, ""} and "reset_at" not in context: try: diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 099cefd36e15c..437b64a9e93ff 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -430,6 +430,66 @@ def _get_continuation_prompt(is_partial_stub: bool, dropped_tools: Optional[List ) +def _format_reset_delta(reset_at: Any) -> Optional[str]: + """Render a human "~Nh Nm" delta from a reset timestamp, or None. + + ``reset_at`` may be an absolute epoch float/int, a numeric string, or an + ISO-8601 string (e.g. "2026-04-12T10:30:00Z"). Returns None when the value + is missing, unparseable, or already in the past. + """ + if reset_at in {None, ""}: + return None + epoch: Optional[float] = None + if isinstance(reset_at, (int, float)): + epoch = float(reset_at) + elif isinstance(reset_at, str): + text = reset_at.strip() + try: + epoch = float(text) + except ValueError: + try: + from datetime import datetime, timezone + epoch = datetime.fromisoformat( + text.replace("Z", "+00:00") + ).timestamp() + except (ValueError, OverflowError): + epoch = None + if epoch is None: + return None + remaining = epoch - time.time() + if remaining <= 0: + return None + hours = int(remaining // 3600) + minutes = int((remaining % 3600) // 60) + if hours and minutes: + return f"~{hours}h {minutes}m" + if hours: + return f"~{hours}h" + if minutes: + return f"~{minutes}m" + return "less than a minute" + + +def _format_usage_limit_message(error_context: Optional[Dict[str, Any]]) -> str: + """Build the user-facing message for a scheduled plan/subscription cap. + + Surfaces the reset time when available and makes clear retrying won't help + (Context Rotation V0-A — usage_limit_reached hard-stop after fallback). + """ + delta = _format_reset_delta( + (error_context or {}).get("reset_at") + ) + if delta: + when = f" It resets in {delta}." + else: + when = "" + return ( + "⏳ Plan usage limit reached. Your provider's usage limit has been hit, " + f"so I've stopped retrying — this won't clear by retrying.{when} " + "You can wait until it resets, or switch models with /model." + ) + + # Shared recovery hint appended to every content-policy refusal message. Both # the HTTP-200 refusal path (``finish_reason=content_filter``) and the # exception path (a provider moderation error classified as @@ -2639,6 +2699,51 @@ def _perform_api_call(next_api_kwargs): "interrupted": True, } + # ── Usage limit reached (scheduled plan/subscription cap) ── + # A plan cap (e.g. GPT-5.5 / openai-codex "usage_limit_reached") + # resets on a clock — retrying the same provider is pointless and + # the classifier already declined same-provider credential + # rotation (should_rotate_credential=False). Policy (Context + # Rotation V0-A): try the configured fallback chain exactly once; + # if no fallback exists or the chain is exhausted, hard-stop and + # report the reset time. No backoff, no sleeps, no retry against + # the capped provider, and no repeated fallback-loop (each + # _try_activate_fallback advances the chain index). + if classified.reason == FailoverReason.usage_limit_reached: + if agent._has_pending_fallback() and agent._try_activate_fallback( + reason=classified.reason + ): + agent._buffer_status( + "⚠️ Plan usage limit reached — switching to fallback provider..." + ) + retry_count = 0 + compression_attempts = 0 + _retry.primary_recovery_attempted = False + continue + # No fallback available or chain exhausted — hard stop. + agent._flush_status_buffer() + _usage_limit_msg = _format_usage_limit_message(error_context) + agent._emit_status(f"❌ {_usage_limit_msg}") + if api_kwargs is not None: + agent._dump_api_request_debug( + api_kwargs, reason="usage_limit_reached", error=api_error, + ) + agent._persist_session(messages, conversation_history) + logger.error( + "%sUsage limit reached — stopped without retry. " + "provider=%s model=%s", + agent.log_prefix, _provider, _model, + ) + return { + "final_response": _usage_limit_msg, + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "failed": True, + "error": _error_summary, + "failure_reason": classified.reason.value, + } + # Check for 413 payload-too-large BEFORE generic 4xx handler. # A 413 is a payload-size error — the correct response is to # compress history and retry, not abort immediately. diff --git a/agent/error_classifier.py b/agent/error_classifier.py index c39c24a6a5d24..6e95b96db0c9e 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -31,6 +31,7 @@ class FailoverReason(enum.Enum): # Billing / quota billing = "billing" # 402 or confirmed credit exhaustion — rotate immediately rate_limit = "rate_limit" # 429 or quota-based throttling — backoff then rotate + usage_limit_reached = "usage_limit_reached" # Scheduled plan/subscription cap (resets on a clock) — do NOT retry the same provider, fallback once then hard-stop # Server-side overloaded = "overloaded" # 503/529 — provider overloaded, backoff @@ -153,6 +154,51 @@ def is_auth(self) -> bool: "window", ] +# Structured error codes/types that unambiguously mean "scheduled plan cap" — +# a subscription/plan usage limit that resets on a clock (e.g. GPT-5.5 / +# openai-codex returns HTTP 429 with body {"error": {"type": +# "usage_limit_reached", "resets_in_seconds": N}}). These are NOT transient +# throttles: retrying the same provider before the reset is pointless, so the +# retry loop must stop immediately (fallback once, then hard-stop). See +# FailoverReason.usage_limit_reached and Context Rotation V0-A. +_USAGE_LIMIT_HARD_CODES = ( + "usage_limit_reached", + "gousagelimit", +) + +# Explicit hard-cap PHRASES (not single words) used when no structured code is +# present. Deliberately specific so they match the codex/Anthropic plan-cap +# wording ("The usage limit has been reached", "You hit your usage limit") +# without swallowing the generic credit-exhaustion phrasing ("usage limit +# reached") that already classifies as ``billing``. +_USAGE_LIMIT_HARD_PHRASES = ( + "usage limit has been reached", + "plan's usage limit", + "plan usage limit", + "hit your usage limit", + "reached your usage limit", + "usage limit reached. it resets", +) + + +def _is_hard_usage_limit(error_code: str, error_msg: str) -> bool: + """True when the error is a confirmed scheduled plan/subscription cap. + + Highest-confidence signal is the structured error code/type + (``usage_limit_reached``). When only prose is available, require an explicit + hard-cap phrase AND the *absence* of any transient signal — so genuine + "usage limit, try again in 20s" throttles keep their retryable + ``rate_limit`` classification and bare "usage limit reached" credit + exhaustion keeps its ``billing`` classification. + """ + code_lower = (error_code or "").strip().lower() + if any(c in code_lower for c in _USAGE_LIMIT_HARD_CODES): + return True + if any(p in error_msg for p in _USAGE_LIMIT_HARD_PHRASES): + has_transient = any(p in error_msg for p in _USAGE_LIMIT_TRANSIENT_SIGNALS) + return not has_transient + return False + # Payload-too-large patterns detected from message text (no status_code attr). # Proxies and some backends embed the HTTP status in the error message. _PAYLOAD_TOO_LARGE_PATTERNS = [ @@ -549,6 +595,22 @@ def _result(reason: FailoverReason, **overrides) -> ClassifiedError: should_fallback=True, ) + # Scheduled plan/subscription usage cap (e.g. GPT-5.5 / openai-codex + # "usage_limit_reached", commonly HTTP 429 with resets_in_seconds). Must run + # before status-based classification so the 429/402 handlers don't downgrade + # it to a retryable ``rate_limit``. Policy (Context Rotation V0-A): NOT + # retryable, do NOT rotate the same provider's credentials (the plan cap is + # shared across keys), allow the fallback chain once, then hard-stop with the + # reset time. Transient "usage limit, try again" throttles are excluded by + # _is_hard_usage_limit and keep their retryable rate_limit classification. + if _is_hard_usage_limit(error_code, error_msg): + return _result( + FailoverReason.usage_limit_reached, + retryable=False, + should_rotate_credential=False, + should_fallback=True, + ) + # Anthropic thinking block recovery (400). Two distinct failure modes, # same recovery (strip all reasoning_details and retry without thinking # blocks — see the thinking_signature handler in conversation_loop.py): diff --git a/cli.py b/cli.py index bc4f4a76befb4..838318abb44f4 100644 --- a/cli.py +++ b/cli.py @@ -13936,7 +13936,7 @@ def _signal_handler_q(signum, frame): _exit_code = 1 if os.environ.get("HERMES_KANBAN_TASK") and result.get( "failure_reason" - ) in ("rate_limit", "billing"): + ) in ("rate_limit", "billing", "usage_limit_reached"): try: from hermes_cli.kanban_db import ( KANBAN_RATE_LIMIT_EXIT_CODE as _RL_CODE, diff --git a/tests/agent/test_error_classifier.py b/tests/agent/test_error_classifier.py index 9708d7aadc325..ef750b1dd106e 100644 --- a/tests/agent/test_error_classifier.py +++ b/tests/agent/test_error_classifier.py @@ -53,6 +53,7 @@ def test_all_reasons_have_string_values(self): def test_enum_members_exist(self): expected = { "auth", "auth_permanent", "billing", "rate_limit", + "usage_limit_reached", "overloaded", "server_error", "timeout", "context_overflow", "payload_too_large", "image_too_large", "model_not_found", "format_error", @@ -308,6 +309,57 @@ def test_429_rate_limit(self): assert result.reason == FailoverReason.rate_limit assert result.should_fallback is True + # ── Usage limit reached (scheduled plan/subscription cap) ── + + def test_429_usage_limit_reached_body_type_not_retryable(self): + """GPT-5.5 / openai-codex plan cap: 429 + body type usage_limit_reached. + + Must NOT be classified as a retryable rate_limit, and must NOT request + same-provider credential rotation (the plan cap is shared across keys). + Fallback to a different provider is allowed. + """ + e = MockAPIError( + "Too Many Requests", + status_code=429, + body={"error": {"type": "usage_limit_reached", "resets_in_seconds": 7200}}, + ) + result = classify_api_error(e, provider="openai-codex", model="gpt-5.5") + assert result.reason == FailoverReason.usage_limit_reached + assert result.retryable is False + assert result.should_rotate_credential is False + assert result.should_fallback is True + + def test_402_usage_limit_reached_body_type_not_retryable(self): + """Same scheduled cap surfaced as 402 is still non-retryable.""" + e = MockAPIError( + "Payment Required", + status_code=402, + body={"error": {"code": "usage_limit_reached"}}, + ) + result = classify_api_error(e) + assert result.reason == FailoverReason.usage_limit_reached + assert result.retryable is False + + def test_usage_limit_reached_prose_without_transient_is_hard_cap(self): + """Explicit plan-cap prose with no transient signal → hard cap.""" + e = MockAPIError( + "You hit your usage limit.", + status_code=429, + ) + result = classify_api_error(e) + assert result.reason == FailoverReason.usage_limit_reached + assert result.retryable is False + + def test_usage_limit_reached_prose_with_transient_stays_rate_limit(self): + """Plan-cap prose that includes a transient signal stays retryable.""" + e = MockAPIError( + "The usage limit has been reached, try again in 30s.", + status_code=429, + ) + result = classify_api_error(e) + assert result.reason == FailoverReason.rate_limit + assert result.retryable is True + def test_alibaba_rate_increased_too_quickly(self): """Alibaba/DashScope returns a unique throttling message. diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index 827bc0ef690af..d0c5c35d60f85 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -3807,6 +3807,136 @@ def _mock_fallback(): assert result["final_response"] != "(empty)" assert "No reply:" in result["final_response"] + def test_usage_limit_reached_stops_without_retry_or_backoff(self, agent): + """A scheduled plan cap (usage_limit_reached) must abort after exactly + one API attempt — no retries, no backoff sleep — and surface the reset + time. (Context Rotation V0-A.)""" + from agent import conversation_loop + + self._setup_agent(agent) + agent.base_url = "http://127.0.0.1:1234/v1" + agent._fallback_chain = [] + agent._fallback_index = 0 + + class _UsageLimitError(Exception): + def __init__(self): + super().__init__("Too Many Requests") + self.status_code = 429 + self.body = { + "error": { + "type": "usage_limit_reached", + "resets_in_seconds": 7200, + } + } + + agent.client.chat.completions.create.side_effect = _UsageLimitError() + + with ( + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + patch.object(conversation_loop, "jittered_backoff") as mock_backoff, + ): + result = agent.run_conversation("do something") + + assert result["completed"] is False + assert result["failed"] is True + assert result["api_calls"] == 1 # one attempt, no retries + assert result["failure_reason"] == "usage_limit_reached" + assert "usage limit" in (result["final_response"] or "").lower() + mock_backoff.assert_not_called() # no backoff for a plan cap + + def test_usage_limit_reached_falls_back_once_then_continues(self, agent): + """With a fallback chain configured, a plan cap activates the fallback + provider exactly once and continues there.""" + self._setup_agent(agent) + agent.base_url = "http://127.0.0.1:1234/v1" + agent._fallback_chain = [ + {"provider": "openrouter", "model": "anthropic/claude-sonnet-4"} + ] + agent._fallback_index = 0 + agent._fallback_activated = False + + class _UsageLimitError(Exception): + def __init__(self): + super().__init__("Too Many Requests") + self.status_code = 429 + self.body = {"error": {"type": "usage_limit_reached"}} + + content_resp = _mock_response(content="Fallback answer.", finish_reason="stop") + agent.client.chat.completions.create.side_effect = [ + _UsageLimitError(), content_resp, + ] + + fallback_called = {"called": False} + + def _mock_fallback(*args, **kwargs): + fallback_called["called"] = True + agent._fallback_index = 1 + agent._fallback_activated = True + agent.model = "anthropic/claude-sonnet-4" + agent.provider = "openrouter" + return True + + with ( + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + patch.object(agent, "_has_pending_fallback", return_value=True), + patch.object(agent, "_try_activate_fallback", side_effect=_mock_fallback), + ): + result = agent.run_conversation("do something") + + assert fallback_called["called"], "Fallback should have been triggered once" + assert result["completed"] is True + assert result["final_response"] == "Fallback answer." + + def test_usage_limit_reached_fallback_also_capped_hard_stops(self, agent): + """If the fallback provider is also capped and the chain is exhausted, + hard-stop with failure_reason — no fallback-loop.""" + self._setup_agent(agent) + agent.base_url = "http://127.0.0.1:1234/v1" + agent._fallback_chain = [ + {"provider": "openrouter", "model": "anthropic/claude-sonnet-4"} + ] + agent._fallback_index = 0 + agent._fallback_activated = False + + class _UsageLimitError(Exception): + def __init__(self): + super().__init__("Too Many Requests") + self.status_code = 429 + self.body = {"error": {"type": "usage_limit_reached"}} + + agent.client.chat.completions.create.side_effect = [ + _UsageLimitError(), _UsageLimitError(), + ] + + def _mock_has_pending(): + return agent._fallback_index < len(agent._fallback_chain) + + def _mock_fallback(*args, **kwargs): + if agent._fallback_index >= len(agent._fallback_chain): + return False + agent._fallback_index += 1 + agent._fallback_activated = True + agent.model = "anthropic/claude-sonnet-4" + agent.provider = "openrouter" + return True + + with ( + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + patch.object(agent, "_has_pending_fallback", side_effect=_mock_has_pending), + patch.object(agent, "_try_activate_fallback", side_effect=_mock_fallback), + ): + result = agent.run_conversation("do something") + + assert result["completed"] is False + assert result["failed"] is True + assert result["failure_reason"] == "usage_limit_reached" + def test_empty_response_emits_status_for_gateway(self, agent): """_emit_status is called during empty retries so gateway users see feedback.""" self._setup_agent(agent) @@ -5035,6 +5165,28 @@ def test_extract_api_error_context_parses_resets_in_hours_and_minutes(self, agen assert context["reason"] == "GoUsageLimitError" assert context["reset_at"] == 1_000.0 + (6 * 60 * 60) + (29 * 60) + def test_extract_api_error_context_parses_resets_in_seconds(self, agent, monkeypatch): + """Numeric resets_in_seconds body field (codex usage_limit_reached) → + an absolute reset_at relative to now.""" + from agent import agent_runtime_helpers + + monkeypatch.setattr(agent_runtime_helpers.time, "time", lambda: 1_000.0) + error = SimpleNamespace( + body={ + "error": { + "type": "usage_limit_reached", + "message": "The usage limit has been reached", + "resets_in_seconds": 7200, + } + }, + response=SimpleNamespace(headers={}), + ) + + context = agent._extract_api_error_context(error) + + assert context["reason"] == "usage_limit_reached" + assert context["reset_at"] == 1_000.0 + 7200 + def test_recover_with_pool_passes_error_context_on_rotated_429(self, agent): next_entry = SimpleNamespace(label="secondary") captured = {}