Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions agent/agent_runtime_helpers.py
Original file line number Diff line number Diff line change
Expand Up @@ -2400,6 +2400,18 @@ def extract_api_error_context(error: Exception) -> Dict[str, Any]:
if value not in {None, ""}:
context["reset_at"] = value
break
# Numeric relative reset (e.g. Anthropic/codex usage_limit_reached body
# carries "resets_in_seconds": N rather than an absolute timestamp).
if "reset_at" not in context:
for key in ("resets_in_seconds", "reset_in_seconds"):
resets_in = payload.get(key)
if resets_in not in {None, ""}:
try:
context["reset_at"] = time.time() + float(resets_in)
except (TypeError, ValueError):
pass
else:
break
retry_after = payload.get("retry_after")
if retry_after not in {None, ""} and "reset_at" not in context:
try:
Expand Down
105 changes: 105 additions & 0 deletions agent/conversation_loop.py
Original file line number Diff line number Diff line change
Expand Up @@ -430,6 +430,66 @@ def _get_continuation_prompt(is_partial_stub: bool, dropped_tools: Optional[List
)


def _format_reset_delta(reset_at: Any) -> Optional[str]:
"""Render a human "~Nh Nm" delta from a reset timestamp, or None.

``reset_at`` may be an absolute epoch float/int, a numeric string, or an
ISO-8601 string (e.g. "2026-04-12T10:30:00Z"). Returns None when the value
is missing, unparseable, or already in the past.
"""
if reset_at in {None, ""}:
return None
epoch: Optional[float] = None
if isinstance(reset_at, (int, float)):
epoch = float(reset_at)
elif isinstance(reset_at, str):
text = reset_at.strip()
try:
epoch = float(text)
except ValueError:
try:
from datetime import datetime, timezone
epoch = datetime.fromisoformat(
text.replace("Z", "+00:00")
).timestamp()
except (ValueError, OverflowError):
epoch = None
if epoch is None:
return None
remaining = epoch - time.time()
if remaining <= 0:
return None
hours = int(remaining // 3600)
minutes = int((remaining % 3600) // 60)
if hours and minutes:
return f"~{hours}h {minutes}m"
if hours:
return f"~{hours}h"
if minutes:
return f"~{minutes}m"
return "less than a minute"


def _format_usage_limit_message(error_context: Optional[Dict[str, Any]]) -> str:
"""Build the user-facing message for a scheduled plan/subscription cap.

Surfaces the reset time when available and makes clear retrying won't help
(Context Rotation V0-A — usage_limit_reached hard-stop after fallback).
"""
delta = _format_reset_delta(
(error_context or {}).get("reset_at")
)
if delta:
when = f" It resets in {delta}."
else:
when = ""
return (
"⏳ Plan usage limit reached. Your provider's usage limit has been hit, "
f"so I've stopped retrying — this won't clear by retrying.{when} "
"You can wait until it resets, or switch models with /model."
)


# Shared recovery hint appended to every content-policy refusal message. Both
# the HTTP-200 refusal path (``finish_reason=content_filter``) and the
# exception path (a provider moderation error classified as
Expand Down Expand Up @@ -2639,6 +2699,51 @@ def _perform_api_call(next_api_kwargs):
"interrupted": True,
}

# ── Usage limit reached (scheduled plan/subscription cap) ──
# A plan cap (e.g. GPT-5.5 / openai-codex "usage_limit_reached")
# resets on a clock — retrying the same provider is pointless and
# the classifier already declined same-provider credential
# rotation (should_rotate_credential=False). Policy (Context
# Rotation V0-A): try the configured fallback chain exactly once;
# if no fallback exists or the chain is exhausted, hard-stop and
# report the reset time. No backoff, no sleeps, no retry against
# the capped provider, and no repeated fallback-loop (each
# _try_activate_fallback advances the chain index).
if classified.reason == FailoverReason.usage_limit_reached:
if agent._has_pending_fallback() and agent._try_activate_fallback(
reason=classified.reason
):
agent._buffer_status(
"⚠️ Plan usage limit reached — switching to fallback provider..."
)
retry_count = 0
compression_attempts = 0
_retry.primary_recovery_attempted = False
continue
# No fallback available or chain exhausted — hard stop.
agent._flush_status_buffer()
_usage_limit_msg = _format_usage_limit_message(error_context)
agent._emit_status(f"❌ {_usage_limit_msg}")
if api_kwargs is not None:
agent._dump_api_request_debug(
api_kwargs, reason="usage_limit_reached", error=api_error,
)
agent._persist_session(messages, conversation_history)
logger.error(
"%sUsage limit reached — stopped without retry. "
"provider=%s model=%s",
agent.log_prefix, _provider, _model,
)
return {
"final_response": _usage_limit_msg,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
"failed": True,
"error": _error_summary,
"failure_reason": classified.reason.value,
}

# Check for 413 payload-too-large BEFORE generic 4xx handler.
# A 413 is a payload-size error — the correct response is to
# compress history and retry, not abort immediately.
Expand Down
62 changes: 62 additions & 0 deletions agent/error_classifier.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,7 @@ class FailoverReason(enum.Enum):
# Billing / quota
billing = "billing" # 402 or confirmed credit exhaustion — rotate immediately
rate_limit = "rate_limit" # 429 or quota-based throttling — backoff then rotate
usage_limit_reached = "usage_limit_reached" # Scheduled plan/subscription cap (resets on a clock) — do NOT retry the same provider, fallback once then hard-stop

# Server-side
overloaded = "overloaded" # 503/529 — provider overloaded, backoff
Expand Down Expand Up @@ -153,6 +154,51 @@ def is_auth(self) -> bool:
"window",
]

# Structured error codes/types that unambiguously mean "scheduled plan cap" —
# a subscription/plan usage limit that resets on a clock (e.g. GPT-5.5 /
# openai-codex returns HTTP 429 with body {"error": {"type":
# "usage_limit_reached", "resets_in_seconds": N}}). These are NOT transient
# throttles: retrying the same provider before the reset is pointless, so the
# retry loop must stop immediately (fallback once, then hard-stop). See
# FailoverReason.usage_limit_reached and Context Rotation V0-A.
_USAGE_LIMIT_HARD_CODES = (
"usage_limit_reached",
"gousagelimit",
)

# Explicit hard-cap PHRASES (not single words) used when no structured code is
# present. Deliberately specific so they match the codex/Anthropic plan-cap
# wording ("The usage limit has been reached", "You hit your usage limit")
# without swallowing the generic credit-exhaustion phrasing ("usage limit
# reached") that already classifies as ``billing``.
_USAGE_LIMIT_HARD_PHRASES = (
"usage limit has been reached",
"plan's usage limit",
"plan usage limit",
"hit your usage limit",
"reached your usage limit",
"usage limit reached. it resets",
)


def _is_hard_usage_limit(error_code: str, error_msg: str) -> bool:
"""True when the error is a confirmed scheduled plan/subscription cap.

Highest-confidence signal is the structured error code/type
(``usage_limit_reached``). When only prose is available, require an explicit
hard-cap phrase AND the *absence* of any transient signal — so genuine
"usage limit, try again in 20s" throttles keep their retryable
``rate_limit`` classification and bare "usage limit reached" credit
exhaustion keeps its ``billing`` classification.
"""
code_lower = (error_code or "").strip().lower()
if any(c in code_lower for c in _USAGE_LIMIT_HARD_CODES):
return True
if any(p in error_msg for p in _USAGE_LIMIT_HARD_PHRASES):
has_transient = any(p in error_msg for p in _USAGE_LIMIT_TRANSIENT_SIGNALS)
return not has_transient
return False

# Payload-too-large patterns detected from message text (no status_code attr).
# Proxies and some backends embed the HTTP status in the error message.
_PAYLOAD_TOO_LARGE_PATTERNS = [
Expand Down Expand Up @@ -549,6 +595,22 @@ def _result(reason: FailoverReason, **overrides) -> ClassifiedError:
should_fallback=True,
)

# Scheduled plan/subscription usage cap (e.g. GPT-5.5 / openai-codex
# "usage_limit_reached", commonly HTTP 429 with resets_in_seconds). Must run
# before status-based classification so the 429/402 handlers don't downgrade
# it to a retryable ``rate_limit``. Policy (Context Rotation V0-A): NOT
# retryable, do NOT rotate the same provider's credentials (the plan cap is
# shared across keys), allow the fallback chain once, then hard-stop with the
# reset time. Transient "usage limit, try again" throttles are excluded by
# _is_hard_usage_limit and keep their retryable rate_limit classification.
if _is_hard_usage_limit(error_code, error_msg):
return _result(
FailoverReason.usage_limit_reached,
retryable=False,
should_rotate_credential=False,
should_fallback=True,
)

# Anthropic thinking block recovery (400). Two distinct failure modes,
# same recovery (strip all reasoning_details and retry without thinking
# blocks — see the thinking_signature handler in conversation_loop.py):
Expand Down
2 changes: 1 addition & 1 deletion cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -13936,7 +13936,7 @@ def _signal_handler_q(signum, frame):
_exit_code = 1
if os.environ.get("HERMES_KANBAN_TASK") and result.get(
"failure_reason"
) in ("rate_limit", "billing"):
) in ("rate_limit", "billing", "usage_limit_reached"):
try:
from hermes_cli.kanban_db import (
KANBAN_RATE_LIMIT_EXIT_CODE as _RL_CODE,
Expand Down
52 changes: 52 additions & 0 deletions tests/agent/test_error_classifier.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,7 @@ def test_all_reasons_have_string_values(self):
def test_enum_members_exist(self):
expected = {
"auth", "auth_permanent", "billing", "rate_limit",
"usage_limit_reached",
"overloaded", "server_error", "timeout",
"context_overflow", "payload_too_large", "image_too_large",
"model_not_found", "format_error",
Expand Down Expand Up @@ -308,6 +309,57 @@ def test_429_rate_limit(self):
assert result.reason == FailoverReason.rate_limit
assert result.should_fallback is True

# ── Usage limit reached (scheduled plan/subscription cap) ──

def test_429_usage_limit_reached_body_type_not_retryable(self):
"""GPT-5.5 / openai-codex plan cap: 429 + body type usage_limit_reached.

Must NOT be classified as a retryable rate_limit, and must NOT request
same-provider credential rotation (the plan cap is shared across keys).
Fallback to a different provider is allowed.
"""
e = MockAPIError(
"Too Many Requests",
status_code=429,
body={"error": {"type": "usage_limit_reached", "resets_in_seconds": 7200}},
)
result = classify_api_error(e, provider="openai-codex", model="gpt-5.5")
assert result.reason == FailoverReason.usage_limit_reached
assert result.retryable is False
assert result.should_rotate_credential is False
assert result.should_fallback is True

def test_402_usage_limit_reached_body_type_not_retryable(self):
"""Same scheduled cap surfaced as 402 is still non-retryable."""
e = MockAPIError(
"Payment Required",
status_code=402,
body={"error": {"code": "usage_limit_reached"}},
)
result = classify_api_error(e)
assert result.reason == FailoverReason.usage_limit_reached
assert result.retryable is False

def test_usage_limit_reached_prose_without_transient_is_hard_cap(self):
"""Explicit plan-cap prose with no transient signal → hard cap."""
e = MockAPIError(
"You hit your usage limit.",
status_code=429,
)
result = classify_api_error(e)
assert result.reason == FailoverReason.usage_limit_reached
assert result.retryable is False

def test_usage_limit_reached_prose_with_transient_stays_rate_limit(self):
"""Plan-cap prose that includes a transient signal stays retryable."""
e = MockAPIError(
"The usage limit has been reached, try again in 30s.",
status_code=429,
)
result = classify_api_error(e)
assert result.reason == FailoverReason.rate_limit
assert result.retryable is True

def test_alibaba_rate_increased_too_quickly(self):
"""Alibaba/DashScope returns a unique throttling message.

Expand Down
Loading
Loading