diff --git a/agent/account_usage.py b/agent/account_usage.py index e5989e26b92a0..33816a7b738fb 100644 --- a/agent/account_usage.py +++ b/agent/account_usage.py @@ -2,6 +2,8 @@ import logging import math +from decimal import Decimal, InvalidOperation +from urllib.parse import urlparse from dataclasses import dataclass from datetime import datetime, timezone from typing import TYPE_CHECKING, Any, Callable, Optional @@ -385,7 +387,8 @@ def _get_json(url: str, headers: dict[str, str], *, timeout: float) -> dict: def _usage_windows( - source: dict, mapping: tuple[tuple[str, str], ...], used_key: str, reset_key: str, *, fraction: bool = False + source: dict, mapping: tuple[tuple[str, str], ...], used_key: str, reset_key: str, *, fraction: bool = False, + label_fn: Optional[Callable[[dict, str], str]] = None, ) -> list[AccountUsageWindow]: """Build windows from ``source[key][used_key]``; ``fraction`` scales values <= 1 to percent.""" windows: list[AccountUsageWindow] = [] @@ -397,7 +400,7 @@ def _usage_windows( used = float(used) if fraction and used <= 1: used *= 100 - windows.append(AccountUsageWindow(label=label, used_percent=used, reset_at=_parse_dt(window.get(reset_key)))) + windows.append(AccountUsageWindow(label=label_fn(window, label) if label_fn else label, used_percent=used, reset_at=_parse_dt(window.get(reset_key)))) return windows @@ -587,7 +590,7 @@ def redeem_codex_reset_credit( def _fetch_anthropic_account_usage( base_url: Optional[str] = None, api_key: Optional[str] = None ) -> Optional[AccountUsageSnapshot]: - token = (resolve_anthropic_token() or "").strip() + token = (str(api_key or "").strip() or (resolve_anthropic_token() or "").strip()) if not token: return None if not _is_oauth_token(token): @@ -645,9 +648,104 @@ def _data(path: str) -> dict: return _snapshot("openrouter", "credits_api", windows, details) +def _money_symbol(currency: str) -> str: + normalized = currency.strip().upper() + if normalized == "CNY": + return "¥" + if normalized == "USD": + return "$" + return f"{normalized} " if normalized else "" + + +def _decimal_or_none(value: Any) -> Optional[Decimal]: + try: + parsed = Decimal(str(value).strip()) + except (InvalidOperation, ValueError, TypeError): + return None + return parsed if parsed.is_finite() else None + + +def _format_money(value: Decimal, currency: str) -> str: + return f"{_money_symbol(currency)}{value:.2f}" + + +def _is_deepseek_base_url(base_url: Optional[str]) -> bool: + try: + host = urlparse(str(base_url or "")).hostname or "" + except Exception: + return False + host = host.lower().strip(".") + return host == "deepseek.com" or host.endswith(".deepseek.com") + + +def _resolve_deepseek_balance_url(base_url: Optional[str]) -> str: + parsed = urlparse(str(base_url or "").strip() or "https://api.deepseek.com") + scheme = parsed.scheme or "https" + netloc = parsed.netloc or parsed.path or "api.deepseek.com" + return f"{scheme}://{netloc.rstrip('/')}/user/balance" + + +def _fetch_deepseek_account_usage(base_url: Optional[str], api_key: Optional[str]) -> Optional[AccountUsageSnapshot]: + if api_key: + runtime = { + "base_url": (base_url or "https://api.deepseek.com").strip(), + "api_key": str(api_key).strip(), + } + else: + runtime = resolve_runtime_provider( + requested="deepseek", + explicit_base_url=base_url, + explicit_api_key=api_key, + ) + token = str(runtime.get("api_key", "") or "").strip() + if not token: + return None + resolved_base_url = str(runtime.get("base_url", "") or base_url or "https://api.deepseek.com") + headers = { + "Authorization": f"Bearer {token}", + "Accept": "application/json", + } + with httpx.Client(timeout=10.0) as client: + response = client.get(_resolve_deepseek_balance_url(resolved_base_url), headers=headers) + response.raise_for_status() + payload = response.json() or {} + details: list[str] = [] + for info in payload.get("balance_infos") or []: + if not isinstance(info, dict): + continue + currency = str(info.get("currency") or "").strip().upper() + total = _decimal_or_none(info.get("total_balance")) + if total is None: + continue + parts = [f"Balance: {_format_money(total, currency)}"] + if currency: + parts[0] += f" {currency}" + subparts: list[str] = [] + granted = _decimal_or_none(info.get("granted_balance")) + topped_up = _decimal_or_none(info.get("topped_up_balance")) + if granted is not None and granted > 0: + subparts.append(f"granted {_format_money(granted, currency)}") + if topped_up is not None and topped_up > 0: + subparts.append(f"topped up {_format_money(topped_up, currency)}") + if subparts: + parts.append(f"({', '.join(subparts)})") + details.append(" ".join(parts)) + unavailable_reason = None + if payload.get("is_available") is False and not details: + unavailable_reason = "DeepSeek API balance is insufficient." + return AccountUsageSnapshot( + provider="deepseek", + source="balance_api", + fetched_at=_utc_now(), + title="Account balance", + details=tuple(details), + unavailable_reason=unavailable_reason, + ) + + _USAGE_FETCHERS: dict[str, Callable[[Optional[str], Optional[str]], Optional[AccountUsageSnapshot]]] = { "openai-codex": _fetch_codex_account_usage, "anthropic": _fetch_anthropic_account_usage, - "openrouter": _fetch_openrouter_account_usage, + "openrouter": _fetch_openrouter_account_usage, "deepseek": _fetch_deepseek_account_usage, } @@ -675,7 +773,12 @@ def _call_plugin_usage_hook(profile, base_url: Optional[str], api_key: Optional[ def fetch_account_usage( provider: Optional[str], *, base_url: Optional[str] = None, api_key: Optional[str] = None, ) -> Optional[AccountUsageSnapshot]: - fetcher = _USAGE_FETCHERS.get(str(provider or "").strip().lower()) + normalized = str(provider or "").strip().lower() + if normalized in {"", "auto", "custom"} and not _is_deepseek_base_url(base_url): + return None + fetcher = _USAGE_FETCHERS.get(normalized) + if fetcher is None and _is_deepseek_base_url(base_url): + fetcher = _fetch_deepseek_account_usage try: if fetcher: return fetcher(base_url, api_key) diff --git a/gateway/run_turn.py b/gateway/run_turn.py index 1484ea2059def..71f5a52d74c5a 100644 --- a/gateway/run_turn.py +++ b/gateway/run_turn.py @@ -1642,18 +1642,44 @@ def _hmwa_prepend_reasoning(self, agent_result, response, source, _intentional_s def _hmwa_runtime_footer_line(self, agent_result, source, _turn_seconds): """Runtime-metadata footer for the FINAL message of the turn; off by default - (display.runtime_footer.enabled=false).""" + (display.runtime_footer.enabled=false). Extends the default footer with + opt-in provider/account/quota/reasoning fields when configured.""" from gateway.run import _load_gateway_config, _platform_config_key, _terminal_scope_cwd try: - from gateway.runtime_footer import build_footer_line as _bfl + from gateway.runtime_footer import build_footer_line as _bfl, resolve_footer_config as _rfc + _user_config = _load_gateway_config() + _platform_key = _platform_config_key(source.platform) + _account_usage = agent_result.get("account_usage") + _account_label = None + if _account_usage is not None: + _account_label = ( + getattr(_account_usage, "account_label", None) + or getattr(_account_usage, "plan", None) + ) + # Usage is resolved by the producer in run_turn_runner.py while the + # routed profile scope and live credential are still available. + _footer_cfg = agent_result.get("footer_config") or _rfc(_user_config, _platform_key) + _reasoning_effort = agent_result.get("reasoning_effort") + if _reasoning_effort is None: + _reasoning_cfg = getattr(self, "_reasoning_config", None) + if isinstance(_reasoning_cfg, dict): + if _reasoning_cfg.get("enabled") is False: + _reasoning_effort = "none" + else: + _reasoning_effort = _reasoning_cfg.get("effort") return _bfl( - user_config=_load_gateway_config(), - platform_key=_platform_config_key(source.platform), model=agent_result.get("model"), + user_config=_user_config, + platform_key=_platform_key, model=agent_result.get("model"), context_tokens=agent_result.get("last_prompt_tokens", 0) or 0, context_length=agent_result.get("context_length") or None, cwd=_terminal_scope_cwd(""), turn_seconds=_turn_seconds, requested_model=agent_result.get("requested_model"), served_model=agent_result.get("served_model"), + provider=agent_result.get("provider"), + account_label=_account_label, + account_usage=_account_usage, + reasoning_effort=_reasoning_effort, + resolved_config=_footer_cfg, ) except Exception as _footer_err: logger.debug("runtime_footer build failed: %s", _footer_err) diff --git a/gateway/run_turn_runner.py b/gateway/run_turn_runner.py index 1b2e2c0bd31b0..eefaeaef02269 100644 --- a/gateway/run_turn_runner.py +++ b/gateway/run_turn_runner.py @@ -72,6 +72,38 @@ class _ExecApprovalDeclined(RuntimeError): """ +def _resolve_runtime_footer_metadata(agent, user_config: dict | None, platform_key: str) -> dict: + """Resolve footer usage while the turn's routed profile scope is active. + + The API credential is used only to key/schedule the background usage fetch; + it is deliberately not returned in the result consumed by ``run_turn.py``. + """ + from gateway.runtime_footer import resolve_footer_config + from gateway.runtime_footer_usage import get_cached + from hermes_constants import get_hermes_home + + footer_config = resolve_footer_config(user_config, platform_key) + fields = set(footer_config.get("fields") or ()) + needs_usage = footer_config.get("enabled") and bool(fields & {"account", "quota"}) + provider = getattr(agent, "provider", None) if agent is not None else None + base_url = getattr(agent, "base_url", None) if agent is not None else None + api_key = getattr(agent, "api_key", None) if agent is not None else None + account_usage = None + if needs_usage and provider: + account_usage = get_cached( + provider, + base_url=base_url, + api_key=api_key, + hermes_home=str(get_hermes_home()), + ) + return { + "provider": provider, + "base_url": base_url, + "account_usage": account_usage, + "footer_config": footer_config, + } + + class TurnRunner: """Per-turn collaborator carrying ``GatewayRunner._run_agent_inner``'s tool-progress callbacks.""" @@ -1968,6 +2000,14 @@ def run_sync(self): "model": getattr(agent, "model", None) if agent else None, "context_length": (getattr(comp, "context_length", 0) or 0) if has_comp else 0, } + footer_metadata = _resolve_runtime_footer_metadata( + agent, + ctx.user_config, + platform_key, + ) + footer_metadata["reasoning_effort"] = ( + getattr(runner, "_reasoning_config", {}) or {} + ).get("effort") compacted_in_place, effective_session_id, history_offset = self._sync_session_after_run(agent_history) # failure_reason must survive the empty-response path too (TUI billing, transient-failure # persistence). compression_deferred (soft lock-contention defer) is distinct from @@ -1984,6 +2024,7 @@ def run_sync(self): "tools": ctx.tools_holder[0] or [], "history_offset": history_offset, "compacted_in_place": compacted_in_place, "session_id": effective_session_id, **usage, + **footer_metadata, } if not final_response: final_response = _normalize_empty_agent_response(result, final_response or "", history_len=len(agent_history)) diff --git a/gateway/runtime_footer.py b/gateway/runtime_footer.py index 54c923c8209da..da1938bf2ec4b 100644 --- a/gateway/runtime_footer.py +++ b/gateway/runtime_footer.py @@ -1,17 +1,25 @@ """Gateway runtime-metadata footer (model · context % · cwd), off by default to keep replies -minimal. Config: ``display.runtime_footer: {enabled: bool, fields: [model, context_pct, cwd]}`` -(order shown; drop any to hide), per-platform override ``display.platforms.

.runtime_footer``, -toggled by ``/footer on|off``. Fields: ``model`` (vendor prefix dropped), ``context_pct`` (last-call -occupancy), ``latency`` (turn wall-clock, opt-in — NOT in the default set so an unset ``fields`` -renders exactly as before), ``served_model`` (opt-in, ``alias → served``: the deployment a routing -proxy reported via ``x-litellm-model-id`` / ``x-litellm-model-api-base``, or Hermes' own fallback -route; skipped when the served model is the requested one), ``cwd`` (home-relative). ``gateway/run.py`` appends the footer to the -final response only (never to tool-progress or streaming partials); when streaming already -delivered the text, it goes out as a trailing message via ``send_trailing_footer()``.""" +minimal. Config: ``display.runtime_footer: {enabled: bool, fields: [model, context_pct, cwd], +underline: bool}`` (order shown; drop any to hide), per-platform override +``display.platforms.

.runtime_footer``, toggled by ``/footer on|off``. Fields: ``model`` (vendor +prefix dropped), ``context_pct`` (last-call occupancy), ``context`` (compact ``ctx used/limit``), +``latency`` (turn wall-clock, opt-in — NOT in the default set so an unset ``fields`` renders exactly +as before), ``served_model`` (opt-in, ``alias → served``: the deployment a routing proxy reported via +``x-litellm-model-id`` / ``x-litellm-model-api-base``, or Hermes' own fallback route; skipped when +the served model is the requested one), ``reasoning`` (compact reasoning-effort label), ``provider`` +(inference provider id), ``account`` (account/plan label, any ``-`` prefix stripped), and +``quota`` (one part per account-usage window — ``5h``/``7d N%`` plus a compact reset — followed by a +compact balance line when the snapshot carries one). ``underline: true`` prefixes the footer with a +separator line. ``gateway/run.py`` appends the footer to the final response only (never to +tool-progress or streaming partials); when streaming already delivered the text, it goes out as a +trailing message via ``send_trailing_footer()``.""" from __future__ import annotations +import math import os +import re +from datetime import datetime, timezone from typing import Any, Iterable, Optional _DEFAULT_FIELDS: tuple[str, ...] = ("model", "context_pct", "cwd") @@ -33,8 +41,10 @@ def _home_relative_cwd(cwd: str) -> str: def _model_short(model: Optional[str]) -> str: - """Drop ``vendor/`` prefix (``openai/gpt-5.4`` → ``gpt-5.4``).""" - return model.rsplit("/", 1)[-1] if model else "" + """Drop ``vendor/`` prefix for readability (``openai/gpt-5.4`` → ``gpt-5.4``).""" + if not model: + return "" + return model.rsplit("/", 1)[-1] def _env_cwd() -> str: @@ -45,20 +55,70 @@ def _env_cwd() -> str: return terminal_env("TERMINAL_CWD", "") -def resolve_footer_config(user_config: dict[str, Any] | None, platform_key: str | None = None) -> dict[str, Any]: - """Resolve effective footer config: defaults (enabled=False) < - ``display.runtime_footer`` < ``display.platforms..runtime_footer``.""" - resolved = {"enabled": False, "fields": list(_DEFAULT_FIELDS)} +# Compact labels for agent.reasoning_effort / runtime reasoning_config. +# Keep these short so the footer stays one line on mobile clients. +_REASONING_ABBREV = { + "none": "off", + "minimal": "min", + "low": "low", + "medium": "med", + "high": "high", + "xhigh": "xhi", + "max": "max", + "ultra": "ult", +} + + +def _reasoning_short(effort: Optional[str]) -> str: + """Return a compact reasoning-effort label, or "" when unknown/empty.""" + raw = str(effort or "").strip().lower() + if not raw: + return "" + if raw in {"false", "disabled", "off"}: + return _REASONING_ABBREV["none"] + if raw in _REASONING_ABBREV: + return _REASONING_ABBREV[raw] + # Unknown but non-empty values still surface compactly so a new level + # is visible before we teach the map about it. + return raw[:6] + + +def resolve_footer_config( + user_config: dict[str, Any] | None, + platform_key: str | None = None, +) -> dict[str, Any]: + """Resolve effective runtime-footer config for *platform_key*. + + Merge order (later wins): + 1. Built-in defaults (enabled=False) + 2. ``display.runtime_footer`` + 3. ``display.platforms..runtime_footer`` + """ + resolved = {"enabled": False, "fields": list(_DEFAULT_FIELDS), "underline": False} cfg = (user_config or {}).get("display") or {} - plat_cfg = (cfg.get("platforms") or {}).get(platform_key) if platform_key else None - sections = [cfg.get("runtime_footer"), plat_cfg.get("runtime_footer") if isinstance(plat_cfg, dict) else None] - for section in sections: - if not isinstance(section, dict): - continue - if "enabled" in section: - resolved["enabled"] = bool(section.get("enabled")) - if isinstance(section.get("fields"), list) and section["fields"]: - resolved["fields"] = [str(f) for f in section["fields"]] + + global_cfg = cfg.get("runtime_footer") + if isinstance(global_cfg, dict): + if "enabled" in global_cfg: + resolved["enabled"] = bool(global_cfg.get("enabled")) + if "underline" in global_cfg: + resolved["underline"] = bool(global_cfg.get("underline")) + if isinstance(global_cfg.get("fields"), list) and global_cfg["fields"]: + resolved["fields"] = [str(f) for f in global_cfg["fields"]] + + if platform_key: + platforms = cfg.get("platforms") or {} + plat_cfg = platforms.get(platform_key) + if isinstance(plat_cfg, dict): + plat_footer = plat_cfg.get("runtime_footer") + if isinstance(plat_footer, dict): + if "enabled" in plat_footer: + resolved["enabled"] = bool(plat_footer.get("enabled")) + if "underline" in plat_footer: + resolved["underline"] = bool(plat_footer.get("underline")) + if isinstance(plat_footer.get("fields"), list) and plat_footer["fields"]: + resolved["fields"] = [str(f) for f in plat_footer["fields"]] + return resolved @@ -73,11 +133,158 @@ def _format_latency(seconds: float) -> str: return f"{m}m{sec:02d}s" +def _compact_number(value: int | float) -> str: + try: + n = float(value) + except Exception: + return str(value) + if abs(n) >= 1_000_000: + text = f"{n / 1_000_000:.1f}M" + elif abs(n) >= 1_000: + text = f"{n / 1_000:.1f}K" + else: + text = str(int(n)) + return text.replace(".0K", "K").replace(".0M", "M") + + +def _compact_reset(dt: Any) -> str: + if not dt: + return "" + if isinstance(dt, str): + try: + dt = datetime.fromisoformat(dt.strip().replace("Z", "+00:00")) + except Exception: + return "" + if not isinstance(dt, datetime): + return "" + if dt.tzinfo is None: + dt = dt.replace(tzinfo=timezone.utc) + seconds = int((dt - datetime.now(timezone.utc)).total_seconds()) + if seconds <= 0: + return "now" + hours, rem = divmod(seconds, 3600) + minutes = rem // 60 + if hours >= 24: + days, rem_hours = divmod(math.ceil(seconds / 3600), 24) + return f"{days}d" + (f"{rem_hours}h" if rem_hours else "") + if hours > 0: + return f"{hours}h" + (f"{minutes}m" if minutes else "") + return f"{minutes}m" if minutes else "<1m" + + +def _quota_label(window: Any, provider: Optional[str] = None, model: Optional[str] = None) -> str: + """Return a compact quota-window label for footer display. + + The detailed ``/usage`` command keeps provider wording. The footer is space + constrained, so normalize the common OAuth/Codex rolling windows to the + short labels users expect while leaving unknown provider windows intact. + """ + raw = str(getattr(window, "label", "") or "quota").strip() or "quota" + label = raw.lower().replace("_", "-") + model_text = str(model or "").lower() + + if "opus" in label: + return "opus7d" + if "sonnet" in label: + return "sonnet7d" + if label in { + "5h", + "5-hour", + "5 hour", + "five-hour", + "five hour", + "current session", + "session", + "primary", + "primary-window", + "primary window", + }: + return "5h" + if label in { + "7d", + "7-day", + "7 day", + "seven-day", + "seven day", + "current week", + "week", + "weekly", + "secondary", + "secondary-window", + "secondary window", + }: + return "7d" + if "week" in label: + if "opus" in model_text: + return "opus7d" + if "sonnet" in model_text: + return "sonnet7d" + return raw + + +def _compact_quota_detail(detail: Any) -> str: + text = str(detail or "").strip() + if not text: + return "" + # Keep footer quota compact. Detailed breakdowns remain available via the + # usage renderer; the footer only needs the immediately useful balance. + if not re.match(r"^(credits\s+)?balance\s*:", text, flags=re.IGNORECASE): + return "" + text = re.sub(r"\s*\([^)]*\)\s*$", "", text).strip() + text = re.sub(r"^(credits\s+)?balance\s*:", "balance", text, flags=re.IGNORECASE).strip() + return text + + +def _format_quota(account_usage: Any, *, provider: Optional[str] = None, model: Optional[str] = None) -> list[str]: + if not account_usage: + return [] + provider = provider or getattr(account_usage, "provider", None) + parts: list[str] = [] + for window in getattr(account_usage, "windows", ()) or (): + used = getattr(window, "used_percent", None) + if used is None: + continue + try: + remaining = max(0, round(100 - float(used))) + except Exception: + continue + label = _quota_label(window, provider=provider, model=model) + text = f"{label} {remaining}%" + reset = _compact_reset(getattr(window, "reset_at", None)) + if reset: + text += f" {reset}" + parts.append(text) + for detail in getattr(account_usage, "details", ()) or (): + compact = _compact_quota_detail(detail) + if compact: + parts.append(compact) + return parts + + +def _account_short(account_label: Optional[str], provider: Optional[str]) -> str: + raw = str(account_label or "").strip() + if not raw: + return "" + prov = str(provider or "").strip() + if prov: + for prefix in (prov, prov.replace("-", "_"), prov.replace("_", "-")): + for sep in ("-", "_"): + marker = prefix + sep + if raw.lower().startswith(marker.lower()): + return raw[len(marker):] + return raw + + def format_runtime_footer(*, model: Optional[str], context_tokens: int, context_length: Optional[int], cwd: Optional[str] = None, turn_seconds: Optional[float] = None, requested_model: Optional[str] = None, served_model: Optional[str] = None, - fields: Iterable[str] = _DEFAULT_FIELDS) -> str: + fields: Iterable[str] = _DEFAULT_FIELDS, + provider: Optional[str] = None, + account_label: Optional[str] = None, + account_usage: Any = None, + reasoning_effort: Optional[str] = None, + underline: bool = False) -> str: """Render the footer line, or "" if no fields have data. Fields whose data is missing (and unknown field names) are skipped silently — a partial footer beats ``?%`` or empty slots.""" def context_pct() -> str: @@ -92,29 +299,76 @@ def served() -> str: return f"{alias} → {served_model}" return "" + def context_compact() -> str: + if context_length and context_length > 0 and context_tokens >= 0: + return f"ctx {_compact_number(context_tokens)}/{_compact_number(context_length)}" + return "" + + def account() -> str: + label = (account_label + or getattr(account_usage, "account_label", None) + or getattr(account_usage, "plan", None)) + return _account_short(str(label), provider) if label else "" + renderers = { "model": lambda: _model_short(model), "served_model": served, + "reasoning": lambda: _reasoning_short(reasoning_effort), + "reasoning_effort": lambda: _reasoning_short(reasoning_effort), + "effort": lambda: _reasoning_short(reasoning_effort), + "provider": lambda: str(provider) if provider else "", + "account": account, + "context": context_compact, "context_pct": context_pct, # Skipped when the caller did not measure (None) or the value is negative. "latency": lambda: _format_latency(turn_seconds) if turn_seconds is not None and turn_seconds >= 0 else "", "cwd": lambda: _home_relative_cwd(cwd or _env_cwd()), } - return _SEP.join(v for field in fields if (render := renderers.get(field)) and (v := render())) + + parts: list[str] = [] + for field in fields: + # ``quota`` expands to one part per account-usage window (plus any + # compact balance line), so it cannot be a single-string renderer. + if field == "quota": + parts.extend(_format_quota(account_usage, provider=provider, model=model)) + continue + render = renderers.get(field) + if render is None: + continue + value = render() + if value: + parts.append(value) + + if not parts: + return "" + line = _SEP.join(parts) + return f"──────────────\n{line}" if underline else line def build_footer_line(*, user_config: dict[str, Any] | None, platform_key: str | None, model: Optional[str], context_tokens: int, context_length: Optional[int], cwd: Optional[str] = None, turn_seconds: Optional[float] = None, - requested_model: Optional[str] = None, served_model: Optional[str] = None) -> str: + requested_model: Optional[str] = None, served_model: Optional[str] = None, + provider: Optional[str] = None, account_label: Optional[str] = None, + account_usage: Any = None, reasoning_effort: Optional[str] = None, + resolved_config: Optional[dict[str, Any]] = None) -> str: """Entry point for gateway/run.py: footer text, or "" when disabled / no data. Callers append it to the final response themselves, preserving a single blank line of separation. ``turn_seconds`` is the caller-measured (``time.monotonic()``) run duration; ``None`` skips the - ``latency`` field.""" - cfg = resolve_footer_config(user_config, platform_key) + ``latency`` field. ``resolved_config`` lets the turn runner hand over a footer config it already + resolved, instead of resolving it a second time.""" + cfg = ( + resolved_config + if isinstance(resolved_config, dict) + else resolve_footer_config(user_config, platform_key) + ) if not cfg.get("enabled"): return "" return format_runtime_footer(model=model, context_tokens=context_tokens, context_length=context_length, cwd=cwd, turn_seconds=turn_seconds, requested_model=requested_model, served_model=served_model, - fields=cfg.get("fields") or _DEFAULT_FIELDS) + fields=cfg.get("fields") or _DEFAULT_FIELDS, + provider=provider, account_label=account_label, + account_usage=account_usage, + reasoning_effort=reasoning_effort, + underline=bool(cfg.get("underline"))) diff --git a/gateway/runtime_footer_usage.py b/gateway/runtime_footer_usage.py new file mode 100644 index 0000000000000..f01c3d7e734f2 --- /dev/null +++ b/gateway/runtime_footer_usage.py @@ -0,0 +1,95 @@ +"""Non-blocking account-usage cache for the runtime footer. + +This module deliberately contains no gateway-runner state. Usage requests are +slow and optional, so callers receive the last snapshot immediately while a +single daemon thread refreshes stale entries in the background. +""" + +from __future__ import annotations + +import hashlib +import threading +import time +from pathlib import Path +from typing import Any + +from agent.account_usage import fetch_account_usage +from hermes_constants import get_hermes_home + +_CACHE: dict[tuple[str, str, str, str], tuple[float, Any]] = {} +_REFRESHING: set[tuple[str, str, str, str]] = set() +_LOCK = threading.Lock() +_TTL_SECONDS = 90.0 +_MAX_ENTRIES = 64 + + +def cache_key(provider: str | None, *, base_url: str | None = None, + api_key: str | None = None, + hermes_home: str | Path | None = None) -> tuple[str, str, str, str]: + token = str(api_key or "").strip() + digest = hashlib.sha256(token.encode()).hexdigest()[:16] if token else "" + return ( + str(hermes_home if hermes_home is not None else get_hermes_home()), + str(provider or "").strip().lower(), + str(base_url or "").strip().rstrip("/").lower(), + digest, + ) + + +def _refresh(key, provider, base_url, api_key) -> None: + try: + snapshot = fetch_account_usage(provider, base_url=base_url, api_key=api_key) + except Exception: + snapshot = None + with _LOCK: + previous = _CACHE.get(key) + # Stale-while-revalidate: a transient provider failure must not erase a + # previously useful quota value. + value = snapshot if snapshot is not None else (previous[1] if previous else None) + if key not in _CACHE and len(_CACHE) >= _MAX_ENTRIES: + oldest = min(_CACHE, key=lambda item: _CACHE[item][0]) + _CACHE.pop(oldest, None) + _CACHE[key] = (time.monotonic(), value) + _REFRESHING.discard(key) + + +def _start_refresh(key, provider, base_url, api_key) -> None: + try: + threading.Thread( + target=_refresh, + args=(key, provider, base_url, api_key), + name=f"runtime-footer-usage-{key[1] or 'unknown'}", + daemon=True, + ).start() + except Exception: + with _LOCK: + _REFRESHING.discard(key) + + +def get_cached(provider: str | None, *, base_url: str | None = None, + api_key: str | None = None, + hermes_home: str | Path | None = None): + """Return cached usage and schedule at most one non-blocking refresh.""" + normalized = str(provider or "").strip().lower() + if normalized in {"", "auto"}: + return None + key = cache_key(provider, base_url=base_url, api_key=api_key, hermes_home=hermes_home) + now = time.monotonic() + start = False + with _LOCK: + cached = _CACHE.get(key) + value = cached[1] if cached else None + fresh = cached is not None and now - cached[0] < _TTL_SECONDS + if not fresh and key not in _REFRESHING: + _REFRESHING.add(key) + start = True + if start: + _start_refresh(key, provider, base_url, api_key) + return value + + +def clear() -> None: + """Clear cache, primarily for tests and profile teardown.""" + with _LOCK: + _CACHE.clear() + _REFRESHING.clear() diff --git a/tests/agent/test_account_usage.py b/tests/agent/test_account_usage.py index 5566299d3ee72..591f5717ba89f 100644 --- a/tests/agent/test_account_usage.py +++ b/tests/agent/test_account_usage.py @@ -45,10 +45,12 @@ def codex_usage_payload(): "primary_window": { "used_percent": 21, "reset_at": 1779846359, + "limit_window_seconds": 18000, }, "secondary_window": { "used_percent": 4, "reset_at": 1780230796, + "limit_window_seconds": 604800, }, }, "credits": {"has_credits": False}, diff --git a/tests/agent/test_account_usage_fetch.py b/tests/agent/test_account_usage_fetch.py index 439ca1ae18bd8..21277b98f95c8 100644 --- a/tests/agent/test_account_usage_fetch.py +++ b/tests/agent/test_account_usage_fetch.py @@ -72,6 +72,22 @@ def get(self, url, headers=None): return _Response(self._payloads[url]) + +class _RecordingClient: + def __init__(self, payload): + self._payload = payload + self.requests = [] + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc, tb): + return False + + def get(self, url, headers=None): + self.requests.append((url, dict(headers or {}))) + return _Response(self._payload) + def test_fetch_account_usage_codex(monkeypatch): monkeypatch.setattr( "agent.account_usage.resolve_codex_runtime_credentials", @@ -112,7 +128,7 @@ def test_fetch_account_usage_codex(monkeypatch): assert snapshot is not None assert snapshot.plan == "Pro" assert len(snapshot.windows) == 2 - assert snapshot.windows[0].label == "Session" + assert [window.label for window in snapshot.windows] == ["Session", "Weekly"] assert snapshot.windows[0].used_percent == 15.0 assert snapshot.windows[0].reset_at == datetime.fromtimestamp(1_900_000_000, tz=timezone.utc) assert "Credits balance: $12.50" in snapshot.details @@ -158,6 +174,33 @@ def test_fetch_account_usage_prefers_builtin_fetcher_over_profile(monkeypatch): assert fetch_account_usage("openrouter") is builtin assert profile.calls == 0 +def test_fetch_account_usage_codex_labels_single_primary_by_actual_duration(monkeypatch): + monkeypatch.setattr( + "agent.account_usage.httpx.Client", + lambda timeout=15.0: _Client( + { + "plan_type": "pro", + "rate_limit": { + "primary_window": { + "used_percent": 95, + "reset_at": 1_900_000_000, + "limit_window_seconds": 604800, + }, + "secondary_window": None, + }, + } + ), + ) + + snapshot = fetch_account_usage( + "openai-codex", + base_url="https://chatgpt.com/backend-api/codex", + api_key="access-token", + ) + + assert snapshot is not None + assert [(window.label, window.used_percent) for window in snapshot.windows] == [("Weekly", 95.0)] + def test_fetch_account_usage_openrouter_uses_limit_remaining_and_ignores_deprecated_rate_limit(monkeypatch): monkeypatch.setattr( @@ -339,3 +382,90 @@ def probing_portal_fetch(*, force_fresh): finally: marker.reset(token) assert seen == {"force_fresh": True, "marker": "profile-scope"} + +def test_fetch_account_usage_anthropic_prefers_explicit_runtime_key_over_resolver(monkeypatch): + client = _RecordingClient( + { + "five_hour": {"utilization": 0.25}, + } + ) + monkeypatch.setattr("agent.account_usage.httpx.Client", lambda timeout=15.0: client) + monkeypatch.setattr("agent.account_usage._is_oauth_token", lambda token: True) + monkeypatch.setattr( + "agent.account_usage.resolve_anthropic_token", + lambda: "rotated-pool-token", + ) + + snapshot = fetch_account_usage("anthropic", api_key="runtime-token") + + assert snapshot is not None + assert snapshot.provider == "anthropic" + assert client.requests[0][1]["Authorization"] == "Bearer runtime-token" + +def test_fetch_account_usage_custom_deepseek_base_url_is_supported(monkeypatch): + client = _RecordingClient( + { + "is_available": True, + "balance_infos": [ + { + "currency": "CNY", + "total_balance": "25.50", + "granted_balance": "0.00", + "topped_up_balance": "25.50", + } + ], + } + ) + monkeypatch.setattr("agent.account_usage.httpx.Client", lambda timeout=10.0: client) + + snapshot = fetch_account_usage( + "custom", + base_url="https://api.deepseek.com/beta/v1/", + api_key="deepseek-token", + ) + + assert snapshot is not None + assert snapshot.provider == "deepseek" + assert snapshot.details == ("Balance: ¥25.50 CNY (topped up ¥25.50)",) + +def test_fetch_account_usage_deepseek_balance_uses_official_endpoint(monkeypatch): + client = _RecordingClient( + { + "is_available": True, + "balance_infos": [ + { + "currency": "CNY", + "total_balance": "110.00", + "granted_balance": "10.00", + "topped_up_balance": "100.00", + }, + { + "currency": "USD", + "total_balance": "2.50", + "granted_balance": "0.00", + "topped_up_balance": "2.50", + }, + ], + } + ) + monkeypatch.setattr("agent.account_usage.httpx.Client", lambda timeout=10.0: client) + + snapshot = fetch_account_usage( + "deepseek", + base_url="https://api.deepseek.com/v1", + api_key="deepseek-token", + ) + + assert snapshot is not None + assert snapshot.provider == "deepseek" + assert snapshot.source == "balance_api" + assert snapshot.details == ( + "Balance: ¥110.00 CNY (granted ¥10.00, topped up ¥100.00)", + "Balance: $2.50 USD (topped up $2.50)", + ) + assert client.requests == [ + ( + "https://api.deepseek.com/user/balance", + {"Authorization": "Bearer deepseek-token", "Accept": "application/json"}, + ) + ] diff --git a/tests/gateway/test_runtime_footer.py b/tests/gateway/test_runtime_footer.py index d4b619afe8a50..c2dd83d5a067e 100644 --- a/tests/gateway/test_runtime_footer.py +++ b/tests/gateway/test_runtime_footer.py @@ -3,10 +3,14 @@ from __future__ import annotations +import os +from datetime import datetime, timedelta, timezone import pytest +from agent.account_usage import AccountUsageSnapshot, AccountUsageWindow from gateway.runtime_footer import ( + _compact_reset, _home_relative_cwd, _model_short, build_footer_line, @@ -41,6 +45,16 @@ def test_home_relative_cwd_collapses_home(tmp_path, monkeypatch): assert result == "~/projects/hermes" +def test_home_relative_cwd_leaves_abs_path_alone(tmp_path, monkeypatch): + monkeypatch.setenv("HOME", str(tmp_path / "other")) + result = _home_relative_cwd(str(tmp_path / "outside" / "dir")) + assert result == str(tmp_path / "outside" / "dir") + + +def test_home_relative_cwd_empty_returns_empty(): + assert _home_relative_cwd("") == "" + + # --------------------------------------------------------------------------- # format_runtime_footer # --------------------------------------------------------------------------- @@ -73,10 +87,226 @@ def test_format_footer_skips_missing_context_length(): assert "/tmp/wd" in out +def test_format_footer_context_pct_clamped_to_100(): + out = format_runtime_footer( + model="m", + context_tokens=500_000, # way over + context_length=100_000, + cwd="", + fields=("context_pct",), + ) + assert out == "100%" + + +def test_format_footer_context_pct_never_negative(): + out = format_runtime_footer( + model="m", + context_tokens=-50, + context_length=100, + cwd="", + fields=("context_pct",), + ) + # Negative input => no field emitted (we require context_tokens >= 0) + assert out == "" + + +def test_format_footer_empty_fields_returns_empty(): + out = format_runtime_footer( + model="m", context_tokens=0, context_length=100, + cwd="/x", fields=(), + ) + assert out == "" + + +def test_format_footer_drops_cwd_when_empty(monkeypatch): + monkeypatch.delenv("TERMINAL_CWD", raising=False) + out = format_runtime_footer( + model="openai/gpt-5.4", + context_tokens=50, context_length=100, + cwd="", + fields=("model", "context_pct", "cwd"), + ) + # cwd silently dropped; model + pct remain + assert out == "gpt-5.4 · 50%" + + +def test_format_footer_custom_field_order(): + out = format_runtime_footer( + model="openai/gpt-5.4", + context_tokens=50, context_length=100, + cwd="/opt/project", + fields=("context_pct", "model"), # swapped + no cwd + ) + assert out == "50% · gpt-5.4" + + + +def test_format_footer_extended_fields_with_quota_and_underline(): + snapshot = AccountUsageSnapshot( + provider="openai-codex", + source="usage_api", + fetched_at=datetime.now(timezone.utc), + plan="Team", + windows=( + AccountUsageWindow(label="5h", used_percent=20, reset_at=None), + AccountUsageWindow(label="7d", used_percent=30, reset_at=None), + ), + ) + out = format_runtime_footer( + model="openai/gpt-5.5", + provider="openai-codex", + account_label="openai-codex-team-main", + context_tokens=21_800, + context_length=200_000, + account_usage=snapshot, + reasoning_effort="ultra", + cwd="", + fields=("provider", "account", "model", "reasoning", "context", "quota"), + underline=True, + ) + assert out == "──────────────\nopenai-codex · team-main · gpt-5.5 · ult · ctx 21.8K/200K · 5h 80% · 7d 70%" + + +def test_format_footer_compacts_reset_times_for_quota(): + snapshot = AccountUsageSnapshot( + provider="anthropic", + source="oauth_usage_api", + fetched_at=datetime.now(timezone.utc), + windows=( + AccountUsageWindow( + label="Current session", + used_percent=32.2, + reset_at=datetime.now(timezone.utc) + timedelta(hours=2, minutes=10), + ), + ), + ) + out = format_runtime_footer( + model="anthropic/claude-sonnet-4-6", + context_tokens=21_800, + context_length=200_000, + account_usage=snapshot, + cwd="", + fields=("quota",), + ) + assert out.startswith("5h 68% 2h") + + +def test_compact_reset_omits_zero_day_and_minute_components(monkeypatch): + import gateway.runtime_footer as runtime_footer + + fixed_now = datetime(2026, 1, 1, tzinfo=timezone.utc) + + class FixedDateTime(datetime): + @classmethod + def now(cls, tz=None): + return fixed_now if tz is not None else fixed_now.replace(tzinfo=None) + + monkeypatch.setattr(runtime_footer, "datetime", FixedDateTime) + fixed_target = FixedDateTime(2026, 1, 1, tzinfo=timezone.utc) + + assert _compact_reset(fixed_target + timedelta(days=1)) == "1d" + assert _compact_reset(fixed_target + timedelta(hours=1)) == "1h" + assert _compact_reset(fixed_target + timedelta(seconds=30)) == "<1m" + + + +def test_format_footer_uses_provider_specific_compact_quota_labels(): + snapshot = AccountUsageSnapshot( + provider="anthropic", + source="oauth_usage_api", + fetched_at=datetime.now(timezone.utc), + windows=( + AccountUsageWindow(label="Current week", used_percent=40, reset_at=None), + AccountUsageWindow(label="Opus week", used_percent=50, reset_at=None), + AccountUsageWindow(label="Sonnet week", used_percent=60, reset_at=None), + ), + ) + out = format_runtime_footer( + model="anthropic/claude-opus-4-5", + provider="anthropic", + context_tokens=0, + context_length=200_000, + account_usage=snapshot, + fields=("quota",), + ) + assert out == "7d 60% · opus7d 50% · sonnet7d 40%" + + +def test_format_footer_preserves_nonstandard_quota_durations(): + snapshot = AccountUsageSnapshot( + provider="openai-codex", + source="usage_api", + fetched_at=datetime.now(timezone.utc), + windows=( + AccountUsageWindow(label="15d", used_percent=10, reset_at=None), + AccountUsageWindow(label="17h", used_percent=20, reset_at=None), + ), + ) + + out = format_runtime_footer( + model="openai/gpt-5.6-sol", + provider="openai-codex", + context_tokens=0, + context_length=200_000, + account_usage=snapshot, + fields=("quota",), + ) + + assert out == "15d 90% · 17h 80%" + + +def test_format_footer_includes_balance_details_for_quota(): + snapshot = AccountUsageSnapshot( + provider="deepseek", + source="balance_api", + fetched_at=datetime.now(timezone.utc), + details=("Balance: ¥110.00 CNY (granted ¥10.00, topped up ¥100.00)",), + ) + + out = format_runtime_footer( + model="deepseek/deepseek-chat", + provider="deepseek", + context_tokens=0, + context_length=200_000, + account_usage=snapshot, + fields=("quota",), + ) + + assert out == "balance ¥110.00 CNY" + + +def test_format_footer_unknown_field_silently_ignored(): + out = format_runtime_footer( + model="openai/gpt-5.4", + context_tokens=50, context_length=100, + cwd="/x", + fields=("model", "bogus", "context_pct"), + ) + assert out == "gpt-5.4 · 50%" + + # --------------------------------------------------------------------------- # resolve_footer_config # --------------------------------------------------------------------------- +def test_resolve_defaults_off_empty_config(): + cfg = resolve_footer_config({}, "telegram") + assert cfg == {"enabled": False, "fields": ["model", "context_pct", "cwd"], "underline": False} + + +def test_resolve_global_enable(): + user = {"display": {"runtime_footer": {"enabled": True}}} + cfg = resolve_footer_config(user, "telegram") + assert cfg["enabled"] is True + assert cfg["fields"] == ["model", "context_pct", "cwd"] + + + +def test_resolve_supports_underline_flag(): + user = {"display": {"runtime_footer": {"enabled": True, "underline": True}}} + cfg = resolve_footer_config(user, "feishu") + assert cfg["underline"] is True + def test_resolve_platform_override_wins(): user = { @@ -110,10 +340,144 @@ def test_resolve_platform_can_add_fields_only(): assert dc["fields"] == ["context_pct"] +def test_resolve_ignores_malformed_config(): + # Non-dict runtime_footer shouldn't crash + user = {"display": {"runtime_footer": "on"}} + cfg = resolve_footer_config(user, "telegram") + assert cfg["enabled"] is False + + # --------------------------------------------------------------------------- + + +def test_format_footer_reasoning_abbreviations(): + cases = { + "none": "off", + "minimal": "min", + "low": "low", + "medium": "med", + "high": "high", + "xhigh": "xhi", + "max": "max", + "ultra": "ult", + "false": "off", + "DISABLED": "off", + } + for effort, expected in cases.items(): + out = format_runtime_footer( + model="m", + context_tokens=0, + context_length=100, + reasoning_effort=effort, + fields=("reasoning",), + ) + assert out == expected, (effort, out) + + +def test_format_footer_reasoning_omitted_when_missing(): + out = format_runtime_footer( + model="openai/gpt-5.5", + context_tokens=10, + context_length=100, + fields=("model", "reasoning"), + ) + assert out == "gpt-5.5" + + +def test_build_footer_passes_reasoning_effort(): + out = build_footer_line( + user_config={ + "display": { + "runtime_footer": { + "enabled": True, + "fields": ["model", "reasoning"], + } + } + }, + platform_key=None, + model="openai/gpt-5.5", + context_tokens=0, + context_length=None, + reasoning_effort="xhigh", + ) + assert out == "gpt-5.5 · xhi" + + # build_footer_line — top-level entry point used by gateway/run.py # --------------------------------------------------------------------------- +def test_build_footer_empty_when_disabled(): + out = build_footer_line( + user_config={}, + platform_key="telegram", + model="openai/gpt-5.4", + context_tokens=10, context_length=100, + cwd="/tmp", + ) + assert out == "" + + +def test_build_footer_returns_rendered_when_enabled(monkeypatch, tmp_path): + monkeypatch.setenv("HOME", str(tmp_path)) + out = build_footer_line( + user_config={"display": {"runtime_footer": {"enabled": True}}}, + platform_key="telegram", + model="openai/gpt-5.4", + context_tokens=25, context_length=100, + cwd=str(tmp_path / "proj"), + ) + (tmp_path / "proj").mkdir(exist_ok=True) + assert "gpt-5.4" in out + assert "25%" in out + + + +def test_build_footer_passes_extended_runtime_metadata(): + snapshot = AccountUsageSnapshot( + provider="openai-codex", + source="usage_api", + fetched_at=datetime.now(timezone.utc), + plan="Team", + windows=( + AccountUsageWindow(label="5h", used_percent=20, reset_at=None), + AccountUsageWindow(label="7d", used_percent=30, reset_at=None), + ), + ) + out = build_footer_line( + user_config={ + "display": { + "runtime_footer": { + "enabled": True, + "underline": True, + "fields": ["provider", "account", "model", "context", "quota"], + } + } + }, + platform_key="feishu", + model="openai/gpt-5.5", + provider="openai-codex", + account_label="team-main", + context_tokens=18_200, + context_length=400_000, + account_usage=snapshot, + cwd="", + ) + assert out == "──────────────\nopenai-codex · team-main · gpt-5.5 · ctx 18.2K/400K · 5h 80% · 7d 70%" + + +def test_build_footer_prefers_profile_scoped_resolved_config(): + out = build_footer_line( + user_config={"display": {"runtime_footer": {"enabled": False}}}, + platform_key="telegram", + resolved_config={"enabled": True, "fields": ["model"], "underline": False}, + model="openai/gpt-5.5", + context_tokens=0, + context_length=None, + cwd="", + ) + + assert out == "gpt-5.5" + def test_build_footer_per_platform_off_suppresses(): user = { @@ -279,3 +643,16 @@ def test_format_footer_served_model_is_opt_in_and_skips_same_model(): assert format_runtime_footer( model="gpt-5.4", context_tokens=0, context_length=None, cwd="/x", served_model=None, fields=["served_model"]) == "" + +def test_build_footer_no_data_returns_empty_even_when_enabled(): + # Enabled, but context_length is None AND cwd empty AND model empty ⇒ no fields + out = build_footer_line( + user_config={"display": {"runtime_footer": {"enabled": True}}}, + platform_key="telegram", + model="", + context_tokens=0, context_length=None, + cwd="", + ) + # With no TERMINAL_CWD env either + if not os.environ.get("TERMINAL_CWD"): + assert out == "" diff --git a/tests/gateway/test_runtime_footer_runner_wiring.py b/tests/gateway/test_runtime_footer_runner_wiring.py new file mode 100644 index 0000000000000..33b1d6f3566a9 --- /dev/null +++ b/tests/gateway/test_runtime_footer_runner_wiring.py @@ -0,0 +1,61 @@ +"""Regression tests for runtime-footer producer/consumer wiring.""" + +from pathlib import Path +from types import SimpleNamespace + +import gateway.runtime_footer_usage as usage +import hermes_constants +from gateway.run_turn_runner import _resolve_runtime_footer_metadata + + +def test_runner_resolves_footer_usage_inside_runtime_scope(monkeypatch): + calls = [] + monkeypatch.setattr( + usage, + "get_cached", + lambda *args, **kwargs: calls.append((args, kwargs)) or "snapshot", + ) + monkeypatch.setattr(hermes_constants, "get_hermes_home", lambda: Path("/profiles/routed")) + + result = _resolve_runtime_footer_metadata( + SimpleNamespace( + provider="anthropic", + base_url="https://api.anthropic.com", + api_key="runtime-secret", + ), + {"display": {"runtime_footer": {"enabled": True, "fields": ["account"]}}}, + "feishu", + ) + + assert result["provider"] == "anthropic" + assert result["base_url"] == "https://api.anthropic.com" + assert result["account_usage"] == "snapshot" + assert result["footer_config"]["fields"] == ["account"] + assert "api_key" not in result + assert calls == [ + ( + ("anthropic",), + { + "base_url": "https://api.anthropic.com", + "api_key": "runtime-secret", + "hermes_home": "/profiles/routed", + }, + ) + ] + + +def test_runner_does_not_fetch_when_footer_usage_is_not_requested(monkeypatch): + monkeypatch.setattr( + usage, + "get_cached", + lambda *_args, **_kwargs: (_ for _ in ()).throw(AssertionError("unexpected refresh")), + ) + + result = _resolve_runtime_footer_metadata( + SimpleNamespace(provider="anthropic", base_url="", api_key="secret"), + {"display": {"runtime_footer": {"enabled": True, "fields": ["model"]}}}, + "feishu", + ) + + assert result["account_usage"] is None + assert "api_key" not in result diff --git a/tests/gateway/test_runtime_footer_usage_cache.py b/tests/gateway/test_runtime_footer_usage_cache.py new file mode 100644 index 0000000000000..87d435a5e266c --- /dev/null +++ b/tests/gateway/test_runtime_footer_usage_cache.py @@ -0,0 +1,191 @@ +"""Regression tests for runtime-footer provider/account/quota wiring helpers.""" + +from __future__ import annotations + +from pathlib import Path +from types import SimpleNamespace + +import gateway.runtime_footer_usage as usage +from agent.account_usage import AccountUsageSnapshot + + +def _reset_cache(): + usage.clear() + + +def test_footer_account_usage_cache_key_is_profile_and_credential_scoped(monkeypatch): + monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/alice")) + a = usage.cache_key( + "openai-codex", + base_url="https://chatgpt.com/backend-api/codex", + api_key="token-a", + ) + b = usage.cache_key( + "openai-codex", + base_url="https://chatgpt.com/backend-api/codex", + api_key="token-b", + ) + c = usage.cache_key( + "openai-codex", + base_url="https://chatgpt.com/backend-api/codex/", + api_key="token-a", + ) + monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/bob")) + d = usage.cache_key( + "openai-codex", + base_url="https://chatgpt.com/backend-api/codex", + api_key="token-a", + ) + + assert a != b + assert a == c + assert a != d + assert "token-a" not in repr(a) + + +def test_footer_account_usage_cache_key_accepts_captured_profile_home(monkeypatch): + monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/default")) + + key = usage.cache_key( + "openai-codex", + base_url="https://chatgpt.com/backend-api/codex", + api_key="token-a", + hermes_home=Path("/profiles/alice"), + ) + + assert key[0] == "/profiles/alice" + + +def test_cold_cache_schedules_once_and_returns_immediately(monkeypatch): + _reset_cache() + started = [] + monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/alice")) + monkeypatch.setattr(usage.time, "monotonic", lambda: 100.0) + monkeypatch.setattr( + usage, + "_start_refresh", + lambda *args: started.append(args), + ) + + first = usage.get_cached( + "openai-codex", base_url="https://example.invalid", api_key="runtime-token" + ) + second = usage.get_cached( + "openai-codex", base_url="https://example.invalid", api_key="runtime-token" + ) + + assert first is None + assert second is None + assert len(started) == 1 + assert started[0][1:] == ( + "openai-codex", + "https://example.invalid", + "runtime-token", + ) + _reset_cache() + + +def test_fresh_cache_returns_snapshot_without_scheduling(monkeypatch): + _reset_cache() + snapshot = AccountUsageSnapshot( + provider="openai-codex", + source="usage_api", + fetched_at=None, + plan="Plus", + windows=(), + ) + monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/alice")) + key = usage.cache_key( + "openai-codex", base_url="https://example.invalid", api_key="runtime-token" + ) + usage._CACHE[key] = (100.0, snapshot) + monkeypatch.setattr(usage.time, "monotonic", lambda: 110.0) + monkeypatch.setattr( + usage, + "_start_refresh", + lambda *_args: (_ for _ in ()).throw(AssertionError("fresh cache refreshed")), + ) + + result = usage.get_cached( + "openai-codex", base_url="https://example.invalid", api_key="runtime-token" + ) + + assert result is snapshot + _reset_cache() + + +def test_stale_cache_returns_stale_snapshot_while_refreshing(monkeypatch): + _reset_cache() + snapshot = SimpleNamespace(windows=(SimpleNamespace(label="Session"),)) + started = [] + monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/alice")) + key = usage.cache_key( + "openai-codex", base_url="https://example.invalid", api_key="runtime-token" + ) + usage._CACHE[key] = (1.0, snapshot) + monkeypatch.setattr(usage.time, "monotonic", lambda: 500.0) + monkeypatch.setattr( + usage, + "_start_refresh", + lambda *args: started.append(args), + ) + + result = usage.get_cached( + "openai-codex", base_url="https://example.invalid", api_key="runtime-token" + ) + + assert result is snapshot + assert len(started) == 1 + _reset_cache() + + +def test_refresh_uses_live_credential_and_updates_cache(monkeypatch): + _reset_cache() + calls = [] + snapshot = SimpleNamespace(windows=(SimpleNamespace(label="Session"),)) + monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/alice")) + key = usage.cache_key( + "openai-codex", base_url="https://example.invalid", api_key="runtime-token" + ) + usage._REFRESHING.add(key) + monkeypatch.setattr( + usage, + "fetch_account_usage", + lambda *args, **kwargs: calls.append((args, kwargs)) or snapshot, + ) + monkeypatch.setattr(usage.time, "monotonic", lambda: 321.0) + + usage._refresh( + key, "openai-codex", "https://example.invalid", "runtime-token" + ) + + assert usage._CACHE[key] == (321.0, snapshot) + assert key not in usage._REFRESHING + assert calls == [ + ( + ("openai-codex",), + {"base_url": "https://example.invalid", "api_key": "runtime-token"}, + ) + ] + _reset_cache() + + +def test_refresh_failure_preserves_last_good_snapshot(monkeypatch): + _reset_cache() + stale = SimpleNamespace(windows=(SimpleNamespace(label="Session"),)) + monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/alice")) + key = usage.cache_key( + "openai-codex", base_url="https://example.invalid", api_key="runtime-token" + ) + usage._CACHE[key] = (1.0, stale) + usage._REFRESHING.add(key) + monkeypatch.setattr(usage, "fetch_account_usage", lambda *_a, **_kw: None) + monkeypatch.setattr(usage.time, "monotonic", lambda: 500.0) + + usage._refresh( + key, "openai-codex", "https://example.invalid", "runtime-token" + ) + + assert usage._CACHE[key] == (500.0, stale) + assert key not in usage._REFRESHING + _reset_cache() diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 2dec21b039bba..b6a28ee9df297 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -2324,6 +2324,7 @@ display: runtime_footer: enabled: true fields: ["model", "context_pct", "cwd"] # order shown; drop any to hide + underline: false # add a separator before the footer ``` Supported fields: @@ -2335,8 +2336,13 @@ Supported fields: | `latency` | Wall-clock duration of the turn | `22s`, `1m05s` | | `served_model` | The model that actually answered, when it differs from the one you configured: the deployment a routing proxy reported in its `x-litellm-model-id` (or `x-litellm-model-api-base`) response header, or the fallback model Hermes switched to for the turn | `hermes-router → gpt-4o-2024-11-20` | | `cwd` | Home-relative working directory | `~` | +| `provider` | Provider serving the turn | `anthropic` | +| `account` | Account or plan label from provider usage | `Pro` | +| `context` | Last-call context tokens as used/total | `8.2K/128K` | +| `quota` | Provider usage window and remaining percentage | `5h 72%` | +| `reasoning` | Reasoning effort used for the turn | `high` | -The default field set is `["model", "context_pct", "cwd"]`. `latency` and `served_model` are opt-in — add them to `fields` to use them. `served_model` renders nothing when the served model is the configured one (or when the proxy sends no such header), so behind a routing proxy or an active fallback it is the field that makes the switch visible. Fields whose data is unavailable are skipped silently rather than rendering an empty slot. +The default field set is `["model", "context_pct", "cwd"]`. `latency`, `served_model`, `provider`, `account`, `context`, `quota`, and `reasoning` are opt-in — add them to `fields` to use them. `served_model` renders nothing when the served model is the configured one (or when the proxy sends no such header), so behind a routing proxy or an active fallback it is the field that makes the switch visible. Account and quota data is fetched with a non-blocking stale-while-revalidate cache; it may be absent on the first turn and appear after the background refresh completes. Fields whose data is unavailable are skipped silently rather than rendering an empty slot. Set `underline: true` to insert a separator before the footer. The `/footer` slash command toggles this at runtime in any session.