diff --git a/agent/account_usage.py b/agent/account_usage.py
index e5989e26b92a0..33816a7b738fb 100644
--- a/agent/account_usage.py
+++ b/agent/account_usage.py
@@ -2,6 +2,8 @@
import logging
import math
+from decimal import Decimal, InvalidOperation
+from urllib.parse import urlparse
from dataclasses import dataclass
from datetime import datetime, timezone
from typing import TYPE_CHECKING, Any, Callable, Optional
@@ -385,7 +387,8 @@ def _get_json(url: str, headers: dict[str, str], *, timeout: float) -> dict:
def _usage_windows(
- source: dict, mapping: tuple[tuple[str, str], ...], used_key: str, reset_key: str, *, fraction: bool = False
+ source: dict, mapping: tuple[tuple[str, str], ...], used_key: str, reset_key: str, *, fraction: bool = False,
+ label_fn: Optional[Callable[[dict, str], str]] = None,
) -> list[AccountUsageWindow]:
"""Build windows from ``source[key][used_key]``; ``fraction`` scales values <= 1 to percent."""
windows: list[AccountUsageWindow] = []
@@ -397,7 +400,7 @@ def _usage_windows(
used = float(used)
if fraction and used <= 1:
used *= 100
- windows.append(AccountUsageWindow(label=label, used_percent=used, reset_at=_parse_dt(window.get(reset_key))))
+ windows.append(AccountUsageWindow(label=label_fn(window, label) if label_fn else label, used_percent=used, reset_at=_parse_dt(window.get(reset_key))))
return windows
@@ -587,7 +590,7 @@ def redeem_codex_reset_credit(
def _fetch_anthropic_account_usage(
base_url: Optional[str] = None, api_key: Optional[str] = None
) -> Optional[AccountUsageSnapshot]:
- token = (resolve_anthropic_token() or "").strip()
+ token = (str(api_key or "").strip() or (resolve_anthropic_token() or "").strip())
if not token:
return None
if not _is_oauth_token(token):
@@ -645,9 +648,104 @@ def _data(path: str) -> dict:
return _snapshot("openrouter", "credits_api", windows, details)
+def _money_symbol(currency: str) -> str:
+ normalized = currency.strip().upper()
+ if normalized == "CNY":
+ return "¥"
+ if normalized == "USD":
+ return "$"
+ return f"{normalized} " if normalized else ""
+
+
+def _decimal_or_none(value: Any) -> Optional[Decimal]:
+ try:
+ parsed = Decimal(str(value).strip())
+ except (InvalidOperation, ValueError, TypeError):
+ return None
+ return parsed if parsed.is_finite() else None
+
+
+def _format_money(value: Decimal, currency: str) -> str:
+ return f"{_money_symbol(currency)}{value:.2f}"
+
+
+def _is_deepseek_base_url(base_url: Optional[str]) -> bool:
+ try:
+ host = urlparse(str(base_url or "")).hostname or ""
+ except Exception:
+ return False
+ host = host.lower().strip(".")
+ return host == "deepseek.com" or host.endswith(".deepseek.com")
+
+
+def _resolve_deepseek_balance_url(base_url: Optional[str]) -> str:
+ parsed = urlparse(str(base_url or "").strip() or "https://api.deepseek.com")
+ scheme = parsed.scheme or "https"
+ netloc = parsed.netloc or parsed.path or "api.deepseek.com"
+ return f"{scheme}://{netloc.rstrip('/')}/user/balance"
+
+
+def _fetch_deepseek_account_usage(base_url: Optional[str], api_key: Optional[str]) -> Optional[AccountUsageSnapshot]:
+ if api_key:
+ runtime = {
+ "base_url": (base_url or "https://api.deepseek.com").strip(),
+ "api_key": str(api_key).strip(),
+ }
+ else:
+ runtime = resolve_runtime_provider(
+ requested="deepseek",
+ explicit_base_url=base_url,
+ explicit_api_key=api_key,
+ )
+ token = str(runtime.get("api_key", "") or "").strip()
+ if not token:
+ return None
+ resolved_base_url = str(runtime.get("base_url", "") or base_url or "https://api.deepseek.com")
+ headers = {
+ "Authorization": f"Bearer {token}",
+ "Accept": "application/json",
+ }
+ with httpx.Client(timeout=10.0) as client:
+ response = client.get(_resolve_deepseek_balance_url(resolved_base_url), headers=headers)
+ response.raise_for_status()
+ payload = response.json() or {}
+ details: list[str] = []
+ for info in payload.get("balance_infos") or []:
+ if not isinstance(info, dict):
+ continue
+ currency = str(info.get("currency") or "").strip().upper()
+ total = _decimal_or_none(info.get("total_balance"))
+ if total is None:
+ continue
+ parts = [f"Balance: {_format_money(total, currency)}"]
+ if currency:
+ parts[0] += f" {currency}"
+ subparts: list[str] = []
+ granted = _decimal_or_none(info.get("granted_balance"))
+ topped_up = _decimal_or_none(info.get("topped_up_balance"))
+ if granted is not None and granted > 0:
+ subparts.append(f"granted {_format_money(granted, currency)}")
+ if topped_up is not None and topped_up > 0:
+ subparts.append(f"topped up {_format_money(topped_up, currency)}")
+ if subparts:
+ parts.append(f"({', '.join(subparts)})")
+ details.append(" ".join(parts))
+ unavailable_reason = None
+ if payload.get("is_available") is False and not details:
+ unavailable_reason = "DeepSeek API balance is insufficient."
+ return AccountUsageSnapshot(
+ provider="deepseek",
+ source="balance_api",
+ fetched_at=_utc_now(),
+ title="Account balance",
+ details=tuple(details),
+ unavailable_reason=unavailable_reason,
+ )
+
+
_USAGE_FETCHERS: dict[str, Callable[[Optional[str], Optional[str]], Optional[AccountUsageSnapshot]]] = {
"openai-codex": _fetch_codex_account_usage, "anthropic": _fetch_anthropic_account_usage,
- "openrouter": _fetch_openrouter_account_usage,
+ "openrouter": _fetch_openrouter_account_usage, "deepseek": _fetch_deepseek_account_usage,
}
@@ -675,7 +773,12 @@ def _call_plugin_usage_hook(profile, base_url: Optional[str], api_key: Optional[
def fetch_account_usage(
provider: Optional[str], *, base_url: Optional[str] = None, api_key: Optional[str] = None,
) -> Optional[AccountUsageSnapshot]:
- fetcher = _USAGE_FETCHERS.get(str(provider or "").strip().lower())
+ normalized = str(provider or "").strip().lower()
+ if normalized in {"", "auto", "custom"} and not _is_deepseek_base_url(base_url):
+ return None
+ fetcher = _USAGE_FETCHERS.get(normalized)
+ if fetcher is None and _is_deepseek_base_url(base_url):
+ fetcher = _fetch_deepseek_account_usage
try:
if fetcher:
return fetcher(base_url, api_key)
diff --git a/gateway/run_turn.py b/gateway/run_turn.py
index 1484ea2059def..71f5a52d74c5a 100644
--- a/gateway/run_turn.py
+++ b/gateway/run_turn.py
@@ -1642,18 +1642,44 @@ def _hmwa_prepend_reasoning(self, agent_result, response, source, _intentional_s
def _hmwa_runtime_footer_line(self, agent_result, source, _turn_seconds):
"""Runtime-metadata footer for the FINAL message of the turn; off by default
- (display.runtime_footer.enabled=false)."""
+ (display.runtime_footer.enabled=false). Extends the default footer with
+ opt-in provider/account/quota/reasoning fields when configured."""
from gateway.run import _load_gateway_config, _platform_config_key, _terminal_scope_cwd
try:
- from gateway.runtime_footer import build_footer_line as _bfl
+ from gateway.runtime_footer import build_footer_line as _bfl, resolve_footer_config as _rfc
+ _user_config = _load_gateway_config()
+ _platform_key = _platform_config_key(source.platform)
+ _account_usage = agent_result.get("account_usage")
+ _account_label = None
+ if _account_usage is not None:
+ _account_label = (
+ getattr(_account_usage, "account_label", None)
+ or getattr(_account_usage, "plan", None)
+ )
+ # Usage is resolved by the producer in run_turn_runner.py while the
+ # routed profile scope and live credential are still available.
+ _footer_cfg = agent_result.get("footer_config") or _rfc(_user_config, _platform_key)
+ _reasoning_effort = agent_result.get("reasoning_effort")
+ if _reasoning_effort is None:
+ _reasoning_cfg = getattr(self, "_reasoning_config", None)
+ if isinstance(_reasoning_cfg, dict):
+ if _reasoning_cfg.get("enabled") is False:
+ _reasoning_effort = "none"
+ else:
+ _reasoning_effort = _reasoning_cfg.get("effort")
return _bfl(
- user_config=_load_gateway_config(),
- platform_key=_platform_config_key(source.platform), model=agent_result.get("model"),
+ user_config=_user_config,
+ platform_key=_platform_key, model=agent_result.get("model"),
context_tokens=agent_result.get("last_prompt_tokens", 0) or 0,
context_length=agent_result.get("context_length") or None,
cwd=_terminal_scope_cwd(""), turn_seconds=_turn_seconds,
requested_model=agent_result.get("requested_model"),
served_model=agent_result.get("served_model"),
+ provider=agent_result.get("provider"),
+ account_label=_account_label,
+ account_usage=_account_usage,
+ reasoning_effort=_reasoning_effort,
+ resolved_config=_footer_cfg,
)
except Exception as _footer_err:
logger.debug("runtime_footer build failed: %s", _footer_err)
diff --git a/gateway/run_turn_runner.py b/gateway/run_turn_runner.py
index 1b2e2c0bd31b0..eefaeaef02269 100644
--- a/gateway/run_turn_runner.py
+++ b/gateway/run_turn_runner.py
@@ -72,6 +72,38 @@ class _ExecApprovalDeclined(RuntimeError):
"""
+def _resolve_runtime_footer_metadata(agent, user_config: dict | None, platform_key: str) -> dict:
+ """Resolve footer usage while the turn's routed profile scope is active.
+
+ The API credential is used only to key/schedule the background usage fetch;
+ it is deliberately not returned in the result consumed by ``run_turn.py``.
+ """
+ from gateway.runtime_footer import resolve_footer_config
+ from gateway.runtime_footer_usage import get_cached
+ from hermes_constants import get_hermes_home
+
+ footer_config = resolve_footer_config(user_config, platform_key)
+ fields = set(footer_config.get("fields") or ())
+ needs_usage = footer_config.get("enabled") and bool(fields & {"account", "quota"})
+ provider = getattr(agent, "provider", None) if agent is not None else None
+ base_url = getattr(agent, "base_url", None) if agent is not None else None
+ api_key = getattr(agent, "api_key", None) if agent is not None else None
+ account_usage = None
+ if needs_usage and provider:
+ account_usage = get_cached(
+ provider,
+ base_url=base_url,
+ api_key=api_key,
+ hermes_home=str(get_hermes_home()),
+ )
+ return {
+ "provider": provider,
+ "base_url": base_url,
+ "account_usage": account_usage,
+ "footer_config": footer_config,
+ }
+
+
class TurnRunner:
"""Per-turn collaborator carrying ``GatewayRunner._run_agent_inner``'s tool-progress callbacks."""
@@ -1968,6 +2000,14 @@ def run_sync(self):
"model": getattr(agent, "model", None) if agent else None,
"context_length": (getattr(comp, "context_length", 0) or 0) if has_comp else 0,
}
+ footer_metadata = _resolve_runtime_footer_metadata(
+ agent,
+ ctx.user_config,
+ platform_key,
+ )
+ footer_metadata["reasoning_effort"] = (
+ getattr(runner, "_reasoning_config", {}) or {}
+ ).get("effort")
compacted_in_place, effective_session_id, history_offset = self._sync_session_after_run(agent_history)
# failure_reason must survive the empty-response path too (TUI billing, transient-failure
# persistence). compression_deferred (soft lock-contention defer) is distinct from
@@ -1984,6 +2024,7 @@ def run_sync(self):
"tools": ctx.tools_holder[0] or [],
"history_offset": history_offset, "compacted_in_place": compacted_in_place, "session_id": effective_session_id,
**usage,
+ **footer_metadata,
}
if not final_response:
final_response = _normalize_empty_agent_response(result, final_response or "", history_len=len(agent_history))
diff --git a/gateway/runtime_footer.py b/gateway/runtime_footer.py
index 54c923c8209da..da1938bf2ec4b 100644
--- a/gateway/runtime_footer.py
+++ b/gateway/runtime_footer.py
@@ -1,17 +1,25 @@
"""Gateway runtime-metadata footer (model · context % · cwd), off by default to keep replies
-minimal. Config: ``display.runtime_footer: {enabled: bool, fields: [model, context_pct, cwd]}``
-(order shown; drop any to hide), per-platform override ``display.platforms.
.runtime_footer``,
-toggled by ``/footer on|off``. Fields: ``model`` (vendor prefix dropped), ``context_pct`` (last-call
-occupancy), ``latency`` (turn wall-clock, opt-in — NOT in the default set so an unset ``fields``
-renders exactly as before), ``served_model`` (opt-in, ``alias → served``: the deployment a routing
-proxy reported via ``x-litellm-model-id`` / ``x-litellm-model-api-base``, or Hermes' own fallback
-route; skipped when the served model is the requested one), ``cwd`` (home-relative). ``gateway/run.py`` appends the footer to the
-final response only (never to tool-progress or streaming partials); when streaming already
-delivered the text, it goes out as a trailing message via ``send_trailing_footer()``."""
+minimal. Config: ``display.runtime_footer: {enabled: bool, fields: [model, context_pct, cwd],
+underline: bool}`` (order shown; drop any to hide), per-platform override
+``display.platforms.
.runtime_footer``, toggled by ``/footer on|off``. Fields: ``model`` (vendor
+prefix dropped), ``context_pct`` (last-call occupancy), ``context`` (compact ``ctx used/limit``),
+``latency`` (turn wall-clock, opt-in — NOT in the default set so an unset ``fields`` renders exactly
+as before), ``served_model`` (opt-in, ``alias → served``: the deployment a routing proxy reported via
+``x-litellm-model-id`` / ``x-litellm-model-api-base``, or Hermes' own fallback route; skipped when
+the served model is the requested one), ``reasoning`` (compact reasoning-effort label), ``provider``
+(inference provider id), ``account`` (account/plan label, any ``-`` prefix stripped), and
+``quota`` (one part per account-usage window — ``5h``/``7d N%`` plus a compact reset — followed by a
+compact balance line when the snapshot carries one). ``underline: true`` prefixes the footer with a
+separator line. ``gateway/run.py`` appends the footer to the final response only (never to
+tool-progress or streaming partials); when streaming already delivered the text, it goes out as a
+trailing message via ``send_trailing_footer()``."""
from __future__ import annotations
+import math
import os
+import re
+from datetime import datetime, timezone
from typing import Any, Iterable, Optional
_DEFAULT_FIELDS: tuple[str, ...] = ("model", "context_pct", "cwd")
@@ -33,8 +41,10 @@ def _home_relative_cwd(cwd: str) -> str:
def _model_short(model: Optional[str]) -> str:
- """Drop ``vendor/`` prefix (``openai/gpt-5.4`` → ``gpt-5.4``)."""
- return model.rsplit("/", 1)[-1] if model else ""
+ """Drop ``vendor/`` prefix for readability (``openai/gpt-5.4`` → ``gpt-5.4``)."""
+ if not model:
+ return ""
+ return model.rsplit("/", 1)[-1]
def _env_cwd() -> str:
@@ -45,20 +55,70 @@ def _env_cwd() -> str:
return terminal_env("TERMINAL_CWD", "")
-def resolve_footer_config(user_config: dict[str, Any] | None, platform_key: str | None = None) -> dict[str, Any]:
- """Resolve effective footer config: defaults (enabled=False) <
- ``display.runtime_footer`` < ``display.platforms..runtime_footer``."""
- resolved = {"enabled": False, "fields": list(_DEFAULT_FIELDS)}
+# Compact labels for agent.reasoning_effort / runtime reasoning_config.
+# Keep these short so the footer stays one line on mobile clients.
+_REASONING_ABBREV = {
+ "none": "off",
+ "minimal": "min",
+ "low": "low",
+ "medium": "med",
+ "high": "high",
+ "xhigh": "xhi",
+ "max": "max",
+ "ultra": "ult",
+}
+
+
+def _reasoning_short(effort: Optional[str]) -> str:
+ """Return a compact reasoning-effort label, or "" when unknown/empty."""
+ raw = str(effort or "").strip().lower()
+ if not raw:
+ return ""
+ if raw in {"false", "disabled", "off"}:
+ return _REASONING_ABBREV["none"]
+ if raw in _REASONING_ABBREV:
+ return _REASONING_ABBREV[raw]
+ # Unknown but non-empty values still surface compactly so a new level
+ # is visible before we teach the map about it.
+ return raw[:6]
+
+
+def resolve_footer_config(
+ user_config: dict[str, Any] | None,
+ platform_key: str | None = None,
+) -> dict[str, Any]:
+ """Resolve effective runtime-footer config for *platform_key*.
+
+ Merge order (later wins):
+ 1. Built-in defaults (enabled=False)
+ 2. ``display.runtime_footer``
+ 3. ``display.platforms..runtime_footer``
+ """
+ resolved = {"enabled": False, "fields": list(_DEFAULT_FIELDS), "underline": False}
cfg = (user_config or {}).get("display") or {}
- plat_cfg = (cfg.get("platforms") or {}).get(platform_key) if platform_key else None
- sections = [cfg.get("runtime_footer"), plat_cfg.get("runtime_footer") if isinstance(plat_cfg, dict) else None]
- for section in sections:
- if not isinstance(section, dict):
- continue
- if "enabled" in section:
- resolved["enabled"] = bool(section.get("enabled"))
- if isinstance(section.get("fields"), list) and section["fields"]:
- resolved["fields"] = [str(f) for f in section["fields"]]
+
+ global_cfg = cfg.get("runtime_footer")
+ if isinstance(global_cfg, dict):
+ if "enabled" in global_cfg:
+ resolved["enabled"] = bool(global_cfg.get("enabled"))
+ if "underline" in global_cfg:
+ resolved["underline"] = bool(global_cfg.get("underline"))
+ if isinstance(global_cfg.get("fields"), list) and global_cfg["fields"]:
+ resolved["fields"] = [str(f) for f in global_cfg["fields"]]
+
+ if platform_key:
+ platforms = cfg.get("platforms") or {}
+ plat_cfg = platforms.get(platform_key)
+ if isinstance(plat_cfg, dict):
+ plat_footer = plat_cfg.get("runtime_footer")
+ if isinstance(plat_footer, dict):
+ if "enabled" in plat_footer:
+ resolved["enabled"] = bool(plat_footer.get("enabled"))
+ if "underline" in plat_footer:
+ resolved["underline"] = bool(plat_footer.get("underline"))
+ if isinstance(plat_footer.get("fields"), list) and plat_footer["fields"]:
+ resolved["fields"] = [str(f) for f in plat_footer["fields"]]
+
return resolved
@@ -73,11 +133,158 @@ def _format_latency(seconds: float) -> str:
return f"{m}m{sec:02d}s"
+def _compact_number(value: int | float) -> str:
+ try:
+ n = float(value)
+ except Exception:
+ return str(value)
+ if abs(n) >= 1_000_000:
+ text = f"{n / 1_000_000:.1f}M"
+ elif abs(n) >= 1_000:
+ text = f"{n / 1_000:.1f}K"
+ else:
+ text = str(int(n))
+ return text.replace(".0K", "K").replace(".0M", "M")
+
+
+def _compact_reset(dt: Any) -> str:
+ if not dt:
+ return ""
+ if isinstance(dt, str):
+ try:
+ dt = datetime.fromisoformat(dt.strip().replace("Z", "+00:00"))
+ except Exception:
+ return ""
+ if not isinstance(dt, datetime):
+ return ""
+ if dt.tzinfo is None:
+ dt = dt.replace(tzinfo=timezone.utc)
+ seconds = int((dt - datetime.now(timezone.utc)).total_seconds())
+ if seconds <= 0:
+ return "now"
+ hours, rem = divmod(seconds, 3600)
+ minutes = rem // 60
+ if hours >= 24:
+ days, rem_hours = divmod(math.ceil(seconds / 3600), 24)
+ return f"{days}d" + (f"{rem_hours}h" if rem_hours else "")
+ if hours > 0:
+ return f"{hours}h" + (f"{minutes}m" if minutes else "")
+ return f"{minutes}m" if minutes else "<1m"
+
+
+def _quota_label(window: Any, provider: Optional[str] = None, model: Optional[str] = None) -> str:
+ """Return a compact quota-window label for footer display.
+
+ The detailed ``/usage`` command keeps provider wording. The footer is space
+ constrained, so normalize the common OAuth/Codex rolling windows to the
+ short labels users expect while leaving unknown provider windows intact.
+ """
+ raw = str(getattr(window, "label", "") or "quota").strip() or "quota"
+ label = raw.lower().replace("_", "-")
+ model_text = str(model or "").lower()
+
+ if "opus" in label:
+ return "opus7d"
+ if "sonnet" in label:
+ return "sonnet7d"
+ if label in {
+ "5h",
+ "5-hour",
+ "5 hour",
+ "five-hour",
+ "five hour",
+ "current session",
+ "session",
+ "primary",
+ "primary-window",
+ "primary window",
+ }:
+ return "5h"
+ if label in {
+ "7d",
+ "7-day",
+ "7 day",
+ "seven-day",
+ "seven day",
+ "current week",
+ "week",
+ "weekly",
+ "secondary",
+ "secondary-window",
+ "secondary window",
+ }:
+ return "7d"
+ if "week" in label:
+ if "opus" in model_text:
+ return "opus7d"
+ if "sonnet" in model_text:
+ return "sonnet7d"
+ return raw
+
+
+def _compact_quota_detail(detail: Any) -> str:
+ text = str(detail or "").strip()
+ if not text:
+ return ""
+ # Keep footer quota compact. Detailed breakdowns remain available via the
+ # usage renderer; the footer only needs the immediately useful balance.
+ if not re.match(r"^(credits\s+)?balance\s*:", text, flags=re.IGNORECASE):
+ return ""
+ text = re.sub(r"\s*\([^)]*\)\s*$", "", text).strip()
+ text = re.sub(r"^(credits\s+)?balance\s*:", "balance", text, flags=re.IGNORECASE).strip()
+ return text
+
+
+def _format_quota(account_usage: Any, *, provider: Optional[str] = None, model: Optional[str] = None) -> list[str]:
+ if not account_usage:
+ return []
+ provider = provider or getattr(account_usage, "provider", None)
+ parts: list[str] = []
+ for window in getattr(account_usage, "windows", ()) or ():
+ used = getattr(window, "used_percent", None)
+ if used is None:
+ continue
+ try:
+ remaining = max(0, round(100 - float(used)))
+ except Exception:
+ continue
+ label = _quota_label(window, provider=provider, model=model)
+ text = f"{label} {remaining}%"
+ reset = _compact_reset(getattr(window, "reset_at", None))
+ if reset:
+ text += f" {reset}"
+ parts.append(text)
+ for detail in getattr(account_usage, "details", ()) or ():
+ compact = _compact_quota_detail(detail)
+ if compact:
+ parts.append(compact)
+ return parts
+
+
+def _account_short(account_label: Optional[str], provider: Optional[str]) -> str:
+ raw = str(account_label or "").strip()
+ if not raw:
+ return ""
+ prov = str(provider or "").strip()
+ if prov:
+ for prefix in (prov, prov.replace("-", "_"), prov.replace("_", "-")):
+ for sep in ("-", "_"):
+ marker = prefix + sep
+ if raw.lower().startswith(marker.lower()):
+ return raw[len(marker):]
+ return raw
+
+
def format_runtime_footer(*, model: Optional[str], context_tokens: int,
context_length: Optional[int], cwd: Optional[str] = None,
turn_seconds: Optional[float] = None,
requested_model: Optional[str] = None, served_model: Optional[str] = None,
- fields: Iterable[str] = _DEFAULT_FIELDS) -> str:
+ fields: Iterable[str] = _DEFAULT_FIELDS,
+ provider: Optional[str] = None,
+ account_label: Optional[str] = None,
+ account_usage: Any = None,
+ reasoning_effort: Optional[str] = None,
+ underline: bool = False) -> str:
"""Render the footer line, or "" if no fields have data. Fields whose data is missing (and
unknown field names) are skipped silently — a partial footer beats ``?%`` or empty slots."""
def context_pct() -> str:
@@ -92,29 +299,76 @@ def served() -> str:
return f"{alias} → {served_model}"
return ""
+ def context_compact() -> str:
+ if context_length and context_length > 0 and context_tokens >= 0:
+ return f"ctx {_compact_number(context_tokens)}/{_compact_number(context_length)}"
+ return ""
+
+ def account() -> str:
+ label = (account_label
+ or getattr(account_usage, "account_label", None)
+ or getattr(account_usage, "plan", None))
+ return _account_short(str(label), provider) if label else ""
+
renderers = {
"model": lambda: _model_short(model),
"served_model": served,
+ "reasoning": lambda: _reasoning_short(reasoning_effort),
+ "reasoning_effort": lambda: _reasoning_short(reasoning_effort),
+ "effort": lambda: _reasoning_short(reasoning_effort),
+ "provider": lambda: str(provider) if provider else "",
+ "account": account,
+ "context": context_compact,
"context_pct": context_pct,
# Skipped when the caller did not measure (None) or the value is negative.
"latency": lambda: _format_latency(turn_seconds) if turn_seconds is not None and turn_seconds >= 0 else "",
"cwd": lambda: _home_relative_cwd(cwd or _env_cwd()),
}
- return _SEP.join(v for field in fields if (render := renderers.get(field)) and (v := render()))
+
+ parts: list[str] = []
+ for field in fields:
+ # ``quota`` expands to one part per account-usage window (plus any
+ # compact balance line), so it cannot be a single-string renderer.
+ if field == "quota":
+ parts.extend(_format_quota(account_usage, provider=provider, model=model))
+ continue
+ render = renderers.get(field)
+ if render is None:
+ continue
+ value = render()
+ if value:
+ parts.append(value)
+
+ if not parts:
+ return ""
+ line = _SEP.join(parts)
+ return f"──────────────\n{line}" if underline else line
def build_footer_line(*, user_config: dict[str, Any] | None, platform_key: str | None,
model: Optional[str], context_tokens: int, context_length: Optional[int],
cwd: Optional[str] = None, turn_seconds: Optional[float] = None,
- requested_model: Optional[str] = None, served_model: Optional[str] = None) -> str:
+ requested_model: Optional[str] = None, served_model: Optional[str] = None,
+ provider: Optional[str] = None, account_label: Optional[str] = None,
+ account_usage: Any = None, reasoning_effort: Optional[str] = None,
+ resolved_config: Optional[dict[str, Any]] = None) -> str:
"""Entry point for gateway/run.py: footer text, or "" when disabled / no data. Callers append it
to the final response themselves, preserving a single blank line of separation.
``turn_seconds`` is the caller-measured (``time.monotonic()``) run duration; ``None`` skips the
- ``latency`` field."""
- cfg = resolve_footer_config(user_config, platform_key)
+ ``latency`` field. ``resolved_config`` lets the turn runner hand over a footer config it already
+ resolved, instead of resolving it a second time."""
+ cfg = (
+ resolved_config
+ if isinstance(resolved_config, dict)
+ else resolve_footer_config(user_config, platform_key)
+ )
if not cfg.get("enabled"):
return ""
return format_runtime_footer(model=model, context_tokens=context_tokens,
context_length=context_length, cwd=cwd, turn_seconds=turn_seconds,
requested_model=requested_model, served_model=served_model,
- fields=cfg.get("fields") or _DEFAULT_FIELDS)
+ fields=cfg.get("fields") or _DEFAULT_FIELDS,
+ provider=provider, account_label=account_label,
+ account_usage=account_usage,
+ reasoning_effort=reasoning_effort,
+ underline=bool(cfg.get("underline")))
diff --git a/gateway/runtime_footer_usage.py b/gateway/runtime_footer_usage.py
new file mode 100644
index 0000000000000..f01c3d7e734f2
--- /dev/null
+++ b/gateway/runtime_footer_usage.py
@@ -0,0 +1,95 @@
+"""Non-blocking account-usage cache for the runtime footer.
+
+This module deliberately contains no gateway-runner state. Usage requests are
+slow and optional, so callers receive the last snapshot immediately while a
+single daemon thread refreshes stale entries in the background.
+"""
+
+from __future__ import annotations
+
+import hashlib
+import threading
+import time
+from pathlib import Path
+from typing import Any
+
+from agent.account_usage import fetch_account_usage
+from hermes_constants import get_hermes_home
+
+_CACHE: dict[tuple[str, str, str, str], tuple[float, Any]] = {}
+_REFRESHING: set[tuple[str, str, str, str]] = set()
+_LOCK = threading.Lock()
+_TTL_SECONDS = 90.0
+_MAX_ENTRIES = 64
+
+
+def cache_key(provider: str | None, *, base_url: str | None = None,
+ api_key: str | None = None,
+ hermes_home: str | Path | None = None) -> tuple[str, str, str, str]:
+ token = str(api_key or "").strip()
+ digest = hashlib.sha256(token.encode()).hexdigest()[:16] if token else ""
+ return (
+ str(hermes_home if hermes_home is not None else get_hermes_home()),
+ str(provider or "").strip().lower(),
+ str(base_url or "").strip().rstrip("/").lower(),
+ digest,
+ )
+
+
+def _refresh(key, provider, base_url, api_key) -> None:
+ try:
+ snapshot = fetch_account_usage(provider, base_url=base_url, api_key=api_key)
+ except Exception:
+ snapshot = None
+ with _LOCK:
+ previous = _CACHE.get(key)
+ # Stale-while-revalidate: a transient provider failure must not erase a
+ # previously useful quota value.
+ value = snapshot if snapshot is not None else (previous[1] if previous else None)
+ if key not in _CACHE and len(_CACHE) >= _MAX_ENTRIES:
+ oldest = min(_CACHE, key=lambda item: _CACHE[item][0])
+ _CACHE.pop(oldest, None)
+ _CACHE[key] = (time.monotonic(), value)
+ _REFRESHING.discard(key)
+
+
+def _start_refresh(key, provider, base_url, api_key) -> None:
+ try:
+ threading.Thread(
+ target=_refresh,
+ args=(key, provider, base_url, api_key),
+ name=f"runtime-footer-usage-{key[1] or 'unknown'}",
+ daemon=True,
+ ).start()
+ except Exception:
+ with _LOCK:
+ _REFRESHING.discard(key)
+
+
+def get_cached(provider: str | None, *, base_url: str | None = None,
+ api_key: str | None = None,
+ hermes_home: str | Path | None = None):
+ """Return cached usage and schedule at most one non-blocking refresh."""
+ normalized = str(provider or "").strip().lower()
+ if normalized in {"", "auto"}:
+ return None
+ key = cache_key(provider, base_url=base_url, api_key=api_key, hermes_home=hermes_home)
+ now = time.monotonic()
+ start = False
+ with _LOCK:
+ cached = _CACHE.get(key)
+ value = cached[1] if cached else None
+ fresh = cached is not None and now - cached[0] < _TTL_SECONDS
+ if not fresh and key not in _REFRESHING:
+ _REFRESHING.add(key)
+ start = True
+ if start:
+ _start_refresh(key, provider, base_url, api_key)
+ return value
+
+
+def clear() -> None:
+ """Clear cache, primarily for tests and profile teardown."""
+ with _LOCK:
+ _CACHE.clear()
+ _REFRESHING.clear()
diff --git a/tests/agent/test_account_usage.py b/tests/agent/test_account_usage.py
index 5566299d3ee72..591f5717ba89f 100644
--- a/tests/agent/test_account_usage.py
+++ b/tests/agent/test_account_usage.py
@@ -45,10 +45,12 @@ def codex_usage_payload():
"primary_window": {
"used_percent": 21,
"reset_at": 1779846359,
+ "limit_window_seconds": 18000,
},
"secondary_window": {
"used_percent": 4,
"reset_at": 1780230796,
+ "limit_window_seconds": 604800,
},
},
"credits": {"has_credits": False},
diff --git a/tests/agent/test_account_usage_fetch.py b/tests/agent/test_account_usage_fetch.py
index 439ca1ae18bd8..21277b98f95c8 100644
--- a/tests/agent/test_account_usage_fetch.py
+++ b/tests/agent/test_account_usage_fetch.py
@@ -72,6 +72,22 @@ def get(self, url, headers=None):
return _Response(self._payloads[url])
+
+class _RecordingClient:
+ def __init__(self, payload):
+ self._payload = payload
+ self.requests = []
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, exc_type, exc, tb):
+ return False
+
+ def get(self, url, headers=None):
+ self.requests.append((url, dict(headers or {})))
+ return _Response(self._payload)
+
def test_fetch_account_usage_codex(monkeypatch):
monkeypatch.setattr(
"agent.account_usage.resolve_codex_runtime_credentials",
@@ -112,7 +128,7 @@ def test_fetch_account_usage_codex(monkeypatch):
assert snapshot is not None
assert snapshot.plan == "Pro"
assert len(snapshot.windows) == 2
- assert snapshot.windows[0].label == "Session"
+ assert [window.label for window in snapshot.windows] == ["Session", "Weekly"]
assert snapshot.windows[0].used_percent == 15.0
assert snapshot.windows[0].reset_at == datetime.fromtimestamp(1_900_000_000, tz=timezone.utc)
assert "Credits balance: $12.50" in snapshot.details
@@ -158,6 +174,33 @@ def test_fetch_account_usage_prefers_builtin_fetcher_over_profile(monkeypatch):
assert fetch_account_usage("openrouter") is builtin
assert profile.calls == 0
+def test_fetch_account_usage_codex_labels_single_primary_by_actual_duration(monkeypatch):
+ monkeypatch.setattr(
+ "agent.account_usage.httpx.Client",
+ lambda timeout=15.0: _Client(
+ {
+ "plan_type": "pro",
+ "rate_limit": {
+ "primary_window": {
+ "used_percent": 95,
+ "reset_at": 1_900_000_000,
+ "limit_window_seconds": 604800,
+ },
+ "secondary_window": None,
+ },
+ }
+ ),
+ )
+
+ snapshot = fetch_account_usage(
+ "openai-codex",
+ base_url="https://chatgpt.com/backend-api/codex",
+ api_key="access-token",
+ )
+
+ assert snapshot is not None
+ assert [(window.label, window.used_percent) for window in snapshot.windows] == [("Weekly", 95.0)]
+
def test_fetch_account_usage_openrouter_uses_limit_remaining_and_ignores_deprecated_rate_limit(monkeypatch):
monkeypatch.setattr(
@@ -339,3 +382,90 @@ def probing_portal_fetch(*, force_fresh):
finally:
marker.reset(token)
assert seen == {"force_fresh": True, "marker": "profile-scope"}
+
+def test_fetch_account_usage_anthropic_prefers_explicit_runtime_key_over_resolver(monkeypatch):
+ client = _RecordingClient(
+ {
+ "five_hour": {"utilization": 0.25},
+ }
+ )
+ monkeypatch.setattr("agent.account_usage.httpx.Client", lambda timeout=15.0: client)
+ monkeypatch.setattr("agent.account_usage._is_oauth_token", lambda token: True)
+ monkeypatch.setattr(
+ "agent.account_usage.resolve_anthropic_token",
+ lambda: "rotated-pool-token",
+ )
+
+ snapshot = fetch_account_usage("anthropic", api_key="runtime-token")
+
+ assert snapshot is not None
+ assert snapshot.provider == "anthropic"
+ assert client.requests[0][1]["Authorization"] == "Bearer runtime-token"
+
+def test_fetch_account_usage_custom_deepseek_base_url_is_supported(monkeypatch):
+ client = _RecordingClient(
+ {
+ "is_available": True,
+ "balance_infos": [
+ {
+ "currency": "CNY",
+ "total_balance": "25.50",
+ "granted_balance": "0.00",
+ "topped_up_balance": "25.50",
+ }
+ ],
+ }
+ )
+ monkeypatch.setattr("agent.account_usage.httpx.Client", lambda timeout=10.0: client)
+
+ snapshot = fetch_account_usage(
+ "custom",
+ base_url="https://api.deepseek.com/beta/v1/",
+ api_key="deepseek-token",
+ )
+
+ assert snapshot is not None
+ assert snapshot.provider == "deepseek"
+ assert snapshot.details == ("Balance: ¥25.50 CNY (topped up ¥25.50)",)
+
+def test_fetch_account_usage_deepseek_balance_uses_official_endpoint(monkeypatch):
+ client = _RecordingClient(
+ {
+ "is_available": True,
+ "balance_infos": [
+ {
+ "currency": "CNY",
+ "total_balance": "110.00",
+ "granted_balance": "10.00",
+ "topped_up_balance": "100.00",
+ },
+ {
+ "currency": "USD",
+ "total_balance": "2.50",
+ "granted_balance": "0.00",
+ "topped_up_balance": "2.50",
+ },
+ ],
+ }
+ )
+ monkeypatch.setattr("agent.account_usage.httpx.Client", lambda timeout=10.0: client)
+
+ snapshot = fetch_account_usage(
+ "deepseek",
+ base_url="https://api.deepseek.com/v1",
+ api_key="deepseek-token",
+ )
+
+ assert snapshot is not None
+ assert snapshot.provider == "deepseek"
+ assert snapshot.source == "balance_api"
+ assert snapshot.details == (
+ "Balance: ¥110.00 CNY (granted ¥10.00, topped up ¥100.00)",
+ "Balance: $2.50 USD (topped up $2.50)",
+ )
+ assert client.requests == [
+ (
+ "https://api.deepseek.com/user/balance",
+ {"Authorization": "Bearer deepseek-token", "Accept": "application/json"},
+ )
+ ]
diff --git a/tests/gateway/test_runtime_footer.py b/tests/gateway/test_runtime_footer.py
index d4b619afe8a50..c2dd83d5a067e 100644
--- a/tests/gateway/test_runtime_footer.py
+++ b/tests/gateway/test_runtime_footer.py
@@ -3,10 +3,14 @@
from __future__ import annotations
+import os
+from datetime import datetime, timedelta, timezone
import pytest
+from agent.account_usage import AccountUsageSnapshot, AccountUsageWindow
from gateway.runtime_footer import (
+ _compact_reset,
_home_relative_cwd,
_model_short,
build_footer_line,
@@ -41,6 +45,16 @@ def test_home_relative_cwd_collapses_home(tmp_path, monkeypatch):
assert result == "~/projects/hermes"
+def test_home_relative_cwd_leaves_abs_path_alone(tmp_path, monkeypatch):
+ monkeypatch.setenv("HOME", str(tmp_path / "other"))
+ result = _home_relative_cwd(str(tmp_path / "outside" / "dir"))
+ assert result == str(tmp_path / "outside" / "dir")
+
+
+def test_home_relative_cwd_empty_returns_empty():
+ assert _home_relative_cwd("") == ""
+
+
# ---------------------------------------------------------------------------
# format_runtime_footer
# ---------------------------------------------------------------------------
@@ -73,10 +87,226 @@ def test_format_footer_skips_missing_context_length():
assert "/tmp/wd" in out
+def test_format_footer_context_pct_clamped_to_100():
+ out = format_runtime_footer(
+ model="m",
+ context_tokens=500_000, # way over
+ context_length=100_000,
+ cwd="",
+ fields=("context_pct",),
+ )
+ assert out == "100%"
+
+
+def test_format_footer_context_pct_never_negative():
+ out = format_runtime_footer(
+ model="m",
+ context_tokens=-50,
+ context_length=100,
+ cwd="",
+ fields=("context_pct",),
+ )
+ # Negative input => no field emitted (we require context_tokens >= 0)
+ assert out == ""
+
+
+def test_format_footer_empty_fields_returns_empty():
+ out = format_runtime_footer(
+ model="m", context_tokens=0, context_length=100,
+ cwd="/x", fields=(),
+ )
+ assert out == ""
+
+
+def test_format_footer_drops_cwd_when_empty(monkeypatch):
+ monkeypatch.delenv("TERMINAL_CWD", raising=False)
+ out = format_runtime_footer(
+ model="openai/gpt-5.4",
+ context_tokens=50, context_length=100,
+ cwd="",
+ fields=("model", "context_pct", "cwd"),
+ )
+ # cwd silently dropped; model + pct remain
+ assert out == "gpt-5.4 · 50%"
+
+
+def test_format_footer_custom_field_order():
+ out = format_runtime_footer(
+ model="openai/gpt-5.4",
+ context_tokens=50, context_length=100,
+ cwd="/opt/project",
+ fields=("context_pct", "model"), # swapped + no cwd
+ )
+ assert out == "50% · gpt-5.4"
+
+
+
+def test_format_footer_extended_fields_with_quota_and_underline():
+ snapshot = AccountUsageSnapshot(
+ provider="openai-codex",
+ source="usage_api",
+ fetched_at=datetime.now(timezone.utc),
+ plan="Team",
+ windows=(
+ AccountUsageWindow(label="5h", used_percent=20, reset_at=None),
+ AccountUsageWindow(label="7d", used_percent=30, reset_at=None),
+ ),
+ )
+ out = format_runtime_footer(
+ model="openai/gpt-5.5",
+ provider="openai-codex",
+ account_label="openai-codex-team-main",
+ context_tokens=21_800,
+ context_length=200_000,
+ account_usage=snapshot,
+ reasoning_effort="ultra",
+ cwd="",
+ fields=("provider", "account", "model", "reasoning", "context", "quota"),
+ underline=True,
+ )
+ assert out == "──────────────\nopenai-codex · team-main · gpt-5.5 · ult · ctx 21.8K/200K · 5h 80% · 7d 70%"
+
+
+def test_format_footer_compacts_reset_times_for_quota():
+ snapshot = AccountUsageSnapshot(
+ provider="anthropic",
+ source="oauth_usage_api",
+ fetched_at=datetime.now(timezone.utc),
+ windows=(
+ AccountUsageWindow(
+ label="Current session",
+ used_percent=32.2,
+ reset_at=datetime.now(timezone.utc) + timedelta(hours=2, minutes=10),
+ ),
+ ),
+ )
+ out = format_runtime_footer(
+ model="anthropic/claude-sonnet-4-6",
+ context_tokens=21_800,
+ context_length=200_000,
+ account_usage=snapshot,
+ cwd="",
+ fields=("quota",),
+ )
+ assert out.startswith("5h 68% 2h")
+
+
+def test_compact_reset_omits_zero_day_and_minute_components(monkeypatch):
+ import gateway.runtime_footer as runtime_footer
+
+ fixed_now = datetime(2026, 1, 1, tzinfo=timezone.utc)
+
+ class FixedDateTime(datetime):
+ @classmethod
+ def now(cls, tz=None):
+ return fixed_now if tz is not None else fixed_now.replace(tzinfo=None)
+
+ monkeypatch.setattr(runtime_footer, "datetime", FixedDateTime)
+ fixed_target = FixedDateTime(2026, 1, 1, tzinfo=timezone.utc)
+
+ assert _compact_reset(fixed_target + timedelta(days=1)) == "1d"
+ assert _compact_reset(fixed_target + timedelta(hours=1)) == "1h"
+ assert _compact_reset(fixed_target + timedelta(seconds=30)) == "<1m"
+
+
+
+def test_format_footer_uses_provider_specific_compact_quota_labels():
+ snapshot = AccountUsageSnapshot(
+ provider="anthropic",
+ source="oauth_usage_api",
+ fetched_at=datetime.now(timezone.utc),
+ windows=(
+ AccountUsageWindow(label="Current week", used_percent=40, reset_at=None),
+ AccountUsageWindow(label="Opus week", used_percent=50, reset_at=None),
+ AccountUsageWindow(label="Sonnet week", used_percent=60, reset_at=None),
+ ),
+ )
+ out = format_runtime_footer(
+ model="anthropic/claude-opus-4-5",
+ provider="anthropic",
+ context_tokens=0,
+ context_length=200_000,
+ account_usage=snapshot,
+ fields=("quota",),
+ )
+ assert out == "7d 60% · opus7d 50% · sonnet7d 40%"
+
+
+def test_format_footer_preserves_nonstandard_quota_durations():
+ snapshot = AccountUsageSnapshot(
+ provider="openai-codex",
+ source="usage_api",
+ fetched_at=datetime.now(timezone.utc),
+ windows=(
+ AccountUsageWindow(label="15d", used_percent=10, reset_at=None),
+ AccountUsageWindow(label="17h", used_percent=20, reset_at=None),
+ ),
+ )
+
+ out = format_runtime_footer(
+ model="openai/gpt-5.6-sol",
+ provider="openai-codex",
+ context_tokens=0,
+ context_length=200_000,
+ account_usage=snapshot,
+ fields=("quota",),
+ )
+
+ assert out == "15d 90% · 17h 80%"
+
+
+def test_format_footer_includes_balance_details_for_quota():
+ snapshot = AccountUsageSnapshot(
+ provider="deepseek",
+ source="balance_api",
+ fetched_at=datetime.now(timezone.utc),
+ details=("Balance: ¥110.00 CNY (granted ¥10.00, topped up ¥100.00)",),
+ )
+
+ out = format_runtime_footer(
+ model="deepseek/deepseek-chat",
+ provider="deepseek",
+ context_tokens=0,
+ context_length=200_000,
+ account_usage=snapshot,
+ fields=("quota",),
+ )
+
+ assert out == "balance ¥110.00 CNY"
+
+
+def test_format_footer_unknown_field_silently_ignored():
+ out = format_runtime_footer(
+ model="openai/gpt-5.4",
+ context_tokens=50, context_length=100,
+ cwd="/x",
+ fields=("model", "bogus", "context_pct"),
+ )
+ assert out == "gpt-5.4 · 50%"
+
+
# ---------------------------------------------------------------------------
# resolve_footer_config
# ---------------------------------------------------------------------------
+def test_resolve_defaults_off_empty_config():
+ cfg = resolve_footer_config({}, "telegram")
+ assert cfg == {"enabled": False, "fields": ["model", "context_pct", "cwd"], "underline": False}
+
+
+def test_resolve_global_enable():
+ user = {"display": {"runtime_footer": {"enabled": True}}}
+ cfg = resolve_footer_config(user, "telegram")
+ assert cfg["enabled"] is True
+ assert cfg["fields"] == ["model", "context_pct", "cwd"]
+
+
+
+def test_resolve_supports_underline_flag():
+ user = {"display": {"runtime_footer": {"enabled": True, "underline": True}}}
+ cfg = resolve_footer_config(user, "feishu")
+ assert cfg["underline"] is True
+
def test_resolve_platform_override_wins():
user = {
@@ -110,10 +340,144 @@ def test_resolve_platform_can_add_fields_only():
assert dc["fields"] == ["context_pct"]
+def test_resolve_ignores_malformed_config():
+ # Non-dict runtime_footer shouldn't crash
+ user = {"display": {"runtime_footer": "on"}}
+ cfg = resolve_footer_config(user, "telegram")
+ assert cfg["enabled"] is False
+
+
# ---------------------------------------------------------------------------
+
+
+def test_format_footer_reasoning_abbreviations():
+ cases = {
+ "none": "off",
+ "minimal": "min",
+ "low": "low",
+ "medium": "med",
+ "high": "high",
+ "xhigh": "xhi",
+ "max": "max",
+ "ultra": "ult",
+ "false": "off",
+ "DISABLED": "off",
+ }
+ for effort, expected in cases.items():
+ out = format_runtime_footer(
+ model="m",
+ context_tokens=0,
+ context_length=100,
+ reasoning_effort=effort,
+ fields=("reasoning",),
+ )
+ assert out == expected, (effort, out)
+
+
+def test_format_footer_reasoning_omitted_when_missing():
+ out = format_runtime_footer(
+ model="openai/gpt-5.5",
+ context_tokens=10,
+ context_length=100,
+ fields=("model", "reasoning"),
+ )
+ assert out == "gpt-5.5"
+
+
+def test_build_footer_passes_reasoning_effort():
+ out = build_footer_line(
+ user_config={
+ "display": {
+ "runtime_footer": {
+ "enabled": True,
+ "fields": ["model", "reasoning"],
+ }
+ }
+ },
+ platform_key=None,
+ model="openai/gpt-5.5",
+ context_tokens=0,
+ context_length=None,
+ reasoning_effort="xhigh",
+ )
+ assert out == "gpt-5.5 · xhi"
+
+
# build_footer_line — top-level entry point used by gateway/run.py
# ---------------------------------------------------------------------------
+def test_build_footer_empty_when_disabled():
+ out = build_footer_line(
+ user_config={},
+ platform_key="telegram",
+ model="openai/gpt-5.4",
+ context_tokens=10, context_length=100,
+ cwd="/tmp",
+ )
+ assert out == ""
+
+
+def test_build_footer_returns_rendered_when_enabled(monkeypatch, tmp_path):
+ monkeypatch.setenv("HOME", str(tmp_path))
+ out = build_footer_line(
+ user_config={"display": {"runtime_footer": {"enabled": True}}},
+ platform_key="telegram",
+ model="openai/gpt-5.4",
+ context_tokens=25, context_length=100,
+ cwd=str(tmp_path / "proj"),
+ )
+ (tmp_path / "proj").mkdir(exist_ok=True)
+ assert "gpt-5.4" in out
+ assert "25%" in out
+
+
+
+def test_build_footer_passes_extended_runtime_metadata():
+ snapshot = AccountUsageSnapshot(
+ provider="openai-codex",
+ source="usage_api",
+ fetched_at=datetime.now(timezone.utc),
+ plan="Team",
+ windows=(
+ AccountUsageWindow(label="5h", used_percent=20, reset_at=None),
+ AccountUsageWindow(label="7d", used_percent=30, reset_at=None),
+ ),
+ )
+ out = build_footer_line(
+ user_config={
+ "display": {
+ "runtime_footer": {
+ "enabled": True,
+ "underline": True,
+ "fields": ["provider", "account", "model", "context", "quota"],
+ }
+ }
+ },
+ platform_key="feishu",
+ model="openai/gpt-5.5",
+ provider="openai-codex",
+ account_label="team-main",
+ context_tokens=18_200,
+ context_length=400_000,
+ account_usage=snapshot,
+ cwd="",
+ )
+ assert out == "──────────────\nopenai-codex · team-main · gpt-5.5 · ctx 18.2K/400K · 5h 80% · 7d 70%"
+
+
+def test_build_footer_prefers_profile_scoped_resolved_config():
+ out = build_footer_line(
+ user_config={"display": {"runtime_footer": {"enabled": False}}},
+ platform_key="telegram",
+ resolved_config={"enabled": True, "fields": ["model"], "underline": False},
+ model="openai/gpt-5.5",
+ context_tokens=0,
+ context_length=None,
+ cwd="",
+ )
+
+ assert out == "gpt-5.5"
+
def test_build_footer_per_platform_off_suppresses():
user = {
@@ -279,3 +643,16 @@ def test_format_footer_served_model_is_opt_in_and_skips_same_model():
assert format_runtime_footer(
model="gpt-5.4", context_tokens=0, context_length=None, cwd="/x",
served_model=None, fields=["served_model"]) == ""
+
+def test_build_footer_no_data_returns_empty_even_when_enabled():
+ # Enabled, but context_length is None AND cwd empty AND model empty ⇒ no fields
+ out = build_footer_line(
+ user_config={"display": {"runtime_footer": {"enabled": True}}},
+ platform_key="telegram",
+ model="",
+ context_tokens=0, context_length=None,
+ cwd="",
+ )
+ # With no TERMINAL_CWD env either
+ if not os.environ.get("TERMINAL_CWD"):
+ assert out == ""
diff --git a/tests/gateway/test_runtime_footer_runner_wiring.py b/tests/gateway/test_runtime_footer_runner_wiring.py
new file mode 100644
index 0000000000000..33b1d6f3566a9
--- /dev/null
+++ b/tests/gateway/test_runtime_footer_runner_wiring.py
@@ -0,0 +1,61 @@
+"""Regression tests for runtime-footer producer/consumer wiring."""
+
+from pathlib import Path
+from types import SimpleNamespace
+
+import gateway.runtime_footer_usage as usage
+import hermes_constants
+from gateway.run_turn_runner import _resolve_runtime_footer_metadata
+
+
+def test_runner_resolves_footer_usage_inside_runtime_scope(monkeypatch):
+ calls = []
+ monkeypatch.setattr(
+ usage,
+ "get_cached",
+ lambda *args, **kwargs: calls.append((args, kwargs)) or "snapshot",
+ )
+ monkeypatch.setattr(hermes_constants, "get_hermes_home", lambda: Path("/profiles/routed"))
+
+ result = _resolve_runtime_footer_metadata(
+ SimpleNamespace(
+ provider="anthropic",
+ base_url="https://api.anthropic.com",
+ api_key="runtime-secret",
+ ),
+ {"display": {"runtime_footer": {"enabled": True, "fields": ["account"]}}},
+ "feishu",
+ )
+
+ assert result["provider"] == "anthropic"
+ assert result["base_url"] == "https://api.anthropic.com"
+ assert result["account_usage"] == "snapshot"
+ assert result["footer_config"]["fields"] == ["account"]
+ assert "api_key" not in result
+ assert calls == [
+ (
+ ("anthropic",),
+ {
+ "base_url": "https://api.anthropic.com",
+ "api_key": "runtime-secret",
+ "hermes_home": "/profiles/routed",
+ },
+ )
+ ]
+
+
+def test_runner_does_not_fetch_when_footer_usage_is_not_requested(monkeypatch):
+ monkeypatch.setattr(
+ usage,
+ "get_cached",
+ lambda *_args, **_kwargs: (_ for _ in ()).throw(AssertionError("unexpected refresh")),
+ )
+
+ result = _resolve_runtime_footer_metadata(
+ SimpleNamespace(provider="anthropic", base_url="", api_key="secret"),
+ {"display": {"runtime_footer": {"enabled": True, "fields": ["model"]}}},
+ "feishu",
+ )
+
+ assert result["account_usage"] is None
+ assert "api_key" not in result
diff --git a/tests/gateway/test_runtime_footer_usage_cache.py b/tests/gateway/test_runtime_footer_usage_cache.py
new file mode 100644
index 0000000000000..87d435a5e266c
--- /dev/null
+++ b/tests/gateway/test_runtime_footer_usage_cache.py
@@ -0,0 +1,191 @@
+"""Regression tests for runtime-footer provider/account/quota wiring helpers."""
+
+from __future__ import annotations
+
+from pathlib import Path
+from types import SimpleNamespace
+
+import gateway.runtime_footer_usage as usage
+from agent.account_usage import AccountUsageSnapshot
+
+
+def _reset_cache():
+ usage.clear()
+
+
+def test_footer_account_usage_cache_key_is_profile_and_credential_scoped(monkeypatch):
+ monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/alice"))
+ a = usage.cache_key(
+ "openai-codex",
+ base_url="https://chatgpt.com/backend-api/codex",
+ api_key="token-a",
+ )
+ b = usage.cache_key(
+ "openai-codex",
+ base_url="https://chatgpt.com/backend-api/codex",
+ api_key="token-b",
+ )
+ c = usage.cache_key(
+ "openai-codex",
+ base_url="https://chatgpt.com/backend-api/codex/",
+ api_key="token-a",
+ )
+ monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/bob"))
+ d = usage.cache_key(
+ "openai-codex",
+ base_url="https://chatgpt.com/backend-api/codex",
+ api_key="token-a",
+ )
+
+ assert a != b
+ assert a == c
+ assert a != d
+ assert "token-a" not in repr(a)
+
+
+def test_footer_account_usage_cache_key_accepts_captured_profile_home(monkeypatch):
+ monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/default"))
+
+ key = usage.cache_key(
+ "openai-codex",
+ base_url="https://chatgpt.com/backend-api/codex",
+ api_key="token-a",
+ hermes_home=Path("/profiles/alice"),
+ )
+
+ assert key[0] == "/profiles/alice"
+
+
+def test_cold_cache_schedules_once_and_returns_immediately(monkeypatch):
+ _reset_cache()
+ started = []
+ monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/alice"))
+ monkeypatch.setattr(usage.time, "monotonic", lambda: 100.0)
+ monkeypatch.setattr(
+ usage,
+ "_start_refresh",
+ lambda *args: started.append(args),
+ )
+
+ first = usage.get_cached(
+ "openai-codex", base_url="https://example.invalid", api_key="runtime-token"
+ )
+ second = usage.get_cached(
+ "openai-codex", base_url="https://example.invalid", api_key="runtime-token"
+ )
+
+ assert first is None
+ assert second is None
+ assert len(started) == 1
+ assert started[0][1:] == (
+ "openai-codex",
+ "https://example.invalid",
+ "runtime-token",
+ )
+ _reset_cache()
+
+
+def test_fresh_cache_returns_snapshot_without_scheduling(monkeypatch):
+ _reset_cache()
+ snapshot = AccountUsageSnapshot(
+ provider="openai-codex",
+ source="usage_api",
+ fetched_at=None,
+ plan="Plus",
+ windows=(),
+ )
+ monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/alice"))
+ key = usage.cache_key(
+ "openai-codex", base_url="https://example.invalid", api_key="runtime-token"
+ )
+ usage._CACHE[key] = (100.0, snapshot)
+ monkeypatch.setattr(usage.time, "monotonic", lambda: 110.0)
+ monkeypatch.setattr(
+ usage,
+ "_start_refresh",
+ lambda *_args: (_ for _ in ()).throw(AssertionError("fresh cache refreshed")),
+ )
+
+ result = usage.get_cached(
+ "openai-codex", base_url="https://example.invalid", api_key="runtime-token"
+ )
+
+ assert result is snapshot
+ _reset_cache()
+
+
+def test_stale_cache_returns_stale_snapshot_while_refreshing(monkeypatch):
+ _reset_cache()
+ snapshot = SimpleNamespace(windows=(SimpleNamespace(label="Session"),))
+ started = []
+ monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/alice"))
+ key = usage.cache_key(
+ "openai-codex", base_url="https://example.invalid", api_key="runtime-token"
+ )
+ usage._CACHE[key] = (1.0, snapshot)
+ monkeypatch.setattr(usage.time, "monotonic", lambda: 500.0)
+ monkeypatch.setattr(
+ usage,
+ "_start_refresh",
+ lambda *args: started.append(args),
+ )
+
+ result = usage.get_cached(
+ "openai-codex", base_url="https://example.invalid", api_key="runtime-token"
+ )
+
+ assert result is snapshot
+ assert len(started) == 1
+ _reset_cache()
+
+
+def test_refresh_uses_live_credential_and_updates_cache(monkeypatch):
+ _reset_cache()
+ calls = []
+ snapshot = SimpleNamespace(windows=(SimpleNamespace(label="Session"),))
+ monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/alice"))
+ key = usage.cache_key(
+ "openai-codex", base_url="https://example.invalid", api_key="runtime-token"
+ )
+ usage._REFRESHING.add(key)
+ monkeypatch.setattr(
+ usage,
+ "fetch_account_usage",
+ lambda *args, **kwargs: calls.append((args, kwargs)) or snapshot,
+ )
+ monkeypatch.setattr(usage.time, "monotonic", lambda: 321.0)
+
+ usage._refresh(
+ key, "openai-codex", "https://example.invalid", "runtime-token"
+ )
+
+ assert usage._CACHE[key] == (321.0, snapshot)
+ assert key not in usage._REFRESHING
+ assert calls == [
+ (
+ ("openai-codex",),
+ {"base_url": "https://example.invalid", "api_key": "runtime-token"},
+ )
+ ]
+ _reset_cache()
+
+
+def test_refresh_failure_preserves_last_good_snapshot(monkeypatch):
+ _reset_cache()
+ stale = SimpleNamespace(windows=(SimpleNamespace(label="Session"),))
+ monkeypatch.setattr(usage, "get_hermes_home", lambda: Path("/profiles/alice"))
+ key = usage.cache_key(
+ "openai-codex", base_url="https://example.invalid", api_key="runtime-token"
+ )
+ usage._CACHE[key] = (1.0, stale)
+ usage._REFRESHING.add(key)
+ monkeypatch.setattr(usage, "fetch_account_usage", lambda *_a, **_kw: None)
+ monkeypatch.setattr(usage.time, "monotonic", lambda: 500.0)
+
+ usage._refresh(
+ key, "openai-codex", "https://example.invalid", "runtime-token"
+ )
+
+ assert usage._CACHE[key] == (500.0, stale)
+ assert key not in usage._REFRESHING
+ _reset_cache()
diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md
index 2dec21b039bba..b6a28ee9df297 100644
--- a/website/docs/user-guide/configuration.md
+++ b/website/docs/user-guide/configuration.md
@@ -2324,6 +2324,7 @@ display:
runtime_footer:
enabled: true
fields: ["model", "context_pct", "cwd"] # order shown; drop any to hide
+ underline: false # add a separator before the footer
```
Supported fields:
@@ -2335,8 +2336,13 @@ Supported fields:
| `latency` | Wall-clock duration of the turn | `22s`, `1m05s` |
| `served_model` | The model that actually answered, when it differs from the one you configured: the deployment a routing proxy reported in its `x-litellm-model-id` (or `x-litellm-model-api-base`) response header, or the fallback model Hermes switched to for the turn | `hermes-router → gpt-4o-2024-11-20` |
| `cwd` | Home-relative working directory | `~` |
+| `provider` | Provider serving the turn | `anthropic` |
+| `account` | Account or plan label from provider usage | `Pro` |
+| `context` | Last-call context tokens as used/total | `8.2K/128K` |
+| `quota` | Provider usage window and remaining percentage | `5h 72%` |
+| `reasoning` | Reasoning effort used for the turn | `high` |
-The default field set is `["model", "context_pct", "cwd"]`. `latency` and `served_model` are opt-in — add them to `fields` to use them. `served_model` renders nothing when the served model is the configured one (or when the proxy sends no such header), so behind a routing proxy or an active fallback it is the field that makes the switch visible. Fields whose data is unavailable are skipped silently rather than rendering an empty slot.
+The default field set is `["model", "context_pct", "cwd"]`. `latency`, `served_model`, `provider`, `account`, `context`, `quota`, and `reasoning` are opt-in — add them to `fields` to use them. `served_model` renders nothing when the served model is the configured one (or when the proxy sends no such header), so behind a routing proxy or an active fallback it is the field that makes the switch visible. Account and quota data is fetched with a non-blocking stale-while-revalidate cache; it may be absent on the first turn and appear after the background refresh completes. Fields whose data is unavailable are skipped silently rather than rendering an empty slot. Set `underline: true` to insert a separator before the footer.
The `/footer` slash command toggles this at runtime in any session.