Skip to content
Open
3 changes: 3 additions & 0 deletions agent/agent_init.py
Original file line number Diff line number Diff line change
Expand Up @@ -2363,6 +2363,9 @@ def init_agent(
_build_client(agent, api_key, base_url, fallback_model)
_init_fallback_chain(agent, fallback_model)
_load_tools(agent, enabled_toolsets, disabled_toolsets)
from providers.reasoning import resolve_provider_reasoning_config
reasoning_config = resolve_provider_reasoning_config(agent.provider, agent.model, reasoning_config)
agent.reasoning_config = reasoning_config
_init_session_state(
agent, session_id, session_db, parent_session_id, reasoning_config, max_tokens,
checkpoints_enabled, checkpoint_max_snapshots, checkpoint_max_total_size_mb, checkpoint_max_file_size_mb,
Expand Down
4 changes: 3 additions & 1 deletion agent/agent_runtime_helpers.py
Original file line number Diff line number Diff line change
Expand Up @@ -2268,7 +2268,9 @@ def switch_model(
try:
from hermes_constants import resolve_reasoning_config
from hermes_cli.config import load_config as _sm_load_config
agent.reasoning_config = resolve_reasoning_config(_sm_load_config() or {}, agent.model)
from providers.reasoning import resolve_provider_reasoning_config
agent.reasoning_config = resolve_provider_reasoning_config(
agent.provider, agent.model, resolve_reasoning_config(_sm_load_config() or {}, agent.model))
logger.info(
"switch_model: reasoning_config resolved for %s: %s", agent.model, agent.reasoning_config
)
Expand Down
69 changes: 69 additions & 0 deletions agent/gemini_catalog_reasoning.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,69 @@
"""AI Studio generateContent controls from cached models.dev metadata.

Model names carry no policy. Vertex retains its separate transport policy.
Missing metadata leaves the API in charge; discovery never makes paid probes.
"""
from agent.models_dev import get_model_info


def describe_thinking_control(model: str) -> dict:
name = model.strip().removeprefix("google/").removeprefix("models/")
info = get_model_info("gemini", name, allow_network=False)
result = {"reasoning_control": "unknown", "reasoning_efforts": [],
"can_disable_reasoning": False, "fast": False}
if info is None:
return result
result["reasoning"] = info.reasoning
if not name.startswith("gemini-"):
return result # e.g. Gemma's toggle is not Gemini's thinkingConfig protocol
if not info.reasoning:
result["reasoning_control"] = "unsupported"
return result
if info.reasoning_options is None:
return result
efforts = []
for option in info.reasoning_options:
if option["type"] == "effort":
efforts.extend(v for v in option["values"] if v not in (None, "default"))
elif option["type"] == "toggle":
efforts.append("none")
elif option["type"] == "budget_tokens":
# These bounds are for the Thinking-on numeric input, not a copy of
# the raw API range. Zero is Off (requires a declared toggle), and
# -1 is Dynamic; neither may bypass those separate controls here.
# Preserve positive minima and never invent a missing bound.
if "min" in option and "max" in option and option["max"] > 0:
result["reasoning_budget"] = {"min": max(1, option["min"]), "max": option["max"], "dynamic": True}
# -1 is generateContent's dynamic sentinel, not a model effort.
result["reasoning_efforts"] = list(dict.fromkeys(efforts))
result["can_disable_reasoning"] = "none" in efforts
result["reasoning_control"] = "adjustable" if efforts or "reasoning_budget" in result else "default"
return result


def supported_efforts(model: str | None) -> tuple[str, ...]:
return tuple(describe_thinking_control(model or "")["reasoning_efforts"])


def build_thinking_config(model: str, reasoning_config: dict | None) -> dict | None:
from providers.reasoning import resolve_provider_reasoning_config
if not isinstance(reasoning_config, dict):
return None
descriptor = describe_thinking_control(model)
if descriptor["reasoning_control"] in ("unknown", "unsupported"):
return None
# The resolver shares the picker's declared Off capability. Inherited Off
# on a mandatory-thinking model becomes an unset override before this branch;
# explicit selections are rejected by the same resolver at the gateway.
config = resolve_provider_reasoning_config("gemini", model, reasoning_config)
if not config:
return None
if config.get("enabled") is False:
return {"includeThoughts": False, "thinkingBudget": 0}
effort = config.get("effort", "")
if effort.startswith("budget:"):
return {"includeThoughts": True, "thinkingBudget": int(effort[7:])}
if effort:
return {"includeThoughts": True, "thinkingLevel": effort}
# Provider default means no level/budget override (in particular for Lite).
return {"includeThoughts": True}
67 changes: 67 additions & 0 deletions agent/gemini_model_catalog.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,67 @@
"""Free native Google model discovery, including every page.

Google determines listed IDs; models.dev supplies text/tool capabilities.
Unknown capabilities remain usable by explicit model ID, not advertised as
verified agent models. Failed pages never return a partial/authoritative catalog;
malformed individual entries are skipped.
"""
import json
import logging
import time
from urllib.parse import urlencode
from urllib.request import Request

from agent.models_dev import fetch_models_dev, get_model_info
from hermes_cli.urllib_security import open_credentialed_url


def fetch_models(api_key: str | None, *, timeout: float = 8.0) -> list[str] | None:
if not api_key:
return None
if not fetch_models_dev():
return None
deadline = time.monotonic() + timeout
result = []
tokens = set()
token = ""
try:
for _ in range(20):
query = {"pageSize": 1000, **({"pageToken": token} if token else {})}
request = Request("https://generativelanguage.googleapis.com/v1beta/models?" + urlencode(query),
headers={"x-goog-api-key": api_key, "Accept": "application/json"})
remaining = deadline - time.monotonic()
if remaining <= 0:
return None
with open_credentialed_url(request, timeout=remaining) as response:
data = json.load(response)
if not isinstance(data, dict) or not isinstance(data.get("models"), list):
return None
for entry in data["models"]:
if not isinstance(entry, dict) or not isinstance(entry.get("name"), str):
continue
model = entry["name"].removeprefix("models/")
methods = entry.get("supportedGenerationMethods")
if not model or not isinstance(methods, list):
continue
# Dedicated Computer Use routes require Google's built-in tool;
# Hermes' generic function tools cannot invoke them. The native
# list and models.dev's tool_call flag do not encode this prerequisite.
if "-computer-use-" in model:
continue
info = get_model_info("gemini", model, allow_network=False)
if ("generateContent" in methods
and info and info.tool_call and info.output_modalities == ("text",)
and info.status != "deprecated"):
result.append(model)
token = data.get("nextPageToken")
if not token:
return list(dict.fromkeys(result))
if not isinstance(token, str) or token in tokens:
return None
tokens.add(token)
except Exception:
return None # no credential-bearing exceptions in logs
logging.getLogger(__name__).warning(
"Gemini model discovery reached its pagination limit; using the fallback catalog."
)
return None
35 changes: 35 additions & 0 deletions agent/models_dev.py
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,8 @@ class ModelInfo:
provider_id: str # models.dev provider ID (e.g. "anthropic")
# Capabilities
reasoning: bool = False
# None means absent/invalid metadata; () explicitly means no caller controls.
reasoning_options: Optional[Tuple[Dict[str, Any], ...]] = None
tool_call: bool = False
attachment: bool = False # supports image/file attachments (vision)
temperature: bool = False
Expand Down Expand Up @@ -878,6 +880,38 @@ def list_agentic_models(provider: str, *, allow_network: bool = True) -> List[st
] if models is not None else []


def _parse_reasoning_options(value: Any) -> Optional[Tuple[Dict[str, Any], ...]]:
"""Read the models.dev tagged union without guessing missing capabilities."""
if not isinstance(value, list):
return None
result = []
seen = set()
for option in value:
if not isinstance(option, dict):
return None
kind = option.get("type")
if kind not in ("toggle", "effort", "budget_tokens") or kind in seen:
return None
seen.add(kind)
parsed = {"type": kind}
if kind == "effort":
values = option.get("values")
allowed = (None, "none", "minimal", "low", "medium", "high", "xhigh", "max", "default")
if not isinstance(values, list) or not values or any(v not in allowed for v in values):
return None
parsed["values"] = list(dict.fromkeys(values))
if kind == "budget_tokens":
for key, lower in (("min", -1), ("max", 0)):
if key in option:
if type(option[key]) is not int or option[key] < lower:
return None
parsed[key] = option[key]
if "min" in parsed and "max" in parsed and parsed["min"] > parsed["max"]:
return None
result.append(parsed)
return tuple(result)


def _parse_model_info(model_id: str, raw: Dict[str, Any], provider_id: str) -> ModelInfo:
"""Convert a raw models.dev model entry dict into a ModelInfo dataclass."""
cost = _dict_or_empty(raw.get("cost"))
Expand All @@ -890,6 +924,7 @@ def _cost(key: str) -> Optional[float]:
return ModelInfo(
id=model_id, name=raw.get("name", "") or model_id, family=raw.get("family", "") or "", provider_id=provider_id,
**{k: bool(raw.get(k, False)) for k in ("reasoning", "tool_call", "attachment", "temperature", "structured_output", "open_weights")},
reasoning_options=_parse_reasoning_options(raw.get("reasoning_options")),
input_modalities=_mods("input"), output_modalities=_mods("output"),
context_window=_extract_limit(raw, "context") or 0, max_output=_extract_limit(raw, "output") or 0, max_input=_extract_limit(raw, "input"),
cost_input=float(cost.get("input", 0) or 0), cost_output=float(cost.get("output", 0) or 0),
Expand Down
14 changes: 9 additions & 5 deletions agent/transports/chat_completions.py
Original file line number Diff line number Diff line change
Expand Up @@ -220,9 +220,11 @@ def _snake_case_gemini_thinking_config(config: dict | None) -> dict | None:
return translated or None


def _raise_gemini_thinking_max_tokens(model: str, reasoning_config: dict | None, requested: Any) -> Any:
def _raise_gemini_thinking_max_tokens(model: str, reasoning_config: dict | None, requested: Any, provider: str = "") -> Any:
"""Raise Gemini output caps that thinking tokens (billed against max_tokens) would otherwise exhaust."""
thinking_config = _build_gemini_thinking_config(model, reasoning_config)
from agent.gemini_catalog_reasoning import build_thinking_config
builder = build_thinking_config if provider == "gemini" else _build_gemini_thinking_config
thinking_config = builder(model, reasoning_config)
if not thinking_config:
return requested
from agent.gemini_native_adapter import _effective_gemini_max_output_tokens
Expand Down Expand Up @@ -331,12 +333,13 @@ def _swap_developer_role(sanitized: list, model_lower: str) -> list:
def _apply_max_tokens(api_kwargs: dict, model: str, reasoning_config: Any, params: dict, profile_max: Any = None) -> None:
"""Preserve internal task/recovery budgets and provider protocol exceptions."""
max_tokens_fn = params.get("max_tokens_param_fn")
provider = getattr(params.get("provider_profile"), "name", None) or params.get("provider_name", "")
for candidate in (params.get("ephemeral_max_output_tokens"), params.get("max_tokens")):
if candidate is not None and max_tokens_fn:
api_kwargs.update(max_tokens_fn(_raise_gemini_thinking_max_tokens(model, reasoning_config, candidate)))
api_kwargs.update(max_tokens_fn(_raise_gemini_thinking_max_tokens(model, reasoning_config, candidate, provider)))
return
if profile_max and max_tokens_fn:
api_kwargs.update(max_tokens_fn(_raise_gemini_thinking_max_tokens(model, reasoning_config, profile_max)))
api_kwargs.update(max_tokens_fn(_raise_gemini_thinking_max_tokens(model, reasoning_config, profile_max, provider)))



Expand Down Expand Up @@ -508,7 +511,8 @@ def build_kwargs(
extra_body["reasoning"] = {"enabled": not off, "effort": "none" if off else _effort}

if str(params.get("provider_name") or "").strip().lower() == "gemini":
raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config)
from agent.gemini_catalog_reasoning import build_thinking_config
raw_thinking_config = build_thinking_config(model, reasoning_config)
if _is_gemini_openai_compat_base_url(base_url):
thinking_config = _snake_case_gemini_thinking_config(raw_thinking_config)
if thinking_config:
Expand Down
11 changes: 11 additions & 0 deletions apps/desktop/src/app/chat/composer/reasoning-pill.test.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,17 @@ afterEach(() => {
})

describe('ReasoningPill', () => {
it('offers the effort menu without inventing a level for an unset override', () => {
$defaultReasoningEffort.set('high')
render(
<SessionViewProvider value={tileView('auto')}>
<ReasoningPill disabled={false} model={modelState()} />
</SessionViewProvider>
)
expect(screen.getByTestId('reasoning-pill').textContent).toBe('Effort')
expect(screen.getByTestId('reasoning-pill').getAttribute('aria-label')).toBe('Effort')
})

it('shows a clamped pick as what the route sends, never as a distinct level (#61634)', () => {
// The gateway says this route clamps `ultra` to `max`: compact "Ultra→Max",
// tooltip in the CLI's `/reasoning` wording.
Expand Down
4 changes: 2 additions & 2 deletions apps/desktop/src/app/chat/composer/reasoning-pill.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -51,7 +51,7 @@ export function ReasoningPill({ disabled, model }: { disabled: boolean; model: C

const title = clamp
? `${copy.effort}: ${copy[clamp.effort]} (${copy.sendsOnRoute(copy[clamp.wire])})`
: `${copy.effort}: ${label}`
: label ? `${copy.effort}: ${label}` : copy.effort

// Closing the menu ends its claim on the keyboard: Radix restores focus to
// this pill (a toolbar button), so without the release the Enter that
Expand All @@ -76,7 +76,7 @@ export function ReasoningPill({ disabled, model }: { disabled: boolean; model: C
type="button"
variant="ghost"
>
<span>{label}</span>
<span>{label || copy.effort}</span>
<ChevronDown className="size-2.5 shrink-0 opacity-50" />
</Button>
</DropdownMenuTrigger>
Expand Down
22 changes: 18 additions & 4 deletions apps/desktop/src/app/settings/model-settings.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,7 @@ import { useI18n } from '@/i18n'
import { isCodeSkewRestartRequired } from '@/lib/code-skew-error'
import { AlertTriangle, Cpu, Loader2 } from '@/lib/icons'
import { isSubmitEnter } from '@/lib/ime'
import { resolveModelReasoningEffort } from '@/lib/reasoning-effort'
import { cn } from '@/lib/utils'
import { setMainModelAssignment } from '@/store/model-assignment'
import { notifyError, readableError } from '@/store/notifications'
Expand Down Expand Up @@ -572,7 +573,15 @@ export function ModelSettings({ onMainModelChanged, scopeProfile, subpage }: Mod
.trim()
.toLowerCase()

const effortValue = rawEffort === 'false' || rawEffort === 'disabled' ? 'none' : rawEffort || DEFAULT_REASONING_EFFORT
const effortValue = resolveModelReasoningEffort(
rawEffort === 'false' || rawEffort === 'disabled' ? 'none' : rawEffort,
DEFAULT_REASONING_EFFORT,
mainCaps
)

const effortChoices = REASONING_EFFORT_VALUES.filter(
value => mainCaps?.reasoning_efforts == null || mainCaps.reasoning_efforts.includes(value)
)

const fastOn = isFastTier(getNested(config ?? {}, 'agent.service_tier'))

Expand Down Expand Up @@ -975,13 +984,18 @@ export function ModelSettings({ onMainModelChanged, scopeProfile, subpage }: Mod
{m.reasoning}
<Select
onValueChange={value => void writeAgentDefault('agent.reasoning_effort', value)}
value={effortValue}
value={effortValue === 'auto' ? '' : effortValue}
>
<SelectTrigger className={cn('min-w-28', CONTROL_TEXT)}>
<SelectValue />
<SelectValue placeholder={t.shell.modelOptions.unverified} />
</SelectTrigger>
<SelectContent>
{REASONING_EFFORT_VALUES.map(value => (
{effortValue.startsWith('budget:') ? (
<SelectItem value={effortValue}>
{effortValue === 'budget:-1' ? t.shell.modelOptions.dynamicThinking : effortValue.slice(7)}
</SelectItem>
) : null}
{effortChoices.map(value => (
<SelectItem key={value} value={value}>
{value === 'none' ? m.reasoningOff : t.shell.modelOptions[value]}
</SelectItem>
Expand Down
Loading