From e69429d1129e8886b87ec6111aeeb237a09424b7 Mon Sep 17 00:00:00 2001 From: jkorzeniak Date: Tue, 15 Sep 2026 11:01:59 +0200 Subject: [PATCH 1/4] fix(gemini): derive Studio picker controls from native catalog and metadata --- agent/agent_init.py | 3 + agent/agent_runtime_helpers.py | 4 +- agent/gemini_catalog_reasoning.py | 63 +++++++++ agent/gemini_model_catalog.py | 59 ++++++++ agent/models_dev.py | 35 +++++ agent/transports/chat_completions.py | 16 ++- .../app/chat/composer/reasoning-pill.test.tsx | 10 ++ .../src/app/chat/composer/reasoning-pill.tsx | 4 +- .../src/app/settings/model-settings.tsx | 26 +++- .../src/app/shell/model-catalog-menu.tsx | 24 +++- .../src/app/shell/model-edit-submenu.test.tsx | 132 ++++++++++++++++++ .../src/app/shell/model-edit-submenu.tsx | 94 +++++++++++-- .../src/app/shell/reasoning-budget-input.tsx | 51 +++++++ .../src/app/shell/reasoning-menu-panel.tsx | 3 + apps/desktop/src/i18n/ar.ts | 2 + apps/desktop/src/i18n/en.ts | 4 + apps/desktop/src/i18n/ja.ts | 4 + apps/desktop/src/i18n/ru.ts | 2 + apps/desktop/src/i18n/types.ts | 4 + apps/desktop/src/i18n/zh-hant.ts | 4 + apps/desktop/src/i18n/zh.ts | 4 + apps/desktop/src/lib/reasoning-effort.test.ts | 40 +++++- apps/desktop/src/lib/reasoning-effort.ts | 52 ++++++- apps/shared/src/gateway-contract.generated.ts | 8 ++ apps/shared/src/gateway-contract.openrpc.json | 68 +++++++++ hermes_cli/inventory.py | 42 ++++++ hermes_cli/models.py | 2 + hermes_constants.py | 5 + plugins/model-providers/gemini/__init__.py | 20 ++- providers/base.py | 10 ++ providers/reasoning.py | 67 +++++++++ tests/agent/test_gemini_catalog_reasoning.py | 125 +++++++++++++++++ tests/agent/test_provider_parity.py | 3 + .../agent/transports/test_chat_completions.py | 3 + tests/conftest.py | 24 ++++ .../test_gemini_reasoning_controls.py | 72 ++++++++++ .../contracts/config_free_tier_control.py | 9 ++ tui_gateway/methods_config.py | 2 +- tui_gateway/methods_config_set.py | 16 +++ tui_gateway/methods_session.py | 9 ++ tui_gateway/model_switch.py | 7 + tui_gateway/server.py | 2 +- website/docs/guides/google-gemini.md | 33 +++++ 43 files changed, 1127 insertions(+), 40 deletions(-) create mode 100644 agent/gemini_catalog_reasoning.py create mode 100644 agent/gemini_model_catalog.py create mode 100644 apps/desktop/src/app/shell/reasoning-budget-input.tsx create mode 100644 providers/reasoning.py create mode 100644 tests/agent/test_gemini_catalog_reasoning.py create mode 100644 tests/tui_gateway/test_gemini_reasoning_controls.py diff --git a/agent/agent_init.py b/agent/agent_init.py index 50ff4ae4bf788..1e07d8dadb3cb 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -2278,6 +2278,9 @@ def init_agent( _build_client(agent, api_key, base_url, fallback_model) _init_fallback_chain(agent, fallback_model) _load_tools(agent, enabled_toolsets, disabled_toolsets) + from providers.reasoning import resolve_provider_reasoning_config + reasoning_config = resolve_provider_reasoning_config(agent.provider, agent.model, reasoning_config) + agent.reasoning_config = reasoning_config _init_session_state( agent, session_id, session_db, parent_session_id, reasoning_config, max_tokens, checkpoints_enabled, checkpoint_max_snapshots, checkpoint_max_total_size_mb, checkpoint_max_file_size_mb, diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index cd3e768e1918e..372239a815ab2 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -2185,7 +2185,9 @@ def switch_model( try: from hermes_constants import resolve_reasoning_config from hermes_cli.config import load_config as _sm_load_config - agent.reasoning_config = resolve_reasoning_config(_sm_load_config() or {}, agent.model) + from providers.reasoning import resolve_provider_reasoning_config + agent.reasoning_config = resolve_provider_reasoning_config( + agent.provider, agent.model, resolve_reasoning_config(_sm_load_config() or {}, agent.model)) logger.info( "switch_model: reasoning_config resolved for %s: %s", agent.model, agent.reasoning_config ) diff --git a/agent/gemini_catalog_reasoning.py b/agent/gemini_catalog_reasoning.py new file mode 100644 index 0000000000000..cc8a672b69fa0 --- /dev/null +++ b/agent/gemini_catalog_reasoning.py @@ -0,0 +1,63 @@ +"""AI Studio generateContent controls from cached models.dev metadata. + +Model names carry no policy. Vertex retains its separate transport policy. +Missing metadata leaves the API in charge; discovery never makes paid probes. +""" +from agent.models_dev import get_model_info + + +def describe_thinking_control(model: str) -> dict: + name = model.strip().removeprefix("google/").removeprefix("models/") + info = get_model_info("gemini", name, allow_network=False) + result = {"reasoning_control": "unknown", "reasoning_efforts": [], + "can_disable_reasoning": False, "fast": False} + if info is None: + return result + result["reasoning"] = info.reasoning + if not name.startswith("gemini-"): + return result # e.g. Gemma's toggle is not Gemini's thinkingConfig protocol + if not info.reasoning: + result["reasoning_control"] = "unsupported" + return result + if info.reasoning_options is None: + return result + efforts = [] + for option in info.reasoning_options: + if option["type"] == "effort": + efforts.extend(v for v in option["values"] if v not in (None, "default")) + elif option["type"] == "toggle": + efforts.append("none") + elif option["type"] == "budget_tokens": + # Do not invent a missing bound or expose zero as "thinking on". + if "min" in option and "max" in option and option["max"] > 0: + result["reasoning_budget"] = {"min": max(1, option["min"]), "max": option["max"], "dynamic": True} + # -1 is generateContent's dynamic sentinel, not a model effort. + result["reasoning_efforts"] = list(dict.fromkeys(efforts)) + result["can_disable_reasoning"] = "none" in efforts + result["reasoning_control"] = "adjustable" if efforts or "reasoning_budget" in result else "default" + return result + + +def supported_efforts(model: str | None) -> tuple[str, ...]: + return tuple(describe_thinking_control(model or "")["reasoning_efforts"]) + + +def build_thinking_config(model: str, reasoning_config: dict | None) -> dict | None: + from providers.reasoning import resolve_provider_reasoning_config + if not isinstance(reasoning_config, dict): + return None + descriptor = describe_thinking_control(model) + if descriptor["reasoning_control"] in ("unknown", "unsupported"): + return None + config = resolve_provider_reasoning_config("gemini", model, reasoning_config) + if not config: + return None + if config.get("enabled") is False: + return {"includeThoughts": False, "thinkingBudget": 0} + effort = config.get("effort", "") + if effort.startswith("budget:"): + return {"includeThoughts": True, "thinkingBudget": int(effort[7:])} + if effort: + return {"includeThoughts": True, "thinkingLevel": effort} + # Provider default means no level/budget override (in particular for Lite). + return {"includeThoughts": True} diff --git a/agent/gemini_model_catalog.py b/agent/gemini_model_catalog.py new file mode 100644 index 0000000000000..4a442158a9ec0 --- /dev/null +++ b/agent/gemini_model_catalog.py @@ -0,0 +1,59 @@ +"""Free native Google model discovery, including every page. + +Google determines listed IDs; models.dev supplies text/tool capabilities. +Unknown capabilities remain usable by explicit model ID, not advertised as +verified agent models. Failures never return a partial/authoritative catalog. +""" +import json +import time +from urllib.parse import urlencode +from urllib.request import Request + +from agent.models_dev import fetch_models_dev, get_model_info +from hermes_cli.urllib_security import open_credentialed_url + + +def fetch_models(api_key: str | None, *, timeout: float = 8.0) -> list[str] | None: + if not api_key: + return None + if not fetch_models_dev(): + return None + deadline = time.monotonic() + timeout + result = [] + tokens = set() + token = "" + try: + for _ in range(20): + query = {"pageSize": 1000, **({"pageToken": token} if token else {})} + request = Request("https://generativelanguage.googleapis.com/v1beta/models?" + urlencode(query), + headers={"x-goog-api-key": api_key, "Accept": "application/json"}) + remaining = deadline - time.monotonic() + if remaining <= 0: + return None + with open_credentialed_url(request, timeout=remaining) as response: + data = json.load(response) + if not isinstance(data, dict) or not isinstance(data.get("models"), list): + return None + for entry in data["models"]: + if not isinstance(entry, dict) or not isinstance(entry.get("name"), str): + return None + model = entry["name"].removeprefix("models/") + # Dedicated Computer Use routes require Google's built-in tool; + # Hermes' generic function tools cannot invoke them. The native + # list and models.dev's tool_call flag do not encode this prerequisite. + if "-computer-use-" in model: + continue + info = get_model_info("gemini", model, allow_network=False) + if ("generateContent" in (entry.get("supportedGenerationMethods") or []) + and info and info.tool_call and info.output_modalities == ("text",) + and info.status != "deprecated"): + result.append(model) + token = data.get("nextPageToken") + if not token: + return list(dict.fromkeys(result)) + if not isinstance(token, str) or token in tokens: + return None + tokens.add(token) + except Exception: + return None # no credential-bearing exceptions in logs + return None diff --git a/agent/models_dev.py b/agent/models_dev.py index fd40784b9b708..3daf298eaffac 100644 --- a/agent/models_dev.py +++ b/agent/models_dev.py @@ -45,6 +45,8 @@ class ModelInfo: provider_id: str # models.dev provider ID (e.g. "anthropic") # Capabilities reasoning: bool = False + # None means absent/invalid metadata; () explicitly means no caller controls. + reasoning_options: Optional[Tuple[Dict[str, Any], ...]] = None tool_call: bool = False attachment: bool = False # supports image/file attachments (vision) temperature: bool = False @@ -798,6 +800,38 @@ def list_agentic_models(provider: str, *, allow_network: bool = True) -> List[st ] if models is not None else [] +def _parse_reasoning_options(value: Any) -> Optional[Tuple[Dict[str, Any], ...]]: + """Read the models.dev tagged union without guessing missing capabilities.""" + if not isinstance(value, list): + return None + result = [] + seen = set() + for option in value: + if not isinstance(option, dict): + return None + kind = option.get("type") + if kind not in ("toggle", "effort", "budget_tokens") or kind in seen: + return None + seen.add(kind) + parsed = {"type": kind} + if kind == "effort": + values = option.get("values") + allowed = (None, "none", "minimal", "low", "medium", "high", "xhigh", "max", "default") + if not isinstance(values, list) or not values or any(v not in allowed for v in values): + return None + parsed["values"] = list(dict.fromkeys(values)) + if kind == "budget_tokens": + for key, lower in (("min", -1), ("max", 0)): + if key in option: + if type(option[key]) is not int or option[key] < lower: + return None + parsed[key] = option[key] + if "min" in parsed and "max" in parsed and parsed["min"] > parsed["max"]: + return None + result.append(parsed) + return tuple(result) + + def _parse_model_info(model_id: str, raw: Dict[str, Any], provider_id: str) -> ModelInfo: """Convert a raw models.dev model entry dict into a ModelInfo dataclass.""" cost = _dict_or_empty(raw.get("cost")) @@ -810,6 +844,7 @@ def _cost(key: str) -> Optional[float]: return ModelInfo( id=model_id, name=raw.get("name", "") or model_id, family=raw.get("family", "") or "", provider_id=provider_id, **{k: bool(raw.get(k, False)) for k in ("reasoning", "tool_call", "attachment", "temperature", "structured_output", "open_weights")}, + reasoning_options=_parse_reasoning_options(raw.get("reasoning_options")), input_modalities=_mods("input"), output_modalities=_mods("output"), context_window=_extract_limit(raw, "context") or 0, max_output=_extract_limit(raw, "output") or 0, max_input=_extract_limit(raw, "input"), cost_input=float(cost.get("input", 0) or 0), cost_output=float(cost.get("output", 0) or 0), diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index 2e4a6d95e199e..dd3707dc6628f 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -122,6 +122,8 @@ def _reasoning_config_for_model(model: str, reasoning_config: dict | None) -> di if not isinstance(reasoning_config, dict): return reasoning_config effort = str(reasoning_config.get("effort") or "").strip().lower() + if effort.startswith("budget:"): + return reasoning_config # validated by the provider, not an effort ladder value clamped = clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS) if effort else effort return {**reasoning_config, "effort": clamped} if clamped != effort else reasoning_config @@ -185,9 +187,11 @@ def _snake_case_gemini_thinking_config(config: dict | None) -> dict | None: return translated or None -def _raise_gemini_thinking_max_tokens(model: str, reasoning_config: dict | None, requested: Any) -> Any: +def _raise_gemini_thinking_max_tokens(model: str, reasoning_config: dict | None, requested: Any, provider: str = "") -> Any: """Raise Gemini output caps that thinking tokens (billed against max_tokens) would otherwise exhaust.""" - thinking_config = _build_gemini_thinking_config(model, reasoning_config) + from agent.gemini_catalog_reasoning import build_thinking_config + builder = build_thinking_config if provider == "gemini" else _build_gemini_thinking_config + thinking_config = builder(model, reasoning_config) if not thinking_config: return requested from agent.gemini_native_adapter import _effective_gemini_max_output_tokens @@ -263,12 +267,13 @@ def _swap_developer_role(sanitized: list, model_lower: str) -> list: def _apply_max_tokens(api_kwargs: dict, model: str, reasoning_config: Any, params: dict, profile_max: Any = None) -> None: """Preserve internal task/recovery budgets and provider protocol exceptions.""" max_tokens_fn = params.get("max_tokens_param_fn") + provider = getattr(params.get("provider_profile"), "name", None) or params.get("provider_name", "") for candidate in (params.get("ephemeral_max_output_tokens"), params.get("max_tokens")): if candidate is not None and max_tokens_fn: - api_kwargs.update(max_tokens_fn(_raise_gemini_thinking_max_tokens(model, reasoning_config, candidate))) + api_kwargs.update(max_tokens_fn(_raise_gemini_thinking_max_tokens(model, reasoning_config, candidate, provider))) return if profile_max and max_tokens_fn: - api_kwargs.update(max_tokens_fn(_raise_gemini_thinking_max_tokens(model, reasoning_config, profile_max))) + api_kwargs.update(max_tokens_fn(_raise_gemini_thinking_max_tokens(model, reasoning_config, profile_max, provider))) @@ -434,7 +439,8 @@ def build_kwargs( extra_body["reasoning"] = {"enabled": not off, "effort": "none" if off else _effort} if str(params.get("provider_name") or "").strip().lower() == "gemini": - raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config) + from agent.gemini_catalog_reasoning import build_thinking_config + raw_thinking_config = build_thinking_config(model, reasoning_config) if _is_gemini_openai_compat_base_url(base_url): thinking_config = _snake_case_gemini_thinking_config(raw_thinking_config) if thinking_config: diff --git a/apps/desktop/src/app/chat/composer/reasoning-pill.test.tsx b/apps/desktop/src/app/chat/composer/reasoning-pill.test.tsx index c79aa7d9965e5..9fcb4c679840b 100644 --- a/apps/desktop/src/app/chat/composer/reasoning-pill.test.tsx +++ b/apps/desktop/src/app/chat/composer/reasoning-pill.test.tsx @@ -39,6 +39,16 @@ afterEach(() => { }) describe('ReasoningPill', () => { + it('offers the effort menu without inventing a level for an unset override', () => { + $defaultReasoningEffort.set('high') + render( + + + + ) + expect(screen.getByTestId('reasoning-pill').textContent).toBe('Effort') + expect(screen.getByTestId('reasoning-pill').getAttribute('aria-label')).toBe('Effort') + }) it("shows THIS surface's live effort, falling back to the profile default when the session has none", () => { $defaultReasoningEffort.set('high') diff --git a/apps/desktop/src/app/chat/composer/reasoning-pill.tsx b/apps/desktop/src/app/chat/composer/reasoning-pill.tsx index d13700ced178d..815e33e5cc7de 100644 --- a/apps/desktop/src/app/chat/composer/reasoning-pill.tsx +++ b/apps/desktop/src/app/chat/composer/reasoning-pill.tsx @@ -42,7 +42,7 @@ export function ReasoningPill({ disabled, model }: { disabled: boolean; model: C } const label = reasoningEffortLabel(reasoningEffort || defaultEffort || DEFAULT_REASONING_EFFORT) - const title = `${copy.effort}: ${label}` + const title = label ? `${copy.effort}: ${label}` : copy.effort // Closing the menu ends its claim on the keyboard: Radix restores focus to // this pill (a toolbar button), so without the release the Enter that @@ -67,7 +67,7 @@ export function ReasoningPill({ disabled, model }: { disabled: boolean; model: C type="button" variant="ghost" > - {label} + {label || copy.effort} diff --git a/apps/desktop/src/app/settings/model-settings.tsx b/apps/desktop/src/app/settings/model-settings.tsx index 5f7ad82ebfc9d..8b9f0f0d78c65 100644 --- a/apps/desktop/src/app/settings/model-settings.tsx +++ b/apps/desktop/src/app/settings/model-settings.tsx @@ -1,4 +1,4 @@ -import type { ModelOptionProvider } from '@hermes/shared' +import type { ModelOptionProvider, ReasoningEffort } from '@hermes/shared' import { DEFAULT_REASONING_EFFORT, REASONING_EFFORT_VALUES } from '@hermes/shared' import { useCallback, useEffect, useMemo, useRef, useState } from 'react' @@ -29,6 +29,7 @@ import { useI18n } from '@/i18n' import { isCodeSkewRestartRequired } from '@/lib/code-skew-error' import { AlertTriangle, Cpu, Loader2 } from '@/lib/icons' import { isSubmitEnter } from '@/lib/ime' +import { resolveModelReasoningEffort } from '@/lib/reasoning-effort' import { cn } from '@/lib/utils' import { setMainModelAssignment } from '@/store/cron-model-impact' import { notifyError, readableError } from '@/store/notifications' @@ -562,7 +563,15 @@ export function ModelSettings({ onMainModelChanged, scopeProfile }: ModelSetting .trim() .toLowerCase() - const effortValue = rawEffort === 'false' || rawEffort === 'disabled' ? 'none' : rawEffort || DEFAULT_REASONING_EFFORT + const effortValue = resolveModelReasoningEffort( + rawEffort === 'false' || rawEffort === 'disabled' ? 'none' : rawEffort, + DEFAULT_REASONING_EFFORT, + mainCaps + ) + + const effortChoices = REASONING_EFFORT_VALUES.filter( + value => mainCaps?.reasoning_efforts == null || mainCaps.reasoning_efforts.includes(value) + ) const fastOn = isFastTier(getNested(config ?? {}, 'agent.service_tier')) @@ -940,13 +949,18 @@ export function ModelSettings({ onMainModelChanged, scopeProfile }: ModelSetting {m.reasoning} setValue(event.target.value)} + placeholder={`${bounds.min.toLocaleString()}–${bounds.max.toLocaleString()}`} + step={1} + type="number" + value={value} + /> + + + + ) +} diff --git a/apps/desktop/src/app/shell/reasoning-menu-panel.tsx b/apps/desktop/src/app/shell/reasoning-menu-panel.tsx index b50078df87723..67ee6bb306c1c 100644 --- a/apps/desktop/src/app/shell/reasoning-menu-panel.tsx +++ b/apps/desktop/src/app/shell/reasoning-menu-panel.tsx @@ -40,6 +40,9 @@ export function ReasoningMenuPanel(props: ModelMenuHostProps) { return ( { + it('preserves valid saved budgets and refuses to transfer them to effort-only models', () => { + const caps = { reasoning: true, reasoning_efforts: [], reasoning_budget: { min: 128, max: 32768 } } + expect(resolveModelReasoningEffort('budget:4096', '', caps)).toBe('budget:4096') + expect(resolveModelReasoningEffort('', 'budget:4096', caps)).toBe('budget:4096') + expect(resolveModelReasoningEffort('budget:127', '', caps)).toBe('') + expect(resolveModelReasoningEffort('budget:4096', '', { reasoning_efforts: ['high'] })).toBe('') + expect(reasoningEffortLabel('budget:4096')).toContain('tok') + }) it('labels every level it claims to support', () => { for (const effort of REASONING_EFFORT_VALUES) { expect(reasoningEffortLabel(effort)).not.toBe('') @@ -14,6 +27,20 @@ describe('reasoning-effort', () => { expect(reasoningEffortLabel('bogus')).toBe('bogus') }) + it('keeps the unset state out of visible effort labels', () => { + expect(reasoningEffortLabel('auto')).toBe('') + expect(REASONING_EFFORT_VALUES).not.toContain('auto') + expect(isReasoningEffort('auto')).toBe(false) + expect(resolveModelReasoningEffort('auto', 'high', { reasoning: true, reasoning_efforts: [] })).toBe('auto') + }) + + it('recognizes only real scale levels', () => { + expect(isReasoningEffort(DEFAULT_REASONING_EFFORT)).toBe(true) + expect(isReasoningEffort('HIGH')).toBe(true) + expect(isReasoningEffort('none')).toBe(false) + expect(isReasoningEffort('bogus')).toBe(false) + }) + it('treats empty as inherit and only `none` as off', () => { expect(isThinkingEnabled('none')).toBe(false) expect(isThinkingEnabled('high')).toBe(true) @@ -31,3 +58,12 @@ describe('reasoning-effort', () => { expect(resolveReasoningEffort('bogus')).toBe(DEFAULT_REASONING_EFFORT) }) }) + +it('keeps inheritance, provider default and unsupported settings distinct', () => { + const caps = { reasoning_efforts: ['low', 'high'] } + expect(resolveModelReasoningEffort('', 'high', caps)).toBe('high') + expect(resolveModelReasoningEffort('auto', 'high', caps)).toBe('auto') + expect(resolveModelReasoningEffort('medium', 'high', caps)).toBe('') + expect(resolveModelReasoningEffort('auto', 'none', caps)).toBe('auto') + expect(resolveModelReasoningEffort('', 'high')).toBe(resolveReasoningEffort('', 'high')) +}) diff --git a/apps/desktop/src/lib/reasoning-effort.ts b/apps/desktop/src/lib/reasoning-effort.ts index 87e968ff4c051..b8118c85fd770 100644 --- a/apps/desktop/src/lib/reasoning-effort.ts +++ b/apps/desktop/src/lib/reasoning-effort.ts @@ -1,10 +1,13 @@ import { DEFAULT_REASONING_EFFORT, isReasoningEffort } from '@hermes/shared' +import type { ModelCapabilities } from '@hermes/shared' import { normalize } from '@/lib/text' /** Compact labels for chrome where space is tight (pill, picker rows). Menus * and settings use the translated `shell.modelOptions` strings instead. */ const SHORT_LABELS: Record = { + 'budget:-1': 'Dynamic', + auto: '', // No explicit level; do not invent an effort label. none: 'Off', minimal: 'Min', low: 'Low', @@ -18,7 +21,54 @@ const SHORT_LABELS: Record = { export function reasoningEffortLabel(effort: string): string { const key = normalize(effort) - return key ? (SHORT_LABELS[key] ?? effort) : '' + return /^budget:\d+$/.test(key) + ? `${Number(key.slice(7)).toLocaleString()} tok` + : key + ? (SHORT_LABELS[key] ?? effort) + : '' +} + +/** Unknown saved values stay unselected instead of acquiring a false label. + * Empty inherits; auto explicitly leaves the level to the provider. */ +export function resolveModelReasoningEffort( + effort: string, + inherited: string = '', + capabilities?: Partial< + Pick + > +): string { + if (capabilities?.reasoning === false || capabilities?.reasoning_control === 'unsupported') { + return '' + } + + const value = normalize(effort || inherited) + const allowed = capabilities?.reasoning_efforts + if (value.startsWith('budget:')) { + const budget = capabilities?.reasoning_budget + if (value === 'budget:-1' && budget?.dynamic) return value + const tokens = Number(value.slice(7)) + return budget && + /^budget:\d+$/.test(value) && + Number.isSafeInteger(tokens) && + tokens >= budget.min && + tokens <= budget.max + ? value + : '' + } + + if (value === 'auto') { + return value + } + + if (allowed == null) { + return value === 'none' ? value : resolveReasoningEffort(value) + } + + if (!value) { + return 'auto' + } + + return allowed.includes(value) ? value : '' } /** Thinking is on unless a level explicitly says otherwise; an empty value diff --git a/apps/shared/src/gateway-contract.generated.ts b/apps/shared/src/gateway-contract.generated.ts index e3b1baa0f9eb4..71278f397435c 100644 --- a/apps/shared/src/gateway-contract.generated.ts +++ b/apps/shared/src/gateway-contract.generated.ts @@ -740,6 +740,14 @@ export interface ModelCapabilities { fast: boolean reasoning: boolean can_disable_reasoning?: boolean | null + reasoning_control?: 'adjustable' | 'default' | 'unsupported' | 'unknown' | null + reasoning_efforts?: string[] | null + reasoning_budget?: ReasoningBudget | null +} +export interface ReasoningBudget { + min: number + max: number + dynamic?: boolean } /** ``hermes_cli/inventory.py::_apply_pricing`` — formatted $/Mtok strings (``""`` unknown, ``"free"``); the sale fields are Nous Portal-only. */ export interface ModelPricing { diff --git a/apps/shared/src/gateway-contract.openrpc.json b/apps/shared/src/gateway-contract.openrpc.json index 76bad2cddc0c6..0d54ce3537090 100644 --- a/apps/shared/src/gateway-contract.openrpc.json +++ b/apps/shared/src/gateway-contract.openrpc.json @@ -15148,6 +15148,50 @@ ], "default": null, "title": "Can Disable Reasoning" + }, + "reasoning_control": { + "anyOf": [ + { + "enum": [ + "adjustable", + "default", + "unsupported", + "unknown" + ], + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Reasoning Control" + }, + "reasoning_efforts": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Reasoning Efforts" + }, + "reasoning_budget": { + "anyOf": [ + { + "$ref": "#/components/schemas/ReasoningBudget" + }, + { + "type": "null" + } + ], + "default": null } }, "required": [ @@ -22008,6 +22052,30 @@ "title": "ReadRangeRequestParams", "type": "object" }, + "ReasoningBudget": { + "additionalProperties": false, + "properties": { + "min": { + "title": "Min", + "type": "integer" + }, + "max": { + "title": "Max", + "type": "integer" + }, + "dynamic": { + "default": false, + "title": "Dynamic", + "type": "boolean" + } + }, + "required": [ + "min", + "max" + ], + "title": "ReasoningBudget", + "type": "object" + }, "RecordRepoItem": { "additionalProperties": false, "properties": { diff --git a/hermes_cli/inventory.py b/hermes_cli/inventory.py index 2d4f80047ff87..bc79954f3e545 100644 --- a/hermes_cli/inventory.py +++ b/hermes_cli/inventory.py @@ -269,6 +269,45 @@ def _reasoning_catalog_reader(slug: str): return read +def _validate_model_description(value: dict) -> dict: + from hermes_constants import VALID_REASONING_EFFORTS + + clean = {key: value[key] for key in ('fast', 'reasoning', 'can_disable_reasoning') + if type(value.get(key)) is bool} + state = value.get('reasoning_control') + if state in ('adjustable', 'default', 'unsupported', 'unknown'): + clean['reasoning_control'] = state + efforts = value.get('reasoning_efforts') + budget = value.get('reasoning_budget') + if (isinstance(budget, dict) and type(budget.get('min')) is int + and type(budget.get('max')) is int and 1 <= budget['min'] <= budget['max']): + clean['reasoning_budget'] = {key: budget[key] for key in ('min', 'max')} + if type(budget.get('dynamic')) is bool: + clean['reasoning_budget']['dynamic'] = budget['dynamic'] + if isinstance(efforts, list) and all( + isinstance(effort, str) and effort in ('none', *VALID_REASONING_EFFORTS) + for effort in efforts): + clean['reasoning_efforts'] = list(dict.fromkeys(efforts)) + return clean + + +def _profile_model_descriptions(slug: str, model_ids: list[str]) -> dict: + from providers import get_provider_profile + + profile = get_provider_profile(slug) + if profile is None: + return {} + try: + descriptions = profile.describe_models(model_ids=model_ids) + except Exception: + # Optional metadata must not take down every provider's picker. + return {} + if not isinstance(descriptions, dict): + return {} + return {model: _validate_model_description(value) + for model, value in descriptions.items() if model in model_ids and isinstance(value, dict)} + + def _apply_capabilities(rows: list[dict]) -> None: """Attach ``{model: {fast, reasoning, ...}}`` per row. ``reasoning`` defaults True when the catalog is silent (the dial is a no-op on models that ignore it; hiding it from a capable model is worse). A @@ -286,6 +325,8 @@ def _apply_capabilities(rows: list[dict]) -> None: caps: dict[str, dict[str, Any]] = {} read_reasoning_catalog = _reasoning_catalog_reader(slug.lower()) + descriptions = _profile_model_descriptions(slug, row.get("models") or []) + for model in row.get("models") or []: reasoning = True if get_model_capabilities is not None and slug: @@ -310,6 +351,7 @@ def _apply_capabilities(rows: list[dict]) -> None: elif detail: entry["can_disable_reasoning"] = not detail.get("mandatory") + entry.update(descriptions.get(model, {})) caps[model] = entry row["capabilities"] = caps diff --git a/hermes_cli/models.py b/hermes_cli/models.py index b975b1dee73fc..f54008fe86511 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -1455,6 +1455,8 @@ def _profile_live_catalog(normalized: str) -> Optional[list[str]]: return None api_key, base_url = _api_key_credentials(normalized) live = profile.fetch_models(api_key=api_key, base_url=base_url or profile.base_url or None) if api_key else None + if live is not None and profile.model_catalog_authoritative: + return live if not live: return list(profile.fallback_models) if profile.fallback_models else None curated = list(_PROVIDER_MODELS.get(normalized, [])) or list(profile.fallback_models or ()) diff --git a/hermes_constants.py b/hermes_constants.py index dced36c03e9a0..bc71ea65582ce 100644 --- a/hermes_constants.py +++ b/hermes_constants.py @@ -932,14 +932,19 @@ def parse_reasoning_effort(effort) -> dict | None: ``None`` for empty/unrecognized input (caller uses the default); ``{"enabled": False}`` for "none"/"false"/"disabled"/YAML False — ``reasoning_effort: false`` must mean disabled. + "auto" returns ``{"enabled": True}``, explicitly leaving the level to the provider. """ if effort is None or effort is True: return None effort = str(effort).strip().lower() # False -> "false" -> disabled; "" matches neither set + if effort == "auto": + return {"enabled": True} if effort in {"none", "false", "disabled"}: return {"enabled": False} if effort in VALID_REASONING_EFFORTS: return {"enabled": True, "effort": effort} + if effort.startswith("budget:") and len(effort[7:]) <= 10 and (effort[7:] == "-1" or (effort[7:].isascii() and effort[7:].isdigit())): + return {"enabled": True, "effort": f"budget:{int(effort[7:])}"} return None diff --git a/plugins/model-providers/gemini/__init__.py b/plugins/model-providers/gemini/__init__.py index ad99a7b3f367a..75efe6923d901 100644 --- a/plugins/model-providers/gemini/__init__.py +++ b/plugins/model-providers/gemini/__init__.py @@ -13,16 +13,31 @@ class GeminiProfile(ProviderProfile): """Gemini — translate reasoning_config to thinking_config in extra_body.""" + def fetch_models(self, *, api_key=None, base_url=None, timeout=8.0): + from urllib.parse import urlparse + from agent.gemini_model_catalog import fetch_models + if base_url and urlparse(base_url).hostname != "generativelanguage.googleapis.com": + return super().fetch_models(api_key=api_key, base_url=base_url, timeout=timeout) + return fetch_models(api_key, timeout=timeout) + + def describe_models(self, *, model_ids: list[str]) -> dict: + from agent.gemini_catalog_reasoning import describe_thinking_control + return {model: describe_thinking_control(model) for model in model_ids} + + def supported_reasoning_efforts(self, model: str | None) -> tuple[str, ...]: + from agent.gemini_catalog_reasoning import supported_efforts + return supported_efforts(model) + def build_extra_body(self, *, session_id: str | None = None, **context: Any) -> dict[str, Any]: """Native: ``thinking_config``; OpenAI-compat /openai subpath: ``extra_body.google.thinking_config`` (snake_case).""" from agent.transports.chat_completions import ( - _build_gemini_thinking_config, _is_gemini_openai_compat_base_url, _snake_case_gemini_thinking_config, ) - raw = _build_gemini_thinking_config(context.get("model") or "", context.get("reasoning_config")) + from agent.gemini_catalog_reasoning import build_thinking_config + raw = build_thinking_config(context.get("model") or "", context.get("reasoning_config")) if not raw: return {} if self.name == "gemini" and _is_gemini_openai_compat_base_url(context.get("base_url") or self.base_url): @@ -33,6 +48,7 @@ def build_extra_body(self, *, session_id: str | None = None, **context: Any) -> gemini = GeminiProfile( name="gemini", aliases=("google", "google-gemini", "google-ai-studio"), api_mode="chat_completions", + validate_reasoning_selection=True, model_catalog_authoritative=True, env_vars=("GOOGLE_API_KEY", "GEMINI_API_KEY"), base_url="https://generativelanguage.googleapis.com/v1beta", auth_type="api_key", default_aux_model="gemini-3.6-flash", diff --git a/providers/base.py b/providers/base.py index 0941c5d8d7c43..c5d610c81131a 100644 --- a/providers/base.py +++ b/providers/base.py @@ -57,6 +57,8 @@ class ProviderProfile: supports_health_check: bool = True # False → doctor skips /models probe for this provider # False → fetch_models returns None without a network call (catalog comes from an SDK/subprocess). supports_model_listing: bool = True + model_catalog_authoritative: bool = False # successful discovery replaces curated IDs + validate_reasoning_selection: bool = False # opt in; other providers keep their existing policy # ── Vision support ──────────────────────────────────────── # True when the provider's API accepts image content inside @@ -114,6 +116,14 @@ class ProviderProfile: # ── Hooks (override in subclass for complex providers) ─── + def describe_models(self, *, model_ids: list[str]) -> dict[str, dict[str, Any]]: + """Optional model controls, from cached metadata only (no network/credentials). + + Explicit reasoning_efforts constrain selectable levels, including an empty + list. Providers must use the same metadata when building requests. + """ + return {} + def resolve_aux_model(self, *, vision: bool = False) -> str: """Return a LIVE cheap-model id for auxiliary tasks, or "". diff --git a/providers/reasoning.py b/providers/reasoning.py new file mode 100644 index 0000000000000..bebec53e4a888 --- /dev/null +++ b/providers/reasoning.py @@ -0,0 +1,67 @@ +"""Resolve settings for providers that opt into validated reasoning selection.""" + +import logging + +from agent.reasoning_effort import clamp_effort +from providers import get_provider_profile + + +def reasoning_selection_efforts(provider: str, model: str) -> tuple[str, ...] | None: + profile = get_provider_profile(provider) + if profile is None or not profile.validate_reasoning_selection: + return None + return profile.supported_reasoning_efforts(model) + + +def resolve_provider_reasoning_config( + provider: str, model: str, config: dict | None, *, explicit: bool = False +) -> dict | None: + effort = "none" if config and config.get("enabled") is False else str((config or {}).get("effort") or "").strip().lower() + if effort.startswith("budget:"): + profile = get_provider_profile(provider) + bounds = (profile.describe_models(model_ids=[model]).get(model, {}).get("reasoning_budget") + if profile else None) + value = effort[7:] + if bounds and value == "-1" and bounds.get("dynamic") is True: + return config + if (bounds and value.isascii() and value.isdigit() + and bounds["min"] <= int(value) <= bounds["max"]): + return config + if explicit: + raise ValueError(f"Unsupported thinking token budget for {provider}/{model}.") + return {"enabled": True} + supported = reasoning_selection_efforts(provider, model) + if supported is None: + return config + if config is None: + return {"enabled": True} # provider default, distinct from inheritance + effort = "none" if config.get("enabled") is False else str(config.get("effort") or "").strip().lower() + if not effort or effort in supported: + return config + if explicit: + if not supported: + raise ValueError( + f"No verified reasoning effort controls in Hermes for {provider}/{model}; choose auto." + ) + raise ValueError( + f"Unsupported reasoning effort {effort!r} for {provider}/{model}; choose auto" + + (f" or {', '.join(supported)}" if supported else "") + "." + ) + # Old presets keep the existing nearest-weaker mapping; report the actual + # level instead of leaving e.g. medium in the UI while sending low. + levels = tuple(level for level in supported if level != "none") + resolved = clamp_effort(effort, levels) if effort != "none" and levels else None + result = {"enabled": True, **({"effort": resolved} if resolved in levels else {})} + logging.getLogger(__name__).warning( + "Reasoning setting for %s/%s changed from %s to %s", + provider, model, effort, result.get("effort", "auto"), + ) + return result + + +def sync_primary_reasoning(agent) -> None: + """Keep accepted edits when the primary runtime is restored after a fallback.""" + primary = getattr(agent, "_primary_runtime", None) + if primary and (primary.get("provider"), primary.get("model")) == (agent.provider, agent.model): + config = getattr(agent, "reasoning_config", None) + primary["reasoning_config"] = dict(config) if config is not None else None diff --git a/tests/agent/test_gemini_catalog_reasoning.py b/tests/agent/test_gemini_catalog_reasoning.py new file mode 100644 index 0000000000000..5374e236200da --- /dev/null +++ b/tests/agent/test_gemini_catalog_reasoning.py @@ -0,0 +1,125 @@ +"""Metadata drives UI choices, selection validation and both Gemini wires.""" +import io +import json + +import pytest + +from agent import models_dev +from agent.gemini_catalog_reasoning import describe_thinking_control +from hermes_constants import parse_reasoning_effort +from providers import get_provider_profile +from providers.reasoning import resolve_provider_reasoning_config + +from agent.transports.chat_completions import ChatCompletionsTransport + + +def select_reasoning(model, effort): + return resolve_provider_reasoning_config( + "gemini", model, parse_reasoning_effort(effort), explicit=True) + + +@pytest.fixture +def catalog(monkeypatch): + entries = { + "gemini-future-levels": [{"type": "effort", "values": ["low", "high"]}], + "gemini-future-budget": [{"type": "toggle"}, {"type": "budget_tokens", "min": 512, "max": 24576}], + "gemini-fixed": [], "gemini-unknown": None, + "gemini-mandatory-budget": [{"type": "budget_tokens", "min": 128, "max": 32768}], + } + models = {key: {"reasoning": True, "reasoning_options": value, "tool_call": True, + "modalities": {"output": ["text"]}} for key, value in entries.items()} + monkeypatch.setattr(models_dev, "fetch_models_dev", lambda **kwargs: {"google": {"models": models}}) + return models + + +def test_metadata_drives_both_wires_and_rejects_impossible_choices(catalog): + profile = get_provider_profile("gemini") + descriptor = describe_thinking_control("gemini-future-levels") + assert descriptor["reasoning_efforts"] == ["low", "high"] + assert not descriptor["can_disable_reasoning"] + for effort in ("low", "high"): + config = select_reasoning("gemini-future-levels", effort) + assert profile.build_extra_body(model="gemini-future-levels", reasoning_config=config) == { + "thinking_config": {"includeThoughts": True, "thinkingLevel": effort}} + with pytest.raises(ValueError): + select_reasoning("gemini-future-levels", "none") + for effort in ("budget:-1", "budget:512", "budget:4096", "budget:24576"): + config = select_reasoning("gemini-future-budget", effort) + kwargs = ChatCompletionsTransport().build_kwargs( + model="gemini-future-budget", messages=[{"role": "user", "content": "hi"}], + provider_profile=profile, provider_name="gemini", base_url=profile.base_url + "/openai", + reasoning_config=config) + assert kwargs["extra_body"]["extra_body"]["google"]["thinking_config"] == { + "include_thoughts": True, "thinking_budget": int(effort[7:])} + for effort in ("budget:0", "budget:511", "budget:24577", "high"): + with pytest.raises(ValueError): + select_reasoning("gemini-future-budget", effort) + assert profile.build_extra_body(model="gemini-future-budget", reasoning_config=parse_reasoning_effort("none")) == { + "thinking_config": {"includeThoughts": False, "thinkingBudget": 0}} + assert describe_thinking_control("gemini-mandatory-budget")["can_disable_reasoning"] is False + assert describe_thinking_control("gemini-fixed")["reasoning_control"] == "default" + assert describe_thinking_control("gemini-unknown")["reasoning_control"] == "unknown" + catalog["gemma-4-fixture"] = {"reasoning": True, "reasoning_options": [{"type": "toggle"}]} + assert profile.build_extra_body(model="gemma-4-fixture", reasoning_config={"enabled": False}) == {} + assert profile.build_extra_body(model="gemini-unknown", reasoning_config={"enabled": False}) == {} + # A metadata update changes options without a release or family-name guesses. + catalog["gemini-future-levels"]["reasoning_options"][0]["values"] = ["medium"] + assert profile.supported_reasoning_efforts("gemini-future-levels") == ("medium",) + + +def test_native_catalog_pagination_filters_and_never_returns_partial(catalog, monkeypatch): + import agent.gemini_model_catalog as native + monkeypatch.setattr(native, "fetch_models_dev", models_dev.fetch_models_dev) + catalog["gemini-future-computer-use-preview"] = catalog["gemini-future-levels"] + pages = [ + {"models": [{"name": "models/gemini-future-levels", "supportedGenerationMethods": ["generateContent"]}], + "nextPageToken": "opaque +/token"}, + {"models": [{"name": "models/gemini-future-budget", "supportedGenerationMethods": ["generateContent"]}, + {"name": "models/gemini-future-computer-use-preview", "supportedGenerationMethods": ["generateContent"]}, + {"name": "models/gemini-fixed", "supportedGenerationMethods": ["embedContent"]}]}, + ] + requests = [] + def open_url(req, **kwargs): + requests.append(req) + return io.StringIO(json.dumps(pages[len(requests) - 1])) + monkeypatch.setattr(native, "open_credentialed_url", open_url) + assert native.fetch_models("fixture-secret") == ["gemini-future-levels", "gemini-future-budget"] + assert "pageToken=opaque+%2B%2Ftoken" in requests[1].full_url + assert "fixture-secret" not in requests[0].full_url + assert requests[0].get_header("X-goog-api-key") == "fixture-secret" + requests.clear() + pages[1] = {"error": "unavailable"} + assert native.fetch_models("fixture-secret") is None + + +def test_successful_native_list_does_not_reintroduce_curated_ids(monkeypatch): + from hermes_cli import models + profile = get_provider_profile("gemini") + monkeypatch.setattr(models, "_api_key_credentials", lambda name: ("fixture-key", profile.base_url)) + monkeypatch.setattr(profile, "fetch_models", lambda **kwargs: ["gemini-current-fixture"]) + assert models._profile_live_catalog("gemini") == ["gemini-current-fixture"] + monkeypatch.setattr(profile, "fetch_models", lambda **kwargs: []) + assert models._profile_live_catalog("gemini") == [] + + +def test_inventory_publishes_the_same_controls_as_the_wire(catalog): + from hermes_cli.inventory import _apply_capabilities + from tui_gateway.contracts.config_free_tier_control import ModelCapabilities + rows = [{"slug": "gemini", "models": list(catalog)}] + _apply_capabilities(rows) + caps = rows[0]["capabilities"] + assert caps["gemini-future-levels"]["reasoning_efforts"] == ["low", "high"] + assert caps["gemini-future-levels"]["fast"] is False + assert caps["gemini-future-budget"]["reasoning_budget"] == { + "min": 512, "max": 24576, "dynamic": True} + assert caps["gemini-unknown"]["reasoning_control"] == "unknown" + for value in caps.values(): + ModelCapabilities.model_validate(value) + + +@pytest.mark.parametrize("options", [None, {}, [{"type": "new"}], + [{"type": "effort", "values": ["bogus"]}], + [{"type": "budget_tokens", "min": True, "max": 100}], + [{"type": "budget_tokens", "min": 101, "max": 100}]]) +def test_invalid_metadata_is_unknown(options): + assert models_dev._parse_reasoning_options(options) is None diff --git a/tests/agent/test_provider_parity.py b/tests/agent/test_provider_parity.py index 5deba0ebcd1a4..ee041d868f5a4 100644 --- a/tests/agent/test_provider_parity.py +++ b/tests/agent/test_provider_parity.py @@ -922,3 +922,6 @@ def test_codex_default_medium(self, monkeypatch): + + +pytestmark = pytest.mark.usefixtures("gemini_reasoning_catalog") diff --git a/tests/agent/transports/test_chat_completions.py b/tests/agent/transports/test_chat_completions.py index 4a4541d0e221c..a7ae004a8978a 100644 --- a/tests/agent/transports/test_chat_completions.py +++ b/tests/agent/transports/test_chat_completions.py @@ -1036,3 +1036,6 @@ def test_whitespace_only_caller_key_is_dropped(self, transport): request_overrides={"prompt_cache_key": " "}, ) assert "prompt_cache_key" not in kwargs + + +pytestmark = pytest.mark.usefixtures("gemini_reasoning_catalog") diff --git a/tests/conftest.py b/tests/conftest.py index 45d67cf7a9633..002cd48a92a9e 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1803,3 +1803,27 @@ def _moa_caches_isolated(): yield moa._preset_cache.clear() moa._runtime_cache.clear() + + +@pytest.fixture +def gemini_reasoning_catalog(monkeypatch): + """Hermetic models.dev fixture for tests of Gemini's catalog-dependent wire.""" + from agent import models_dev + levels = { + "gemini-3-pro-preview": ["low", "high"], + "gemini-3.1-pro": ["low", "medium", "high"], + "gemini-3.1-pro-preview": ["low", "medium", "high"], + "gemini-3.1-pro-preview-customtools": ["low", "medium", "high"], + "gemini-3-flash-preview": ["minimal", "low", "medium", "high"], + "gemini-3.6-flash": ["minimal", "low", "medium", "high"], + "gemini-3.7-flash": ["low", "medium", "high"], + "gemini-3.8-flash": ["low", "medium", "high"], + "gemini-flash-latest": ["low", "medium", "high"], + "gemini-3-pro-image-preview": ["high"], + } + models = {key: {"reasoning": True, "reasoning_options": [{"type": "effort", "values": value}]} + for key, value in levels.items()} + models["gemini-2.5-flash"] = {"reasoning": True, "reasoning_options": [ + {"type": "toggle"}, {"type": "budget_tokens", "min": 0, "max": 24576}]} + monkeypatch.setattr(models_dev, "fetch_models_dev", lambda **kwargs: {"google": {"models": models}}) + return models diff --git a/tests/tui_gateway/test_gemini_reasoning_controls.py b/tests/tui_gateway/test_gemini_reasoning_controls.py new file mode 100644 index 0000000000000..1dae7a0cdc529 --- /dev/null +++ b/tests/tui_gateway/test_gemini_reasoning_controls.py @@ -0,0 +1,72 @@ +"""AI Studio selections cross the real gateway boundary without paid requests.""" +from types import SimpleNamespace +from unittest.mock import patch + +import pytest + +import tui_gateway.server as server + +pytestmark = pytest.mark.usefixtures("gemini_reasoning_catalog") + + +def make_session(model="gemini-2.5-flash"): + agent = SimpleNamespace(model=model, provider="gemini", session_id="test-key", + service_tier=None, reasoning_config={"enabled": True}, + _primary_runtime={"provider": "gemini", "model": model}) + return {"session_key": "test-key", "agent": agent} + + +@pytest.mark.parametrize("effort", ["auto", "budget:-1", "budget:4096", "none"]) +def test_reasoning_round_trip_and_primary_runtime(effort): + from hermes_constants import parse_reasoning_effort + session = make_session() + with patch.dict(server._sessions, {"studio-test": session}), \ + patch.object(server, "_write_config_key") as write, \ + patch.object(server, "_persist_live_session_runtime") as persist, \ + patch.object(server, "_emit"): + response = server._methods["config.set"]("r", { + "key": "reasoning", "session_id": "studio-test", "value": effort}) + assert response["result"]["value"] == effort + snapshot = server._session_info(session["agent"]) + assert snapshot["reasoning_effort"] == effort + assert session["create_reasoning_override"] == parse_reasoning_effort(effort) + assert session["agent"]._primary_runtime["reasoning_config"] == parse_reasoning_effort(effort) + persist.assert_called_once() + write.assert_not_called() + + +@pytest.mark.parametrize("effort", ["budget:24577", "high", "budget:0"]) +def test_invalid_budget_or_effort_does_not_mutate_session(effort): + session = make_session() + before = dict(session["agent"].reasoning_config) + with patch.dict(server._sessions, {"studio-test": session}), \ + patch.object(server, "_write_config_key") as write, \ + patch.object(server, "_persist_live_session_runtime") as persist: + response = server._methods["config.set"]("r", { + "key": "reasoning", "session_id": "studio-test", "value": effort}) + assert response["error"]["code"] == 4002 + assert session["agent"].reasoning_config == before + assert "create_reasoning_override" not in session + write.assert_not_called() + persist.assert_not_called() + + +def test_mandatory_thinking_cannot_be_disabled(): + session = make_session("gemini-3.8-flash") + with patch.dict(server._sessions, {"studio-test": session}): + response = server._methods["config.set"]("r", { + "key": "reasoning", "session_id": "studio-test", "value": "none"}) + assert response["error"]["code"] == 4002 + assert session["agent"].reasoning_config == {"enabled": True} + + +def test_running_turn_cannot_receive_controls_for_a_pending_different_model(): + session = make_session("gemini-3.8-flash") + session.update(running=True, pending_model_switch={ + "display_model": "gemini-2.5-flash", "display_provider": "gemini"}) + with patch.dict(server._sessions, {"studio-test": session}): + response = server._methods["config.set"]("r", { + "key": "reasoning", "session_id": "studio-test", "value": "budget:4096"}) + assert response["error"]["code"] == 4009 + assert session["agent"].reasoning_config == {"enabled": True} + assert "create_reasoning_override" not in session diff --git a/tui_gateway/contracts/config_free_tier_control.py b/tui_gateway/contracts/config_free_tier_control.py index afe5a22cb65c6..e4e45ec1df19f 100644 --- a/tui_gateway/contracts/config_free_tier_control.py +++ b/tui_gateway/contracts/config_free_tier_control.py @@ -236,12 +236,21 @@ class ModelPricing(Result): was_output: str | None = None +class ReasoningBudget(Result): + min: int + max: int + dynamic: bool = False + + class ModelCapabilities(Result): """``hermes_cli/inventory.py::_apply_capabilities``.""" fast: bool reasoning: bool can_disable_reasoning: bool | None = None + reasoning_control: Literal['adjustable', 'default', 'unsupported', 'unknown'] | None = None + reasoning_efforts: list[str] | None = None + reasoning_budget: ReasoningBudget | None = None class ModelOptionProvider(OpenModel): diff --git a/tui_gateway/methods_config.py b/tui_gateway/methods_config.py index 424fb48210395..6d54f319e0b55 100644 --- a/tui_gateway/methods_config.py +++ b/tui_gateway/methods_config.py @@ -156,7 +156,7 @@ def _cfg_get_reasoning(params): reasoning_config = getattr(session.get("agent"), "reasoning_config", None) if isinstance(reasoning_config, dict): enabled = reasoning_config.get("enabled") is not False - effort = str(reasoning_config.get("effort") or "medium") if enabled else "none" + effort = str(reasoning_config.get("effort") or "auto") if enabled else "none" else: raw_effort = (cfg.get("agent") or {}).get("reasoning_effort", "") # YAML `reasoning_effort: false` means thinking disabled, not "unset". diff --git a/tui_gateway/methods_config_set.py b/tui_gateway/methods_config_set.py index 3163089bc8077..b42ce31b0d086 100644 --- a/tui_gateway/methods_config_set.py +++ b/tui_gateway/methods_config_set.py @@ -307,6 +307,20 @@ def _set_reasoning(rid, params, key, value, session): parsed = parse_reasoning_effort(arg) if parsed is None: return _err(rid, 4002, f"unknown reasoning value: {value}") + if session is not None: + from providers import get_provider_profile + from providers.reasoning import resolve_provider_reasoning_config + agent = session.get("agent") + target = session.get("pending_model_switch") or session.get("model_override") or {} + model = target.get("display_model") or target.get("model") or getattr(agent, "model", "") + provider = target.get("display_provider") or target.get("provider") or getattr(agent, "provider", "") + try: + parsed = resolve_provider_reasoning_config(provider, model, parsed, explicit=True) + except ValueError as exc: + return _err(rid, 4002, str(exc)) + profile = get_provider_profile(provider) + if profile and profile.validate_reasoning_selection and session.get("running"): + return _err(rid, 4009, "Wait for the current turn before changing reasoning controls.") if scope == "global" or session is None: _write_config_key("agent.reasoning_effort", arg) if session is not None: @@ -320,6 +334,8 @@ def _set_reasoning(rid, params, key, value, session): session["create_reasoning_override"] = parsed if session and session.get("agent") is not None: session["agent"].reasoning_config = parsed + from providers.reasoning import sync_primary_reasoning + sync_primary_reasoning(session["agent"]) _persist_live_session_runtime(session) _emit_session_info(params.get("session_id", ""), session) return _kv(rid, key, arg) diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index d111623a89094..df46a82b291d5 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -330,6 +330,15 @@ def _(rid, params: dict) -> dict: # ``profile`` (app-global remote mode): stored so the build and every turn re-bind HERMES_HOME. profile_home = _profile_home(profile := (params.get("profile") or "").strip() or None) session_model_override, create_reasoning_override, create_service_tier_override = _create_overrides(params) + if session_model_override and create_reasoning_override is not None: + from providers.reasoning import resolve_provider_reasoning_config + try: + with _profile_build_scope(profile_home): + create_reasoning_override = resolve_provider_reasoning_config( + session_model_override.get("provider") or "", session_model_override["model"], + create_reasoning_override, explicit=True) + except ValueError as exc: + return _err(rid, 4002, str(exc)) now = time.time() with _sessions_lock: _sessions[sid] = { diff --git a/tui_gateway/model_switch.py b/tui_gateway/model_switch.py index a45804cc9dcc8..a205a9d461983 100644 --- a/tui_gateway/model_switch.py +++ b/tui_gateway/model_switch.py @@ -227,6 +227,11 @@ def _apply_model_switch( custom_providers=custom_provs) if not result.success: raise ValueError(result.error_message or "model switch failed") + if reasoning_effort: + from hermes_constants import parse_reasoning_effort + from providers.reasoning import resolve_provider_reasoning_config + resolve_provider_reasoning_config(result.target_provider, result.new_model, + parse_reasoning_effort(reasoning_effort), explicit=True) restore_snapshot = _snapshot_agent_model_runtime(agent) if (one_turn and agent) else None if agent: _merge_preflight_warning(result, agent, session, cfg, custom_provs) @@ -265,6 +270,8 @@ def _apply_switch_reasoning(sid: str, session, agent, effort: str, *, persist_gl return if agent is not None: agent.reasoning_config = parsed + from providers.reasoning import sync_primary_reasoning + sync_primary_reasoning(agent) if one_turn or not isinstance(session, dict): return if persist_global: diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 99ef6639f111d..76bd3a12e3efb 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -2078,7 +2078,7 @@ def _session_info(agent, session: dict | None = None) -> dict: reasoning_effort = "" if isinstance(reasoning_config, dict): # Disabled must differ from unset ("" = provider default) or the desktop loses "thinking off" after turn 1. - reasoning_effort = "none" if reasoning_config.get("enabled") is False else str(reasoning_config.get("effort", "") or "") + reasoning_effort = "none" if reasoning_config.get("enabled") is False else str(reasoning_config.get("effort") or "auto") service_tier = getattr(agent, "service_tier", None) or mirror.get("service_tier") or "" # yolo ORs the same three sources check_all_command_guards() does (approvals.mode=off, the process # --yolo env, the per-session flag): the session flag alone would show "off" while config auto-approves. diff --git a/website/docs/guides/google-gemini.md b/website/docs/guides/google-gemini.md index d047916b7d354..80e21bcc4e635 100644 --- a/website/docs/guides/google-gemini.md +++ b/website/docs/guides/google-gemini.md @@ -8,6 +8,39 @@ description: "Use Hermes Agent with Google Gemini — native AI Studio API, API- Hermes Agent supports Google Gemini as a native provider using the **Google AI Studio / Gemini API** — not the OpenAI-compatible endpoint. This lets Hermes translate its internal OpenAI-shaped message and tool loop into Gemini's native `generateContent` API while preserving tool calling, streaming, multimodal inputs, and Gemini-specific response metadata. +## Model catalog and thinking controls + +The model picker reads Google's native model list, including pagination, and +uses Hermes' cached [models.dev](https://models.dev) metadata to keep text-output +models with function calling. A successful discovery replaces the curated list; +on failure Hermes retains its existing fallback behavior. Models that require +the dedicated Computer Use tool are excluded from this general chat picker. +Being listed does not guarantee that your project has access to a model. + +Desktop shows the thinking controls described for each model: + +- **Effort:** only the model's declared levels. +- **Thinking off:** only when disabling thinking is declared. +- **Thinking budget:** a token count within the model's bounds, with **Dynamic** + for Google's `thinkingBudget: -1` mode. + +When no explicit level or budget is set, the picker leaves the options +unselected rather than displaying a synthetic "Provider default" effort. +Omitting the override differs from Dynamic, particularly for models whose +default is to leave thinking off. + +Missing or unrecognized metadata leaves control to Google and is shown as +unverified. A model's name containing “Flash” does not enable a Fast switch. +The picker and the AI Studio request adapter use the same cached description; +catalog refreshes do not make generation requests. + +For configuration, `agent.reasoning_effort` additionally accepts `auto` and +`budget:N` (including `budget:-1`). Explicit unsupported session choices are +rejected before saving. Inherited settings from another model are normalized +to a supported level or provider default. These controls apply to AI Studio; +Vertex retains its separate policy. See Google's +[generateContent thinking guide](https://ai.google.dev/gemini-api/docs/generate-content/thinking). + ## Prerequisites - **Google AI Studio API key** — create one at [aistudio.google.com/apikey](https://aistudio.google.com/apikey) From 1dca4f60c8057a2f09edb091a856c25b34870033 Mon Sep 17 00:00:00 2001 From: jkorzeniak Date: Tue, 15 Sep 2026 18:36:53 +0200 Subject: [PATCH 2/4] fix(gemini): isolate malformed catalog entries and verify thinking gates A-061 review follow-up: preserve valid paginated discovery results, document budget sentinels, and cover Off validation through inventory, gateway and request builders. --- agent/gemini_catalog_reasoning.py | 8 +- agent/gemini_model_catalog.py | 10 ++- tests/agent/test_gemini_catalog_reasoning.py | 75 +++++++++++++++++++ .../test_gemini_reasoning_controls.py | 18 ++++- website/docs/guides/google-gemini.md | 8 ++ 5 files changed, 112 insertions(+), 7 deletions(-) diff --git a/agent/gemini_catalog_reasoning.py b/agent/gemini_catalog_reasoning.py index cc8a672b69fa0..3acdf80a20d71 100644 --- a/agent/gemini_catalog_reasoning.py +++ b/agent/gemini_catalog_reasoning.py @@ -28,7 +28,10 @@ def describe_thinking_control(model: str) -> dict: elif option["type"] == "toggle": efforts.append("none") elif option["type"] == "budget_tokens": - # Do not invent a missing bound or expose zero as "thinking on". + # These bounds are for the Thinking-on numeric input, not a copy of + # the raw API range. Zero is Off (requires a declared toggle), and + # -1 is Dynamic; neither may bypass those separate controls here. + # Preserve positive minima and never invent a missing bound. if "min" in option and "max" in option and option["max"] > 0: result["reasoning_budget"] = {"min": max(1, option["min"]), "max": option["max"], "dynamic": True} # -1 is generateContent's dynamic sentinel, not a model effort. @@ -49,6 +52,9 @@ def build_thinking_config(model: str, reasoning_config: dict | None) -> dict | N descriptor = describe_thinking_control(model) if descriptor["reasoning_control"] in ("unknown", "unsupported"): return None + # The resolver shares the picker's declared Off capability. Inherited Off + # on a mandatory-thinking model becomes an unset override before this branch; + # explicit selections are rejected by the same resolver at the gateway. config = resolve_provider_reasoning_config("gemini", model, reasoning_config) if not config: return None diff --git a/agent/gemini_model_catalog.py b/agent/gemini_model_catalog.py index 4a442158a9ec0..f433f1bd5e1e5 100644 --- a/agent/gemini_model_catalog.py +++ b/agent/gemini_model_catalog.py @@ -2,7 +2,8 @@ Google determines listed IDs; models.dev supplies text/tool capabilities. Unknown capabilities remain usable by explicit model ID, not advertised as -verified agent models. Failures never return a partial/authoritative catalog. +verified agent models. Failed pages never return a partial/authoritative catalog; +malformed individual entries are skipped. """ import json import time @@ -36,15 +37,18 @@ def fetch_models(api_key: str | None, *, timeout: float = 8.0) -> list[str] | No return None for entry in data["models"]: if not isinstance(entry, dict) or not isinstance(entry.get("name"), str): - return None + continue model = entry["name"].removeprefix("models/") + methods = entry.get("supportedGenerationMethods") + if not model or not isinstance(methods, list): + continue # Dedicated Computer Use routes require Google's built-in tool; # Hermes' generic function tools cannot invoke them. The native # list and models.dev's tool_call flag do not encode this prerequisite. if "-computer-use-" in model: continue info = get_model_info("gemini", model, allow_network=False) - if ("generateContent" in (entry.get("supportedGenerationMethods") or []) + if ("generateContent" in methods and info and info.tool_call and info.output_modalities == ("text",) and info.status != "deprecated"): result.append(model) diff --git a/tests/agent/test_gemini_catalog_reasoning.py b/tests/agent/test_gemini_catalog_reasoning.py index 5374e236200da..50f4bdcebe664 100644 --- a/tests/agent/test_gemini_catalog_reasoning.py +++ b/tests/agent/test_gemini_catalog_reasoning.py @@ -102,6 +102,81 @@ def test_successful_native_list_does_not_reintroduce_curated_ids(monkeypatch): assert models._profile_live_catalog("gemini") == [] +@pytest.mark.parametrize("bad_entry", [ + None, {}, {"name": None}, {"name": 42}, {"name": ""}, + {"name": "models/gemini-fixed", "supportedGenerationMethods": 42}, + {"name": "models/gemini-fixed", "supportedGenerationMethods": "generateContent"}, +]) +def test_bad_catalog_entry_does_not_hide_valid_models_on_other_pages(catalog, monkeypatch, bad_entry): + import agent.gemini_model_catalog as native + + monkeypatch.setattr(native, "fetch_models_dev", models_dev.fetch_models_dev) + pages = iter([ + {"models": [bad_entry, {"name": "models/gemini-future-levels", + "supportedGenerationMethods": ["generateContent"]}], + "nextPageToken": "next"}, + {"models": [{"name": "models/gemini-future-budget", + "supportedGenerationMethods": ["generateContent"]}]}, + ]) + monkeypatch.setattr(native, "open_credentialed_url", lambda *args, **kwargs: io.StringIO(json.dumps(next(pages)))) + assert native.fetch_models("fixture-key") == ["gemini-future-levels", "gemini-future-budget"] + + +@pytest.mark.parametrize("minimum", [-1, 0, 512]) +@pytest.mark.parametrize("can_disable", [False, True]) +def test_off_and_dynamic_stay_separate_from_positive_budgets_on_both_wires(catalog, minimum, can_disable): + from agent.gemini_native_adapter import build_gemini_request + from hermes_cli.inventory import _apply_capabilities + + model = "gemini-budget-fixture" + options = [{"type": "budget_tokens", "min": minimum, "max": 8192}] + if can_disable: + options.append({"type": "toggle"}) + catalog[model] = {"reasoning": True, "reasoning_options": options} + rows = [{"slug": "gemini", "models": [model]}] + _apply_capabilities(rows) + caps = rows[0]["capabilities"][model] + # Preserve source metadata, but expose only positive values in the On input. + assert models_dev.get_model_info("gemini", model, allow_network=False).reasoning_options[0]["min"] == minimum + assert caps["reasoning_budget"]["min"] == max(1, minimum) + assert caps["can_disable_reasoning"] is can_disable + with pytest.raises(ValueError): + select_reasoning(model, "budget:0") + if can_disable: + assert select_reasoning(model, "none") == {"enabled": False} + else: + with pytest.raises(ValueError): + select_reasoning(model, "none") + + profile = get_provider_profile("gemini") + # Include inherited Off, bypassing explicit selection, to exercise the + # request-time normalizer as well as the gateway's selection gate. + choices = [("none", 0 if can_disable else None), ("budget:0", None), + ("budget:-1", -1), (f"budget:{max(1, minimum)}", max(1, minimum))] + for effort, expected_budget in choices: + for use_profile in (False, True): + for compat in (False, True): + kwargs = ChatCompletionsTransport().build_kwargs( + model=model, messages=[{"role": "user", "content": "hi"}], + provider_profile=profile if use_profile else None, provider_name="gemini", + base_url=profile.base_url + ("/openai" if compat else ""), + reasoning_config=parse_reasoning_effort(effort)) + if compat: + wire = kwargs["extra_body"]["extra_body"]["google"]["thinking_config"] + budget_key, include_key = "thinking_budget", "include_thoughts" + else: + request = build_gemini_request( + model=model, messages=kwargs["messages"], + thinking_config=kwargs["extra_body"]["thinking_config"]) + wire = request["generationConfig"]["thinkingConfig"] + budget_key, include_key = "thinkingBudget", "includeThoughts" + if expected_budget is None: + assert budget_key not in wire + else: + assert wire[budget_key] == expected_budget + assert wire[include_key] is (expected_budget != 0) + + def test_inventory_publishes_the_same_controls_as_the_wire(catalog): from hermes_cli.inventory import _apply_capabilities from tui_gateway.contracts.config_free_tier_control import ModelCapabilities diff --git a/tests/tui_gateway/test_gemini_reasoning_controls.py b/tests/tui_gateway/test_gemini_reasoning_controls.py index 1dae7a0cdc529..f2a7e4660ecc5 100644 --- a/tests/tui_gateway/test_gemini_reasoning_controls.py +++ b/tests/tui_gateway/test_gemini_reasoning_controls.py @@ -51,13 +51,25 @@ def test_invalid_budget_or_effort_does_not_mutate_session(effort): persist.assert_not_called() -def test_mandatory_thinking_cannot_be_disabled(): - session = make_session("gemini-3.8-flash") - with patch.dict(server._sessions, {"studio-test": session}): +@pytest.mark.parametrize("options", [ + [{"type": "effort", "values": ["low", "high"]}], + [{"type": "budget_tokens", "min": 0, "max": 8192}], +]) +def test_mandatory_thinking_cannot_be_disabled(gemini_reasoning_catalog, options): + model = "gemini-mandatory-fixture" + gemini_reasoning_catalog[model] = {"reasoning": True, "reasoning_options": options} + session = make_session(model) + with patch.dict(server._sessions, {"studio-test": session}), \ + patch.object(server, "_write_config_key") as write, \ + patch.object(server, "_persist_live_session_runtime") as persist: response = server._methods["config.set"]("r", { "key": "reasoning", "session_id": "studio-test", "value": "none"}) assert response["error"]["code"] == 4002 assert session["agent"].reasoning_config == {"enabled": True} + assert "create_reasoning_override" not in session + assert "reasoning_config" not in session["agent"]._primary_runtime + write.assert_not_called() + persist.assert_not_called() def test_running_turn_cannot_receive_controls_for_a_pending_different_model(): diff --git a/website/docs/guides/google-gemini.md b/website/docs/guides/google-gemini.md index 80e21bcc4e635..b461111b50ceb 100644 --- a/website/docs/guides/google-gemini.md +++ b/website/docs/guides/google-gemini.md @@ -15,6 +15,8 @@ uses Hermes' cached [models.dev](https://models.dev) metadata to keep text-outpu models with function calling. A successful discovery replaces the curated list; on failure Hermes retains its existing fallback behavior. Models that require the dedicated Computer Use tool are excluded from this general chat picker. +Malformed individual entries are skipped; a failed page still invalidates the +discovery result, so a truncated catalog never replaces the fallback. Being listed does not guarantee that your project has access to a model. Desktop shows the thinking controls described for each model: @@ -24,6 +26,12 @@ Desktop shows the thinking controls described for each model: - **Thinking budget:** a token count within the model's bounds, with **Dynamic** for Google's `thinkingBudget: -1` mode. +The numeric input accepts positive thinking budgets. When the source range +starts at zero, the input starts at one: zero means **Thinking off**, available +only when the model declares that capability. Use that control (or `none` in +configuration) instead of `budget:0`. Dynamic remains a separate choice; +positive model-specific minimum budgets are preserved. + When no explicit level or budget is set, the picker leaves the options unselected rather than displaying a synthetic "Provider default" effort. Omitting the override differs from Dynamic, particularly for models whose From 3840f233efaa0812f8f47700d3a3b9f3d0eee0b0 Mon Sep 17 00:00:00 2001 From: jkorzeniak Date: Tue, 15 Sep 2026 19:18:07 +0200 Subject: [PATCH 3/4] fix(reasoning): preserve provider opt-in and hide unset TUI effort A-061 audit follow-up: leave budgets unchanged for providers that do not opt into validation, hide auto in TUI status, and report catalog pagination exhaustion without sensitive data. --- agent/gemini_model_catalog.py | 4 ++++ providers/reasoning.py | 6 ++--- tests/agent/test_gemini_catalog_reasoning.py | 24 +++++++++++++++++++ tests/providers/test_reasoning_selection.py | 19 +++++++++++++++ .../__tests__/appChromeStatusRule.test.tsx | 13 ++++++++++ ui-tui/src/components/appChrome.tsx | 2 +- 6 files changed, 64 insertions(+), 4 deletions(-) create mode 100644 tests/providers/test_reasoning_selection.py diff --git a/agent/gemini_model_catalog.py b/agent/gemini_model_catalog.py index f433f1bd5e1e5..338569e6acb9a 100644 --- a/agent/gemini_model_catalog.py +++ b/agent/gemini_model_catalog.py @@ -6,6 +6,7 @@ malformed individual entries are skipped. """ import json +import logging import time from urllib.parse import urlencode from urllib.request import Request @@ -60,4 +61,7 @@ def fetch_models(api_key: str | None, *, timeout: float = 8.0) -> list[str] | No tokens.add(token) except Exception: return None # no credential-bearing exceptions in logs + logging.getLogger(__name__).warning( + "Gemini model discovery reached its pagination limit; using the fallback catalog." + ) return None diff --git a/providers/reasoning.py b/providers/reasoning.py index bebec53e4a888..e152eb8b4e19b 100644 --- a/providers/reasoning.py +++ b/providers/reasoning.py @@ -16,6 +16,9 @@ def reasoning_selection_efforts(provider: str, model: str) -> tuple[str, ...] | def resolve_provider_reasoning_config( provider: str, model: str, config: dict | None, *, explicit: bool = False ) -> dict | None: + supported = reasoning_selection_efforts(provider, model) + if supported is None: + return config effort = "none" if config and config.get("enabled") is False else str((config or {}).get("effort") or "").strip().lower() if effort.startswith("budget:"): profile = get_provider_profile(provider) @@ -30,9 +33,6 @@ def resolve_provider_reasoning_config( if explicit: raise ValueError(f"Unsupported thinking token budget for {provider}/{model}.") return {"enabled": True} - supported = reasoning_selection_efforts(provider, model) - if supported is None: - return config if config is None: return {"enabled": True} # provider default, distinct from inheritance effort = "none" if config.get("enabled") is False else str(config.get("effort") or "").strip().lower() diff --git a/tests/agent/test_gemini_catalog_reasoning.py b/tests/agent/test_gemini_catalog_reasoning.py index 50f4bdcebe664..ea58da2a0454e 100644 --- a/tests/agent/test_gemini_catalog_reasoning.py +++ b/tests/agent/test_gemini_catalog_reasoning.py @@ -102,6 +102,30 @@ def test_successful_native_list_does_not_reintroduce_curated_ids(monkeypatch): assert models._profile_live_catalog("gemini") == [] +def test_catalog_page_limit_reports_fallback_without_sensitive_data(catalog, monkeypatch, caplog): + import agent.gemini_model_catalog as native + + monkeypatch.setattr(native, "fetch_models_dev", models_dev.fetch_models_dev) + page_count = 0 + + def next_page(*args, **kwargs): + nonlocal page_count + page_count += 1 + return io.StringIO(json.dumps({ + "models": [{"name": "models/gemini-future-levels", + "supportedGenerationMethods": ["generateContent"]}], + "nextPageToken": f"sensitive-page-token-{page_count}", + })) + + monkeypatch.setattr(native, "open_credentialed_url", next_page) + assert native.fetch_models("fixture-secret-key") is None + assert page_count > 1 + assert "pagination limit" in caplog.text + assert "fallback" in caplog.text + assert "fixture-secret-key" not in caplog.text + assert "sensitive-page-token" not in caplog.text + + @pytest.mark.parametrize("bad_entry", [ None, {}, {"name": None}, {"name": 42}, {"name": ""}, {"name": "models/gemini-fixed", "supportedGenerationMethods": 42}, diff --git a/tests/providers/test_reasoning_selection.py b/tests/providers/test_reasoning_selection.py new file mode 100644 index 0000000000000..6ea6d783f5483 --- /dev/null +++ b/tests/providers/test_reasoning_selection.py @@ -0,0 +1,19 @@ +"""Providers must opt into reasoning validation before preferences are changed.""" + +import pytest + +from providers import get_provider_profile +from providers.reasoning import resolve_provider_reasoning_config + + +@pytest.mark.parametrize("provider", ["anthropic", "openrouter", "unregistered-fixture"]) +@pytest.mark.parametrize("explicit", [False, True]) +def test_non_validating_providers_keep_their_reasoning_config(provider, explicit): + profile = get_provider_profile(provider) + assert profile is None or not profile.validate_reasoning_selection + for effort in ("budget:5000", "budget:-1", "high", "none", "auto"): + config = {"enabled": effort != "none", "effort": effort} + assert resolve_provider_reasoning_config( + provider, "fixture-model", config, explicit=explicit + ) is config + assert resolve_provider_reasoning_config(provider, "fixture-model", None, explicit=explicit) is None diff --git a/ui-tui/src/__tests__/appChromeStatusRule.test.tsx b/ui-tui/src/__tests__/appChromeStatusRule.test.tsx index 8b8ce4285250e..7075e0dd02ca2 100644 --- a/ui-tui/src/__tests__/appChromeStatusRule.test.tsx +++ b/ui-tui/src/__tests__/appChromeStatusRule.test.tsx @@ -105,6 +105,19 @@ const baseProps = { } describe('StatusRule session title', () => { + it('keeps an unset effort label empty while retaining explicit effort and fast', () => { + for (const model of ['gemini-fixture', 'anthropic/claude-fixture']) { + const props = { ...baseProps, model, modelFast: true } + const unset = textContent(StatusRule(props)) + + expect(textContent(StatusRule({ ...props, modelReasoningEffort: ' Auto ' }))).toBe(unset) + + for (const effort of ['high', 'none']) { + expect(textContent(StatusRule({ ...props, modelReasoningEffort: effort }))).toContain(`${effort} fast`) + } + } + }) + it('marks only estimated context occupancy at every visible width', () => { for (const cols of [80, 120, 200]) { for (const estimated of [true, false]) { diff --git a/ui-tui/src/components/appChrome.tsx b/ui-tui/src/components/appChrome.tsx index d5ece234736d0..8e673c925dab5 100644 --- a/ui-tui/src/components/appChrome.tsx +++ b/ui-tui/src/components/appChrome.tsx @@ -443,7 +443,7 @@ const effortLabel = (effort?: string) => { .trim() .toLowerCase() - return value && value !== 'medium' && value !== 'normal' && value !== 'default' ? value : '' + return value && value !== 'medium' && value !== 'normal' && value !== 'default' && value !== 'auto' ? value : '' } const shortModelLabel = (model: string) => From a09101573a5ba39e1f96436a99baad08bf57137b Mon Sep 17 00:00:00 2001 From: jkorzeniak Date: Sun, 20 Sep 2026 08:57:35 +0200 Subject: [PATCH 4/4] Preserve authoritative empty catalogs through setup fallback --- hermes_cli/model_setup_flows.py | 2 +- tests/hermes_cli/test_provider_live_curated_merge.py | 1 + 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/hermes_cli/model_setup_flows.py b/hermes_cli/model_setup_flows.py index 8f39d8d6f1a44..769aa5a92f759 100644 --- a/hermes_cli/model_setup_flows.py +++ b/hermes_cli/model_setup_flows.py @@ -896,7 +896,7 @@ def _api_key_provider_model_list(provider_id: str, pconfig, existing_key: str, k # ``/model`` picker; when neither live nor fallback_models yields rows, the curated # ``_PROVIDER_MODELS`` row still applies (built-in providers with a short curated list). model_list = probe_profile_catalog(provider_id, profile, api_key_for_probe, effective_base) - if model_list: + if model_list or (model_list is not None and profile.model_catalog_authoritative): _report_live_models(model_list, f"{pconfig.name} catalog") return model_list _show_curated(curated) diff --git a/tests/hermes_cli/test_provider_live_curated_merge.py b/tests/hermes_cli/test_provider_live_curated_merge.py index 27a5621a25b4c..a962013758628 100644 --- a/tests/hermes_cli/test_provider_live_curated_merge.py +++ b/tests/hermes_cli/test_provider_live_curated_merge.py @@ -121,6 +121,7 @@ def test_authoritative_catalog_is_shared_with_setup(monkeypatch, live): monkeypatch.setattr("providers.get_provider_profile", lambda name: profile) monkeypatch.setattr(models, "_api_key_credentials", lambda name: ("test-key", profile.base_url)) monkeypatch.setattr(model_setup_flows, "_models_dev_merged", lambda *a: []) + monkeypatch.setitem(models._PROVIDER_MODELS, profile.name, ["stale-curated"]) expected = ["fallback"] if live is None else live assert models._profile_live_catalog(profile.name) == expected assert model_setup_flows._api_key_provider_model_list(