From ac7b3a07ccc0513f1fc0c6b4284b6ecbc3401d29 Mon Sep 17 00:00:00 2001 From: Dominic Wehrmann Date: Fri, 20 Mar 2026 20:03:39 +0000 Subject: [PATCH] feat(web): expose cheap_model and smart_routing_cascade in settings UI The LLM_CHEAP_MODEL and SMART_ROUTING_CASCADE options from #1081 were only configurable via env vars. This adds them to the Settings struct and web UI so users can configure smart routing from the browser. Resolution order: env var > settings > default (None / true). Co-Authored-By: Claude Opus 4.6 (1M context) --- src/channels/web/static/app.js | 4 +++- src/channels/web/static/i18n/en.js | 4 ++++ src/channels/web/static/i18n/zh-CN.js | 4 ++++ src/config/llm.rs | 21 +++++++++++++++------ src/settings.rs | 9 +++++++++ 5 files changed, 35 insertions(+), 7 deletions(-) diff --git a/src/channels/web/static/app.js b/src/channels/web/static/app.js index 0b247a63164..e3b7e8061c4 100644 --- a/src/channels/web/static/app.js +++ b/src/channels/web/static/app.js @@ -4705,6 +4705,8 @@ var INFERENCE_SETTINGS = [ { key: 'llm_backend', label: 'cfg.llm_backend.label', description: 'cfg.llm_backend.desc', type: 'select', options: ['nearai', 'anthropic', 'openai', 'ollama', 'openai_compatible', 'tinfoil', 'bedrock'] }, { key: 'selected_model', label: 'cfg.selected_model.label', description: 'cfg.selected_model.desc', type: 'text' }, + { key: 'cheap_model', label: 'cfg.cheap_model.label', description: 'cfg.cheap_model.desc', type: 'text' }, + { key: 'smart_routing_cascade', label: 'cfg.smart_routing_cascade.label', description: 'cfg.smart_routing_cascade.desc', type: 'boolean' }, { key: 'ollama_base_url', label: 'cfg.ollama_base_url.label', description: 'cfg.ollama_base_url.desc', type: 'text', showWhen: { key: 'llm_backend', value: 'ollama' } }, { key: 'openai_compatible_base_url', label: 'cfg.openai_compatible_base_url.label', description: 'cfg.openai_compatible_base_url.desc', type: 'text', @@ -5097,7 +5099,7 @@ function renderStructuredSettingsRow(def, value, activeValue) { return row; } -var RESTART_REQUIRED_KEYS = ['llm_backend', 'selected_model', 'ollama_base_url', 'openai_compatible_base_url', +var RESTART_REQUIRED_KEYS = ['llm_backend', 'selected_model', 'cheap_model', 'smart_routing_cascade', 'ollama_base_url', 'openai_compatible_base_url', 'bedrock_region', 'bedrock_cross_region', 'bedrock_profile', 'embeddings.enabled', 'embeddings.provider', 'embeddings.model', 'agent.auto_approve_tools', 'tunnel.provider', 'tunnel.public_url', 'gateway.rate_limit', 'gateway.max_connections']; diff --git a/src/channels/web/static/i18n/en.js b/src/channels/web/static/i18n/en.js index 6029075d2ff..61e47eecb68 100644 --- a/src/channels/web/static/i18n/en.js +++ b/src/channels/web/static/i18n/en.js @@ -403,6 +403,10 @@ I18n.register('en', { 'cfg.llm_backend.desc': 'LLM inference provider', 'cfg.selected_model.label': 'Model', 'cfg.selected_model.desc': 'Model name or ID for the selected backend', + 'cfg.cheap_model.label': 'Cheap Model', + 'cfg.cheap_model.desc': 'Cheap/fast model for smart routing (lightweight tasks)', + 'cfg.smart_routing_cascade.label': 'Smart Routing Cascade', + 'cfg.smart_routing_cascade.desc': 'Retry with primary model if cheap model response seems uncertain', 'cfg.ollama_base_url.label': 'Ollama URL', 'cfg.ollama_base_url.desc': 'Base URL for Ollama API', 'cfg.openai_compatible_base_url.label': 'OpenAI-compatible URL', diff --git a/src/channels/web/static/i18n/zh-CN.js b/src/channels/web/static/i18n/zh-CN.js index 480724c9b0a..69773dbfb16 100644 --- a/src/channels/web/static/i18n/zh-CN.js +++ b/src/channels/web/static/i18n/zh-CN.js @@ -402,6 +402,10 @@ I18n.register('zh-CN', { 'cfg.llm_backend.desc': 'LLM 推理提供商', 'cfg.selected_model.label': '模型', 'cfg.selected_model.desc': '所选后端的模型名称或 ID', + 'cfg.cheap_model.label': '廉价模型', + 'cfg.cheap_model.desc': '用于智能路由的廉价/快速模型(轻量级任务)', + 'cfg.smart_routing_cascade.label': '智能路由级联', + 'cfg.smart_routing_cascade.desc': '当廉价模型回答不确定时,使用主模型重试', 'cfg.ollama_base_url.label': 'Ollama URL', 'cfg.ollama_base_url.desc': 'Ollama API 基础 URL', 'cfg.openai_compatible_base_url.label': 'OpenAI 兼容 URL', diff --git a/src/config/llm.rs b/src/config/llm.rs index 03ce1f8590c..0225551f4d9 100644 --- a/src/config/llm.rs +++ b/src/config/llm.rs @@ -213,13 +213,22 @@ impl LlmConfig { let request_timeout_secs = parse_optional_env("LLM_REQUEST_TIMEOUT_SECS", 120)?; - // Generic cheap model (works with any backend). + // Generic cheap model: env var > settings > None. // Falls back to NearAI-specific cheap_model in provider chain logic. - let cheap_model = optional_env("LLM_CHEAP_MODEL")?; - - // Generic smart routing cascade flag. - // Defaults to true. Overrides NearAI-specific smart_routing_cascade. - let smart_routing_cascade = parse_optional_env("SMART_ROUTING_CASCADE", true)?; + let cheap_model = optional_env("LLM_CHEAP_MODEL")? + .or_else(|| settings.cheap_model.clone()); + + // Generic smart routing cascade flag: env var > settings > true. + // Overrides NearAI-specific smart_routing_cascade. + let smart_routing_cascade = optional_env("SMART_ROUTING_CASCADE")? + .map(|v| v.parse::()) + .transpose() + .map_err(|_| ConfigError::InvalidValue { + key: "SMART_ROUTING_CASCADE".into(), + message: "expected true or false".into(), + })? + .or(settings.smart_routing_cascade) + .unwrap_or(true); Ok(Self { backend: if is_nearai { diff --git a/src/settings.rs b/src/settings.rs index 15437f446b5..7ccb3e6ed4e 100644 --- a/src/settings.rs +++ b/src/settings.rs @@ -84,6 +84,15 @@ pub struct Settings { #[serde(default)] pub selected_model: Option, + /// Cheap/fast model for smart routing (lightweight tasks like heartbeat, routing). + #[serde(default)] + pub cheap_model: Option, + + /// Enable cascade mode for smart routing (retry with primary if cheap model + /// response seems uncertain). When None, defaults to true. + #[serde(default)] + pub smart_routing_cascade: Option, + // === Step 5: Embeddings === /// Embeddings configuration. #[serde(default)]