Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion src/channels/web/static/app.js
Original file line number Diff line number Diff line change
Expand Up @@ -4705,6 +4705,8 @@ var INFERENCE_SETTINGS = [
{ key: 'llm_backend', label: 'cfg.llm_backend.label', description: 'cfg.llm_backend.desc',
type: 'select', options: ['nearai', 'anthropic', 'openai', 'ollama', 'openai_compatible', 'tinfoil', 'bedrock'] },
{ key: 'selected_model', label: 'cfg.selected_model.label', description: 'cfg.selected_model.desc', type: 'text' },
{ key: 'cheap_model', label: 'cfg.cheap_model.label', description: 'cfg.cheap_model.desc', type: 'text' },
{ key: 'smart_routing_cascade', label: 'cfg.smart_routing_cascade.label', description: 'cfg.smart_routing_cascade.desc', type: 'boolean' },
{ key: 'ollama_base_url', label: 'cfg.ollama_base_url.label', description: 'cfg.ollama_base_url.desc', type: 'text',
showWhen: { key: 'llm_backend', value: 'ollama' } },
{ key: 'openai_compatible_base_url', label: 'cfg.openai_compatible_base_url.label', description: 'cfg.openai_compatible_base_url.desc', type: 'text',
Expand Down Expand Up @@ -5097,7 +5099,7 @@ function renderStructuredSettingsRow(def, value, activeValue) {
return row;
}

var RESTART_REQUIRED_KEYS = ['llm_backend', 'selected_model', 'ollama_base_url', 'openai_compatible_base_url',
var RESTART_REQUIRED_KEYS = ['llm_backend', 'selected_model', 'cheap_model', 'smart_routing_cascade', 'ollama_base_url', 'openai_compatible_base_url',
'bedrock_region', 'bedrock_cross_region', 'bedrock_profile', 'embeddings.enabled', 'embeddings.provider', 'embeddings.model',
'agent.auto_approve_tools', 'tunnel.provider', 'tunnel.public_url', 'gateway.rate_limit', 'gateway.max_connections'];

Expand Down
4 changes: 4 additions & 0 deletions src/channels/web/static/i18n/en.js
Original file line number Diff line number Diff line change
Expand Up @@ -403,6 +403,10 @@ I18n.register('en', {
'cfg.llm_backend.desc': 'LLM inference provider',
'cfg.selected_model.label': 'Model',
'cfg.selected_model.desc': 'Model name or ID for the selected backend',
'cfg.cheap_model.label': 'Cheap Model',
'cfg.cheap_model.desc': 'Cheap/fast model for smart routing (lightweight tasks)',
'cfg.smart_routing_cascade.label': 'Smart Routing Cascade',
'cfg.smart_routing_cascade.desc': 'Retry with primary model if cheap model response seems uncertain',
'cfg.ollama_base_url.label': 'Ollama URL',
'cfg.ollama_base_url.desc': 'Base URL for Ollama API',
'cfg.openai_compatible_base_url.label': 'OpenAI-compatible URL',
Expand Down
4 changes: 4 additions & 0 deletions src/channels/web/static/i18n/zh-CN.js
Original file line number Diff line number Diff line change
Expand Up @@ -402,6 +402,10 @@ I18n.register('zh-CN', {
'cfg.llm_backend.desc': 'LLM 推理提供商',
'cfg.selected_model.label': '模型',
'cfg.selected_model.desc': '所选后端的模型名称或 ID',
'cfg.cheap_model.label': '廉价模型',
'cfg.cheap_model.desc': '用于智能路由的廉价/快速模型(轻量级任务)',
'cfg.smart_routing_cascade.label': '智能路由级联',
'cfg.smart_routing_cascade.desc': '当廉价模型回答不确定时,使用主模型重试',
'cfg.ollama_base_url.label': 'Ollama URL',
'cfg.ollama_base_url.desc': 'Ollama API 基础 URL',
'cfg.openai_compatible_base_url.label': 'OpenAI 兼容 URL',
Expand Down
21 changes: 15 additions & 6 deletions src/config/llm.rs
Original file line number Diff line number Diff line change
Expand Up @@ -213,13 +213,22 @@ impl LlmConfig {

let request_timeout_secs = parse_optional_env("LLM_REQUEST_TIMEOUT_SECS", 120)?;

// Generic cheap model (works with any backend).
// Generic cheap model: env var > settings > None.
// Falls back to NearAI-specific cheap_model in provider chain logic.
let cheap_model = optional_env("LLM_CHEAP_MODEL")?;

// Generic smart routing cascade flag.
// Defaults to true. Overrides NearAI-specific smart_routing_cascade.
let smart_routing_cascade = parse_optional_env("SMART_ROUTING_CASCADE", true)?;
let cheap_model = optional_env("LLM_CHEAP_MODEL")?
.or_else(|| settings.cheap_model.clone());

// Generic smart routing cascade flag: env var > settings > true.
// Overrides NearAI-specific smart_routing_cascade.
let smart_routing_cascade = optional_env("SMART_ROUTING_CASCADE")?
.map(|v| v.parse::<bool>())
.transpose()
.map_err(|_| ConfigError::InvalidValue {
key: "SMART_ROUTING_CASCADE".into(),
message: "expected true or false".into(),
})?
.or(settings.smart_routing_cascade)
.unwrap_or(true);

Ok(Self {
backend: if is_nearai {
Expand Down
9 changes: 9 additions & 0 deletions src/settings.rs
Original file line number Diff line number Diff line change
Expand Up @@ -84,6 +84,15 @@ pub struct Settings {
#[serde(default)]
pub selected_model: Option<String>,

/// Cheap/fast model for smart routing (lightweight tasks like heartbeat, routing).
#[serde(default)]
pub cheap_model: Option<String>,

/// Enable cascade mode for smart routing (retry with primary if cheap model
/// response seems uncertain). When None, defaults to true.
#[serde(default)]
pub smart_routing_cascade: Option<bool>,

// === Step 5: Embeddings ===
/// Embeddings configuration.
#[serde(default)]
Expand Down
Loading