Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .env.example
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@ DATABASE_POOL_SIZE=10
# LLM Provider
# LLM_BACKEND=nearai # default
# Possible values: nearai, ollama, openai_compatible, openai, anthropic, tinfoil
# LLM_REQUEST_TIMEOUT_SECS=120 # Increase for local LLMs (Ollama, vLLM, LM Studio)

# === Anthropic Direct ===
# Two auth modes:
Expand Down
34 changes: 34 additions & 0 deletions src/config/llm.rs
Original file line number Diff line number Diff line change
Expand Up @@ -103,6 +103,10 @@ pub struct LlmConfig {
/// Resolved provider config for registry-based providers.
/// `None` when backend is "nearai".
pub provider: Option<RegistryProviderConfig>,
/// HTTP request timeout in seconds for LLM API calls.
/// Default: 120. Increase for local LLMs (Ollama, vLLM, LM Studio) that
/// need more time for prompt evaluation on consumer hardware.
pub request_timeout_secs: u64,
}

/// NEAR AI configuration.
Expand Down Expand Up @@ -165,6 +169,7 @@ impl LlmConfig {
smart_routing_cascade: false,
},
provider: None,
request_timeout_secs: 120,
}
}

Expand Down Expand Up @@ -254,6 +259,8 @@ impl LlmConfig {
)?)
};

let request_timeout_secs = parse_optional_env("LLM_REQUEST_TIMEOUT_SECS", 120)?;

Ok(Self {
backend: if is_nearai {
"nearai".to_string()
Expand All @@ -265,6 +272,7 @@ impl LlmConfig {
session,
nearai,
provider,
request_timeout_secs,
})
}

Expand Down Expand Up @@ -1016,4 +1024,30 @@ mod tests {
assert_eq!(parsed, variant, "round-trip failed for {s}");
}
}

#[test]
fn test_request_timeout_defaults_to_120() {
let _guard = ENV_MUTEX.lock().expect("env mutex poisoned");
// SAFETY: Under ENV_MUTEX.
unsafe {
std::env::remove_var("LLM_REQUEST_TIMEOUT_SECS");
}
let config = LlmConfig::resolve(&Settings::default()).expect("resolve");
assert_eq!(config.request_timeout_secs, 120);
}

#[test]
fn test_request_timeout_configurable() {
let _guard = ENV_MUTEX.lock().expect("env mutex poisoned");
// SAFETY: Under ENV_MUTEX.
unsafe {
std::env::set_var("LLM_REQUEST_TIMEOUT_SECS", "300");
}
let config = LlmConfig::resolve(&Settings::default()).expect("resolve");
assert_eq!(config.request_timeout_secs, 300);
// SAFETY: Cleanup
unsafe {
std::env::remove_var("LLM_REQUEST_TIMEOUT_SECS");
}
}
}
25 changes: 21 additions & 4 deletions src/llm/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -58,8 +58,10 @@ pub fn create_llm_provider(
config: &LlmConfig,
session: Arc<SessionManager>,
) -> Result<Arc<dyn LlmProvider>, LlmError> {
let timeout = config.request_timeout_secs;

if config.backend == "nearai" || config.backend == "near_ai" || config.backend == "near" {
return create_llm_provider_with_config(&config.nearai, session);
return create_llm_provider_with_config(&config.nearai, session, timeout);
}

let reg_config = config
Expand All @@ -79,6 +81,7 @@ pub fn create_llm_provider(
pub fn create_llm_provider_with_config(
config: &NearAiConfig,
session: Arc<SessionManager>,
request_timeout_secs: u64,
) -> Result<Arc<dyn LlmProvider>, LlmError> {
let auth_mode = if config.api_key.is_some() {
"API key"
Expand All @@ -89,9 +92,14 @@ pub fn create_llm_provider_with_config(
model = %config.model,
base_url = %config.base_url,
auth = auth_mode,
timeout_secs = request_timeout_secs,
"Using NEAR AI (Chat Completions API)"
);
Ok(Arc::new(NearAiChatProvider::new(config.clone(), session)?))
Ok(Arc::new(NearAiChatProvider::new_with_timeout(
config.clone(),
session,
request_timeout_secs,
)?))
}

/// Create a provider from a registry-resolved config.
Expand Down Expand Up @@ -365,7 +373,11 @@ pub fn build_provider_chain(
let llm: Arc<dyn LlmProvider> = if let Some(ref cheap_model) = config.nearai.cheap_model {
let mut cheap_config = config.nearai.clone();
cheap_config.model = cheap_model.clone();
let cheap = create_llm_provider_with_config(&cheap_config, session.clone())?;
let cheap = create_llm_provider_with_config(
&cheap_config,
session.clone(),
config.request_timeout_secs,
)?;
let cheap: Arc<dyn LlmProvider> = if retry_config.max_retries > 0 {
Arc::new(RetryProvider::new(cheap, retry_config.clone()))
} else {
Expand Down Expand Up @@ -397,7 +409,11 @@ pub fn build_provider_chain(
}
let mut fallback_config = config.nearai.clone();
fallback_config.model = fallback_model.clone();
let fallback = create_llm_provider_with_config(&fallback_config, session.clone())?;
let fallback = create_llm_provider_with_config(
&fallback_config,
session.clone(),
config.request_timeout_secs,
)?;
tracing::info!(
primary = %llm.model_name(),
fallback = %fallback.model_name(),
Expand Down Expand Up @@ -503,6 +519,7 @@ mod tests {
session: SessionConfig::default(),
nearai: test_nearai_config(),
provider: None,
request_timeout_secs: 120,
}
}

Expand Down
19 changes: 15 additions & 4 deletions src/llm/nearai_chat.rs
Original file line number Diff line number Diff line change
Expand Up @@ -58,17 +58,28 @@ impl NearAiChatProvider {
/// By default this enables tool-message flattening for compatibility with
/// providers that reject `role: "tool"` messages.
pub fn new(config: NearAiConfig, session: Arc<SessionManager>) -> Result<Self, LlmError> {
Self::new_with_flatten(config, session, true)
Comment on lines 60 to -61

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

medium

Consider using the new_with_options function directly to avoid the extra function call.

        Self::new_with_options(config, session, true, 120)

Self::new_with_options(config, session, true, 120)
}

/// Create a chat completions provider with configurable tool-message flattening.
pub fn new_with_flatten(
/// Create a new provider with a custom request timeout.
pub fn new_with_timeout(
config: NearAiConfig,
session: Arc<SessionManager>,
request_timeout_secs: u64,
) -> Result<Self, LlmError> {
Self::new_with_options(config, session, true, request_timeout_secs)
}

/// Create a chat completions provider with configurable tool-message flattening
/// and request timeout.
pub fn new_with_options(
config: NearAiConfig,
session: Arc<SessionManager>,
flatten_tool_messages: bool,
request_timeout_secs: u64,
) -> Result<Self, LlmError> {
let client = Client::builder()
.timeout(std::time::Duration::from_secs(120))
.timeout(std::time::Duration::from_secs(request_timeout_secs))
.build()
.map_err(|e| LlmError::RequestFailed {
provider: "nearai_chat".to_string(),
Expand Down
1 change: 1 addition & 0 deletions src/setup/wizard.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1476,6 +1476,7 @@ impl SetupWizard {
smart_routing_cascade: true,
},
provider: None,
request_timeout_secs: 120,
};

match create_llm_provider(&config, session) {
Expand Down