diff --git a/.config/nextest.toml b/.config/nextest.toml index dcdce03eddb..a5698ff8edf 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -64,3 +64,15 @@ slow-timeout = { period = "300s", terminate-after = 2 } [[profile.ci.overrides]] filter = "binary(~reborn_group_)" slow-timeout = { period = "28m", terminate-after = 1 } + +# ironclaw_architecture_tests: reborn_{extension_contract,loop_port,product_contract}_location_scan +# and reborn_sealed_evidence_mint_ratchet each contain one whole-crates/-tree +# scan test. On the last green run the location-scan tests finished at +# 176.8s against the default 60s/3 = 180s hard kill; the very next run was +# terminated at 180.008s. reborn_sealed_evidence_mint_ratchet rides the same +# ladder to 136-141s. Locally these take ~110s. 60s/6 = 360s keeps the same +# SLOW cadence but gives real headroom instead of raising the default for +# every binary in the crate. +[[profile.ci.overrides]] +filter = 'binary(/_location_scan$/) | binary(~reborn_sealed_evidence_mint_ratchet)' +slow-timeout = { period = "60s", terminate-after = 6 } diff --git a/Cargo.toml b/Cargo.toml index 6471fa15111..fed0b265fe1 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -332,6 +332,10 @@ path = "tests/integration/backend_matrix.rs" name = "reborn_integration_budget" path = "tests/integration/budget.rs" +[[test]] +name = "reborn_integration_context_budget" +path = "tests/integration/context_budget.rs" + [[test]] name = "reborn_integration_cancel" path = "tests/integration/cancel.rs" diff --git a/crates/app/ironclaw_architecture_tests/tests/reborn_dependency_boundaries.rs b/crates/app/ironclaw_architecture_tests/tests/reborn_dependency_boundaries.rs index f40058fd1e0..4541796bf1b 100644 --- a/crates/app/ironclaw_architecture_tests/tests/reborn_dependency_boundaries.rs +++ b/crates/app/ironclaw_architecture_tests/tests/reborn_dependency_boundaries.rs @@ -1039,7 +1039,15 @@ fn reborn_contracts_crates_carry_a_checked_size_ceiling() { // production prompt validation checks structural limits and control // characters only; decoded Basic-auth samples remain test-only. Count // read from this test's own failure message after merging #7416 and #6985. - ("ironclaw_loop_contracts", 13_608), + // 13_608 -> 13_773 (2026-09-02, model-derived prompt context budget): + // `PromptContextTokenBudget::from_advertised_window` derives the + // per-run budget on the type this crate owns, and `LoopRunContext` + // carries the optional resolved budget; derivation stays a pure + // function of the DTO, resolution and consumption stay in + // `ironclaw_loop_host` / `ironclaw_turn_runner` / `ironclaw_agent_loop`. + // `main` already sat 3 lines under the effective ceiling. Count read + // from this test's own failure message. + ("ironclaw_loop_contracts", 13_773), // Raised 15_685 -> 15_758 by #7220 (operator inspector API): the growth // is bounded, output-only read-view descriptors. Capture, retention, // authorization, and transport behavior remain in their owning diff --git a/crates/contracts/ironclaw_loop_contracts/src/context_budget.rs b/crates/contracts/ironclaw_loop_contracts/src/context_budget.rs index 51e26d4c8a9..b8b84257cf7 100644 --- a/crates/contracts/ironclaw_loop_contracts/src/context_budget.rs +++ b/crates/contracts/ironclaw_loop_contracts/src/context_budget.rs @@ -4,7 +4,7 @@ /// Storage still scans transcript context by message count. Host adapters use /// this budget after that scan, and compaction strategies use the same budget /// shape to decide when the observed prompt is near its context ceiling. -#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize)] +#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)] pub struct PromptContextTokenBudget { pub context_limit_tokens: u64, pub reserve_tokens: u64, @@ -16,6 +16,13 @@ impl PromptContextTokenBudget { pub const DEFAULT_RESERVE_TOKENS: u64 = 20_000; pub const DEFAULT_MAIN_LOOP_MAX_OUTPUT_TOKENS: u64 = 0; + /// Fraction of a provider-advertised window we are willing to fill. + /// + /// This margin exists to absorb error in the chars/4 token estimate + /// (`estimate_tokens_from_chars`), which is the only reason for it. Room + /// for the model's *response* is a separate axis — `reserve_tokens`. + pub const DEFAULT_USABLE_FRACTION_PERCENT: u64 = 90; + pub const fn new( context_limit_tokens: u64, reserve_tokens: u64, @@ -32,6 +39,34 @@ impl PromptContextTokenBudget { self.context_limit_tokens .saturating_sub(self.reserve_tokens.max(self.main_loop_max_output_tokens)) } + + /// Derive a budget from a provider-advertised total context window. + /// + /// `None` (or a nonsense zero) reproduces the compiled-in default + /// exactly, so a provider that advertises nothing behaves as it always + /// has. Never guess a window for an unknown model: guessing high + /// produces the provider rejection this mechanism exists to avoid. A + /// window too small to leave any visible transcript after the reserve + /// (today: below 2 tokens) is treated as unknown, like `None` and `0`. + pub fn from_advertised_window(advertised_tokens: Option) -> Self { + let Some(advertised) = advertised_tokens.filter(|tokens| *tokens > 0) else { + return Self::default(); + }; + let context_limit_tokens = + advertised.saturating_mul(Self::DEFAULT_USABLE_FRACTION_PERCENT) / 100; + // A small-window model would otherwise have its whole budget consumed + // by the flat response reserve, leaving zero visible transcript. + let reserve_tokens = Self::DEFAULT_RESERVE_TOKENS.min(context_limit_tokens / 4); + let candidate = Self { + context_limit_tokens, + reserve_tokens, + main_loop_max_output_tokens: Self::DEFAULT_MAIN_LOOP_MAX_OUTPUT_TOKENS, + }; + if candidate.visible_transcript_tokens() == 0 { + return Self::default(); + } + candidate + } } impl Default for PromptContextTokenBudget { @@ -45,27 +80,4 @@ impl Default for PromptContextTokenBudget { } #[cfg(test)] -mod tests { - use super::PromptContextTokenBudget; - - #[test] - fn visible_transcript_tokens_reserves_larger_output_buffer() { - let budget = PromptContextTokenBudget::new(100, 10, 30); - - assert_eq!(budget.visible_transcript_tokens(), 70); - } - - #[test] - fn visible_transcript_tokens_saturates_when_reserve_exceeds_limit() { - let budget = PromptContextTokenBudget::new(10, 20, 0); - - assert_eq!(budget.visible_transcript_tokens(), 0); - } - - #[test] - fn visible_transcript_tokens_uses_reserve_when_larger_than_output_budget() { - let budget = PromptContextTokenBudget::new(100, 30, 10); - - assert_eq!(budget.visible_transcript_tokens(), 70); - } -} +mod tests; diff --git a/crates/contracts/ironclaw_loop_contracts/src/context_budget/tests.rs b/crates/contracts/ironclaw_loop_contracts/src/context_budget/tests.rs new file mode 100644 index 00000000000..4cb93019927 --- /dev/null +++ b/crates/contracts/ironclaw_loop_contracts/src/context_budget/tests.rs @@ -0,0 +1,112 @@ +use super::*; + +#[test] +fn visible_transcript_tokens_reserves_larger_output_buffer() { + let budget = PromptContextTokenBudget::new(100, 10, 30); + + assert_eq!(budget.visible_transcript_tokens(), 70); +} + +#[test] +fn visible_transcript_tokens_saturates_when_reserve_exceeds_limit() { + let budget = PromptContextTokenBudget::new(10, 20, 0); + + assert_eq!(budget.visible_transcript_tokens(), 0); +} + +#[test] +fn visible_transcript_tokens_uses_reserve_when_larger_than_output_budget() { + let budget = PromptContextTokenBudget::new(100, 30, 10); + + assert_eq!(budget.visible_transcript_tokens(), 70); +} + +#[test] +fn advertised_window_of_none_reproduces_the_compiled_in_default() { + // A provider that reports nothing must behave exactly as it does + // today. This is the compatibility guarantee of the whole change. + assert_eq!( + PromptContextTokenBudget::from_advertised_window(None), + PromptContextTokenBudget::default() + ); +} + +#[test] +fn advertised_window_of_zero_is_treated_as_unknown() { + assert_eq!( + PromptContextTokenBudget::from_advertised_window(Some(0)), + PromptContextTokenBudget::default() + ); +} + +#[test] +fn large_advertised_window_keeps_the_flat_response_reserve() { + let budget = PromptContextTokenBudget::from_advertised_window(Some(2_000_000)); + + assert_eq!(budget.context_limit_tokens, 1_800_000); + assert_eq!( + budget.reserve_tokens, + PromptContextTokenBudget::DEFAULT_RESERVE_TOKENS + ); + assert_eq!(budget.visible_transcript_tokens(), 1_780_000); +} + +#[test] +fn small_advertised_window_clamps_the_reserve_and_keeps_budget_usable() { + // An 8k model would otherwise have its entire budget consumed by the + // flat 20k response reserve, leaving zero visible transcript and a + // loop that cannot run at all. + let budget = PromptContextTokenBudget::from_advertised_window(Some(8_000)); + + assert_eq!(budget.context_limit_tokens, 7_200); + assert_eq!(budget.reserve_tokens, 1_800); + assert!( + budget.visible_transcript_tokens() > 0, + "a small-window model must still have room for transcript" + ); +} + +#[test] +fn smallest_positive_window_is_treated_as_unknown() { + // Some(1) survives the `> 0` filter but derives a zero visible + // transcript, which must fall back to the default exactly like + // None and Some(0). + assert_eq!( + PromptContextTokenBudget::from_advertised_window(Some(1)), + PromptContextTokenBudget::default() + ); +} + +#[test] +fn smallest_usable_window_keeps_a_nonzero_visible_transcript() { + // Find the smallest advertised window whose derivation does NOT + // fall back to the default, and prove it still leaves visible + // transcript room rather than trusting the arithmetic. + let smallest_non_default = (1..=16) + .find(|&candidate| { + PromptContextTokenBudget::from_advertised_window(Some(candidate)) + != PromptContextTokenBudget::default() + }) + .expect("some small window must derive a non-default budget"); + + let budget = PromptContextTokenBudget::from_advertised_window(Some(smallest_non_default)); + + assert_ne!(budget, PromptContextTokenBudget::default()); + assert!( + budget.visible_transcript_tokens() > 0, + "the smallest non-default derived budget must still leave visible transcript room" + ); +} + +#[test] +fn advertised_window_matching_todays_constant_is_reduced_by_the_margin() { + // 128k advertised is NOT the same as the 128k fallback: the fallback + // is a guess, an advertised value gets the estimate-error margin. + let budget = PromptContextTokenBudget::from_advertised_window(Some(128_000)); + + assert_eq!(budget.context_limit_tokens, 115_200); + assert_eq!( + budget.reserve_tokens, + PromptContextTokenBudget::DEFAULT_RESERVE_TOKENS + ); +} diff --git a/crates/contracts/ironclaw_loop_contracts/src/host/run_context.rs b/crates/contracts/ironclaw_loop_contracts/src/host/run_context.rs index 4e8fe4bb42f..9211577399b 100644 --- a/crates/contracts/ironclaw_loop_contracts/src/host/run_context.rs +++ b/crates/contracts/ironclaw_loop_contracts/src/host/run_context.rs @@ -10,6 +10,7 @@ use ironclaw_host_api::{ }; use serde::{Deserialize, Serialize}; +use crate::context_budget::PromptContextTokenBudget; use crate::refs::{CheckpointSchemaId, LoopDriverId}; use crate::snapshot::ResolvedRunProfile; use ironclaw_host_api::turn::{ @@ -245,6 +246,11 @@ pub struct LoopRunContext { pub resolved_run_profile: ResolvedRunProfile, #[serde(default, skip_serializing_if = "Option::is_none")] pub resolved_model_route: Option, + /// Prompt context budget resolved from this run's model at host + /// construction. `None` — an older serialized context, or a provider that + /// advertises no window — means the compiled-in default. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub resolved_context_budget: Option, pub loop_driver_id: LoopDriverId, pub loop_driver_version: RunProfileVersion, pub checkpoint_schema_id: CheckpointSchemaId, @@ -279,6 +285,7 @@ impl LoopRunContext { run_id, resolved_run_profile, resolved_model_route: None, + resolved_context_budget: None, loop_driver_id, loop_driver_version, checkpoint_schema_id, @@ -347,6 +354,11 @@ impl LoopRunContext { self } + pub fn with_resolved_context_budget(mut self, budget: PromptContextTokenBudget) -> Self { + self.resolved_context_budget = Some(budget); + self + } + pub fn with_product_context(mut self, product_context: ProductTurnContext) -> Self { self.product_context = Some(product_context); self diff --git a/crates/contracts/ironclaw_loop_contracts/src/host/run_context/tests.rs b/crates/contracts/ironclaw_loop_contracts/src/host/run_context/tests.rs index e99dd7d84fd..d9344488dc8 100644 --- a/crates/contracts/ironclaw_loop_contracts/src/host/run_context/tests.rs +++ b/crates/contracts/ironclaw_loop_contracts/src/host/run_context/tests.rs @@ -249,3 +249,59 @@ mod acting_identity_ladder { ); } } + +fn sample_run_context() -> LoopRunContext { + let scope = TurnScope::new( + ironclaw_host_api::ids::TenantId::new("tenant-context-budget").expect("tenant"), + None, + None, + ThreadId::new("thread-context-budget").expect("thread"), + ); + let profile = ResolvedRunProfile::legacy_compatibility( + ironclaw_host_api::turn::RunProfileId::default_profile(), + ironclaw_host_api::turn::RunProfileVersion::new(1), + true, + ); + LoopRunContext::new(scope, TurnId::new(), TurnRunId::new(), profile) +} + +#[test] +fn run_context_defaults_to_no_resolved_context_budget() { + let context = sample_run_context(); + + assert_eq!(context.resolved_context_budget, None); +} + +#[test] +fn run_context_carries_a_resolved_context_budget() { + let budget = PromptContextTokenBudget::from_advertised_window(Some(200_000)); + let context = sample_run_context().with_resolved_context_budget(budget); + + assert_eq!(context.resolved_context_budget, Some(budget)); +} + +#[test] +fn run_context_without_a_budget_field_still_deserializes() { + // Runs recorded before this change must replay, landing on the + // compiled-in default rather than failing to deserialize. + let context = sample_run_context(); + let mut wire = serde_json::to_value(&context).expect("serialize"); + wire.as_object_mut() + .expect("object") + .remove("resolved_context_budget"); + + let restored: LoopRunContext = serde_json::from_value(wire).expect("deserialize"); + + assert_eq!(restored.resolved_context_budget, None); +} + +#[test] +fn resolved_context_budget_round_trips_through_the_wire() { + let budget = PromptContextTokenBudget::from_advertised_window(Some(1_000_000)); + let context = sample_run_context().with_resolved_context_budget(budget); + + let wire = serde_json::to_string(&context).expect("serialize"); + let restored: LoopRunContext = serde_json::from_str(&wire).expect("deserialize"); + + assert_eq!(restored.resolved_context_budget, Some(budget)); +} diff --git a/crates/domains/ironclaw_llm/CONTRACT.md b/crates/domains/ironclaw_llm/CONTRACT.md index 177c6a973b8..186877e5a13 100644 --- a/crates/domains/ironclaw_llm/CONTRACT.md +++ b/crates/domains/ironclaw_llm/CONTRACT.md @@ -236,6 +236,8 @@ Key notes: - `cost_per_token()` returns `(Decimal, Decimal)` using `rust_decimal`. Look up via `costs::model_cost()` in your constructor; fall back to `costs::default_cost()` for unknowns. - `RigAdapter` forwards per-request model overrides through rig-core's typed request model field. Do not put `model` in flattened `additional_params`, which would serialize a duplicate top-level JSON key. - `complete_with_tools()` is never cached (tool calls can have side effects) — `CachedProvider` always passes them through. +- `model_metadata().context_length` is now consumed at runtime: `ironclaw_loop_host`'s `LlmProviderModelGateway::advertised_context_window_tokens` reads it — only when the returned `ModelMetadata::id` matches the model the request will actually be served by — to derive the per-run prompt context budget. A provider that populates `context_length` therefore changes prompt sizing and compaction thresholds for runs served by that model; a provider that leaves it `None` keeps the compiled-in default. A provider that may serve a call from more than one model (routing, fan-out, failover chains — `SmartRoutingProvider`, `FailoverProvider`) advertises the smallest window among them, and a table lookup returns `None` for a model it does not know — a guessed window is worse than none. +- `model_metadata()` must be a static description of the configured model: no network or credential I/O (no token refresh, no discovery call) and no lock shared with an in-flight call — it is awaited on the turn-run host-build critical path (`ironclaw_turn_runner::loop_driver_host`), so any I/O there stalls or serializes behind that path. Decorators (`TokenRefreshingProvider` and peers) must delegate it unchanged rather than wrapping it with pre-call work they add to the other trait methods. To add a new provider: 1. Create `crates/domains/ironclaw_llm/src/myprovider.rs` implementing `LlmProvider` (prescriptive: the file you are about to add, not one that exists) diff --git a/crates/domains/ironclaw_llm/src/failover.rs b/crates/domains/ironclaw_llm/src/failover.rs index a7cd0a9e690..4bd0128dcde 100644 --- a/crates/domains/ironclaw_llm/src/failover.rs +++ b/crates/domains/ironclaw_llm/src/failover.rs @@ -579,9 +579,37 @@ impl LlmProvider for FailoverProvider { } async fn model_metadata(&self) -> Result { - self.providers[self.last_used.load(Ordering::Relaxed)] - .model_metadata() - .await + // Any member of the chain may serve a call after a failover, and the + // derived budget outlives the call that failed, so advertise the + // window that is safe for every member. `id` follows `last_used`, + // which is what `active_model_name()` reports and what the + // gateway's identity check compares against. Static by contract: no + // I/O in any member. One call per provider; stop early once the + // aggregate has gone `None` and `last_used`'s id is captured. + let last_used = self.last_used.load(Ordering::Relaxed); + let mut id = None; + let mut context_length: Option> = None; // outer None = no member seen yet + for (index, provider) in self.providers.iter().enumerate() { + let member = provider.model_metadata().await?; + if index == last_used { + id = Some(member.id); + } + context_length = Some(match (context_length, member.context_length) { + (None, first) => first, + (Some(Some(a)), Some(b)) => Some(a.min(b)), + _ => None, + }); + if context_length == Some(None) && id.is_some() { + break; + } + } + Ok(ModelMetadata { + id: id.ok_or_else(|| LlmError::RequestFailed { + provider: "failover".to_string(), + reason: "FailoverProvider has no providers".to_string(), + })?, + context_length: context_length.unwrap_or(None), + }) } fn calculate_cost(&self, input_tokens: u32, output_tokens: u32) -> Decimal { @@ -871,6 +899,8 @@ mod tests { output_cost: Decimal, complete_result: Mutex>>, tool_complete_result: Mutex>>, + context_length: Option, + model_metadata_calls: AtomicUsize, } impl MockProvider { @@ -900,6 +930,8 @@ mod tests { reasoning: None, reasoning_details: None, }))), + context_length: None, + model_metadata_calls: AtomicUsize::new(0), } } @@ -916,6 +948,18 @@ mod tests { } } + /// Test-only builder: advertise a specific `context_length` from + /// `model_metadata()`. `None` means "unknown," matching the + /// production default. + fn with_context_length(mut self, context_length: Option) -> Self { + self.context_length = context_length; + self + } + + fn model_metadata_calls(&self) -> usize { + self.model_metadata_calls.load(Ordering::Relaxed) + } + fn failing_retryable(name: &str) -> Self { Self { name: name.to_string(), @@ -930,6 +974,8 @@ mod tests { provider: name.to_string(), reason: "server error".to_string(), }))), + context_length: None, + model_metadata_calls: AtomicUsize::new(0), } } @@ -945,6 +991,8 @@ mod tests { tool_complete_result: Mutex::new(Some(Err(LlmError::AuthFailed { provider: name.to_string(), }))), + context_length: None, + model_metadata_calls: AtomicUsize::new(0), } } @@ -962,6 +1010,8 @@ mod tests { provider: name.to_string(), retry_after: Some(Duration::from_secs(30)), }))), + context_length: None, + model_metadata_calls: AtomicUsize::new(0), } } } @@ -1010,6 +1060,14 @@ mod tests { *self.active_model.write().unwrap() = model.to_string(); Ok(()) } + + async fn model_metadata(&self) -> Result { + self.model_metadata_calls.fetch_add(1, Ordering::Relaxed); + Ok(ModelMetadata { + id: self.name.clone(), + context_length: self.context_length, + }) + } } fn make_request() -> CompletionRequest { @@ -1131,6 +1189,117 @@ mod tests { assert_eq!(failover.cost_per_token(), (fallback_cost, fallback_cost)); } + // Test: model_metadata advertises the smallest context window across the + // whole chain, not just the currently `last_used` provider — the derived + // budget outlives the call that may later fail over to a smaller-window + // member. + #[tokio::test] + async fn model_metadata_advertises_the_smallest_window_across_the_chain() { + let primary = Arc::new( + MockProvider::succeeding("primary-model", "ok").with_context_length(Some(2_000_000)), + ); + let fallback = Arc::new( + MockProvider::succeeding("fallback-model", "ok").with_context_length(Some(1_000_000)), + ); + + let failover = FailoverProvider::new(vec![primary, fallback]).unwrap(); + + let metadata = failover.model_metadata().await.unwrap(); + assert_eq!(metadata.context_length, Some(1_000_000)); + assert_eq!(metadata.id, "primary-model"); + } + + // Test: an unknown window on any chain member makes the whole chain's + // advertised window unknown — a guessed window is worse than none. + #[tokio::test] + async fn model_metadata_is_unknown_when_any_chain_member_is_unknown() { + let primary = Arc::new( + MockProvider::succeeding("primary-model", "ok").with_context_length(Some(2_000_000)), + ); + let fallback = + Arc::new(MockProvider::succeeding("fallback-model", "ok").with_context_length(None)); + + let failover = FailoverProvider::new(vec![primary, fallback]).unwrap(); + + let metadata = failover.model_metadata().await.unwrap(); + assert_eq!(metadata.context_length, None); + } + + // Test: `id` tracks `last_used` (what `active_model_name()` reports and + // what the gateway's identity check compares against) even though the + // advertised window is the chain minimum. + #[tokio::test] + async fn model_metadata_id_tracks_last_used_after_failover() { + let primary = Arc::new( + MockProvider::failing_retryable("primary-model").with_context_length(Some(2_000_000)), + ); + let fallback = Arc::new( + MockProvider::succeeding("fallback-model", "ok").with_context_length(Some(1_000_000)), + ); + + let failover = FailoverProvider::new(vec![primary, fallback]).unwrap(); + + let _ = failover.complete(make_request()).await.unwrap(); + + let metadata = failover.model_metadata().await.unwrap(); + assert_eq!(metadata.id, "fallback-model"); + assert_eq!(metadata.context_length, Some(1_000_000)); + } + + // Test: each chain member's `model_metadata()` is awaited exactly once + // per call, and the scan stops the moment the aggregate window has gone + // `None` and `last_used`'s id is already captured — a later member is + // never queried once it can no longer change the outcome. + #[tokio::test] + async fn model_metadata_queries_each_member_once_and_stops_after_an_unknown_window() { + let p0 = Arc::new( + MockProvider::succeeding("p0-model", "ok").with_context_length(Some(2_000_000)), + ); + let p1 = Arc::new(MockProvider::succeeding("p1-model", "ok").with_context_length(None)); + let p2 = Arc::new( + MockProvider::succeeding("p2-model", "ok").with_context_length(Some(1_000_000)), + ); + + let failover = FailoverProvider::new(vec![p0.clone(), p1.clone(), p2.clone()]).unwrap(); + + let metadata = failover.model_metadata().await.unwrap(); + + assert_eq!(metadata.id, "p0-model"); + assert_eq!(metadata.context_length, None); + assert_eq!(p0.model_metadata_calls(), 1); + assert_eq!(p1.model_metadata_calls(), 1); + assert_eq!( + p2.model_metadata_calls(), + 0, + "p2 must never be reached: the aggregate already went None and the id was known at p0" + ); + } + + // Test: when every member reports a known window, every member is still + // queried exactly once (no early exit is possible) and the aggregate is + // the chain minimum. + #[tokio::test] + async fn model_metadata_queries_every_member_once_when_all_windows_are_known() { + let p0 = Arc::new( + MockProvider::succeeding("p0-model", "ok").with_context_length(Some(3_000_000)), + ); + let p1 = Arc::new( + MockProvider::succeeding("p1-model", "ok").with_context_length(Some(2_000_000)), + ); + let p2 = Arc::new( + MockProvider::succeeding("p2-model", "ok").with_context_length(Some(1_000_000)), + ); + + let failover = FailoverProvider::new(vec![p0.clone(), p1.clone(), p2.clone()]).unwrap(); + + let metadata = failover.model_metadata().await.unwrap(); + + assert_eq!(metadata.context_length, Some(1_000_000)); + assert_eq!(p0.model_metadata_calls(), 1); + assert_eq!(p1.model_metadata_calls(), 1); + assert_eq!(p2.model_metadata_calls(), 1); + } + // Test: model reporting is request-scoped under concurrent requests. #[tokio::test] async fn effective_model_name_is_request_scoped_under_concurrency() { diff --git a/crates/domains/ironclaw_llm/src/gemini_oauth.rs b/crates/domains/ironclaw_llm/src/gemini_oauth.rs index 19c3252db5f..5b4b15e3b90 100644 --- a/crates/domains/ironclaw_llm/src/gemini_oauth.rs +++ b/crates/domains/ironclaw_llm/src/gemini_oauth.rs @@ -131,26 +131,27 @@ fn parse_custom_headers() -> std::collections::HashMap { } /// Return the context window length for a known Gemini model. -/// Uses explicit match on known model IDs, with a fallback heuristic -/// for unrecognized models. -fn gemini_context_length(model: &str) -> u32 { +/// Uses explicit match on known model IDs; unrecognized models return `None` +/// rather than a guessed value, since a wrong `Some` produces the provider +/// rejection the context budget exists to avoid. +fn gemini_context_length(model: &str) -> Option { match model { // Pro models — 2M context "gemini-2.5-pro" | "gemini-3-pro-preview" | "gemini-3.1-pro-preview" - | "gemini-3.1-pro-preview-customtools" => 2_000_000, + | "gemini-3.1-pro-preview-customtools" => Some(2_000_000), // Flash / Flash-Lite — 1M context "gemini-2.5-flash" | "gemini-2.5-flash-lite" | "gemini-3-flash-preview" - | "gemini-3.1-flash-lite-preview" => 1_000_000, + | "gemini-3.1-flash-lite-preview" => Some(1_000_000), // Legacy - "gemini-1.5-pro" => 2_000_000, - "gemini-1.5-flash" => 1_000_000, - "gemini-2.0-flash" => 1_000_000, - // Fallback for unknown models - _ => 1_000_000, + "gemini-1.5-pro" => Some(2_000_000), + "gemini-1.5-flash" => Some(1_000_000), + "gemini-2.0-flash" => Some(1_000_000), + // Unknown model — do not guess. + _ => None, } } @@ -2157,7 +2158,7 @@ impl LlmProvider for GeminiOauthProvider { async fn model_metadata(&self) -> Result { let model = self.config.model.as_str(); - let context_length = Some(gemini_context_length(model)); + let context_length = gemini_context_length(model); Ok(ModelMetadata { id: self.config.model.clone(), @@ -2280,6 +2281,12 @@ impl LlmProvider for GeminiOauthProvider { mod tests { use super::*; + #[test] + fn gemini_context_length_is_none_for_unknown_models() { + assert_eq!(gemini_context_length("gemini-9-hypothetical"), None); + assert_eq!(gemini_context_length("gemini-2.5-pro"), Some(2_000_000)); + } + #[tokio::test] async fn adapter_response_path_maps_oauth_http_413_to_context_overflow() { use tokio::io::AsyncWriteExt; diff --git a/crates/domains/ironclaw_llm/src/smart_routing.rs b/crates/domains/ironclaw_llm/src/smart_routing.rs index 5d224b15f33..f875740486e 100644 --- a/crates/domains/ironclaw_llm/src/smart_routing.rs +++ b/crates/domains/ironclaw_llm/src/smart_routing.rs @@ -1070,7 +1070,23 @@ impl LlmProvider for SmartRoutingProvider { } async fn model_metadata(&self) -> Result { - self.primary.model_metadata().await + // `complete`/`complete_streaming` may serve a simple no-tool call from + // `self.cheap` instead of `self.primary` (see the routing above), so + // the advertised context window must be safe for whichever model + // actually answers. `id` stays the primary's so it keeps matching + // `active_model_name()`, which the gateway checks against before + // trusting this metadata. This stays I/O-free: both inner + // `model_metadata()` calls are static by contract. + let primary = self.primary.model_metadata().await?; + let cheap = self.cheap.model_metadata().await?; + let context_length = match (primary.context_length, cheap.context_length) { + (Some(a), Some(b)) => Some(a.min(b)), + _ => None, + }; + Ok(ModelMetadata { + id: primary.id, + context_length, + }) } fn effective_model_name(&self, requested_model: Option<&str>) -> String { @@ -1747,6 +1763,9 @@ mod tests { completion_requests: Mutex>, tool_requests: Mutex>, sinks: Mutex>>, + // Test-only: lets tests control the advertised context window instead + // of always exercising the trait default (`None`). + context_length: Option, } impl StreamingStubLlm { @@ -1764,6 +1783,7 @@ mod tests { completion_requests: Mutex::new(Vec::new()), tool_requests: Mutex::new(Vec::new()), sinks: Mutex::new(Vec::new()), + context_length: None, } } @@ -1772,6 +1792,11 @@ mod tests { self } + fn with_context_length(mut self, context_length: Option) -> Self { + self.context_length = context_length; + self + } + fn calls(&self) -> u32 { self.calls.load(Ordering::Relaxed) } @@ -1799,6 +1824,13 @@ mod tests { (Decimal::ZERO, Decimal::ZERO) } + async fn model_metadata(&self) -> Result { + Ok(ModelMetadata { + id: self.model_name.to_string(), + context_length: self.context_length, + }) + } + async fn complete( &self, _request: CompletionRequest, @@ -1951,6 +1983,40 @@ mod tests { assert_eq!(primary.calls(), 0); } + #[tokio::test] + async fn model_metadata_advertises_the_smaller_window_of_primary_and_cheap() { + let primary = Arc::new( + StreamingStubLlm::new("primary", "primary-response", vec![]) + .with_context_length(Some(2_000_000)), + ); + let cheap = Arc::new( + StreamingStubLlm::new("cheap", "cheap-response", vec![]) + .with_context_length(Some(1_000_000)), + ); + let router = SmartRoutingProvider::new(primary.clone(), cheap.clone(), default_config()); + + let metadata = router.model_metadata().await.unwrap(); + + assert_eq!(metadata.id, "primary"); + assert_eq!(metadata.context_length, Some(1_000_000)); + } + + #[tokio::test] + async fn model_metadata_is_unknown_when_either_route_is_unknown() { + let primary = Arc::new( + StreamingStubLlm::new("primary", "primary-response", vec![]) + .with_context_length(Some(2_000_000)), + ); + let cheap = Arc::new( + StreamingStubLlm::new("cheap", "cheap-response", vec![]).with_context_length(None), + ); + let router = SmartRoutingProvider::new(primary.clone(), cheap.clone(), default_config()); + + let metadata = router.model_metadata().await.unwrap(); + + assert_eq!(metadata.context_length, None); + } + #[tokio::test] async fn complex_task_routes_to_primary() { let primary = Arc::new(StubLlm::new("primary-response").with_model_name("primary")); diff --git a/crates/domains/ironclaw_llm/src/token_refreshing.rs b/crates/domains/ironclaw_llm/src/token_refreshing.rs index 821c0458ae4..002ec051061 100644 --- a/crates/domains/ironclaw_llm/src/token_refreshing.rs +++ b/crates/domains/ironclaw_llm/src/token_refreshing.rs @@ -246,8 +246,11 @@ impl LlmProvider for TokenRefreshingProvider { self.inner.list_model_catalog().await } + // `model_metadata()` is a static, I/O-free description of the configured + // model (CONTRACT.md, "LlmProvider Trait" key notes) -- it must never + // touch the network or take the token-refresh lock, so no pre-emptive + // refresh happens here. async fn model_metadata(&self) -> Result { - self.ensure_fresh_token().await; self.inner.model_metadata().await } @@ -586,6 +589,77 @@ mod tests { assert_eq!(inner.token_updates.load(Ordering::Relaxed), 0); } + /// Binds a local TCP listener that counts connection attempts and replies + /// with a minimal HTTP error response so any client `send()` completes + /// quickly instead of hanging. Used to observe whether `refresh_tokens()` + /// (which POSTs to `{auth_endpoint}/oauth/token`) was ever attempted, + /// without a mock-HTTP dependency. + fn spawn_counting_auth_endpoint() -> (String, Arc) { + let listener = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + listener.set_nonblocking(true).unwrap(); + let listener = tokio::net::TcpListener::from_std(listener).unwrap(); + let addr = listener.local_addr().unwrap(); + let connections = Arc::new(AtomicUsize::new(0)); + let connections_task = connections.clone(); + tokio::spawn(async move { + loop { + let Ok((mut stream, _)) = listener.accept().await else { + break; + }; + connections_task.fetch_add(1, Ordering::SeqCst); + use tokio::io::AsyncWriteExt; + let _ = stream + .write_all(b"HTTP/1.1 400 Bad Request\r\ncontent-length: 0\r\n\r\n") + .await; + } + }); + (format!("http://{addr}"), connections) + } + + #[tokio::test] + async fn model_metadata_does_not_refresh_the_token() { + let dir = tempdir().unwrap(); + let (auth_endpoint, connections) = spawn_counting_auth_endpoint(); + let mut config = test_codex_config(dir.path().join("session.json")); + config.auth_endpoint = auth_endpoint; + let jwt = make_test_jwt("acct_test"); + let inner = Arc::new( + OpenAiCodexProvider::new(&config.model, &config.api_base_url, &jwt, 300) + .expect("provider creation should succeed"), + ); + let session = Arc::new(OpenAiCodexSessionManager::new(config).unwrap()); + // A session that needs a refresh: expired, with a non-empty refresh + // token so `refresh_tokens()` would actually attempt the HTTP call + // rather than short-circuiting on a missing refresh token. + session + .set_session(OpenAiCodexSession { + access_token: make_test_jwt("acct_test"), + refresh_token: "refresh-token".to_string(), + expires_at: chrono::Utc::now() - chrono::Duration::hours(1), + created_at: chrono::Utc::now() - chrono::Duration::hours(2), + }) + .await; + assert!( + session.needs_refresh().await, + "test session must report it needs a refresh" + ); + + let provider = TokenRefreshingProvider::new(inner.clone(), session); + let metadata = provider + .model_metadata() + .await + .expect("model_metadata should succeed without touching the network"); + + assert_eq!( + connections.load(Ordering::SeqCst), + 0, + "model_metadata() must not perform a token refresh" + ); + let inner_metadata = inner.model_metadata().await.unwrap(); + assert_eq!(metadata.id, inner_metadata.id); + assert_eq!(metadata.context_length, inner_metadata.context_length); + } + #[tokio::test] async fn auth_retry_sink_delegates_replacement_capabilities() { let inner = Arc::new(RecordingSink::default()); diff --git a/crates/loop/ironclaw_agent_loop/src/families/mod.rs b/crates/loop/ironclaw_agent_loop/src/families/mod.rs index 2cede7699d0..021116d97ea 100644 --- a/crates/loop/ironclaw_agent_loop/src/families/mod.rs +++ b/crates/loop/ironclaw_agent_loop/src/families/mod.rs @@ -32,7 +32,7 @@ fn default_family_fingerprint(iteration_limit: u32, model_availability_attempts: planner=DefaultPlanner;\ strategies=\ context:DefaultContextStrategy(max_messages=128),\ - compaction:ActiveTaskPreservingCompactionStrategy(context_limit=128000,reserve=20000,preserve_tail=8000,min_compacted=3,min_tail=3,deadline_ms=30000,ineffective_trip_limit=3),\ + compaction:ActiveTaskPreservingCompactionStrategy(context_limit=run_context,reserve=run_context,preserve_tail=min(8000,visible/2),min_compacted=3,min_tail=3,deadline_ms=30000,ineffective_trip_limit=3),\ capability:DefaultCapabilityStrategy(all),\ model:DefaultModelStrategy(primary_or_fallback_index),\ batch:model_emitted_calls(bounded_fanout=4),\ @@ -51,8 +51,8 @@ fn default_family_fingerprint(iteration_limit: u32, model_availability_attempts: /// Update this digest when the default family composition, planner behavior, or /// identity schema changes in a replay-relevant way. pub const DEFAULT_FAMILY_DIGEST: ComponentDigest = ComponentDigest([ - 0x51, 0x7c, 0x91, 0x80, 0xec, 0xca, 0x58, 0xdf, 0x60, 0xad, 0x9a, 0x47, 0x4e, 0x07, 0x0d, 0x14, - 0x29, 0xa9, 0x38, 0xa6, 0x3b, 0xd9, 0x3f, 0x21, 0xd5, 0x16, 0x66, 0x94, 0xee, 0xef, 0xff, 0x06, + 0x33, 0xdb, 0xbd, 0x7a, 0xda, 0xb0, 0x37, 0xd4, 0x81, 0x6f, 0xde, 0x08, 0xad, 0xac, 0x30, 0xba, + 0x6b, 0x5d, 0x47, 0xe8, 0x4d, 0x2e, 0x64, 0xd0, 0x73, 0xe6, 0xc1, 0x7c, 0x94, 0xab, 0x97, 0xd5, ]); /// The default loop family: the text-tool-use baseline. diff --git a/crates/loop/ironclaw_agent_loop/src/families/subagent.rs b/crates/loop/ironclaw_agent_loop/src/families/subagent.rs index c8016f56752..064bddd3551 100644 --- a/crates/loop/ironclaw_agent_loop/src/families/subagent.rs +++ b/crates/loop/ironclaw_agent_loop/src/families/subagent.rs @@ -17,7 +17,7 @@ const SUBAGENT_FAMILY_FINGERPRINT: &[u8] = concat!( "planner=DefaultPlanner;", "strategies=", "context:DefaultContextStrategy(max_messages=128),", - "compaction:ActiveTaskPreservingCompactionStrategy(context_limit=128000,reserve=20000,preserve_tail=8000,min_compacted=3,min_tail=3,deadline_ms=30000,ineffective_trip_limit=3),", + "compaction:ActiveTaskPreservingCompactionStrategy(context_limit=run_context,reserve=run_context,preserve_tail=min(8000,visible/2),min_compacted=3,min_tail=3,deadline_ms=30000,ineffective_trip_limit=3),", "capability:DefaultCapabilityStrategy(all),", "model:DefaultModelStrategy(primary_or_fallback_index),", "batch:model_emitted_calls(bounded_fanout=4),", @@ -31,8 +31,8 @@ const SUBAGENT_FAMILY_FINGERPRINT: &[u8] = concat!( .as_bytes(); pub const SUBAGENT_FAMILY_DIGEST: ComponentDigest = ComponentDigest([ - 0x9a, 0x67, 0xdd, 0xb9, 0x72, 0x7a, 0x5b, 0xf0, 0xda, 0x9f, 0xb8, 0x12, 0xd1, 0x0f, 0x95, 0xd1, - 0x6c, 0xe9, 0x6e, 0x53, 0x0f, 0x4e, 0xa5, 0x8e, 0xec, 0xc7, 0xe5, 0x28, 0xf8, 0x0b, 0x8d, 0x2f, + 0x57, 0xb8, 0x80, 0xcf, 0x93, 0x9f, 0xbf, 0x87, 0x9e, 0xc5, 0x33, 0xa2, 0x76, 0x74, 0xaa, 0x47, + 0x10, 0xd6, 0x02, 0x8d, 0x0b, 0x84, 0x50, 0x2b, 0xce, 0xa4, 0x99, 0xb6, 0x1b, 0xad, 0xff, 0x02, ]); pub fn subagent() -> LoopFamily { diff --git a/crates/loop/ironclaw_agent_loop/src/families/unbound.rs b/crates/loop/ironclaw_agent_loop/src/families/unbound.rs index 954fe0ed408..fde3dbbeb07 100644 --- a/crates/loop/ironclaw_agent_loop/src/families/unbound.rs +++ b/crates/loop/ironclaw_agent_loop/src/families/unbound.rs @@ -21,7 +21,7 @@ const UNBOUND_DEFAULT_FAMILY_FINGERPRINT: &[u8] = concat!( "planner=DefaultPlanner;", "strategies=", "context:DefaultContextStrategy(max_messages=128),", - "compaction:ActiveTaskPreservingCompactionStrategy(context_limit=128000,reserve=20000,preserve_tail=8000,min_compacted=3,min_tail=3,deadline_ms=30000,ineffective_trip_limit=3),", + "compaction:ActiveTaskPreservingCompactionStrategy(context_limit=run_context,reserve=run_context,preserve_tail=min(8000,visible/2),min_compacted=3,min_tail=3,deadline_ms=30000,ineffective_trip_limit=3),", "capability:DefaultCapabilityStrategy(all),", "model:DefaultModelStrategy(primary_or_fallback_index),", "batch:DefaultBatchPolicyStrategy(parallel_unless_exclusive),", @@ -42,7 +42,7 @@ const UNBOUND_STRUCTURED_FAMILY_FINGERPRINT: &[u8] = concat!( "planner=DefaultPlanner;", "strategies=", "context:DefaultContextStrategy(max_messages=128),", - "compaction:ActiveTaskPreservingCompactionStrategy(context_limit=128000,reserve=20000,preserve_tail=8000,min_compacted=3,min_tail=3,deadline_ms=30000,ineffective_trip_limit=3),", + "compaction:ActiveTaskPreservingCompactionStrategy(context_limit=run_context,reserve=run_context,preserve_tail=min(8000,visible/2),min_compacted=3,min_tail=3,deadline_ms=30000,ineffective_trip_limit=3),", "capability:DefaultCapabilityStrategy(all),", "model:StructuredResultModelStrategy(primary_or_fallback_index,force_result_tool_on_structured_repair),", "batch:DefaultBatchPolicyStrategy(parallel_unless_exclusive),", @@ -57,14 +57,14 @@ const UNBOUND_STRUCTURED_FAMILY_FINGERPRINT: &[u8] = concat!( /// Stable digest: BLAKE3-256 of the unbound-default family fingerprint. pub const UNBOUND_DEFAULT_FAMILY_DIGEST: ComponentDigest = ComponentDigest([ - 0xfb, 0x7b, 0x56, 0x08, 0x72, 0x91, 0xd7, 0xd4, 0x9e, 0x8a, 0x80, 0x6b, 0x81, 0xdc, 0x35, 0x1f, - 0x66, 0x09, 0x7b, 0xb9, 0x50, 0x48, 0xf3, 0x01, 0x87, 0x0a, 0xd5, 0x44, 0xd0, 0xfd, 0x37, 0xb2, + 0x3d, 0x53, 0x2e, 0xcf, 0x31, 0x98, 0xfd, 0x07, 0xe3, 0x06, 0x81, 0x09, 0xa7, 0x08, 0xbe, 0xe4, + 0x76, 0x1b, 0x92, 0xba, 0x26, 0x33, 0x48, 0xd9, 0x9c, 0x75, 0x2e, 0x18, 0x35, 0xfe, 0xb7, 0x05, ]); /// Stable digest: BLAKE3-256 of the unbound-structured family fingerprint. pub const UNBOUND_STRUCTURED_FAMILY_DIGEST: ComponentDigest = ComponentDigest([ - 0x27, 0x9d, 0x02, 0xc8, 0xe3, 0x05, 0x7f, 0xb7, 0xcd, 0x55, 0xe4, 0x55, 0xd4, 0xd3, 0x64, 0x25, - 0x84, 0x2a, 0x3f, 0xb8, 0xd9, 0x99, 0xae, 0x1b, 0xf6, 0xb4, 0xfc, 0x95, 0x18, 0xc2, 0xa1, 0x58, + 0xbb, 0x98, 0x16, 0x49, 0xcc, 0xf7, 0xaf, 0xf1, 0x6a, 0xb5, 0xf5, 0x98, 0x47, 0xc1, 0x1c, 0x8f, + 0x59, 0x51, 0x79, 0xb6, 0x3b, 0x80, 0x2b, 0x50, 0x3e, 0xf6, 0x41, 0x98, 0xad, 0xd2, 0xf4, 0x32, ]); fn structured_result_capability() -> Result { diff --git a/crates/loop/ironclaw_agent_loop/src/strategies/active_task_compaction.rs b/crates/loop/ironclaw_agent_loop/src/strategies/active_task_compaction.rs index a831fdbe2b1..b32affd0fa0 100644 --- a/crates/loop/ironclaw_agent_loop/src/strategies/active_task_compaction.rs +++ b/crates/loop/ironclaw_agent_loop/src/strategies/active_task_compaction.rs @@ -43,9 +43,10 @@ impl CompactionStrategy for ActiveTaskPreservingCompactionStrategy { fn should_compact( &self, state: &LoopExecutionState, - _ctx: &LoopRunContext, + ctx: &LoopRunContext, ) -> CompactionDecision { - if !self.base.can_evaluate(state) { + let budget = self.base.effective_budget(ctx); + if !self.base.can_evaluate(state, budget) { return CompactionDecision::Skip; } @@ -59,17 +60,17 @@ impl CompactionStrategy for ActiveTaskPreservingCompactionStrategy { prompt_fingerprint, preserve_from_sequence, ) - .map(|sequence| self.base.trigger_at(state, sequence)) + .map(|sequence| self.base.trigger_at(state, budget, sequence)) .unwrap_or(CompactionDecision::Skip); } active_task_preserving_user_boundary( state, prompt_fingerprint, - self.base.preserve_tail_tokens, + self.base.effective_preserve_tail_tokens(budget), self.minimum_tail_messages, self.minimum_compacted_messages, ) - .map(|sequence| self.base.trigger_at(state, sequence)) + .map(|sequence| self.base.trigger_at(state, budget, sequence)) .unwrap_or(CompactionDecision::Skip) } } @@ -219,6 +220,44 @@ mod tests { ); } + #[test] + fn active_task_strategy_finds_a_boundary_inside_a_small_window() { + // Context-length regression: an 8k-advertised window derives + // visible_transcript_tokens() == 5,400 (half == 2,700), while the + // compiled-in tail is 8,000. Scale active_task_message_index()'s + // eight 10-token entries up 100x (8,000 tokens total) so the + // unclamped 8,000 tail can never be reached mid-walk (no boundary), + // while the clamped 2,700 tail is reached with enough prefix and + // tail messages to find one. + let context = crate::test_support::test_run_context("active-task-small-window") + .with_resolved_context_budget(PromptContextTokenBudget::from_advertised_window(Some( + 8_000, + ))); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_state.force_compact_on_next_iteration = true; + let scaled_index: Vec = active_task_message_index() + .into_iter() + .map(|entry| MessageIndexEntry { + estimated_tokens: entry.estimated_tokens * 100, + ..entry + }) + .collect(); + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(scaled_index); + let strategy = ActiveTaskPreservingCompactionStrategy::from(DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::default(), + preserve_tail_tokens: DefaultCompactionStrategy::DEFAULT_PRESERVE_TAIL_TOKENS, + deadline_ms: 7, + }); + + assert!(matches!( + strategy.should_compact(&state, &context), + CompactionDecision::Trigger { + drop_through_seq, + .. + } if drop_through_seq > 0 + )); + } + #[test] fn forced_compaction_skips_when_only_latest_user_is_safe_candidate() { let context = crate::test_support::test_run_context("active-task-preserving-only-latest"); @@ -251,7 +290,17 @@ mod tests { state.compaction_state.force_compact_on_next_iteration = true; state.compaction_prompt = CompactionPromptSnapshot::from_message_index(active_task_message_index()); - let strategy = active_task_preserving_strategy(60); + // `active_task_preserving_strategy`'s helper budget (100/10 -> 90 + // visible) would clamp a tail of 60 down to 45, changing this test's + // premise (a candidate becomes reachable). Use a budget whose half- + // visible ceiling comfortably exceeds 60 so the configured tail + // survives the clamp intact and this test still proves the tail + // budget alone can defeat every candidate. + let strategy = ActiveTaskPreservingCompactionStrategy::from(DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(200, 10, 0), + preserve_tail_tokens: 60, + deadline_ms: 7, + }); assert_eq!( strategy.should_compact(&state, &context), @@ -460,4 +509,36 @@ mod tests { CompactionDecision::Skip ); } + + #[test] + fn active_task_strategy_honors_the_run_context_budget() { + // The strategy's own budget (100/10 -> 90 visible) leaves headroom + // for this 80-token index, so it would not compact. The run's model + // has a far smaller real window, which puts the prompt over. + let context = crate::test_support::test_run_context("active-task-budget-override") + .with_resolved_context_budget(PromptContextTokenBudget::new(50, 5, 0)); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_prompt = + CompactionPromptSnapshot::from_message_index(active_task_message_index()); + let strategy = active_task_preserving_strategy(1); + + assert!(matches!( + strategy.should_compact(&state, &context), + CompactionDecision::Trigger { .. } + )); + } + + #[test] + fn active_task_strategy_falls_back_to_its_own_budget_without_a_resolved_one() { + let context = crate::test_support::test_run_context("active-task-budget-fallback"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_prompt = + CompactionPromptSnapshot::from_message_index(active_task_message_index()); + let strategy = active_task_preserving_strategy(1); + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Skip + ); + } } diff --git a/crates/loop/ironclaw_agent_loop/src/strategies/compaction.rs b/crates/loop/ironclaw_agent_loop/src/strategies/compaction.rs index de93f2c2517..09b7814dfa7 100644 --- a/crates/loop/ironclaw_agent_loop/src/strategies/compaction.rs +++ b/crates/loop/ironclaw_agent_loop/src/strategies/compaction.rs @@ -53,12 +53,40 @@ impl DefaultCompactionStrategy { pub const DEFAULT_PRESERVE_TAIL_TOKENS: u64 = 8_000; pub const DEFAULT_DEADLINE_MS: u64 = 30_000; - pub(super) fn can_evaluate(&self, state: &LoopExecutionState) -> bool { + /// The budget this run actually runs with: the one resolved from the + /// run's model when present, otherwise this strategy's compiled-in + /// default. + pub(super) fn effective_budget(&self, ctx: &LoopRunContext) -> PromptContextTokenBudget { + ctx.resolved_context_budget + .unwrap_or(self.prompt_context_budget) + } + + /// The tail this run can afford to protect: the configured tail, capped + /// at half the visible transcript so compaction can always drop the + /// other half. + /// + /// `preserve_tail_tokens` is compiled-in, but the visible transcript is + /// run-resolved from the model's advertised context window and can be + /// far smaller (a small-window model can derive a visible transcript + /// below the compiled-in tail). Without this cap, both automatic and + /// forced-recovery compaction would be permanently disabled for that + /// run — the `can_evaluate` guard would never clear and the boundary + /// searches would never find a tail-sized gap to cut before. + pub(super) fn effective_preserve_tail_tokens(&self, budget: PromptContextTokenBudget) -> u64 { + self.preserve_tail_tokens + .min(budget.visible_transcript_tokens() / 2) + } + + pub(super) fn can_evaluate( + &self, + state: &LoopExecutionState, + budget: PromptContextTokenBudget, + ) -> bool { if state.compaction_prompt.message_index.is_empty() { return false; } - let threshold = self.prompt_context_budget.visible_transcript_tokens(); - if threshold <= self.preserve_tail_tokens { + let threshold = budget.visible_transcript_tokens(); + if threshold <= self.effective_preserve_tail_tokens(budget) { return false; } // Forced/recovery compactions (context-overflow retry, byte-cap @@ -77,6 +105,7 @@ impl DefaultCompactionStrategy { pub(super) fn trigger_at( &self, state: &LoopExecutionState, + budget: PromptContextTokenBudget, drop_through_seq: u64, ) -> CompactionDecision { let effectiveness_baseline = if state.compaction_state.force_compact_on_next_iteration { @@ -85,12 +114,12 @@ impl DefaultCompactionStrategy { } } else { CompactionEffectivenessBaseline::TriggerThresholdTokens { - tokens: self.prompt_context_budget.visible_transcript_tokens(), + tokens: budget.visible_transcript_tokens(), } }; CompactionDecision::Trigger { drop_through_seq, - preserve_tail_tokens: self.preserve_tail_tokens, + preserve_tail_tokens: self.effective_preserve_tail_tokens(budget), deadline_ms: self.deadline_ms, effectiveness_baseline, } @@ -111,9 +140,10 @@ impl CompactionStrategy for DefaultCompactionStrategy { fn should_compact( &self, state: &LoopExecutionState, - _ctx: &LoopRunContext, + ctx: &LoopRunContext, ) -> CompactionDecision { - if !self.can_evaluate(state) { + let budget = self.effective_budget(ctx); + if !self.can_evaluate(state, budget) { return CompactionDecision::Skip; } let prompt_fingerprint = state.compaction_prompt.fingerprint(); @@ -122,22 +152,22 @@ impl CompactionStrategy for DefaultCompactionStrategy { == Some(CompactionInitiator::WindowEviction) { return eligible_window_eviction_boundary(state, prompt_fingerprint, None) - .map(|sequence| self.trigger_at(state, sequence)) + .map(|sequence| self.trigger_at(state, budget, sequence)) .unwrap_or(CompactionDecision::Skip); } return latest_eligible_user_boundary(state, prompt_fingerprint) - .map(|sequence| self.trigger_at(state, sequence)) + .map(|sequence| self.trigger_at(state, budget, sequence)) .unwrap_or(CompactionDecision::Skip); } tail_preserving_user_boundary( state, prompt_fingerprint, - self.preserve_tail_tokens, + self.effective_preserve_tail_tokens(budget), 0, |_| true, ) - .map(|sequence| self.trigger_at(state, sequence)) + .map(|sequence| self.trigger_at(state, budget, sequence)) .unwrap_or(CompactionDecision::Skip) } } @@ -320,673 +350,4 @@ impl CompactionForceStrategy for ByteCapStrategy { } #[cfg(test)] -mod tests { - use super::*; - use crate::state::{ - CompactionEffectivenessBaseline, CompactionPromptSnapshot, CompactionStrategyState, - DeferredCompactionWatermark, LoopExecutionState, MessageIndexEntry, - }; - use ironclaw_host_api::ids::CapabilityId; - use ironclaw_loop_contracts::PromptContextTokenBudget; - - #[test] - fn evaluate_skips_when_message_index_is_empty() { - let context = crate::test_support::test_run_context("compaction-strategy-empty"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_state.force_compact_on_next_iteration = true; - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 1, - deadline_ms: 1, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Skip - ); - } - - #[test] - fn evaluate_skips_when_no_eligible_user_message_boundary_exists() { - let context = crate::test_support::test_run_context("compaction-strategy"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_prompt = - CompactionPromptSnapshot::from_message_index(vec![MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 100, - }]); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 1, - deadline_ms: 1, - }; - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Skip - ); - } - - #[test] - fn evaluate_skips_when_below_threshold_with_valid_user_boundary_and_forcing_is_off() { - let context = crate::test_support::test_run_context("compaction-strategy-below-threshold"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ - MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 20, - }, - MessageIndexEntry { - sequence: 2, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 20, - }, - ]); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 60, - deadline_ms: 1, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Skip - ); - } - - #[test] - fn can_evaluate_skips_when_visible_threshold_equals_preserve_tail() { - let context = crate::test_support::test_run_context("compaction-strategy-equal-tail"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_prompt = - CompactionPromptSnapshot::from_message_index(vec![MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 100, - }]); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 90, - deadline_ms: 1, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Skip - ); - } - - #[test] - fn evaluate_triggers_at_latest_user_boundary_outside_tail() { - let context = crate::test_support::test_run_context("compaction-strategy-trigger"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_state = CompactionStrategyState::default(); - state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ - MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 30, - }, - MessageIndexEntry { - sequence: 2, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 30, - }, - MessageIndexEntry { - sequence: 3, - kind: IndexedMessageKind::User, - estimated_tokens: 30, - }, - MessageIndexEntry { - sequence: 4, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 30, - }, - ]); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 60, - deadline_ms: 7, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Trigger { - drop_through_seq: 1, - preserve_tail_tokens: 60, - deadline_ms: 7, - effectiveness_baseline: CompactionEffectivenessBaseline::TriggerThresholdTokens { - tokens: 90, - }, - } - ); - } - - #[test] - fn evaluate_triggers_when_newest_assistant_block_exceeds_tail_budget() { - let context = crate::test_support::test_run_context("compaction-strategy-tail-overflow"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_state = CompactionStrategyState::default(); - state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ - MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 2, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 100, - }, - ]); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 60, - deadline_ms: 7, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Trigger { - drop_through_seq: 1, - preserve_tail_tokens: 60, - deadline_ms: 7, - effectiveness_baseline: CompactionEffectivenessBaseline::TriggerThresholdTokens { - tokens: 90, - }, - } - ); - } - - #[test] - fn evaluate_skips_when_latest_user_boundary_was_already_compacted() { - let context = crate::test_support::test_run_context("compaction-strategy-compacted"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_state.last_compacted_through_seq = Some(3); - state.compaction_state.force_compact_on_next_iteration = true; - state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ - MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 2, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 3, - kind: IndexedMessageKind::User, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 4, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 100, - }, - ]); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 60, - deadline_ms: 7, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Skip - ); - } - - #[test] - fn evaluate_skips_previously_deferred_boundary_when_forced() { - let context = crate::test_support::test_run_context("compaction-strategy-deferred"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_state.force_compact_on_next_iteration = true; - state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ - MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 2, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 3, - kind: IndexedMessageKind::User, - estimated_tokens: 10, - }, - ]); - state.compaction_state.last_deferred = Some(DeferredCompactionWatermark { - through_seq: 3, - prompt_fingerprint: state.compaction_prompt.fingerprint(), - }); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 1, - deadline_ms: 7, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Trigger { - drop_through_seq: 1, - preserve_tail_tokens: 1, - deadline_ms: 7, - effectiveness_baseline: - CompactionEffectivenessBaseline::PreCompactionPromptTokens { tokens: 30 }, - } - ); - } - - #[test] - fn evaluate_skips_deferred_boundary_in_threshold_overflow_path() { - let context = - crate::test_support::test_run_context("compaction-strategy-deferred-threshold"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ - MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 50, - }, - MessageIndexEntry { - sequence: 2, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 50, - }, - MessageIndexEntry { - sequence: 3, - kind: IndexedMessageKind::User, - estimated_tokens: 50, - }, - MessageIndexEntry { - sequence: 4, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 50, - }, - ]); - state.compaction_state.last_deferred = Some(DeferredCompactionWatermark { - through_seq: 3, - prompt_fingerprint: state.compaction_prompt.fingerprint(), - }); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 60, - deadline_ms: 7, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Trigger { - drop_through_seq: 1, - preserve_tail_tokens: 60, - deadline_ms: 7, - effectiveness_baseline: CompactionEffectivenessBaseline::TriggerThresholdTokens { - tokens: 90, - }, - } - ); - } - - #[test] - fn evaluate_skips_when_only_deferred_boundary_is_eligible_in_threshold_overflow_path() { - let context = crate::test_support::test_run_context("compaction-strategy-deferred-skip"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ - MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 50, - }, - MessageIndexEntry { - sequence: 2, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 50, - }, - ]); - state.compaction_state.last_deferred = Some(DeferredCompactionWatermark { - through_seq: 1, - prompt_fingerprint: state.compaction_prompt.fingerprint(), - }); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 60, - deadline_ms: 7, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Skip - ); - } - - #[test] - fn evaluate_retries_deferred_boundary_after_prompt_snapshot_changes() { - let context = crate::test_support::test_run_context("compaction-strategy-deferred-changed"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_state.last_deferred = Some(DeferredCompactionWatermark { - through_seq: 3, - prompt_fingerprint: 42, - }); - state.compaction_state.force_compact_on_next_iteration = true; - state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ - MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 2, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 3, - kind: IndexedMessageKind::User, - estimated_tokens: 10, - }, - ]); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 1, - deadline_ms: 7, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Trigger { - drop_through_seq: 3, - preserve_tail_tokens: 1, - deadline_ms: 7, - effectiveness_baseline: - CompactionEffectivenessBaseline::PreCompactionPromptTokens { tokens: 30 }, - } - ); - } - - #[test] - fn evaluate_retries_after_transcript_advances_past_deferred_boundary() { - let context = crate::test_support::test_run_context("compaction-strategy-deferred-newer"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_state.force_compact_on_next_iteration = true; - state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ - MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 2, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 3, - kind: IndexedMessageKind::User, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 4, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 5, - kind: IndexedMessageKind::User, - estimated_tokens: 10, - }, - ]); - state.compaction_state.last_deferred = Some(DeferredCompactionWatermark { - through_seq: 3, - prompt_fingerprint: state.compaction_prompt.fingerprint(), - }); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 1, - deadline_ms: 7, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Trigger { - drop_through_seq: 5, - preserve_tail_tokens: 1, - deadline_ms: 7, - effectiveness_baseline: - CompactionEffectivenessBaseline::PreCompactionPromptTokens { tokens: 50 }, - } - ); - } - - #[test] - fn evaluate_uses_output_budget_when_larger_than_reserve() { - let context = crate::test_support::test_run_context("compaction-strategy-output-budget"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ - MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 40, - }, - MessageIndexEntry { - sequence: 2, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 35, - }, - ]); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 30), - preserve_tail_tokens: 1, - deadline_ms: 7, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Trigger { - drop_through_seq: 1, - preserve_tail_tokens: 1, - deadline_ms: 7, - effectiveness_baseline: CompactionEffectivenessBaseline::TriggerThresholdTokens { - tokens: 70, - }, - } - ); - } - - #[test] - fn tail_preserving_user_boundary_respects_minimum_tail_message_count() { - let context = crate::test_support::test_run_context("compaction-strategy-min-tail"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ - MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 2, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 3, - kind: IndexedMessageKind::User, - estimated_tokens: 10, - }, - MessageIndexEntry { - sequence: 4, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 10, - }, - ]); - - let boundary = tail_preserving_user_boundary( - &state, - state.compaction_prompt.fingerprint(), - 1, - 2, - |_| true, - ); - - assert_eq!(boundary, Some(1)); - } - - #[test] - fn evaluate_skips_threshold_trigger_when_circuit_is_open() { - let context = crate::test_support::test_run_context("compaction-strategy-circuit-open"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_state.compaction_circuit_open = true; - state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ - MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 100, - }, - MessageIndexEntry { - sequence: 2, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 100, - }, - ]); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 1, - deadline_ms: 7, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Skip - ); - } - - #[test] - fn evaluate_triggers_forced_compaction_even_when_circuit_is_open() { - // BUG B1 regression: force_compact_on_next_iteration is how - // context-overflow recovery and byte-cap overflow request a shrink. - // An open breaker must not suppress it — only automatic - // threshold-triggered compaction is gated. - let context = - crate::test_support::test_run_context("compaction-strategy-circuit-open-forced"); - let mut state = LoopExecutionState::initial_for_run(&context); - state.compaction_state.compaction_circuit_open = true; - state.compaction_state.force_compact_on_next_iteration = true; - state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ - MessageIndexEntry { - sequence: 1, - kind: IndexedMessageKind::User, - estimated_tokens: 100, - }, - MessageIndexEntry { - sequence: 2, - kind: IndexedMessageKind::Assistant, - estimated_tokens: 100, - }, - ]); - let strategy = DefaultCompactionStrategy { - prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), - preserve_tail_tokens: 1, - deadline_ms: 7, - }; - - assert_eq!( - strategy.should_compact(&state, &context), - CompactionDecision::Trigger { - drop_through_seq: 1, - preserve_tail_tokens: 1, - deadline_ms: 7, - effectiveness_baseline: - CompactionEffectivenessBaseline::PreCompactionPromptTokens { tokens: 200 }, - } - ); - } - - // --- ByteCapStrategy tests --- - - #[test] - fn byte_cap_strategy_trips_when_capability_exceeds_cap() { - let context = crate::test_support::test_run_context("byte-cap-policy-trips"); - let mut state = LoopExecutionState::initial_for_run(&context); - let id = CapabilityId::new("builtin.http").expect("valid capability"); - // 32_000 is the cap; 32_001 exceeds it. - state - .post_capability_state - .pending_capability_bytes - .insert(id, 32_001); - - let strategy = ByteCapStrategy::with_defaults(); - assert_eq!( - strategy.should_force_compact(&state), - Some(CompactionInitiator::CapabilityResultOverflow) - ); - } - - #[test] - fn byte_cap_strategy_skips_when_under_threshold() { - let context = crate::test_support::test_run_context("byte-cap-policy-under"); - let mut state = LoopExecutionState::initial_for_run(&context); - let http_id = CapabilityId::new("builtin.http").expect("valid capability"); - let subagent_id = CapabilityId::new("builtin.spawn_subagent").expect("valid capability"); - // Both under their respective caps. - state - .post_capability_state - .pending_capability_bytes - .insert(http_id, 31_999); - state - .post_capability_state - .pending_capability_bytes - .insert(subagent_id, 47_999); - - let strategy = ByteCapStrategy::with_defaults(); - assert_eq!(strategy.should_force_compact(&state), None); - } - - #[test] - fn byte_cap_strategy_uses_default_cap_for_unknown_capability() { - let context = crate::test_support::test_run_context("byte-cap-policy-unknown"); - let mut state = LoopExecutionState::initial_for_run(&context); - let id = CapabilityId::new("custom.unknown_tool").expect("valid capability"); - // DEFAULT_FALLBACK_CAP_BYTES is 32_000; 32_001 exceeds it. - state - .post_capability_state - .pending_capability_bytes - .insert(id, ByteCapStrategy::DEFAULT_FALLBACK_CAP_BYTES + 1); - - let strategy = ByteCapStrategy::with_defaults(); - assert_eq!( - strategy.should_force_compact(&state), - Some(CompactionInitiator::CapabilityResultOverflow) - ); - } - - #[test] - fn byte_cap_strategy_empty_accumulator_returns_none() { - let context = crate::test_support::test_run_context("byte-cap-policy-empty"); - let state = LoopExecutionState::initial_for_run(&context); - // pending_capability_bytes is empty by default. - let strategy = ByteCapStrategy::with_defaults(); - assert_eq!(strategy.should_force_compact(&state), None); - } - - #[test] - fn byte_cap_strategy_with_cap_overrides_default_cap() { - let ctx = crate::test_support::test_run_context("byte-cap-with-cap"); - let mut state = LoopExecutionState::initial_for_run(&ctx); - let id = CapabilityId::new("custom.large_tool").unwrap(); - state - .post_capability_state - .pending_capability_bytes - .insert(id.clone(), 5_000); - // Default cap (32_000) would NOT trip at 5_000; custom cap of 4_000 should trip. - let strategy = ByteCapStrategy::with_defaults().with_cap(id, 4_000); - assert_eq!( - strategy.should_force_compact(&state), - Some(CompactionInitiator::CapabilityResultOverflow) - ); - } -} +mod tests; diff --git a/crates/loop/ironclaw_agent_loop/src/strategies/compaction/tests.rs b/crates/loop/ironclaw_agent_loop/src/strategies/compaction/tests.rs new file mode 100644 index 00000000000..41fe66ccf4a --- /dev/null +++ b/crates/loop/ironclaw_agent_loop/src/strategies/compaction/tests.rs @@ -0,0 +1,844 @@ +use super::*; +use crate::state::{ + CompactionEffectivenessBaseline, CompactionPromptSnapshot, CompactionStrategyState, + DeferredCompactionWatermark, LoopExecutionState, MessageIndexEntry, +}; +use ironclaw_host_api::ids::CapabilityId; +use ironclaw_loop_contracts::PromptContextTokenBudget; + +#[test] +fn evaluate_skips_when_message_index_is_empty() { + let context = crate::test_support::test_run_context("compaction-strategy-empty"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_state.force_compact_on_next_iteration = true; + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), + preserve_tail_tokens: 1, + deadline_ms: 1, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Skip + ); +} + +#[test] +fn evaluate_skips_when_no_eligible_user_message_boundary_exists() { + let context = crate::test_support::test_run_context("compaction-strategy"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_prompt = + CompactionPromptSnapshot::from_message_index(vec![MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 100, + }]); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), + preserve_tail_tokens: 1, + deadline_ms: 1, + }; + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Skip + ); +} + +#[test] +fn evaluate_skips_when_below_threshold_with_valid_user_boundary_and_forcing_is_off() { + let context = crate::test_support::test_run_context("compaction-strategy-below-threshold"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 20, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 20, + }, + ]); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), + preserve_tail_tokens: 60, + deadline_ms: 1, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Skip + ); +} + +#[test] +fn can_evaluate_skips_when_the_budget_leaves_no_visible_transcript() { + // Old assertion (`can_evaluate_skips_when_visible_threshold_equals_preserve_tail`) + // pinned the defect: it relied on the raw `preserve_tail_tokens` (90) + // being >= `threshold` (90) to trip the guard, which is exactly the + // condition the small-advertised-window bug hit (a run-resolved + // threshold now legitimately falls below a compiled-in tail). With the + // tail clamped to half the visible transcript, that guard no longer + // trips here (threshold 90 > clamped tail 45), and forcing would no + // longer be suppressed — which is correct: the guard's real purpose is + // "there is no visible transcript to compact," not "the tail happens to + // be large." This rewrite asserts that real purpose directly with a + // budget whose `visible_transcript_tokens()` is 0, and proves it still + // holds even for the forced/recovery path. + let context = + crate::test_support::test_run_context("compaction-strategy-no-visible-transcript"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_state.force_compact_on_next_iteration = true; + state.compaction_prompt = + CompactionPromptSnapshot::from_message_index(vec![MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 100, + }]); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(10, 10, 0), + preserve_tail_tokens: 90, + deadline_ms: 1, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Skip + ); +} + +#[test] +fn evaluate_triggers_at_latest_user_boundary_outside_tail() { + let context = crate::test_support::test_run_context("compaction-strategy-trigger"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_state = CompactionStrategyState::default(); + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 30, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 30, + }, + MessageIndexEntry { + sequence: 3, + kind: IndexedMessageKind::User, + estimated_tokens: 30, + }, + MessageIndexEntry { + sequence: 4, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 30, + }, + ]); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), + preserve_tail_tokens: 60, + deadline_ms: 7, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Trigger { + drop_through_seq: 1, + // Clamped from the configured 60: visible (90) / 2 = 45. The + // walk-back to sequence 1 is unaffected by the clamp here (it + // still needs to cross two message blocks either way), so this + // is the genuine clamp output, not an incidental input choice. + preserve_tail_tokens: 45, + deadline_ms: 7, + effectiveness_baseline: CompactionEffectivenessBaseline::TriggerThresholdTokens { + tokens: 90, + }, + } + ); +} + +#[test] +fn evaluate_triggers_when_newest_assistant_block_exceeds_tail_budget() { + let context = crate::test_support::test_run_context("compaction-strategy-tail-overflow"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_state = CompactionStrategyState::default(); + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 100, + }, + ]); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), + preserve_tail_tokens: 60, + deadline_ms: 7, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Trigger { + drop_through_seq: 1, + // Clamped from the configured 60 to visible (90) / 2 = 45; the + // boundary walk-back is unaffected (genuine clamp output). + preserve_tail_tokens: 45, + deadline_ms: 7, + effectiveness_baseline: CompactionEffectivenessBaseline::TriggerThresholdTokens { + tokens: 90, + }, + } + ); +} + +#[test] +fn evaluate_skips_when_latest_user_boundary_was_already_compacted() { + let context = crate::test_support::test_run_context("compaction-strategy-compacted"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_state.last_compacted_through_seq = Some(3); + state.compaction_state.force_compact_on_next_iteration = true; + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 3, + kind: IndexedMessageKind::User, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 4, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 100, + }, + ]); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), + preserve_tail_tokens: 60, + deadline_ms: 7, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Skip + ); +} + +#[test] +fn evaluate_skips_previously_deferred_boundary_when_forced() { + let context = crate::test_support::test_run_context("compaction-strategy-deferred"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_state.force_compact_on_next_iteration = true; + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 3, + kind: IndexedMessageKind::User, + estimated_tokens: 10, + }, + ]); + state.compaction_state.last_deferred = Some(DeferredCompactionWatermark { + through_seq: 3, + prompt_fingerprint: state.compaction_prompt.fingerprint(), + }); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), + preserve_tail_tokens: 1, + deadline_ms: 7, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Trigger { + drop_through_seq: 1, + preserve_tail_tokens: 1, + deadline_ms: 7, + effectiveness_baseline: CompactionEffectivenessBaseline::PreCompactionPromptTokens { + tokens: 30 + }, + } + ); +} + +#[test] +fn evaluate_skips_deferred_boundary_in_threshold_overflow_path() { + let context = crate::test_support::test_run_context("compaction-strategy-deferred-threshold"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 50, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 50, + }, + MessageIndexEntry { + sequence: 3, + kind: IndexedMessageKind::User, + estimated_tokens: 50, + }, + MessageIndexEntry { + sequence: 4, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 50, + }, + ]); + state.compaction_state.last_deferred = Some(DeferredCompactionWatermark { + through_seq: 3, + prompt_fingerprint: state.compaction_prompt.fingerprint(), + }); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), + preserve_tail_tokens: 60, + deadline_ms: 7, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Trigger { + drop_through_seq: 1, + // Clamped from the configured 60 to visible (90) / 2 = 45; the + // boundary walk-back (and the deferred-boundary rejection at + // sequence 3) is unaffected (genuine clamp output). + preserve_tail_tokens: 45, + deadline_ms: 7, + effectiveness_baseline: CompactionEffectivenessBaseline::TriggerThresholdTokens { + tokens: 90, + }, + } + ); +} + +#[test] +fn evaluate_skips_when_only_deferred_boundary_is_eligible_in_threshold_overflow_path() { + let context = crate::test_support::test_run_context("compaction-strategy-deferred-skip"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 50, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 50, + }, + ]); + state.compaction_state.last_deferred = Some(DeferredCompactionWatermark { + through_seq: 1, + prompt_fingerprint: state.compaction_prompt.fingerprint(), + }); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), + preserve_tail_tokens: 60, + deadline_ms: 7, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Skip + ); +} + +#[test] +fn evaluate_retries_deferred_boundary_after_prompt_snapshot_changes() { + let context = crate::test_support::test_run_context("compaction-strategy-deferred-changed"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_state.last_deferred = Some(DeferredCompactionWatermark { + through_seq: 3, + prompt_fingerprint: 42, + }); + state.compaction_state.force_compact_on_next_iteration = true; + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 3, + kind: IndexedMessageKind::User, + estimated_tokens: 10, + }, + ]); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), + preserve_tail_tokens: 1, + deadline_ms: 7, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Trigger { + drop_through_seq: 3, + preserve_tail_tokens: 1, + deadline_ms: 7, + effectiveness_baseline: CompactionEffectivenessBaseline::PreCompactionPromptTokens { + tokens: 30 + }, + } + ); +} + +#[test] +fn evaluate_retries_after_transcript_advances_past_deferred_boundary() { + let context = crate::test_support::test_run_context("compaction-strategy-deferred-newer"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_state.force_compact_on_next_iteration = true; + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 3, + kind: IndexedMessageKind::User, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 4, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 5, + kind: IndexedMessageKind::User, + estimated_tokens: 10, + }, + ]); + state.compaction_state.last_deferred = Some(DeferredCompactionWatermark { + through_seq: 3, + prompt_fingerprint: state.compaction_prompt.fingerprint(), + }); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), + preserve_tail_tokens: 1, + deadline_ms: 7, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Trigger { + drop_through_seq: 5, + preserve_tail_tokens: 1, + deadline_ms: 7, + effectiveness_baseline: CompactionEffectivenessBaseline::PreCompactionPromptTokens { + tokens: 50 + }, + } + ); +} + +#[test] +fn evaluate_uses_output_budget_when_larger_than_reserve() { + let context = crate::test_support::test_run_context("compaction-strategy-output-budget"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 40, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 35, + }, + ]); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 30), + preserve_tail_tokens: 1, + deadline_ms: 7, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Trigger { + drop_through_seq: 1, + preserve_tail_tokens: 1, + deadline_ms: 7, + effectiveness_baseline: CompactionEffectivenessBaseline::TriggerThresholdTokens { + tokens: 70, + }, + } + ); +} + +#[test] +fn tail_preserving_user_boundary_respects_minimum_tail_message_count() { + let context = crate::test_support::test_run_context("compaction-strategy-min-tail"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 3, + kind: IndexedMessageKind::User, + estimated_tokens: 10, + }, + MessageIndexEntry { + sequence: 4, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 10, + }, + ]); + + let boundary = + tail_preserving_user_boundary(&state, state.compaction_prompt.fingerprint(), 1, 2, |_| { + true + }); + + assert_eq!(boundary, Some(1)); +} + +#[test] +fn evaluate_skips_threshold_trigger_when_circuit_is_open() { + let context = crate::test_support::test_run_context("compaction-strategy-circuit-open"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_state.compaction_circuit_open = true; + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 100, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 100, + }, + ]); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), + preserve_tail_tokens: 1, + deadline_ms: 7, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Skip + ); +} + +#[test] +fn evaluate_triggers_forced_compaction_even_when_circuit_is_open() { + // BUG B1 regression: force_compact_on_next_iteration is how + // context-overflow recovery and byte-cap overflow request a shrink. + // An open breaker must not suppress it — only automatic + // threshold-triggered compaction is gated. + let context = crate::test_support::test_run_context("compaction-strategy-circuit-open-forced"); + let mut state = LoopExecutionState::initial_for_run(&context); + state.compaction_state.compaction_circuit_open = true; + state.compaction_state.force_compact_on_next_iteration = true; + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: 100, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: 100, + }, + ]); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(100, 10, 0), + preserve_tail_tokens: 1, + deadline_ms: 7, + }; + + assert_eq!( + strategy.should_compact(&state, &context), + CompactionDecision::Trigger { + drop_through_seq: 1, + preserve_tail_tokens: 1, + deadline_ms: 7, + effectiveness_baseline: CompactionEffectivenessBaseline::PreCompactionPromptTokens { + tokens: 200 + }, + } + ); +} + +// --- ByteCapStrategy tests --- + +#[test] +fn byte_cap_strategy_trips_when_capability_exceeds_cap() { + let context = crate::test_support::test_run_context("byte-cap-policy-trips"); + let mut state = LoopExecutionState::initial_for_run(&context); + let id = CapabilityId::new("builtin.http").expect("valid capability"); + // 32_000 is the cap; 32_001 exceeds it. + state + .post_capability_state + .pending_capability_bytes + .insert(id, 32_001); + + let strategy = ByteCapStrategy::with_defaults(); + assert_eq!( + strategy.should_force_compact(&state), + Some(CompactionInitiator::CapabilityResultOverflow) + ); +} + +#[test] +fn byte_cap_strategy_skips_when_under_threshold() { + let context = crate::test_support::test_run_context("byte-cap-policy-under"); + let mut state = LoopExecutionState::initial_for_run(&context); + let http_id = CapabilityId::new("builtin.http").expect("valid capability"); + let subagent_id = CapabilityId::new("builtin.spawn_subagent").expect("valid capability"); + // Both under their respective caps. + state + .post_capability_state + .pending_capability_bytes + .insert(http_id, 31_999); + state + .post_capability_state + .pending_capability_bytes + .insert(subagent_id, 47_999); + + let strategy = ByteCapStrategy::with_defaults(); + assert_eq!(strategy.should_force_compact(&state), None); +} + +#[test] +fn byte_cap_strategy_uses_default_cap_for_unknown_capability() { + let context = crate::test_support::test_run_context("byte-cap-policy-unknown"); + let mut state = LoopExecutionState::initial_for_run(&context); + let id = CapabilityId::new("custom.unknown_tool").expect("valid capability"); + // DEFAULT_FALLBACK_CAP_BYTES is 32_000; 32_001 exceeds it. + state + .post_capability_state + .pending_capability_bytes + .insert(id, ByteCapStrategy::DEFAULT_FALLBACK_CAP_BYTES + 1); + + let strategy = ByteCapStrategy::with_defaults(); + assert_eq!( + strategy.should_force_compact(&state), + Some(CompactionInitiator::CapabilityResultOverflow) + ); +} + +#[test] +fn byte_cap_strategy_empty_accumulator_returns_none() { + let context = crate::test_support::test_run_context("byte-cap-policy-empty"); + let state = LoopExecutionState::initial_for_run(&context); + // pending_capability_bytes is empty by default. + let strategy = ByteCapStrategy::with_defaults(); + assert_eq!(strategy.should_force_compact(&state), None); +} + +#[test] +fn byte_cap_strategy_with_cap_overrides_default_cap() { + let ctx = crate::test_support::test_run_context("byte-cap-with-cap"); + let mut state = LoopExecutionState::initial_for_run(&ctx); + let id = CapabilityId::new("custom.large_tool").unwrap(); + state + .post_capability_state + .pending_capability_bytes + .insert(id.clone(), 5_000); + // Default cap (32_000) would NOT trip at 5_000; custom cap of 4_000 should trip. + let strategy = ByteCapStrategy::with_defaults().with_cap(id, 4_000); + assert_eq!( + strategy.should_force_compact(&state), + Some(CompactionInitiator::CapabilityResultOverflow) + ); +} + +/// Four alternating messages summing to `tokens`, so the index both +/// trips the threshold and offers a compactable user boundary outside +/// the preserved tail. A single entry trips the threshold but can never +/// Trigger — there is no boundary to drop through. +fn state_with_observed_prompt_tokens(tokens: u64, context: &LoopRunContext) -> LoopExecutionState { + let each = tokens / 4; + let mut state = LoopExecutionState::initial_for_run(context); + state.compaction_state = CompactionStrategyState::default(); + state.compaction_prompt = CompactionPromptSnapshot::from_message_index(vec![ + MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: each, + }, + MessageIndexEntry { + sequence: 2, + kind: IndexedMessageKind::Assistant, + estimated_tokens: each, + }, + MessageIndexEntry { + sequence: 3, + kind: IndexedMessageKind::User, + estimated_tokens: each, + }, + MessageIndexEntry { + sequence: 4, + kind: IndexedMessageKind::Assistant, + estimated_tokens: each, + }, + ]); + state +} + +#[test] +fn run_context_budget_overrides_the_strategy_default_for_compaction() { + // The strategy's own budget would not trigger, but the run's model + // has a far smaller real window, so this prompt is already over. + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(128_000, 20_000, 0), + preserve_tail_tokens: 10, + deadline_ms: 30_000, + }; + let ctx = crate::test_support::test_run_context("compaction-budget-override") + .with_resolved_context_budget(PromptContextTokenBudget::new(40_000, 5_000, 0)); + let state = state_with_observed_prompt_tokens(50_000, &ctx); + + assert!(matches!( + strategy.should_compact(&state, &ctx), + CompactionDecision::Trigger { .. } + )); +} + +#[test] +fn absent_run_context_budget_falls_back_to_the_strategy_default() { + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(128_000, 20_000, 0), + preserve_tail_tokens: 10, + deadline_ms: 30_000, + }; + let ctx = crate::test_support::test_run_context("compaction-budget-fallback"); + let state = state_with_observed_prompt_tokens(50_000, &ctx); + + assert_eq!( + strategy.should_compact(&state, &ctx), + CompactionDecision::Skip + ); +} + +// --- Small-advertised-window tail clamp regression (context-length bug) --- +// +// `preserve_tail_tokens` is compiled-in at 8,000, but `threshold` is now +// derived from the run's model-advertised window. An 8k-window model derives +// visible_transcript_tokens() == 5,400 (< 8,000), so both the automatic and +// forced-recovery paths must clamp the tail to half the visible transcript +// instead of disabling compaction outright. + +#[test] +fn small_advertised_window_still_allows_forced_compaction() { + let ctx = crate::test_support::test_run_context("compaction-small-window-forced") + .with_resolved_context_budget(PromptContextTokenBudget::from_advertised_window(Some( + 8_000, + ))); + let mut state = state_with_observed_prompt_tokens(80, &ctx); + state.compaction_state.force_compact_on_next_iteration = true; + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::default(), + preserve_tail_tokens: DefaultCompactionStrategy::DEFAULT_PRESERVE_TAIL_TOKENS, + deadline_ms: 30_000, + }; + + assert_eq!( + strategy.should_compact(&state, &ctx), + CompactionDecision::Trigger { + // Forced/recovery compaction uses `latest_eligible_user_boundary`, + // which is not tail-aware — it always cuts at the most recent + // eligible user message (sequence 3 here), not the earliest. + drop_through_seq: 3, + preserve_tail_tokens: 2_700, + deadline_ms: 30_000, + effectiveness_baseline: CompactionEffectivenessBaseline::PreCompactionPromptTokens { + tokens: 80, + }, + } + ); +} + +#[test] +fn small_advertised_window_still_triggers_automatic_compaction() { + let ctx = crate::test_support::test_run_context("compaction-small-window-automatic") + .with_resolved_context_budget(PromptContextTokenBudget::from_advertised_window(Some( + 8_000, + ))); + let state = state_with_observed_prompt_tokens(5_400, &ctx); + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::default(), + preserve_tail_tokens: DefaultCompactionStrategy::DEFAULT_PRESERVE_TAIL_TOKENS, + deadline_ms: 30_000, + }; + + assert_eq!( + strategy.should_compact(&state, &ctx), + CompactionDecision::Trigger { + drop_through_seq: 1, + preserve_tail_tokens: 2_700, + deadline_ms: 30_000, + effectiveness_baseline: CompactionEffectivenessBaseline::TriggerThresholdTokens { + tokens: 5_400, + }, + } + ); +} + +#[test] +fn default_budget_keeps_the_full_configured_tail() { + let ctx = crate::test_support::test_run_context("compaction-default-budget-tail"); + let state = state_with_observed_prompt_tokens(120_000, &ctx); + let strategy = DefaultCompactionStrategy::default(); + + assert_eq!( + strategy.should_compact(&state, &ctx), + CompactionDecision::Trigger { + // Each 30,000-token message block alone exceeds the 8,000-token + // tail, so the walk-back returns at the first eligible user + // boundary it finds (sequence 3), not the earliest message. + drop_through_seq: 3, + preserve_tail_tokens: 8_000, + deadline_ms: DefaultCompactionStrategy::DEFAULT_DEADLINE_MS, + effectiveness_baseline: CompactionEffectivenessBaseline::TriggerThresholdTokens { + tokens: 108_000, + }, + } + ); +} diff --git a/crates/loop/ironclaw_loop_host/src/lib.rs b/crates/loop/ironclaw_loop_host/src/lib.rs index 183c59a41f5..deab49d1f52 100644 --- a/crates/loop/ironclaw_loop_host/src/lib.rs +++ b/crates/loop/ironclaw_loop_host/src/lib.rs @@ -2306,6 +2306,26 @@ pub trait HostManagedModelGateway: Send + Sync { resolved_model_route.and_then(|route| ProviderModelId::new(route.model_id()).ok()) } + /// Best-effort provider-advertised total context window (input and + /// output), in tokens, for the route this run will use. + /// + /// Gateways that own provider selection should override this. The default + /// returns `None`, which keeps the compiled-in budget — a gateway that + /// knows nothing must not change how any run is budgeted. + /// + /// This is awaited on the turn-run host-build critical path, so an + /// implementation must be cheap and I/O-free — `ironclaw_llm`'s + /// `LlmProvider::model_metadata()` contract makes it a static description + /// for exactly this reason. A slow override stalls every run's + /// construction; return `None` rather than doing work. + async fn advertised_context_window_tokens( + &self, + _model_profile_id: &ModelProfileId, + _resolved_model_route: Option<&HostManagedModelRouteSnapshot>, + ) -> Option { + None + } + async fn stream_model( &self, request: HostManagedModelRequest, diff --git a/crates/loop/ironclaw_loop_host/src/model_gateway.rs b/crates/loop/ironclaw_loop_host/src/model_gateway.rs index 1367c114cff..54a59ab250b 100644 --- a/crates/loop/ironclaw_loop_host/src/model_gateway.rs +++ b/crates/loop/ironclaw_loop_host/src/model_gateway.rs @@ -425,6 +425,32 @@ where .and_then(|route| ProviderModelId::new(route.model).ok()) } + async fn advertised_context_window_tokens( + &self, + model_profile_id: &ModelProfileId, + resolved_model_route: Option<&HostManagedModelRouteSnapshot>, + ) -> Option { + // Advisory only: a provider that cannot report a window for the model + // this run will actually be served must leave the run on the + // compiled-in budget, never fail the run. + let metadata = self.provider.model_metadata().await.ok()?; // silent-ok: advisory context-window probe, run proceeds on compiled-in budget + // `model_metadata()` takes no model argument -- it describes whatever + // model the provider was configured with. The served model is resolved + // per request (a route override wins, and providers that honor + // per-request overrides serve it), so a window borrowed from a + // different model would be worse than none: guessing high produces the + // provider rejection this mechanism exists to avoid. + let route = self.policy.route_for(model_profile_id)?; + let served = request_model_override( + route, + self.provider.as_ref(), + resolved_model_route.map(HostManagedModelRouteSnapshot::model_id), + ) + .ok()?; + (served == metadata.id).then_some(())?; + metadata.context_length.map(u64::from) + } + async fn stream_model( &self, request: HostManagedModelRequest, @@ -3701,4 +3727,189 @@ mod tests { "repaired assistant message must preserve typed reasoning_details" ); } + + // Bespoke double per test, matching this module's convention + // (`StopSequenceRecordingProvider` above): implement only the four + // required `LlmProvider` methods, plus the `model_metadata` override the + // advertised-window tests actually need. + struct WindowReportingProvider { + model_id: String, + context_length: Option, + // Defaults to `false` at every existing call site (Rust has no + // `..Default::default()` shorthand for a plain struct literal, so + // each constructs it explicitly); only + // `gateway_reports_none_when_model_metadata_fails` sets it. + metadata_error: bool, + } + + #[async_trait] + impl LlmProvider for WindowReportingProvider { + fn model_name(&self) -> &str { + &self.model_id + } + + fn cost_per_token(&self) -> (rust_decimal::Decimal, rust_decimal::Decimal) { + Default::default() + } + + async fn model_metadata(&self) -> Result { + if self.metadata_error { + return Err(LlmError::ModelNotAvailable { + provider: "window-reporting-test-provider".to_string(), + model: self.model_id.clone(), + }); + } + Ok(ironclaw_llm::ModelMetadata { + id: self.model_id.clone(), + context_length: self.context_length, + }) + } + + async fn complete( + &self, + _request: CompletionRequest, + ) -> Result { + unreachable!("the advertised-window tests never dispatch a completion") + } + + async fn complete_with_tools( + &self, + _request: ToolCompletionRequest, + ) -> Result { + unreachable!("the advertised-window tests have no tool surface") + } + } + + fn window_test_profile_id() -> ModelProfileId { + ModelProfileId::new("interactive_model").expect("valid profile id") + } + + #[tokio::test] + async fn gateway_reports_the_providers_advertised_context_window() { + let provider = Arc::new(WindowReportingProvider { + model_id: "base-model".to_string(), + context_length: Some(200_000), + metadata_error: false, + }); + let policy = + LlmModelProfilePolicy::new().allow_model_profile(window_test_profile_id(), None); + let gateway = LlmProviderModelGateway::new(provider, policy); + + let window = gateway + .advertised_context_window_tokens(&window_test_profile_id(), None) + .await; + + assert_eq!(window, Some(200_000)); + } + + #[tokio::test] + async fn gateway_reports_none_when_the_provider_advertises_nothing() { + let provider = Arc::new(WindowReportingProvider { + model_id: "base-model".to_string(), + context_length: None, + metadata_error: false, + }); + let policy = + LlmModelProfilePolicy::new().allow_model_profile(window_test_profile_id(), None); + let gateway = LlmProviderModelGateway::new(provider, policy); + + let window = gateway + .advertised_context_window_tokens(&window_test_profile_id(), None) + .await; + + assert_eq!(window, None); + } + + #[tokio::test] + async fn gateway_reports_none_when_the_route_overrides_to_another_model() { + // The provider describes "base-model" but the policy routes this + // profile to a different one. Budgeting from the base model's window + // could hand a small model a budget sized for a large one. + let provider = Arc::new(WindowReportingProvider { + model_id: "base-model".to_string(), + context_length: Some(200_000), + metadata_error: false, + }); + let policy = LlmModelProfilePolicy::new() + .allow_model_profile(window_test_profile_id(), Some("other-model".to_string())); + let gateway = LlmProviderModelGateway::new(provider, policy); + + let window = gateway + .advertised_context_window_tokens(&window_test_profile_id(), None) + .await; + + assert_eq!( + window, None, + "a window for a different model must not be trusted" + ); + } + + #[tokio::test] + async fn gateway_reports_none_when_the_resolved_route_snapshot_names_another_model() { + // The policy route for this profile has no override, so the served + // model would be the provider's base model ("base-model") -- but the + // resolved route snapshot (the run's actual bound route) names a + // different model. The snapshot identity must win the same way the + // route-override case above does. + let provider = Arc::new(WindowReportingProvider { + model_id: "base-model".to_string(), + context_length: Some(200_000), + metadata_error: false, + }); + let policy = + LlmModelProfilePolicy::new().allow_model_profile(window_test_profile_id(), None); + let gateway = LlmProviderModelGateway::new(provider, policy); + let snapshot = HostManagedModelRouteSnapshot::advisory("other-model") + .expect("valid advisory model id"); + + let window = gateway + .advertised_context_window_tokens(&window_test_profile_id(), Some(&snapshot)) + .await; + + assert_eq!( + window, None, + "a window for a model other than the resolved route snapshot must not be trusted" + ); + } + + #[tokio::test] + async fn gateway_reports_the_window_when_the_resolved_route_snapshot_names_the_served_model() { + let provider = Arc::new(WindowReportingProvider { + model_id: "base-model".to_string(), + context_length: Some(200_000), + metadata_error: false, + }); + let policy = + LlmModelProfilePolicy::new().allow_model_profile(window_test_profile_id(), None); + let gateway = LlmProviderModelGateway::new(provider, policy); + let snapshot = + HostManagedModelRouteSnapshot::advisory("base-model").expect("valid advisory model id"); + + let window = gateway + .advertised_context_window_tokens(&window_test_profile_id(), Some(&snapshot)) + .await; + + assert_eq!(window, Some(200_000)); + } + + #[tokio::test] + async fn gateway_reports_none_when_model_metadata_fails() { + let provider = Arc::new(WindowReportingProvider { + model_id: "base-model".to_string(), + context_length: Some(200_000), + metadata_error: true, + }); + let policy = + LlmModelProfilePolicy::new().allow_model_profile(window_test_profile_id(), None); + let gateway = LlmProviderModelGateway::new(provider, policy); + + let window = gateway + .advertised_context_window_tokens(&window_test_profile_id(), None) + .await; + + assert_eq!( + window, None, + "a failed model_metadata() call must fall back to the compiled-in budget, not fail the run" + ); + } } diff --git a/crates/loop/ironclaw_loop_host/src/thread_resolving_model_gateway.rs b/crates/loop/ironclaw_loop_host/src/thread_resolving_model_gateway.rs index 62d6de2f02a..22a635fdbe0 100644 --- a/crates/loop/ironclaw_loop_host/src/thread_resolving_model_gateway.rs +++ b/crates/loop/ironclaw_loop_host/src/thread_resolving_model_gateway.rs @@ -9,7 +9,7 @@ use async_trait::async_trait; use ironclaw_loop_contracts::{ InstructionMaterializationStore, LoopCapabilityPort, LoopModelGateway, LoopModelGatewayError, LoopModelGatewayRequest, LoopModelPort, LoopModelProgressSink, LoopModelResponse, - LoopPromptBundleAuthority, + LoopPromptBundleAuthority, PromptContextTokenBudget, }; use ironclaw_threads::{SessionThreadService, ThreadScope}; @@ -41,6 +41,7 @@ where pub context_window_cache: Option>, pub attachment_read_port: Option>, pub prompt_diagnostic_sink: Option>, + pub prompt_context_budget: PromptContextTokenBudget, } /// Resolves a thread's transcript into a host-managed model request. @@ -68,6 +69,7 @@ where context_window_cache: Option>, attachment_read_port: Option>, prompt_diagnostic_sink: Option>, + prompt_context_budget: PromptContextTokenBudget, } impl ThreadResolvingLoopModelGateway @@ -89,6 +91,7 @@ where context_window_cache, attachment_read_port, prompt_diagnostic_sink, + prompt_context_budget, } = parts; Self { thread_service, @@ -103,6 +106,7 @@ where context_window_cache, attachment_read_port, prompt_diagnostic_sink, + prompt_context_budget, } } } @@ -146,7 +150,8 @@ where Arc::clone(&self.host_gateway), self.max_messages, ) - .with_prompt_bundle_authority(self.prompt_authority.clone()); + .with_prompt_bundle_authority(self.prompt_authority.clone()) + .with_prompt_context_token_budget(self.prompt_context_budget); if let Some(source) = self.skill_context_source.as_ref() { model_port = model_port.with_skill_context_source(source.clone()); } diff --git a/crates/loop/ironclaw_loop_host/tests/llm_gateway.rs b/crates/loop/ironclaw_loop_host/tests/llm_gateway.rs index d5841f4dc00..d691f80e58a 100644 --- a/crates/loop/ironclaw_loop_host/tests/llm_gateway.rs +++ b/crates/loop/ironclaw_loop_host/tests/llm_gateway.rs @@ -12,7 +12,8 @@ use ironclaw_host_api::ids::{ }; use ironclaw_llm::{ CompletionRequest, CompletionResponse, CompletionStreamSink, FailoverProvider, FinishReason, - LlmError, LlmProvider, Role, ToolCall, ToolCompletionRequest, ToolCompletionResponse, + LlmError, LlmProvider, OpenAiCodexConfig, OpenAiCodexProvider, OpenAiCodexSessionManager, Role, + TokenRefreshingProvider, ToolCall, ToolCompletionRequest, ToolCompletionResponse, }; use ironclaw_loop_contracts::{ AgentLoopHostError, AgentLoopHostErrorKind, AgentLoopHostErrorReasonKind, @@ -5688,4 +5689,125 @@ impl LlmProvider for RecordingLlmProvider { }) } } + +// Regression coverage for the gateway-level caller, not just the +// `TokenRefreshingProvider::model_metadata()` leaf covered by +// `ironclaw_llm::token_refreshing::model_metadata_does_not_refresh_the_token`. +// The production caller is `advertised_context_window_tokens` (turn-run host +// construction); a future decorator/gateway wiring change could reintroduce +// the HTTP call while the leaf test alone stays green. + +/// Minimal local equivalent of `ironclaw_llm`'s private +/// `codex_test_helpers::make_test_jwt` (`pub(crate)`, not reachable from this +/// crate): an unsigned JWT-shaped string with the same three-segment +/// header.payload.signature layout. +fn probe_test_jwt(account_id: &str) -> String { + use base64::Engine; + let engine = base64::engine::general_purpose::URL_SAFE_NO_PAD; + let header = engine.encode(b"{\"alg\":\"RS256\",\"typ\":\"JWT\"}"); + let payload_json = serde_json::json!({ + "sub": "user123", + "https://api.openai.com/auth": { + "chatgpt_account_id": account_id, + }, + }); + let payload = engine.encode(payload_json.to_string().as_bytes()); + let sig = engine.encode(b"fake-signature"); + format!("{header}.{payload}.{sig}") +} + +/// Local equivalent of `ironclaw_llm::token_refreshing::tests::spawn_counting_auth_endpoint` +/// (private to that crate): counts connection attempts against a local +/// listener so the test can observe whether a token refresh was ever +/// attempted, without a mock-HTTP dependency. +fn probe_spawn_counting_auth_endpoint() -> (String, Arc) { + let listener = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + listener.set_nonblocking(true).unwrap(); + let listener = tokio::net::TcpListener::from_std(listener).unwrap(); + let addr = listener.local_addr().unwrap(); + let connections = Arc::new(AtomicUsize::new(0)); + let connections_task = connections.clone(); + tokio::spawn(async move { + loop { + let Ok((mut stream, _)) = listener.accept().await else { + break; + }; + connections_task.fetch_add(1, Ordering::SeqCst); + use tokio::io::AsyncWriteExt; + let _ = stream + .write_all(b"HTTP/1.1 400 Bad Request\r\ncontent-length: 0\r\n\r\n") + .await; + } + }); + (format!("http://{addr}"), connections) +} + +#[tokio::test] +async fn advertised_context_window_probe_does_not_refresh_the_token() { + let dir = tempfile::tempdir().unwrap(); + let session_path = dir.path().join("session.json"); + let (auth_endpoint, connections) = probe_spawn_counting_auth_endpoint(); + + // A session that needs a refresh: expired, with a non-empty refresh + // token so `refresh_tokens()` would actually attempt the HTTP call rather + // than short-circuiting on a missing refresh token. Written directly to + // disk (rather than via a private session-mutation helper) because + // `OpenAiCodexSessionManager::new` loads synchronously from + // `config.session_path` during construction — see the read at + // `openai_codex_session.rs`'s `new()`: + // `if let Ok(data) = std::fs::read_to_string(&mgr.config.session_path) + // && let Ok(session) = serde_json::from_str::(&data)` + // — and `OpenAiCodexSession`'s fields are `pub(crate)` to `ironclaw_llm`, + // so this crate cannot construct one directly; it can write the same + // shape (`#[derive(Serialize, Deserialize)]`, plain field names, chrono's + // default RFC3339 date serialization) as JSON instead. + let session_json = serde_json::json!({ + "access_token": probe_test_jwt("acct_test"), + "refresh_token": "refresh-token", + "expires_at": (chrono::Utc::now() - chrono::Duration::hours(1)).to_rfc3339(), + "created_at": (chrono::Utc::now() - chrono::Duration::hours(2)).to_rfc3339(), + }); + std::fs::write(&session_path, session_json.to_string()).unwrap(); + + let config = OpenAiCodexConfig { + model: "gpt-5.3-codex".to_string(), + auth_endpoint, + api_base_url: "https://chatgpt.com/backend-api/codex".to_string(), + client_id: "test_client_id".to_string(), + session_path, + token_refresh_margin_secs: 300, + }; + + let jwt = probe_test_jwt("acct_test"); + let inner = Arc::new( + OpenAiCodexProvider::new(&config.model, &config.api_base_url, &jwt, 300) + .expect("provider creation should succeed"), + ); + let session = Arc::new( + OpenAiCodexSessionManager::new(config).expect("session manager creation should succeed"), + ); + assert!( + session.needs_refresh().await, + "session loaded from disk must report it needs a refresh" + ); + + let provider = Arc::new(TokenRefreshingProvider::new(inner.clone(), session)); + let policy = LlmModelProfilePolicy::new().allow_model_profile(interactive_model(), None); + let gateway = LlmProviderModelGateway::new(provider, policy); + + let window = gateway + .advertised_context_window_tokens(&interactive_model(), None) + .await; + + assert_eq!( + connections.load(Ordering::SeqCst), + 0, + "advertised_context_window_tokens() must not perform a token refresh" + ); + assert_eq!( + window, None, + "the codex provider reports context_length: None" + ); +} + // arch-exempt: large_file, LLM gateway contract coverage remains centralized, plan #6175 diff --git a/crates/loop/ironclaw_loop_host/tests/thread_loop_host_contract.rs b/crates/loop/ironclaw_loop_host/tests/thread_loop_host_contract.rs index 57873fd7199..55ee90b1a03 100644 --- a/crates/loop/ironclaw_loop_host/tests/thread_loop_host_contract.rs +++ b/crates/loop/ironclaw_loop_host/tests/thread_loop_host_contract.rs @@ -26,15 +26,15 @@ use ironclaw_loop_contracts::{ LoopContextBundle, LoopContextCompactionKind, LoopContextMessage, LoopContextPort, LoopContextRequest, LoopContextSnippet, LoopDriverNoteKind, LoopHostMilestoneKind, LoopHostMilestoneSink, LoopInputCursor, LoopInputCursorToken, LoopModelCapabilityView, - LoopModelMessage, LoopModelPort, LoopModelRequest, LoopModelRouteSnapshot, LoopModelUsage, - LoopPromptBundle, LoopPromptBundleAuthority, LoopPromptBundleRef, LoopPromptBundleRequest, - LoopPromptPort, LoopRequest, LoopRequestBatch, LoopRunContext, LoopTranscriptPort, - ModelProfileId, ModelVisibleToolObservation, ObservationTrust, ParentLoopOutput, - PersonalContextPolicy, PromptMode, PromptSkillContextMetadata, ProviderToolCallReference, - ProviderToolCallReplay, ProviderToolDefinition, RunProfileResolutionRequest, - RunProfileResolver, SkillName, SkillTrustLevel, SkillVisibility, ToolObservationDetail, - ToolObservationStatus, UpdateAssistantDraft, VisibleCapabilityRequest, - VisibleCapabilitySurface, resolution, + LoopModelGateway, LoopModelGatewayRequest, LoopModelMessage, LoopModelPort, LoopModelRequest, + LoopModelRouteSnapshot, LoopModelUsage, LoopPromptBundle, LoopPromptBundleAuthority, + LoopPromptBundleRef, LoopPromptBundleRequest, LoopPromptPort, LoopRequest, LoopRequestBatch, + LoopRunContext, LoopTranscriptPort, ModelProfileId, ModelVisibleToolObservation, + ObservationTrust, ParentLoopOutput, PersonalContextPolicy, PromptMode, + PromptSkillContextMetadata, ProviderToolCallReference, ProviderToolCallReplay, + ProviderToolDefinition, RunProfileResolutionRequest, RunProfileResolver, SkillName, + SkillTrustLevel, SkillVisibility, ToolObservationDetail, ToolObservationStatus, + UpdateAssistantDraft, VisibleCapabilityRequest, VisibleCapabilitySurface, resolution, }; use ironclaw_loop_host::{ EmptyLoopCapabilityPort, HostIdentityContextBuildError, HostIdentityContextCandidate, @@ -49,6 +49,7 @@ use ironclaw_loop_host::{ SkillBundleContextSource, SkillBundleDescriptor, SkillBundleId, SkillBundleSource, SkillBundleSourceError, SkillFilePath, SkillSourceKind, ThreadBackedLoopContextPort, ThreadBackedLoopModelPort, ThreadBackedLoopTranscriptPort, ThreadContextWindowCache, + ThreadResolvingLoopModelGateway, ThreadResolvingLoopModelGatewayParts, build_skill_run_snapshot, identity_message_ref, load_canonical_system_inference_context, }; use ironclaw_outbound::{ @@ -398,6 +399,100 @@ async fn model_port_empty_request_applies_prompt_token_budget_to_context_fallbac assert_eq!(calls[0].messages[0].content, "latest short"); } +/// Builds the gateway wrapper the driver host uses in production, with the +/// fixture's thread and the given budget, so the test drives the same +/// construction path as `ironclaw_turn_runner`'s `loop_driver_host`. +fn thread_resolving_gateway( + fixture: &ThreadFixture, + host_gateway: Arc, + prompt_context_budget: PromptContextTokenBudget, +) -> ThreadResolvingLoopModelGateway { + ThreadResolvingLoopModelGateway::new(ThreadResolvingLoopModelGatewayParts { + thread_service: Arc::clone(&fixture.thread_service), + thread_scope: fixture.thread_scope.clone(), + host_gateway, + max_messages: 16, + skill_context_source: None, + identity_context_source: None, + instruction_materialization_store: None, + capabilities: None, + prompt_authority: LoopPromptBundleAuthority::shared(), + context_window_cache: None, + attachment_read_port: None, + prompt_diagnostic_sink: None, + prompt_context_budget, + }) +} + +/// Sends an empty request through the gateway wrapper and returns the +/// transcript messages the host gateway received. +async fn messages_forwarded_by_gateway( + fixture: &ThreadFixture, + prompt_context_budget: PromptContextTokenBudget, +) -> Vec { + let host_gateway = Arc::new(RecordingGateway::reply("model says hi")); + let gateway = thread_resolving_gateway(fixture, host_gateway.clone(), prompt_context_budget); + issue_prompt_grant(&fixture.run_context, &[]); + + LoopModelGateway::stream_model( + &gateway, + LoopModelGatewayRequest { + context: fixture.run_context.clone(), + request: LoopModelRequest { + inline_messages: Vec::new(), + messages: Vec::new(), + surface_version: None, + model_preference: None, + fallback_index: 0, + iteration: 0, + capability_view: None, + tool_choice: None, + }, + }, + ) + .await + .unwrap(); + + let calls = host_gateway.calls.lock().unwrap(); + assert_eq!(calls.len(), 1); + calls[0] + .messages + .iter() + .map(|message| message.content.clone()) + .collect() +} + +#[tokio::test] +async fn thread_resolving_gateway_applies_its_prompt_context_budget_to_message_selection() { + // The gateway wrapper is what the driver host hands the loop, so a budget + // that stops here never reaches the outbound request. The port-level + // sibling above proves the port honors a budget; this one proves the + // wrapper passes its own budget down instead of the port's default. + // Roles alternate because the port coalesces consecutive user text into + // one message; alternating keeps the control count honest. + let fixture = ThreadFixture::new_with_user_content("old short").await; + fixture.append_assistant_reply("assistant reply").await; + fixture + .accept_user_message("event-2", &"large ".repeat(32)) + .await; + fixture + .append_assistant_reply("assistant reply again") + .await; + fixture.accept_user_message("event-3", "latest short").await; + + let unbudgeted = + messages_forwarded_by_gateway(&fixture, PromptContextTokenBudget::default()).await; + assert_eq!( + unbudgeted.len(), + 5, + "control: the default budget must admit the whole seeded transcript" + ); + + let budgeted = + messages_forwarded_by_gateway(&fixture, PromptContextTokenBudget::new(6, 0, 0)).await; + assert_eq!(budgeted, vec!["latest short".to_string()]); +} + #[tokio::test] async fn model_port_empty_request_pins_the_run_accepted_task_on_cache_miss() { let mut fixture = ThreadFixture::new_with_user_content("original accepted task").await; @@ -5988,6 +6083,20 @@ async fn register_reply_attachment( .expect("reply attachment registration"); } +impl ThreadFixture { + async fn append_assistant_reply(&self, content: &str) { + self.thread_service + .append_finalized_assistant_message(AppendFinalizedAssistantMessageRequest { + scope: self.thread_scope.clone(), + thread_id: self.thread_id.clone(), + turn_run_id: self.run_context.run_id.to_string(), + content: MessageContent::text(content), + }) + .await + .unwrap(); + } +} + async fn finalized_assistant_message(fixture: &ThreadFixture) -> ThreadMessageRecord { fixture .thread_service diff --git a/crates/loop/ironclaw_turn_runner/src/loop_driver_host.rs b/crates/loop/ironclaw_turn_runner/src/loop_driver_host.rs index dabf02c3dcd..c94a90ca65a 100644 --- a/crates/loop/ironclaw_turn_runner/src/loop_driver_host.rs +++ b/crates/loop/ironclaw_turn_runner/src/loop_driver_host.rs @@ -133,10 +133,10 @@ use ironclaw_loop_contracts::{ LoopModelUsage, LoopProgressEvent, LoopProgressPort, LoopPromptBundle, LoopPromptBundleAuthority, LoopPromptBundleRequest, LoopPromptPort, LoopRequest, LoopRequestBatch, LoopRunContext, LoopRunInfoPort, LoopRuntimeContext, LoopTranscriptPort, - MemoryPromptContextService, NoOpBudgetAccountant, NoOpPolicyGuard, ProviderToolCall, - ProviderToolDefinition, RegisterProviderToolCallRequest, RunScopedHookMilestoneSink, - StageCheckpointPayloadRequest, SystemInferencePort, UpdateAssistantDraft, - VisibleCapabilityRequest, VisibleCapabilitySurface, + MemoryPromptContextService, NoOpBudgetAccountant, NoOpPolicyGuard, PromptContextTokenBudget, + ProviderToolCall, ProviderToolDefinition, RegisterProviderToolCallRequest, + RunScopedHookMilestoneSink, StageCheckpointPayloadRequest, SystemInferencePort, + UpdateAssistantDraft, VisibleCapabilityRequest, VisibleCapabilitySurface, }; use ironclaw_turns::{ AgentTurnRuntimePort, AgentTurnSpawnTreeRuntimePort, LoopCheckpointStore, RunProfileId, @@ -1601,8 +1601,15 @@ where validate_thread_scope(&effective_scope, &request.loop_run_context)?; let max_messages = self.config.max_messages.max(1); - let prompt_context_budget = self.config.prompt_context_budget; let run_context = self.attach_model_route_snapshot(request.loop_run_context)?; + // Resolve the scope-specific gateway ONCE and reuse the same object + // for both the window query below and the gateway construction later + // in this function. Asking `self.model_gateway` while the run is + // served by a `resolve_for_scope` override would let the budget + // describe a different gateway than the one issuing the request. + // (Production overrides none -- the trait default returns `None` -- + // but a test harness that does would get a silently wrong budget.) + let scoped_gateway = self.model_gateway.resolve_for_scope(&run_context.scope); // Kick off advisory communication-context fetches only for origins that // can use delivery/channel context. WebUI chat renders its origin without @@ -1643,6 +1650,47 @@ where .await }); + // Ask the run's model how much context it really holds. Deliberately + // awaited HERE, after the prefetch kickoffs above, rather than beside + // the route resolution: a blocking await placed before them would + // serialize three fetches the surrounding code deliberately runs in + // parallel. A caller that already supplied a budget is authoritative. + // + // `model_metadata()` is a static, I/O-free description of the + // configured model by the ironclaw_llm contract (CONTRACT.md, + // "LlmProvider Trait" key notes), so this await resolves + // immediately. ponytail: if that contract is ever relaxed, promote + // this to a `tokio::spawn` alongside `user_profile_fetch` instead of + // widening the critical path. + let run_context = if run_context.resolved_context_budget.is_some() { + run_context + } else { + let profile_id = &run_context.resolved_run_profile.model_profile_id; + let route = run_context.resolved_model_route.as_ref(); + let advertised = match scoped_gateway.as_ref() { + Some(gateway) => { + gateway + .advertised_context_window_tokens(profile_id, route) + .await + } + None => { + self.model_gateway + .advertised_context_window_tokens(profile_id, route) + .await + } + }; + match advertised { + Some(window) => run_context.with_resolved_context_budget( + PromptContextTokenBudget::from_advertised_window(Some(window)), + ), + None => run_context, + } + }; + // Derived FROM the field, so the two can never disagree. + let prompt_context_budget = run_context + .resolved_context_budget + .unwrap_or(self.config.prompt_context_budget); + let context_window_cache = Arc::new(ThreadContextWindowCache::default()); let mut context_adapter = ThreadBackedLoopContextPort::new( Arc::clone(&self.thread_service), @@ -1650,7 +1698,8 @@ where run_context.clone(), max_messages, ) - .with_context_window_cache(Arc::clone(&context_window_cache)); + .with_context_window_cache(Arc::clone(&context_window_cache)) + .with_prompt_context_token_budget(prompt_context_budget); // An unbound run's prepared context is its COMPLETE input by // contract: no skill, identity, or memory lane is folded in, so the // caller's declared context is exactly what the model sees. @@ -1956,54 +2005,55 @@ where // own `Arc` (G: ?Sized, not coercible to `Arc`) in the fallback — // see `build_compaction_ports` above. Each arm moves its owned fields. let model_gateway_ports_started_at = ironclaw_observability::live_latency_started_at(); - let model_gateway: Arc = - if let Some(gw) = self.model_gateway.resolve_for_scope(&run_context.scope) { - Arc::new(ThreadResolvingLoopModelGateway::new( - ThreadResolvingLoopModelGatewayParts { - thread_service: Arc::clone(&self.thread_service), - thread_scope: effective_scope.clone(), - host_gateway: gw, - max_messages, - skill_context_source: (!unbound_run) - .then(|| self.skill_context_source.clone()) - .flatten(), - identity_context_source: (!unbound_run) - .then(|| self.identity_context_source.clone()) - .flatten(), - instruction_materialization_store: Some(Arc::clone( - &instruction_materialization_store, - )), - capabilities: Some(Arc::clone(&capabilities)), - prompt_authority, - context_window_cache: Some(context_window_cache), - attachment_read_port: self.attachment_read_port.clone(), - prompt_diagnostic_sink: self.prompt_diagnostic_sink.clone(), - }, - )) - } else { - Arc::new(ThreadResolvingLoopModelGateway::new( - ThreadResolvingLoopModelGatewayParts { - thread_service: Arc::clone(&self.thread_service), - thread_scope: effective_scope.clone(), - host_gateway: Arc::clone(&self.model_gateway), - max_messages, - skill_context_source: (!unbound_run) - .then(|| self.skill_context_source.clone()) - .flatten(), - identity_context_source: (!unbound_run) - .then(|| self.identity_context_source.clone()) - .flatten(), - instruction_materialization_store: Some(Arc::clone( - &instruction_materialization_store, - )), - capabilities: Some(Arc::clone(&capabilities)), - prompt_authority, - context_window_cache: Some(context_window_cache), - attachment_read_port: self.attachment_read_port.clone(), - prompt_diagnostic_sink: self.prompt_diagnostic_sink.clone(), - }, - )) - }; + let model_gateway: Arc = if let Some(gw) = scoped_gateway { + Arc::new(ThreadResolvingLoopModelGateway::new( + ThreadResolvingLoopModelGatewayParts { + thread_service: Arc::clone(&self.thread_service), + thread_scope: effective_scope.clone(), + host_gateway: gw, + max_messages, + skill_context_source: (!unbound_run) + .then(|| self.skill_context_source.clone()) + .flatten(), + identity_context_source: (!unbound_run) + .then(|| self.identity_context_source.clone()) + .flatten(), + instruction_materialization_store: Some(Arc::clone( + &instruction_materialization_store, + )), + capabilities: Some(Arc::clone(&capabilities)), + prompt_authority, + context_window_cache: Some(context_window_cache), + attachment_read_port: self.attachment_read_port.clone(), + prompt_diagnostic_sink: self.prompt_diagnostic_sink.clone(), + prompt_context_budget, + }, + )) + } else { + Arc::new(ThreadResolvingLoopModelGateway::new( + ThreadResolvingLoopModelGatewayParts { + thread_service: Arc::clone(&self.thread_service), + thread_scope: effective_scope.clone(), + host_gateway: Arc::clone(&self.model_gateway), + max_messages, + skill_context_source: (!unbound_run) + .then(|| self.skill_context_source.clone()) + .flatten(), + identity_context_source: (!unbound_run) + .then(|| self.identity_context_source.clone()) + .flatten(), + instruction_materialization_store: Some(Arc::clone( + &instruction_materialization_store, + )), + capabilities: Some(Arc::clone(&capabilities)), + prompt_authority, + context_window_cache: Some(context_window_cache), + attachment_read_port: self.attachment_read_port.clone(), + prompt_diagnostic_sink: self.prompt_diagnostic_sink.clone(), + prompt_context_budget, + }, + )) + }; let structured_finalization: Option> = if run_context.output_contract.is_structured_output() { Some(Arc::new(StructuredFinalizationCoordinator::new( @@ -3196,6 +3246,10 @@ mod compaction_tests; #[path = "loop_driver_host/run_lease_fence_tests.rs"] mod run_lease_fence_tests; +#[cfg(test)] +#[path = "loop_driver_host/context_budget_tests.rs"] +mod context_budget_tests; + #[cfg(test)] mod tests { use super::*; diff --git a/crates/loop/ironclaw_turn_runner/src/loop_driver_host/context_budget_tests.rs b/crates/loop/ironclaw_turn_runner/src/loop_driver_host/context_budget_tests.rs new file mode 100644 index 00000000000..68e02fd98c9 --- /dev/null +++ b/crates/loop/ironclaw_turn_runner/src/loop_driver_host/context_budget_tests.rs @@ -0,0 +1,315 @@ +//! Coverage for the per-run prompt context budget the production host-build +//! seam resolves from the run's model. Split out of `loop_driver_host`'s +//! `mod tests` (sibling pattern, like `compaction_tests`) to keep that file +//! under the architecture size threshold. +//! +//! The behavior under test: `build_text_only_host_with_capabilities` asks the +//! run's gateway for the model's advertised context window and, when the +//! gateway reports one, derives a `PromptContextTokenBudget` from it and +//! carries it on `LoopRunContext`. A gateway that reports nothing must leave +//! the run exactly as it behaved before this seam existed. + +use super::*; + +use ironclaw_host_api::ids::{AgentId, ProjectId, TenantId, ThreadId}; +use ironclaw_host_api::turn::{TurnId, TurnLeaseToken, TurnRunId, TurnScope}; +use ironclaw_loop_contracts::{ + InMemoryLoopHostMilestoneSink, InMemoryRunProfileResolver, LoopPromptBundleRef, + LoopRunInfoPort, RunProfileResolutionRequest, RunProfileResolver, +}; +use ironclaw_threads::{ + AcceptInboundMessageRequest, AppendFinalizedAssistantMessageRequest, EnsureThreadRequest, + MessageContent, +}; +use ironclaw_turns::test_support::{in_memory_agent_turn_runtime, in_memory_loop_checkpoint_store}; + +/// A gateway that advertises a configurable context window and records the +/// transcript content of every model call it receives. +struct WindowAdvertisingGateway { + window: Option, + forwarded_messages: Mutex>>, +} + +impl WindowAdvertisingGateway { + fn advertising(window: Option) -> Self { + Self { + window, + forwarded_messages: Mutex::new(Vec::new()), + } + } +} + +#[async_trait] +impl HostManagedModelGateway for WindowAdvertisingGateway { + async fn stream_model( + &self, + request: ironclaw_loop_host::HostManagedModelRequest, + ) -> Result< + ironclaw_loop_host::HostManagedModelResponse, + ironclaw_loop_host::HostManagedModelError, + > { + self.forwarded_messages.lock().unwrap().push( + request + .messages + .iter() + .map(|message| message.content.clone()) + .collect(), + ); + Ok(ironclaw_loop_host::HostManagedModelResponse::assistant_reply("ack".to_string())) + } + + async fn advertised_context_window_tokens( + &self, + _model_profile_id: &ironclaw_loop_contracts::ModelProfileId, + _resolved_model_route: Option<&ironclaw_loop_host::HostManagedModelRouteSnapshot>, + ) -> Option { + self.window + } +} + +/// Builds a host through the real production seam and hands back the +/// `LoopRunContext` it resolved, so the assertion reads what a real run would. +async fn resolved_budget_for(window: Option) -> Option { + let (host, _) = build_host_for(window).await; + host.run_context().resolved_context_budget +} + +/// Sends an empty model request through the built host — the path the loop +/// takes when it asks the host to assemble the transcript itself — and returns +/// the message contents the gateway received. +async fn messages_forwarded_for(window: Option) -> Vec { + let (host, gateway) = build_host_for(window).await; + let run_context = host.run_context().clone(); + LoopPromptBundleAuthority::shared() + .issue_bundle( + &run_context, + &LoopPromptBundle { + bundle_ref: LoopPromptBundleRef::for_run(&run_context, "ctx-budget").unwrap(), + messages: Vec::new(), + surface_version: None, + compaction_message_index: Vec::new(), + recent_window_truncation: None, + instruction_fingerprint: None, + identity_message_count: 0, + instruction_snippet_count: 0, + }, + ) + .unwrap(); + + LoopModelPort::stream_model( + &host, + LoopModelRequest { + inline_messages: Vec::new(), + messages: Vec::new(), + surface_version: None, + model_preference: None, + fallback_index: 0, + iteration: 0, + capability_view: None, + tool_choice: None, + }, + ) + .await + .expect("model call"); + + let mut calls = gateway.forwarded_messages.lock().unwrap(); + assert_eq!(calls.len(), 1, "exactly one model call reaches the gateway"); + calls.remove(0) +} + +/// Builds a host over a thread seeded with five alternating user/assistant +/// messages (roles alternate because the port coalesces consecutive user +/// text into one message, which would make message counts lie). +async fn build_host_for( + window: Option, +) -> (RebornLoopDriverHost, Arc) { + let thread_service = Arc::new(ironclaw_threads::InMemorySessionThreadService::default()); + let suffix = window.map(|w| w.to_string()).unwrap_or("none".to_string()); + let tenant_id = TenantId::new(format!("tenant-ctx-budget-{suffix}")).unwrap(); + let agent_id = AgentId::new(format!("agent-ctx-budget-{suffix}")).unwrap(); + let project_id = ProjectId::new(format!("project-ctx-budget-{suffix}")).unwrap(); + let thread_id = ThreadId::new(format!("thread-ctx-budget-{suffix}")).unwrap(); + let thread_scope = ThreadScope { + tenant_id: tenant_id.clone(), + agent_id: agent_id.clone(), + project_id: Some(project_id.clone()), + owner_user_id: None, + mission_id: None, + }; + thread_service + .ensure_thread(EnsureThreadRequest { + scope: thread_scope.clone(), + thread_id: Some(thread_id.clone()), + created_by_actor_id: "user-ctx-budget".to_string(), + title: None, + metadata_json: None, + }) + .await + .unwrap(); + let run_id = TurnRunId::new(); + for (index, content) in [ + "old short", + "assistant reply", + &"large ".repeat(32), + "assistant reply again", + "latest short", + ] + .into_iter() + .enumerate() + { + if index % 2 == 0 { + thread_service + .accept_inbound_message(AcceptInboundMessageRequest { + scope: thread_scope.clone(), + thread_id: thread_id.clone(), + actor_id: "user-ctx-budget".to_string(), + source_binding_id: Some("source-web".to_string()), + reply_target_binding_id: Some("reply-web".to_string()), + external_event_id: Some(format!("event-ctx-budget-{suffix}-{index}")), + content: MessageContent::text(content), + }) + .await + .unwrap(); + } else { + thread_service + .append_finalized_assistant_message(AppendFinalizedAssistantMessageRequest { + scope: thread_scope.clone(), + thread_id: thread_id.clone(), + turn_run_id: run_id.to_string(), + content: MessageContent::text(content), + }) + .await + .unwrap(); + } + } + + let turn_scope = TurnScope::new( + tenant_id, + Some(agent_id), + Some(project_id), + thread_id.clone(), + ); + let resolved = InMemoryRunProfileResolver::default() + .resolve_run_profile(RunProfileResolutionRequest::interactive_default()) + .await + .unwrap(); + let run_context = LoopRunContext::new(turn_scope.clone(), TurnId::new(), run_id, resolved); + let claimed_run = claimed_run_for(&run_context, &turn_scope); + + let gateway = Arc::new(WindowAdvertisingGateway::advertising(window)); + let factory = RebornLoopDriverHostFactory::new( + thread_service, + thread_scope, + Arc::clone(&gateway), + Arc::new(in_memory_agent_turn_runtime()) as Arc, + Arc::new(in_memory_loop_checkpoint_store()) as Arc, + Arc::new(InMemoryLoopHostMilestoneSink::default()) as Arc, + TextOnlyLoopHostConfig { + max_messages: 8, + prompt_context_budget: Default::default(), + require_model_route_snapshot: false, + }, + InstructionSafetyContext::non_production_noop(), + ); + + let host = factory + .build_text_only_host_with_capabilities( + RebornLoopDriverHostRequest { + claimed_run, + loop_run_context: run_context, + }, + Arc::new(EmptyLoopCapabilityPort), + ) + .await + .expect("host builds"); + (host, gateway) +} + +fn claimed_run_for( + run_context: &LoopRunContext, + scope: &TurnScope, +) -> ironclaw_turns::runner::ClaimedTurnRun { + use ironclaw_turns::{AcceptedMessageRef, TurnRunnerId, TurnStatus}; + + ironclaw_turns::runner::ClaimedTurnRun { + subagent_activation_provenance: None, + state: ironclaw_turns::TurnRunState { + scope: scope.clone(), + actor: None, + turn_id: run_context.turn_id, + run_id: run_context.run_id, + status: TurnStatus::Running, + accepted_message_ref: AcceptedMessageRef::new("msg:accepted").expect("valid"), // safety: fixed fixture satisfies the bounded-ref grammar. + output_contract: ironclaw_host_api::output::OutputContract::AssistantMessage, + resolved_run_profile_id: persisted_profile_id( + &run_context.resolved_run_profile.profile_id, + ), + resolved_run_profile_version: run_context.resolved_run_profile.profile_version, + allow_steering: true, + resolved_model_route: None, + model_usage: None, + execution_outcome: None, + received_at: chrono::Utc::now(), + checkpoint_id: None, + gate_ref: None, + blocked_activity_id: None, + credential_requirements: Vec::new(), + failure: None, + event_cursor: ironclaw_turns::EventCursor(0), + product_context: None, + resume_disposition: None, + }, + resolved_run_profile: run_context.resolved_run_profile.clone(), + subagent_depth: 0, + spawn_tree_descendant_cap: None, + runner_id: TurnRunnerId::new(), + lease_token: TurnLeaseToken::new(), + } +} + +#[tokio::test] +async fn resolved_budget_reaches_the_run_context_when_the_gateway_advertises_a_window() { + let budget = resolved_budget_for(Some(40_000)).await; + + assert_eq!( + budget, + Some(PromptContextTokenBudget::from_advertised_window(Some( + 40_000 + ))), + "a run whose model advertises a window must carry the derived budget" + ); +} + +#[tokio::test] +async fn run_context_carries_no_budget_when_the_gateway_advertises_nothing() { + let budget = resolved_budget_for(None).await; + + assert_eq!( + budget, None, + "a gateway that advertises nothing must leave the run on the compiled-in default" + ); +} + +#[tokio::test] +async fn derived_budget_sizes_the_request_the_host_sends_to_the_gateway() { + // The link the two tests above cannot see: the host derives the model + // port's budget FROM the run context it resolved. A 40-token window + // derives limit 36 / reserve 9 / visible 27 tokens, which admits only the + // two-message tail of the seeded transcript. + let wide = messages_forwarded_for(None).await; + assert_eq!( + wide.len(), + 5, + "control: an unadvertised run must forward the whole seeded transcript" + ); + + let narrow = messages_forwarded_for(Some(40)).await; + assert_eq!( + narrow, + vec![ + "assistant reply again".to_string(), + "latest short".to_string() + ], + "the advertised window must bound the messages the gateway receives" + ); +} diff --git a/crates/product/ironclaw_webui/frontend/src/lib/api-boundary.test.ts b/crates/product/ironclaw_webui/frontend/src/lib/api-boundary.test.ts index 1ad170a5180..630092e8785 100644 --- a/crates/product/ironclaw_webui/frontend/src/lib/api-boundary.test.ts +++ b/crates/product/ironclaw_webui/frontend/src/lib/api-boundary.test.ts @@ -95,7 +95,7 @@ test("notification setup rejects malformed successful responses", async () => { globalThis.fetch = async () => new Response( JSON.stringify({ - extension_id: "web-push", + extension_id: "web-app", requires_setup: false, enabled: "yes", }), @@ -103,7 +103,7 @@ test("notification setup rejects malformed successful responses", async () => { ); await assert.rejects( - getNotificationSetupStatus({ extensionId: "web-push" }), + getNotificationSetupStatus({ extensionId: "web-app" }), /invalid notification setup response/, ); }); diff --git a/docs/internal/reborn/design/model-derived-context-budget.md b/docs/internal/reborn/design/model-derived-context-budget.md new file mode 100644 index 00000000000..cf29e0b64b2 --- /dev/null +++ b/docs/internal/reborn/design/model-derived-context-budget.md @@ -0,0 +1,339 @@ +# Model-derived prompt context budget + +Shape spec. Resolve the provider-advertised context window once per run, carry +it on `LoopRunContext`, and read it from every consumer that today reaches an +independent `PromptContextTokenBudget::default()`. + +## Extend-vs-fork verdict + +**Extends** `PromptContextTokenBudget` — the existing canonical budget type in +`ironclaw_loop_contracts`. No new budget type, no parallel "dynamic" path, no +cargo feature. `visible_transcript_tokens()` already computes +`limit − max(reserve, max_output)`; `estimate_tokens_from_chars` already does +chars/4. The only new behavior is what populates `context_limit_tokens`. + +## Deletion-first check + +Deleting alone does not solve it — a real per-model number still has to come +from somewhere. But the change **is** net-subtractive at the wiring level: five +independent production `::default()` sites collapse to one resolver, and the +`with_prompt_context_token_budget` builders (today test-only dead weight, +`ironclaw_loop_host/src/lib.rs:482` and `:1592`) gain their first production +callers instead of being deleted. + +## Alternatives considered + +Recorded because this changes a serialized contracts struct and recomputes four +replay-identity digests — both awkward to reverse once runs exist. + +| Option | Mechanism | Why not | +|---|---|---| +| **Do nothing** | Keep 128k for every model. | A 2M-window model compacts at 108k, discarding 95% of usable context. A sub-128k model is sent oversized prompts, and the only overflow remedy (`executor/model.rs:475-481`) forces compaction against a ceiling the prompt already satisfies, so the run cannot converge. Rejected: this is a live correctness bug, not just waste. | +| **B — resolve at family composition, via `FamilyOverrides`** | Add a budget field to the existing overrides seam (`families/mod.rs:78-84`). | `FamilyOverrides` carries only `iteration_limit` and `model_availability_attempts` today, and `default_with_overrides` applies them through `.with_budget()` and `.with_recovery()` (`:118-128`) — it does not reach the compaction strategy at all, so a budget override would mean extending that seam too. More decisively, everything it builds is constructed inside `ironclaw_agent_loop`, which is barred from depending on `ironclaw_loop_host`, so it could never reach the ports at all — the loop-host consumers would still need separate wiring, giving two carriers of one number. And the digest is a pure function of resolved values, so every distinct window yields a distinct family identity, fragmenting replay identity per model. | +| **C — static per-model table consulted directly by the ports** | Generalize `gemini_context_length` and look it up at each consumer. | No async and a tiny diff, but each consumer performs its own lookup — the same multi-carrier problem as B, and it ignores what providers actually report. Adopted *as the data source* for the follow-on slice (see below), rejected as the *transport*. | +| **A — resolve per run, carry on `LoopRunContext`** ✅ | One resolution at host construction; every consumer reads one field. | Chosen. One carrier, one value; the digest stays stable across models; `should_compact` already receives `&LoopRunContext` and ignores it, so the agent-loop side costs almost nothing. | + +Cost to undo: moderate. One `Option` field on a contracts struct (serde-default, +so old runs replay) and one digest bump to revert. + +## Untouched + +- `ironclaw_threads` — its `truncate_context_window` (`contract.rs:873`) is a + message-**count** cap on a different axis. Not this change. +- The durable transcript. This is a read-time projection throughout; root + `AGENTS.md:126` ("LLM data is never deleted") is unaffected. +- `LlmModelCatalogEntry` (`ironclaw_product_contracts/src/operator_llm.rs`) — + modality-only presentation metadata, documented as forbidden from + influencing routing or policy. Do not route context length through it. +- `ironclaw_agent_loop`'s dependency set. It stays contracts-only + (`reborn_dependency_boundaries.rs:308-317`); the budget reaches it as a + contracts-tier value on `LoopRunContext`, never as a provider dependency. +- `MODEL_WORK_ESTIMATED_CHARS_PER_TOKEN` (`model_work.rs:13`) — a second chars/4 + constant, but it feeds `budget_accountant.rs:380` (spend), not a context + ceiling. Different axis; leave it alone. +- **`ThreadBackedLoopModelGateway::issue_host_prompt_bundle`** + (`model_gateway.rs:311-326`) builds a context port with the cache but no + budget, so it keeps the 128k default. It is non-test code exported at + `lib.rs:122`, but `ThreadBackedLoopModelGateway::new` is instantiated only by + `ironclaw_loop_host/tests/llm_gateway.rs` — never by composition, which wires + `ThreadResolvingLoopModelGateway`. Left on the default deliberately: wiring an + unreachable path would add a call site with no production meaning. If it is + ever composed, it must be wired in that change. + +## 1. Derivation — `crates/contracts/ironclaw_loop_contracts/src/context_budget.rs` + +Add `serde::Deserialize` to the derive at `:7` (today it is `Serialize` only; +`LoopRunContext` is deserialized on replay). + +```rust +impl PromptContextTokenBudget { + /// Fraction of the advertised window we will fill. The margin absorbs + /// chars/4 estimate error, which is the only reason it exists — the + /// response headroom is `reserve_tokens`, a separate axis. + pub const DEFAULT_USABLE_FRACTION_PERCENT: u64 = 90; + + /// Derive a budget from a provider-advertised total context window. + /// + /// `None` reproduces today's compiled-in 128k/20k exactly, so a provider + /// that reports nothing behaves as it does now. + pub fn from_advertised_window(advertised_tokens: Option) -> Self { + let Some(advertised) = advertised_tokens.filter(|tokens| *tokens > 0) else { + return Self::default(); + }; + let context_limit_tokens = + advertised.saturating_mul(Self::DEFAULT_USABLE_FRACTION_PERCENT) / 100; + // A small-window model would otherwise have its whole budget consumed + // by the flat response reserve, leaving zero visible transcript. + let reserve_tokens = Self::DEFAULT_RESERVE_TOKENS.min(context_limit_tokens / 4); + Self { + context_limit_tokens, + reserve_tokens, + main_loop_max_output_tokens: Self::DEFAULT_MAIN_LOOP_MAX_OUTPUT_TOKENS, + } + } +} +``` + +## 2. Source — `crates/loop/ironclaw_loop_host/src/lib.rs` + +New defaulted method on `#[async_trait] pub trait HostManagedModelGateway` +(`:2293`), copying the narrow shape of its neighbour `diagnostic_effective_model` +(`:2297`) — best-effort, route-keyed, `None`-defaulting: + +```rust + async fn advertised_context_window_tokens( + &self, + _model_profile_id: &ModelProfileId, + _resolved_model_route: Option<&HostManagedModelRouteSnapshot>, + ) -> Option { + None + } +``` + +Override in `LlmProviderModelGateway` (`model_gateway.rs:406`) — the gateway +composition actually wires (`ironclaw_composition/src/model_gateway_assembly.rs:138`) +— reading `model_metadata().context_length`, the field at +`ironclaw_llm/src/provider.rs:932` that has no consumer today. + +**The route parameter is load-bearing, not decoration.** `model_metadata()` takes +no model argument (`provider.rs:1016`); it describes whatever model the provider +was configured with. But the same gateway's request path resolves the served +model through `request_model_override` (`model_gateway.rs:1221-1248`) as +*route-requested → route override → `provider.active_model_name()`*, and its own +comment records that "providers that honor per-request overrides (e.g. NEAR AI) +serve the requested model." So on a run whose advisory route overrides the model, +`model_metadata()` can describe a **different model than the one served** — and a +window borrowed from the wrong model is worse than no window, because guessing +high produces the provider rejection this work exists to prevent. + +The override must therefore verify identity and fail safe: + +```rust + async fn advertised_context_window_tokens( + &self, + model_profile_id: &ModelProfileId, + resolved_model_route: Option<&HostManagedModelRouteSnapshot>, + ) -> Option { + let metadata = self.provider.model_metadata().await.ok()?; + // Resolve the model this run will actually be served, the same way + // stream_model does, and only trust the window if it describes that + // model. A mismatch falls back to the compiled-in default. + let route = self.policy.route_for(model_profile_id)?; + let served = request_model_override( + route, + self.provider.as_ref(), + resolved_model_route.map(HostManagedModelRouteSnapshot::model_id), + ) + .ok()?; + (served == metadata.id).then_some(())?; + metadata.context_length.map(u64::from) + } +``` + +This is what makes the derived budget genuinely a function of the pinned route. +Where it cannot be — a provider that serves an overridden model without +reporting metadata for it — the run keeps today's behavior rather than a wrong +number. + +## 3. Transport — `crates/contracts/ironclaw_loop_contracts/src/host/run_context.rs:236` + +```rust + #[serde(default, skip_serializing_if = "Option::is_none")] + pub resolved_context_budget: Option, +``` + +Set in `loop_driver_host.rs` beside the existing route resolution +(`attach_model_route_snapshot`, `:2131`), before the ports are built. + +## 4. Consumers — the complete inventory + +There are **five** production `PromptContextTokenBudget::default()` sites across +**three** types. Getting this list wrong is the central risk: the spec's whole +premise is that one number governs, and wiring a subset produces a loop that +*selects* messages against one ceiling while *compacting* against another. + +| # | Site | Type | How it gets the resolved value | +|---|---|---|---| +| 1 | `compaction.rs:103` | `DefaultCompactionStrategy` | reads `ctx.resolved_context_budget` in `should_compact` | +| 2 | `loop_driver_host/config.rs:17` | `TextOnlyLoopHostConfig` | stays the fallback when nothing is advertised | +| 3 | `lib.rs:430` | `ThreadBackedLoopContextPort` | `.with_prompt_context_token_budget(..)` (`:482`) | +| 4 | `lib.rs:1538` | `ThreadBackedLoopModelPort::new` | `.with_prompt_context_token_budget(..)` (`:1592`), threaded through the gateway | +| 5 | `lib.rs:1566` | `ThreadBackedLoopModelPort::with_milestone_sink` | same | + +**a. Compaction trigger** — `ironclaw_agent_loop/src/strategies/compaction.rs`. +`CompactionStrategy::should_compact` (`:21-25`) **already receives +`ctx: &LoopRunContext` and both impls ignore it** (`compaction.rs:114`, +`active_task_compaction.rs:45` — named `_ctx`). Use it: + +```rust +fn effective_budget(&self, ctx: &LoopRunContext) -> PromptContextTokenBudget { + ctx.resolved_context_budget.unwrap_or(self.prompt_context_budget) +} +``` + +`can_evaluate` (`:56`) and `trigger_at` (`:76`) take a `budget` parameter +instead of reading the field. + +**b. Prompt context port** — `loop_driver_host.rs:1647`, via the builder at +`lib.rs:482`. Feeds the prompt's assembled context. + +**c. Model port — the one that sizes the outbound request.** +`ThreadResolvingLoopModelGateway` is what composition wires +(`loop_driver_host.rs:1957-1998`). Its `stream_model_inner` +(`thread_resolving_model_gateway.rs:135-142`) builds +`ThreadBackedLoopModelPort::new(..)` with no budget, so the port keeps the +default; its `resolve_model_messages` (`lib.rs:2010-2021`) then calls +`select_prompt_context_messages(context.messages, self.prompt_context_budget, ..)` +— the call that decides which transcript messages actually reach the provider. +`ThreadResolvingLoopModelGatewayParts` (`:27-44`) has twelve fields and none is a +budget, so it needs a thirteenth, stored on the gateway and applied in +`stream_model_inner` via the builder at `lib.rs:1592`. + +**d. Structured finalization** — `loop_driver_host.rs:2018` reads the same local +as (b) and (c), inside the same function, so it follows automatically. + +## 5. Lifetime of the resolved value — deliberately not persisted + +`LoopRunContext` is **not** the durable run record. `TurnRunState` +(`ironclaw_turns/src/status.rs:180-203`) is, and `create_host` +(`loop_driver_host.rs:2714-2736`) rebuilds `LoopRunContext` from it on every +claim, carrying forward only the five fields `TurnRunState` actually has. So +`resolved_context_budget` is **re-derived once per claim, not persisted**. + +That is the intended design, not an oversight: + +- The budget is a pure function of `resolved_model_route`, which *is* pinned + durably on `TurnRunState:203` and carried forward at `:2732-2733`. Persisting + the budget too would duplicate derivable state — the mirror-DTO pattern the + repo bans. +- Re-deriving self-corrects. A frozen number stays wrong if it was resolved from + a provider having a bad moment. + +The one behavior this accepts: if a provider changes its advertised window for +the same model between claims of one run, the budget changes mid-run. That is +bounded — the route is pinned and §2's identity check rejects a window that +describes a different model — and preferable to freezing a stale value. + +There is also a guard that short-circuits when a caller supplies a budget +explicitly. **Do not describe it as mirroring `attach_model_route_snapshot`'s +"already present" branch.** That branch is live in production because +`create_host` carries `resolved_model_route` forward from `TurnRunState` +(`:2731-2733`); no equivalent carry exists for the budget, so this guard's +precondition never occurs on a production claim. It protects direct callers of +`build_text_only_host_with_capabilities` — test harnesses today — from having a +deliberately-set budget silently overwritten. That is a narrow but real purpose; +it is not resume stability, and claiming the parallel would overstate it. + +## 6. Replay fingerprint + +`context_limit=128000` and `reserve=20000` are hand-typed into four fingerprint +strings: `families/mod.rs:35`, `families/subagent.rs:20`, +`families/unbound.rs:24,45`. Replace both literals with `context_limit=run_context` +and `reserve=run_context` and recompute the four `ComponentDigest` constants. +The digest then stays **stable across models**. +Digest tests (`families/mod.rs:164-173`) recompute from the fingerprint function +and self-heal; the `const` byte arrays are the edit. No production code compares +a persisted digest against a live one, so a revert is a code-only change. + +## 7. Tests + +| Tier | Where | Pins | +|---|---|---| +| 1 | `-p ironclaw_loop_contracts` | `from_advertised_window`: `None` → 128k/20k; 2M → 1.8M/20k; 8k → 7.2k clamped, non-zero visible | +| 1 | `-p ironclaw_agent_loop` | `should_compact` honors `ctx.resolved_context_budget` for both strategies | +| 1 | `-p ironclaw_loop_host` | the routed gateway returns the provider's `context_length`; the model port selects against an injected budget | +| 2 | `tests/integration/` | a small advertised window both compacts earlier **and** shrinks the message set actually sent | +| 3 | `-p ironclaw_architecture_tests` | `ironclaw_agent_loop` still contracts-only | + +The integration assertion must cover the **selected message set**, not only a +compaction milestone. A compaction-only assertion passes while consumer (c) is +still on the default — the exact split this spec exists to prevent. + +## Follow-on slice — populating the providers + +Only `gemini_oauth.rs:2158` populates `ModelMetadata.context_length` today, from +a static name table (`:136-155`). `bedrock.rs:240` and +`openai_codex_provider.rs:416` return `None`; every other provider inherits the +`None` default. **The main slice makes those providers behave exactly as they do +now** and gives them a seam that pays off as each is filled. + +### Why not `/models` alone + +`/models` carries a window for roughly half the fleet — Gemini +(`inputTokenLimit`), OpenRouter / Groq / Together, GitHub Copilot, Ollama via a +second `/api/show`. It carries **nothing** for OpenAI, Anthropic, or Bedrock. +Verify each shape with one `curl` before writing its parser. + +The catalog is also not cached: `list_model_catalog()` has one consumer outside +this crate — `llm_config_service.rs:805`, a live admin probe returned straight to +the settings UI. Binding the run path to it would mean an HTTP round-trip per run +or new cache infrastructure. So the catalog is a *free bonus where it exists*, +not the mechanism. + +### Two parts + +**a. Static table** — new `model_context_windows.rs` in `ironclaw_llm`'s +model-catalog family. Add it to the sub-owner table in `CONTRACT.md` or +`tests/module_charter.rs` fails. + +**Deviation from the neighbouring idiom, deliberately:** `vision_models.rs:14` +and `reasoning_models.rs:31` match with `lower.contains(pattern)`. Do **not** +copy that here. Substring matching is safe for a boolean capability and unsafe +for a magnitude — `"gpt-4"` is a substring of both `gpt-4o` and `gpt-4.1`, whose +windows differ by 8×, and guessing high causes exactly the provider rejection +this work exists to prevent. Match exact model id first, then explicit +longest-prefix family: + +```rust +/// Returns `None` for anything unrecognized — the caller then falls back to +/// the compiled-in default. Never guess a window for an unknown model. +pub fn context_window_for(model_id: &str) -> Option; +``` + +Populate from each provider's published limits at implementation time; do not +carry values over from this document. + +**b. Catalog field** — add `context_length: Option` to `DiscoveredModel` +(`models.rs:36`) and read it where the body is already deserialized. +`parse_nearai_models` (`nearai_chat.rs:56`) is the pattern: alias-tolerant, +`#[serde(default)]`, lossy. + +Resolution order in `model_metadata()`: catalog → static table → `None`. + +### Later, if it earns it + +`error.rs:437-465` already parses the provider's authoritative limit into +`LlmError::ContextLengthExceeded { used, limit }`, and `model_gateway.rs:2692` +matches `{ .. }` and discards it. Carrying `limit` through would make the budget +self-correcting — but only after one overflow, and it needs a durable store. + +## Why this bug is worth fixing beyond wasted headroom + +`executor/model.rs:475-481`: the sole corrective action on a provider +`context_length_exceeded` is `force_compact_on_next_iteration`, aimed at the +**same unchanged 128k ceiling**. For a model whose real window is under 128k, +compaction believes there is headroom, shrinks nothing meaningful, and the run +aborts on the second overflow (`recovery.rs:489-505`). Today that model can +never converge. Note this requires consumer (c) to be wired: fixing only the +compaction trigger changes when compaction fires but not the size of the request +that overflowed. diff --git a/docs/internal/superpowers/plans/2026-09-01-model-derived-context-budget.md b/docs/internal/superpowers/plans/2026-09-01-model-derived-context-budget.md new file mode 100644 index 00000000000..80421ad46c7 --- /dev/null +++ b/docs/internal/superpowers/plans/2026-09-01-model-derived-context-budget.md @@ -0,0 +1,1332 @@ +# Model-Derived Prompt Context Budget Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Make the agent loop's prompt context budget derive from the model's real advertised context window instead of a compiled-in 128,000-token constant applied to every model on every provider. + +**Architecture:** The budget type (`PromptContextTokenBudget`), the "limit minus buffer" formula (`visible_transcript_tokens()`), and the chars/4 estimator (`estimate_tokens_from_chars`) all already exist and are unchanged. So does the provider seam that reports a real window (`ModelMetadata.context_length`), which today has zero consumers outside `ironclaw_llm`. This plan connects them: the loop host asks its gateway for the window once per run, derives a budget, and puts it on `LoopRunContext` — from which all four consumers that today each reach an independent `PromptContextTokenBudget::default()` read it instead. + +**Tech Stack:** Rust 2024 edition, tokio, `async_trait`, serde, BLAKE3 (replay digests), `cargo test` / `cargo clippy`. + +**Spec:** `docs/internal/reborn/design/model-derived-context-budget.md` + +## Global Constraints + +- **`ironclaw_agent_loop` may take normal dependencies on contracts-layer crates only.** Enforced with zero exceptions by `crates/app/ironclaw_architecture_tests/tests/reborn_dependency_boundaries.rs:308-317`. Never add `ironclaw_llm` (or any non-contracts crate) to `crates/loop/ironclaw_agent_loop/Cargo.toml`. The budget reaches this crate as a contracts-tier value on `LoopRunContext`. +- **No `.unwrap()` or `.expect()` in production code.** Tests are fine. Propagate with `.map_err(|e| SomeError::Variant { reason: e.to_string() })?`. +- **Zero clippy warnings.** CI denies warnings: `cargo clippy --all --benches --tests --examples --all-features -- -D warnings`. +- **No new cargo feature.** This is runtime configuration (`.claude/rules/cargo-features.md`). +- **LLM data is never deleted** (root `AGENTS.md:126`). Every change here is a read-time projection. No task may delete, redact, or skip persisting a transcript message because it no longer fits a budget. +- **Preserve existing defaults.** A provider that advertises nothing must behave exactly as it does today: 128,000 limit / 20,000 reserve. +- **All five default sites must end up on one value.** There are five production `PromptContextTokenBudget::default()` sites across three types: `compaction.rs:103` (`DefaultCompactionStrategy`), `loop_driver_host/config.rs:17` (`TextOnlyLoopHostConfig`), `lib.rs:430` (`ThreadBackedLoopContextPort`), and `lib.rs:1538` + `:1566` (`ThreadBackedLoopModelPort`). Wiring a subset produces a loop that *selects* messages against one ceiling while *compacting* against another — the exact failure this work exists to prevent. `ThreadBackedLoopContextPort` and `ThreadBackedLoopModelPort` are different types with nearly identical names; both take a budget, and wiring one reads like wiring both. +- **Structural vs behavioral commits never mix.** Tasks 3 and 6 are structural (no behavior change); every other task is behavioral. Do not combine them in one commit. +- **Every test helper a task names must be real or declared new.** Before writing any test, `rg` for each helper the task names. If it exists, the task cites its `file:line`; if it does not, the task says so outright and names the nearest existing pattern to model it on. Never write a test as though a helper exists when it does not — an implementer who trusts the plan will search, find nothing, and either stall or invent undisclosed shared test infrastructure. Both `ironclaw_loop_host` and `ironclaw_turn_runner` use **one bespoke double per test**: there is no shared `FakeProvider`, no `test_policy()`, no `test_host_factory()`. +- **`main_loop_max_output_tokens` default stays `0`.** Not in scope. + +--- + +### Task 1: Derive a budget from an advertised window + +Adds the pure derivation function to the existing budget type. No caller yet. + +**Files:** +- Modify: `crates/contracts/ironclaw_loop_contracts/src/context_budget.rs:7` (derive), `:14-35` (impl block) +- Test: `crates/contracts/ironclaw_loop_contracts/src/context_budget.rs` (the existing `#[cfg(test)] mod tests` at `:47`) + +**Interfaces:** +- Consumes: nothing. +- Produces: `PromptContextTokenBudget::from_advertised_window(advertised_tokens: Option) -> PromptContextTokenBudget` and the associated const `DEFAULT_USABLE_FRACTION_PERCENT: u64 = 90`. `PromptContextTokenBudget` also gains `serde::Deserialize`, which Task 2 requires. + +**Context you need:** `PromptContextTokenBudget` has three public fields — `context_limit_tokens`, `reserve_tokens`, `main_loop_max_output_tokens` — and one method, `visible_transcript_tokens()`, which returns `context_limit_tokens - max(reserve_tokens, main_loop_max_output_tokens)`. That is the number of tokens of transcript the loop will actually put in a prompt. The two knobs are independent: `reserve_tokens` holds room for the model's *response*, while the new 90% fraction absorbs error in our chars/4 token estimate. Both apply; they are not the same buffer. + +- [ ] **Step 1: Write the failing tests** + +Append to the existing `mod tests` block at the end of `crates/contracts/ironclaw_loop_contracts/src/context_budget.rs`: + +```rust + #[test] + fn advertised_window_of_none_reproduces_the_compiled_in_default() { + // A provider that reports nothing must behave exactly as it does + // today. This is the compatibility guarantee of the whole change. + assert_eq!( + PromptContextTokenBudget::from_advertised_window(None), + PromptContextTokenBudget::default() + ); + } + + #[test] + fn advertised_window_of_zero_is_treated_as_unknown() { + assert_eq!( + PromptContextTokenBudget::from_advertised_window(Some(0)), + PromptContextTokenBudget::default() + ); + } + + #[test] + fn large_advertised_window_keeps_the_flat_response_reserve() { + let budget = PromptContextTokenBudget::from_advertised_window(Some(2_000_000)); + + assert_eq!(budget.context_limit_tokens, 1_800_000); + assert_eq!( + budget.reserve_tokens, + PromptContextTokenBudget::DEFAULT_RESERVE_TOKENS + ); + assert_eq!(budget.visible_transcript_tokens(), 1_780_000); + } + + #[test] + fn small_advertised_window_clamps_the_reserve_and_keeps_budget_usable() { + // An 8k model would otherwise have its entire budget consumed by the + // flat 20k response reserve, leaving zero visible transcript and a + // loop that cannot run at all. + let budget = PromptContextTokenBudget::from_advertised_window(Some(8_000)); + + assert_eq!(budget.context_limit_tokens, 7_200); + assert_eq!(budget.reserve_tokens, 1_800); + assert!( + budget.visible_transcript_tokens() > 0, + "a small-window model must still have room for transcript" + ); + } + + #[test] + fn advertised_window_matching_todays_constant_is_reduced_by_the_margin() { + // 128k advertised is NOT the same as the 128k fallback: the fallback + // is a guess, an advertised value gets the estimate-error margin. + let budget = PromptContextTokenBudget::from_advertised_window(Some(128_000)); + + assert_eq!(budget.context_limit_tokens, 115_200); + assert_eq!( + budget.reserve_tokens, + PromptContextTokenBudget::DEFAULT_RESERVE_TOKENS + ); + } +``` + +- [ ] **Step 2: Run the tests to verify they fail** + +Run: `cargo test -p ironclaw_loop_contracts context_budget` +Expected: FAIL to compile — `no function or associated item named 'from_advertised_window' found`. + +- [ ] **Step 3: Add the `Deserialize` derive** + +In `crates/contracts/ironclaw_loop_contracts/src/context_budget.rs:7`, change: + +```rust +#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize)] +``` + +to: + +```rust +#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)] +``` + +- [ ] **Step 4: Write the derivation** + +In the same file, inside `impl PromptContextTokenBudget` (after `DEFAULT_MAIN_LOOP_MAX_OUTPUT_TOKENS` at `:17`), add: + +```rust + /// Fraction of a provider-advertised window we are willing to fill. + /// + /// This margin exists to absorb error in the chars/4 token estimate + /// (`estimate_tokens_from_chars`), which is the only reason for it. Room + /// for the model's *response* is a separate axis — `reserve_tokens`. + pub const DEFAULT_USABLE_FRACTION_PERCENT: u64 = 90; +``` + +and, after `visible_transcript_tokens` (`:31-34`), add: + +```rust + /// Derive a budget from a provider-advertised total context window. + /// + /// `None` (or a nonsense zero) reproduces the compiled-in default + /// exactly, so a provider that advertises nothing behaves as it always + /// has. Never guess a window for an unknown model: guessing high + /// produces the provider rejection this mechanism exists to avoid. + pub fn from_advertised_window(advertised_tokens: Option) -> Self { + let Some(advertised) = advertised_tokens.filter(|tokens| *tokens > 0) else { + return Self::default(); + }; + let context_limit_tokens = + advertised.saturating_mul(Self::DEFAULT_USABLE_FRACTION_PERCENT) / 100; + // A small-window model would otherwise have its whole budget consumed + // by the flat response reserve, leaving zero visible transcript. + let reserve_tokens = Self::DEFAULT_RESERVE_TOKENS.min(context_limit_tokens / 4); + Self { + context_limit_tokens, + reserve_tokens, + main_loop_max_output_tokens: Self::DEFAULT_MAIN_LOOP_MAX_OUTPUT_TOKENS, + } + } +``` + +- [ ] **Step 5: Run the tests to verify they pass** + +Run: `cargo test -p ironclaw_loop_contracts context_budget` +Expected: PASS, all five new tests plus the pre-existing ones. + +- [ ] **Step 6: Commit** + +```bash +git add crates/contracts/ironclaw_loop_contracts/src/context_budget.rs +git commit -m "feat(loop-contracts): derive a prompt context budget from an advertised window" +``` + +--- + +### Task 2: Carry the resolved budget on the run context + +Adds the transport field. Still no producer or consumer. + +**Files:** +- Modify: `crates/contracts/ironclaw_loop_contracts/src/host/run_context.rs:236-259` (struct), `:273-283` (`new`), `:345-348` (beside `with_resolved_model_route`) +- Test: same file's test module + +**Interfaces:** +- Consumes: `PromptContextTokenBudget` with `Deserialize` (Task 1). +- Produces: `LoopRunContext.resolved_context_budget: Option` and `LoopRunContext::with_resolved_context_budget(self, budget: PromptContextTokenBudget) -> Self`. + +**Context you need:** `LoopRunContext` is the per-run context handed to every loop strategy. It is `Serialize + Deserialize` and is persisted, so runs recorded before this change must still deserialize — hence `#[serde(default)]`. It already carries `resolved_model_route: Option` resolved the same way and at the same point, which is the pattern to copy. + +- [ ] **Step 1: Write the failing tests** + +Add to the test module in `crates/contracts/ironclaw_loop_contracts/src/host/run_context.rs`. If the module imports a helper that builds a `LoopRunContext`, reuse it; otherwise build one with `LoopRunContext::new(...)` following the nearest existing test in that file. + +```rust + #[test] + fn run_context_defaults_to_no_resolved_context_budget() { + let context = sample_run_context(); + + assert_eq!(context.resolved_context_budget, None); + } + + #[test] + fn run_context_carries_a_resolved_context_budget() { + let budget = PromptContextTokenBudget::from_advertised_window(Some(200_000)); + let context = sample_run_context().with_resolved_context_budget(budget); + + assert_eq!(context.resolved_context_budget, Some(budget)); + } + + #[test] + fn run_context_without_a_budget_field_still_deserializes() { + // Runs recorded before this change must replay, landing on the + // compiled-in default rather than failing to deserialize. + let context = sample_run_context(); + let mut wire = serde_json::to_value(&context).expect("serialize"); + wire.as_object_mut() + .expect("object") + .remove("resolved_context_budget"); + + let restored: LoopRunContext = serde_json::from_value(wire).expect("deserialize"); + + assert_eq!(restored.resolved_context_budget, None); + } + + #[test] + fn resolved_context_budget_round_trips_through_the_wire() { + let budget = PromptContextTokenBudget::from_advertised_window(Some(1_000_000)); + let context = sample_run_context().with_resolved_context_budget(budget); + + let wire = serde_json::to_string(&context).expect("serialize"); + let restored: LoopRunContext = serde_json::from_str(&wire).expect("deserialize"); + + assert_eq!(restored.resolved_context_budget, Some(budget)); + } +``` + +If no `sample_run_context()` helper exists in that module, add one modeled on the nearest existing `LoopRunContext::new(...)` construction in the same file, and use it in all four tests. + +- [ ] **Step 2: Run the tests to verify they fail** + +Run: `cargo test -p ironclaw_loop_contracts run_context` +Expected: FAIL to compile — `no field 'resolved_context_budget'` and `no method named 'with_resolved_context_budget'`. + +- [ ] **Step 3: Add the field** + +In `crates/contracts/ironclaw_loop_contracts/src/host/run_context.rs`, in `pub struct LoopRunContext` (after `resolved_model_route` at `:247`): + +```rust + /// Prompt context budget resolved from this run's model at host + /// construction. `None` — an older serialized context, or a provider that + /// advertises no window — means the compiled-in default. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub resolved_context_budget: Option, +``` + +Add `resolved_context_budget: None,` to the struct literal in `LoopRunContext::new` (beside `resolved_model_route: None,` at `:281`). Import `PromptContextTokenBudget` at the top of the file if it is not already in scope (it lives at `crate::context_budget::PromptContextTokenBudget`). + +- [ ] **Step 4: Add the builder** + +Immediately after `with_resolved_model_route` (`:345-348`): + +```rust + pub fn with_resolved_context_budget(mut self, budget: PromptContextTokenBudget) -> Self { + self.resolved_context_budget = Some(budget); + self + } +``` + +- [ ] **Step 5: Run the tests to verify they pass** + +Run: `cargo test -p ironclaw_loop_contracts` +Expected: PASS. If other constructions of `LoopRunContext` in this crate fail to compile because of the new field, add `resolved_context_budget: None,` to each — do not change their behavior. + +- [ ] **Step 6: Commit** + +```bash +git add crates/contracts/ironclaw_loop_contracts/src/host/run_context.rs +git commit -m "feat(loop-contracts): carry a resolved context budget on LoopRunContext" +``` + +--- + +### Task 3: Thread the budget through the compaction helpers (STRUCTURAL) + +**This task must not change behavior.** It makes the budget an argument instead of a field read, so Task 4 can vary it. Every call site passes `self.prompt_context_budget`, so every decision the loop makes is byte-identical before and after. + +**Files:** +- Modify: `crates/loop/ironclaw_agent_loop/src/strategies/compaction.rs:56` (`can_evaluate`), `:77` (`trigger_at`), `:116`/`:125`/`:129`/`:140` (call sites) +- Verify only (no edit): `crates/loop/ironclaw_agent_loop/src/strategies/compaction.rs:398-416` — see Step 4 +- Modify: `crates/loop/ironclaw_agent_loop/src/strategies/active_task_compaction.rs:48`/`:62`/`:72` (call sites) + +**Interfaces:** +- Consumes: nothing new. +- Produces: `DefaultCompactionStrategy::can_evaluate(&self, state: &LoopExecutionState, budget: PromptContextTokenBudget) -> bool` and `DefaultCompactionStrategy::trigger_at(&self, state: &LoopExecutionState, budget: PromptContextTokenBudget, drop_through_seq: u64) -> CompactionDecision`. Both stay `pub(super)`. + +**Context you need:** `DefaultCompactionStrategy` decides *when* the loop should summarize old transcript. It reads `self.prompt_context_budget.visible_transcript_tokens()` in exactly two places — `can_evaluate` (is the observed prompt over the threshold?) and `trigger_at` (what threshold do we record as the effectiveness baseline?). `ActiveTaskPreservingCompactionStrategy` wraps it via a `base` field and calls both. There are exactly 7 call sites across the 2 files (`compaction.rs:116,125,129,140`; `active_task_compaction.rs:48,62,72`) and **no test calls either helper directly** — Step 4 verifies that. + +- [ ] **Step 1: Change the two helper signatures** + +In `crates/loop/ironclaw_agent_loop/src/strategies/compaction.rs`, change `can_evaluate` (`:56`) from reading the field to taking a parameter: + +```rust + pub(super) fn can_evaluate( + &self, + state: &LoopExecutionState, + budget: PromptContextTokenBudget, + ) -> bool { + if state.compaction_prompt.message_index.is_empty() { + return false; + } + let threshold = budget.visible_transcript_tokens(); +``` + +Leave the rest of the function body unchanged. + +Change `trigger_at` (`:77`) the same way: + +```rust + pub(super) fn trigger_at( + &self, + state: &LoopExecutionState, + budget: PromptContextTokenBudget, + drop_through_seq: u64, + ) -> CompactionDecision { +``` + +and inside it, replace `self.prompt_context_budget.visible_transcript_tokens()` with `budget.visible_transcript_tokens()`. + +Keep the `prompt_context_budget` field on the struct — Task 4 uses it as the fallback. + +- [ ] **Step 2: Update the call sites in `compaction.rs`** + +In `DefaultCompactionStrategy::should_compact` (`:111-142`), pass `self.prompt_context_budget` at each of the four sites: + +```rust + if !self.can_evaluate(state, self.prompt_context_budget) { +``` + +and, at `:125`, `:129`, and `:140`: + +```rust + .map(|sequence| self.trigger_at(state, self.prompt_context_budget, sequence)) +``` + +- [ ] **Step 3: Update the call sites in `active_task_compaction.rs`** + +In `ActiveTaskPreservingCompactionStrategy::should_compact` (`:43-74`), at `:48`: + +```rust + if !self.base.can_evaluate(state, self.base.prompt_context_budget) { +``` + +and at `:62` and `:72`: + +```rust + .map(|sequence| self.base.trigger_at(state, self.base.prompt_context_budget, sequence)) +``` + +- [ ] **Step 4: Confirm no test needs editing** + +Run: `rg -n "can_evaluate\(|trigger_at\(" crates/loop/ironclaw_agent_loop/` +Expected: only the definitions (`compaction.rs:56`, `:77`) and the internal call sites you just changed (`compaction.rs:116,125,129,140`; `active_task_compaction.rs:48,62,72`). **No test calls either helper directly** — note that `can_evaluate_skips_when_visible_threshold_equals_preserve_tail` (`compaction.rs:398-416`) is named after `can_evaluate` but actually asserts on `strategy.should_compact(&state, &context)`. If this grep shows a test-level call, stop: the refactor's blast radius is larger than this plan assumes. + +This zero-test-churn result is the proof that Task 3 is purely internal. + +- [ ] **Step 5: Run the full crate suite to prove nothing moved** + +Run: `cargo test -p ironclaw_agent_loop` +Expected: PASS, with **zero test edits in this task**. Compaction behavior is unchanged; if any test fails, you changed behavior — revert and redo. In particular the family digest tests must still pass, because the fingerprint string is untouched in this task. + +- [ ] **Step 6: Verify clippy is clean** + +Run: `cargo clippy -p ironclaw_agent_loop --tests -- -D warnings` +Expected: no warnings. + +- [ ] **Step 7: Commit** + +```bash +git add crates/loop/ironclaw_agent_loop/src/strategies/compaction.rs \ + crates/loop/ironclaw_agent_loop/src/strategies/active_task_compaction.rs +git commit -m "refactor(agent-loop): pass the compaction budget as an argument + +Structural only. Every call site passes the strategy's own +prompt_context_budget, so every compaction decision is identical." +``` + +--- + +### Task 4: Let the run context override the compaction ceiling + +**Files:** +- Modify: `crates/loop/ironclaw_agent_loop/src/strategies/compaction.rs` (new `effective_budget` + `should_compact`), `crates/loop/ironclaw_agent_loop/src/strategies/active_task_compaction.rs:43-74` +- Modify: `crates/loop/ironclaw_agent_loop/src/families/mod.rs:35` and `:53-56`, `crates/loop/ironclaw_agent_loop/src/families/subagent.rs:20` + its digest const, `crates/loop/ironclaw_agent_loop/src/families/unbound.rs:24`/`:45` + both digest consts +- Test: `crates/loop/ironclaw_agent_loop/src/strategies/compaction.rs` test module + +**Interfaces:** +- Consumes: `LoopRunContext.resolved_context_budget` (Task 2), the parameterized helpers (Task 3). +- Produces: compaction that honors a per-run budget. No new public API. + +**Context you need:** `CompactionStrategy::should_compact` already receives `ctx: &LoopRunContext` and both implementations currently name it `_ctx` and ignore it. This task uses it. The literal `context_limit=128000` also appears in four hand-typed "replay fingerprint" strings that get BLAKE3-hashed into family identity digests; those must change to say the ceiling is now run-scoped, and the digest constants must be recomputed. + +- [ ] **Step 1: Write the failing tests** + +Add to the test module in `crates/loop/ironclaw_agent_loop/src/strategies/compaction.rs`. Follow the nearest existing test in that module for how to build a `LoopExecutionState` with a given observed prompt size and a `LoopRunContext`; reuse its helpers rather than inventing new ones. + +```rust + #[test] + fn run_context_budget_overrides_the_strategy_default_for_compaction() { + // The strategy's own budget would not trigger, but the run's model + // has a far smaller real window, so this prompt is already over. + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(128_000, 20_000, 0), + preserve_tail_tokens: 10, + deadline_ms: 30_000, + }; + let ctx = test_run_context("compaction-budget-override") + .with_resolved_context_budget(PromptContextTokenBudget::new(40_000, 5_000, 0)); + let state = state_with_observed_prompt_tokens(50_000, &ctx); + + assert!(matches!( + strategy.should_compact(&state, &ctx), + CompactionDecision::Trigger { .. } + )); + } + + #[test] + fn absent_run_context_budget_falls_back_to_the_strategy_default() { + let strategy = DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(128_000, 20_000, 0), + preserve_tail_tokens: 10, + deadline_ms: 30_000, + }; + let ctx = test_run_context("compaction-budget-fallback"); + let state = state_with_observed_prompt_tokens(50_000, &ctx); + + assert_eq!( + strategy.should_compact(&state, &ctx), + CompactionDecision::Skip + ); + } +``` + +Add the equivalent pair to `active_task_compaction.rs`'s test module against `ActiveTaskPreservingCompactionStrategy`, so the wrapper is proven too and not just the base: + +```rust + #[test] + fn active_task_strategy_honors_the_run_context_budget() { + let strategy = ActiveTaskPreservingCompactionStrategy::from(DefaultCompactionStrategy { + prompt_context_budget: PromptContextTokenBudget::new(128_000, 20_000, 0), + preserve_tail_tokens: 10, + deadline_ms: 30_000, + }); + let ctx = test_run_context("active-task-budget-override") + .with_resolved_context_budget(PromptContextTokenBudget::new(40_000, 5_000, 0)); + let state = state_with_observed_prompt_tokens(50_000, &ctx); + + assert!(matches!( + strategy.should_compact(&state, &ctx), + CompactionDecision::Trigger { .. } + )); + } +``` + +`state_with_observed_prompt_tokens` does not exist yet. Write it in the test module using the real helpers the neighbouring test at `compaction.rs:398-416` already uses — `CompactionPromptSnapshot::from_message_index` derives `observed_prompt_tokens` as the sum of the entries' `estimated_tokens` (`state/compaction.rs:129-138`), so one entry carrying the whole figure is enough: + +```rust + fn state_with_observed_prompt_tokens( + tokens: u64, + context: &LoopRunContext, + ) -> LoopExecutionState { + let mut state = LoopExecutionState::initial_for_run(context); + state.compaction_prompt = + CompactionPromptSnapshot::from_message_index(vec![MessageIndexEntry { + sequence: 1, + kind: IndexedMessageKind::User, + estimated_tokens: tokens, + }]); + state + } +``` + +Build the context with `crate::test_support::test_run_context("