From 133d14991ce8e9c103f655d99386f86ba234c1c7 Mon Sep 17 00:00:00 2001 From: Illia Polosukhin Date: Sat, 11 Jul 2026 03:33:02 +0000 Subject: [PATCH 1/6] feat(reborn): per-run token usage + USD cost on OpenAI-compatible Responses API MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The OpenAI-compatible Responses (and Chat) surface hard-coded `usage: None` and Reborn captured no per-run token totals anywhere — token counts existed only per-LLM-call, were spent transiently on budget/stop heuristics, and were never aggregated, persisted, or projected. Callers had no view of tokens or cost. This lands the shared per-run usage backbone and surfaces it as `usage` + an IronClaw `cost` extension. - ironclaw_turns: LoopModelUsage gains cache token fields (previously dropped at the gateway) + add_assign/total_tokens. New additive `model_usage: Option` on TurnRunState/TurnRunRecord/RunRecord rides the JSON-blob snapshot like resolved_model_route — no table, column, migration, or per-backend change. The dead usage_summary_ref/LoopUsageSummaryRef (never set/read; pointed at a store that never existed) is replaced by an inline model_usage value on LoopCompleted/LoopFailed, extracted in LoopExitApplier::apply and accumulated onto the run record at the terminal transition (block/resume legs sum). - ironclaw_runner: reply paths preserve provider cache token counts. - ironclaw_agent_loop: LoopExecutionState accumulates cumulative usage at both assistant-reply finalize paths; completed/failed exits carry it. - ironclaw_reborn_openai_compat: OpenAiResponseUsage/OpenAiUsage gain input_tokens_details.cached_tokens (OpenAI-standard) + a namespaced cost object (input/cached-input/output/total USD, decimal strings). No new rust_decimal/ironclaw_llm dependency on the route crate. - ironclaw_reborn_composition: the projection reader reads persisted model_usage via get_run_state, prices it through ironclaw_llm::costs (cache-read at the provider discount, unknown models fall back to default rate not zero), and fills usage. Cost gated behind root-llm-provider; tokens always reported. Follow-up (Phase 2): route the requested model through the turn so model selection actually takes effect (and cost prices the model that ran). Tests: token-breakout, cost pricing incl. cache discount, and unknown-model default-rate fallback; existing responses/chat/DTO/streaming contract suites updated for the new optional usage fields. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../src/executor/assistant_reply.rs | 1 + .../src/executor/exit_helpers.rs | 6 +- .../src/executor/loop_exit.rs | 1 + crates/ironclaw_agent_loop/src/state.rs | 23 ++ .../src/budget_accountant.rs | 1 + .../src/cancellation_port/tests.rs | 1 + .../src/subagent_spawn_port/tests.rs | 1 + .../src/auth_continuation.rs | 1 + .../tests/approval_interaction_contract.rs | 1 + .../tests/auth_interaction_contract.rs | 1 + .../tests/reborn_services_contract.rs | 2 + .../trigger_poller/active_run_lookup.rs | 1 + .../src/blocked_auth_resume.rs | 1 + .../src/factory/auth_tests.rs | 1 + .../src/llm_admin/openai_compat_serve.rs | 199 +++++++++++++++++- .../llm_admin/openai_compat_serve/tests.rs | 64 ++++++ .../src/projection/tests.rs | 1 + .../src/slack/slack_delivery.rs | 1 + .../src/slack/slack_serve/e2e_tests.rs | 1 + .../src/test_support/budget_gateway.rs | 1 + .../tests/auth_lifecycle.rs | 1 + .../ironclaw_reborn_openai_compat/src/chat.rs | 17 +- .../ironclaw_reborn_openai_compat/src/cost.rs | 34 +++ .../ironclaw_reborn_openai_compat/src/lib.rs | 12 +- .../src/responses.rs | 17 +- .../tests/chat_workflow_handlers_contract.rs | 2 + .../tests/dto_contract.rs | 2 + .../responses_workflow_handlers_contract.rs | 2 + .../tests/streaming_handlers_contract.rs | 2 + .../src/loop_exit_applier/tests/mod.rs | 20 +- .../src/loop_exit_applier/tests/support.rs | 5 +- crates/ironclaw_runner/src/model_gateway.rs | 8 + .../src/subagent/await_edge/resolver.rs | 1 + .../ironclaw_runner/src/text_loop_driver.rs | 2 +- .../ironclaw_runner/src/turn_run_executor.rs | 6 +- .../src/turn_scheduler/tests.rs | 1 + .../tests/concurrent_workers.rs | 2 +- .../tests/hooks_integration.rs | 1 + .../ironclaw_runner/tests/loop_driver_host.rs | 5 +- .../tests/loop_milestone_event_projection.rs | 1 + crates/ironclaw_turns/src/events.rs | 2 + .../src/filesystem_store/projection.rs | 1 + .../src/filesystem_store/runner_lease.rs | 1 + crates/ironclaw_turns/src/ids.rs | 1 - crates/ironclaw_turns/src/lib.rs | 6 +- crates/ironclaw_turns/src/loop_exit.rs | 33 ++- .../ironclaw_turns/src/loop_exit/tests/mod.rs | 32 ++- crates/ironclaw_turns/src/memory/mod.rs | 19 +- crates/ironclaw_turns/src/run_profile/host.rs | 34 +++ crates/ironclaw_turns/src/runner.rs | 5 + crates/ironclaw_turns/src/status.rs | 6 + crates/ironclaw_turns/src/store.rs | 6 + .../tests/agent_loop_host_contract.rs | 5 +- .../tests/checkpoint_state_store_contract.rs | 4 + .../tests/filesystem_turn_state_contract.rs | 2 + .../tests/per_inbound_type_concurrency_cap.rs | 2 + .../tests/per_user_concurrency_cap.rs | 2 + .../tests/retry_failed_turn_store_contract.rs | 2 + .../tests/turn_coordinator_contract.rs | 3 + 59 files changed, 551 insertions(+), 65 deletions(-) create mode 100644 crates/ironclaw_reborn_openai_compat/src/cost.rs diff --git a/crates/ironclaw_agent_loop/src/executor/assistant_reply.rs b/crates/ironclaw_agent_loop/src/executor/assistant_reply.rs index dca0f7353cb..cd7c75233df 100644 --- a/crates/ironclaw_agent_loop/src/executor/assistant_reply.rs +++ b/crates/ironclaw_agent_loop/src/executor/assistant_reply.rs @@ -40,6 +40,7 @@ impl ExecutorStage for AssistantReplyStage { })?; state.assistant_refs.push(reply_ref.clone()); state.recent_output_token_counts.push(output_tokens); + state.accumulate_model_usage(input.usage); state = match CheckpointStage.cancel_if_requested(ctx, state).await? { CancelCheck::Continue(state) => *state, CancelCheck::Exit(exit) => return Ok(TurnCompletedStep::Exit(exit)), diff --git a/crates/ironclaw_agent_loop/src/executor/exit_helpers.rs b/crates/ironclaw_agent_loop/src/executor/exit_helpers.rs index 330a46d9e9d..d871d873d27 100644 --- a/crates/ironclaw_agent_loop/src/executor/exit_helpers.rs +++ b/crates/ironclaw_agent_loop/src/executor/exit_helpers.rs @@ -20,12 +20,13 @@ pub(super) fn completed_exit( } else { LoopCompletionKind::NoReply }; + let model_usage = state.cumulative_model_usage; Ok(LoopExit::Completed(LoopCompleted { completion_kind, reply_message_refs: state.assistant_refs, result_refs: state.result_refs, final_checkpoint_id, - usage_summary_ref: None, + model_usage, exit_id: exit_id(host, "completed")?, })) } @@ -37,10 +38,11 @@ pub(super) fn failed_exit( checkpoint_id: Option, details: FailedExitDetails, ) -> Result { + let model_usage = state.cumulative_model_usage; Ok(LoopExit::Failed(LoopFailed { reason_kind, checkpoint_id, - usage_summary_ref: None, + model_usage, diagnostic_ref: details.diagnostic_ref, exit_id: exit_id(host, "failed")?, explanation_message_refs: failure_message_refs(&state, details.explanation_message_ref), diff --git a/crates/ironclaw_agent_loop/src/executor/loop_exit.rs b/crates/ironclaw_agent_loop/src/executor/loop_exit.rs index 5271316d649..f49cb21b394 100644 --- a/crates/ironclaw_agent_loop/src/executor/loop_exit.rs +++ b/crates/ironclaw_agent_loop/src/executor/loop_exit.rs @@ -132,6 +132,7 @@ pub(super) async fn try_final_answer_nudge( Err(error) => return nudge_bail("transcript", error), }; state.recent_output_token_counts.push(output_tokens); + state.accumulate_model_usage(usage); Ok(Some(reply_ref)) } // Admission rejected it (empty / artifact) — give up; the caller diff --git a/crates/ironclaw_agent_loop/src/state.rs b/crates/ironclaw_agent_loop/src/state.rs index f87db9ae5b3..dcd0434c226 100644 --- a/crates/ironclaw_agent_loop/src/state.rs +++ b/crates/ironclaw_agent_loop/src/state.rs @@ -70,6 +70,14 @@ pub struct LoopExecutionState { /// (#3841 follow-up F1). pub recent_output_token_counts: BoundedRing, + /// Cumulative provider-reported token usage across this run's model calls, + /// summed from `LoopModelResponse::usage`. Carried into the terminal + /// `LoopExit` so the run record persists per-run usage for the + /// OpenAI-compatible surfaces. `None` until the first call that reports + /// usage (replay stubs and usage-less providers leave it `None`). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub cumulative_model_usage: Option, + /// Count of final-answer nudges issued this run (driver-specific nudge, /// gated by `SteeringPolicy.allow_driver_specific_nudges`). Capped so the /// loop can't issue unbounded extra model calls. `#[serde(default)]` keeps @@ -243,6 +251,20 @@ impl PendingExternalToolResume { } impl LoopExecutionState { + /// Accumulate one model call's reported usage into the run's cumulative + /// total. No-op when the call reported no usage (replay stubs, usage-less + /// providers), leaving any prior total intact. + pub(crate) fn accumulate_model_usage( + &mut self, + usage: Option, + ) { + if let Some(usage) = usage { + self.cumulative_model_usage + .get_or_insert_with(Default::default) + .add_assign(&usage); + } + } + /// Builds the initial state at the start of a fresh run. /// /// The `input_cursor` field is populated via @@ -263,6 +285,7 @@ impl LoopExecutionState { seen_capability_output_digests: BoundedRing::new(), recent_failure_kinds: BoundedRing::new(), recent_output_token_counts: BoundedRing::new(), + cumulative_model_usage: None, final_answer_nudges_used: 0, context_state: ContextStrategyState::default(), capability_state: CapabilityStrategyState::default(), diff --git a/crates/ironclaw_loop_support/src/budget_accountant.rs b/crates/ironclaw_loop_support/src/budget_accountant.rs index 51f1acd9787..6ab26a95ec1 100644 --- a/crates/ironclaw_loop_support/src/budget_accountant.rs +++ b/crates/ironclaw_loop_support/src/budget_accountant.rs @@ -1091,6 +1091,7 @@ mod tests { usage: Some(ironclaw_turns::run_profile::LoopModelUsage { input_tokens: 7, output_tokens: 3, + ..Default::default() }), }; accountant diff --git a/crates/ironclaw_loop_support/src/cancellation_port/tests.rs b/crates/ironclaw_loop_support/src/cancellation_port/tests.rs index 5ba146c6fc2..5a5b447c3c8 100644 --- a/crates/ironclaw_loop_support/src/cancellation_port/tests.rs +++ b/crates/ironclaw_loop_support/src/cancellation_port/tests.rs @@ -164,6 +164,7 @@ fn test_run_state(status: TurnStatus) -> TurnRunState { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref: None, diff --git a/crates/ironclaw_loop_support/src/subagent_spawn_port/tests.rs b/crates/ironclaw_loop_support/src/subagent_spawn_port/tests.rs index 19cb90b1f65..5360607ffff 100644 --- a/crates/ironclaw_loop_support/src/subagent_spawn_port/tests.rs +++ b/crates/ironclaw_loop_support/src/subagent_spawn_port/tests.rs @@ -1138,6 +1138,7 @@ fn turn_record(run_context: &LoopRunContext, subagent_depth: u32) -> TurnRunReco status: TurnStatus::Queued, profile: TurnRunProfile::from_resolved(run_context.resolved_run_profile.clone()), resolved_model_route: None, + model_usage: None, checkpoint_id: None, gate_ref: None, blocked_activity_id: None, diff --git a/crates/ironclaw_product_workflow/src/auth_continuation.rs b/crates/ironclaw_product_workflow/src/auth_continuation.rs index 438c54073bc..05d3a94cb31 100644 --- a/crates/ironclaw_product_workflow/src/auth_continuation.rs +++ b/crates/ironclaw_product_workflow/src/auth_continuation.rs @@ -547,6 +547,7 @@ mod tests { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref: gate_ref.map(|value| GateRef::new(value).unwrap()), diff --git a/crates/ironclaw_product_workflow/tests/approval_interaction_contract.rs b/crates/ironclaw_product_workflow/tests/approval_interaction_contract.rs index 52a9a401cd0..242f0898dbc 100644 --- a/crates/ironclaw_product_workflow/tests/approval_interaction_contract.rs +++ b/crates/ironclaw_product_workflow/tests/approval_interaction_contract.rs @@ -593,6 +593,7 @@ impl TurnCoordinator for FakeTurnCoordinator { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref: self.gate_ref.lock().expect("lock").clone(), diff --git a/crates/ironclaw_product_workflow/tests/auth_interaction_contract.rs b/crates/ironclaw_product_workflow/tests/auth_interaction_contract.rs index 5331b0c0615..4ee6bd6c240 100644 --- a/crates/ironclaw_product_workflow/tests/auth_interaction_contract.rs +++ b/crates/ironclaw_product_workflow/tests/auth_interaction_contract.rs @@ -374,6 +374,7 @@ impl TurnCoordinator for RecordingTurnCoordinator { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref: self.gate_ref.lock().expect("lock").clone(), diff --git a/crates/ironclaw_product_workflow/tests/reborn_services_contract.rs b/crates/ironclaw_product_workflow/tests/reborn_services_contract.rs index 2b1dfe3ce40..80a98cbdf57 100644 --- a/crates/ironclaw_product_workflow/tests/reborn_services_contract.rs +++ b/crates/ironclaw_product_workflow/tests/reborn_services_contract.rs @@ -537,6 +537,7 @@ impl TurnCoordinator for FakeTurnCoordinator { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref, @@ -641,6 +642,7 @@ impl TurnCoordinator for BlockingSubmitCoordinator { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref: None, diff --git a/crates/ironclaw_reborn_composition/src/automation/trigger_poller/active_run_lookup.rs b/crates/ironclaw_reborn_composition/src/automation/trigger_poller/active_run_lookup.rs index 39da1c72020..35c26e42cd0 100644 --- a/crates/ironclaw_reborn_composition/src/automation/trigger_poller/active_run_lookup.rs +++ b/crates/ironclaw_reborn_composition/src/automation/trigger_poller/active_run_lookup.rs @@ -403,6 +403,7 @@ mod tests { status, profile: TurnRunProfile::from_resolved(resolved_run_profile()), resolved_model_route: None, + model_usage: None, checkpoint_id: None, gate_ref: None, blocked_activity_id: None, diff --git a/crates/ironclaw_reborn_composition/src/blocked_auth_resume.rs b/crates/ironclaw_reborn_composition/src/blocked_auth_resume.rs index 120dac9f1b6..ed3b803af1b 100644 --- a/crates/ironclaw_reborn_composition/src/blocked_auth_resume.rs +++ b/crates/ironclaw_reborn_composition/src/blocked_auth_resume.rs @@ -385,6 +385,7 @@ mod tests { status: TurnStatus::BlockedAuth, profile: TurnRunProfile::from_resolved(resolved_run_profile()), resolved_model_route: None, + model_usage: None, checkpoint_id: None, gate_ref: Some(GateRef::new(format!("gate-{run_id}")).expect("gate ref")), blocked_activity_id: None, diff --git a/crates/ironclaw_reborn_composition/src/factory/auth_tests.rs b/crates/ironclaw_reborn_composition/src/factory/auth_tests.rs index 8ef74665b39..48162c4c320 100644 --- a/crates/ironclaw_reborn_composition/src/factory/auth_tests.rs +++ b/crates/ironclaw_reborn_composition/src/factory/auth_tests.rs @@ -87,6 +87,7 @@ fn auth_error_mapping_run_state(request: &GetRunStateRequest) -> TurnRunState { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref: Some(GateRef::new("gate:auth-error").unwrap()), // safety: fixed test gate literal is valid. diff --git a/crates/ironclaw_reborn_composition/src/llm_admin/openai_compat_serve.rs b/crates/ironclaw_reborn_composition/src/llm_admin/openai_compat_serve.rs index 339a92e1bdf..bdc7ce27c76 100644 --- a/crates/ironclaw_reborn_composition/src/llm_admin/openai_compat_serve.rs +++ b/crates/ironclaw_reborn_composition/src/llm_admin/openai_compat_serve.rs @@ -44,9 +44,11 @@ use ironclaw_reborn_openai_compat::{ OpenAiCompatRouterState, OpenAiResponseErrorObject, OpenAiResponseId, OpenAiResponseObject, OpenAiResponseOutputItem, OpenAiResponseOutputItemStatus, OpenAiResponseProjection, OpenAiResponseProjectionStreamRequest, OpenAiResponseReadRequest, OpenAiResponseStatus, - OpenAiResponseWaitRequest, OpenAiResponsesMessageRole, OpenAiResponsesProjectionReader, - OpenAiResponsesWorkflow, openai_compat_router_with_state, openai_compat_routes, + OpenAiResponseUsage, OpenAiResponseWaitRequest, OpenAiResponsesMessageRole, + OpenAiResponsesProjectionReader, OpenAiResponsesWorkflow, openai_compat_router_with_state, + openai_compat_routes, }; +use ironclaw_reborn_openai_compat::{OpenAiCompatCost, OpenAiResponseInputTokensDetails}; use ironclaw_reborn_openai_compat::{ OpenAiCompatExternalToolResume, OpenAiCompatExternalToolResumeRequest, OpenAiCompatExternalToolSpec, OpenAiCompatExternalToolStore, OpenAiCompatTurnRunRef, @@ -62,7 +64,7 @@ use ironclaw_threads::{ use ironclaw_turns::{ ExternalToolCatalog, ExternalToolCatalogError, ExternalToolSpec, GateRef, GetRunStateRequest, IdempotencyKey, ResumeTurnPrecondition, ResumeTurnRequest, TurnCoordinator, TurnError, - TurnErrorCategory, TurnRunId, TurnScope, TurnStatus, + TurnErrorCategory, TurnRunId, TurnScope, TurnStatus, run_profile::LoopModelUsage, }; use sha2::{Digest, Sha256}; @@ -566,6 +568,35 @@ impl OpenAiResponsesThreadProjectionReader { } } + /// Read the run's cumulative token usage from persisted run state and render + /// it as an OpenAI-compatible `usage` object (with USD cost). Best-effort: + /// returns `None` when no coordinator is wired, the run can't be read, or the + /// run reported no usage (replay stubs, usage-less providers). The model is + /// priced by the caller's requested model — which, once model selection + /// routes it, is also the model that actually ran. + async fn read_run_usage( + &self, + projection_read: &ProjectionReadRequest, + actor_scope: &OpenAiCompatActorScope, + submitted_run_id: &str, + requested_model: &str, + ) -> Option { + let coordinator = self.turn_coordinator.as_ref()?; + let run_id = TurnRunId::parse(submitted_run_id).ok()?; + let state = coordinator + .get_run_state(GetRunStateRequest { + scope: openai_compat_resume_turn_scope( + actor_scope, + projection_read.scope.thread_id.clone(), + ), + run_id, + }) + .await + .ok()?; + let usage = state.model_usage?; + Some(response_usage_from_model_usage(&usage, requested_model)) + } + /// Project a single response run into OpenAI Responses `output` items: every /// tool call paired with its raw output, then the run's finalized assistant /// message. Tool items always precede the assistant message regardless of @@ -883,6 +914,14 @@ impl OpenAiResponsesProjectionReader for OpenAiResponsesThreadProjectionReader { .run_completed(&request.projection_read, submitted_run_id.clone()) .await? { + let usage = self + .read_run_usage( + &request.projection_read, + &request.actor_scope, + &submitted_run_id, + &request.requested_model, + ) + .await; let projection = self .read_run_output( &request.projection_read, @@ -896,6 +935,7 @@ impl OpenAiResponsesProjectionReader for OpenAiResponsesThreadProjectionReader { request.requested_model, OpenAiResponseStatus::Completed, projection.items, + usage, ))); } let projected = self @@ -924,12 +964,21 @@ impl OpenAiResponsesProjectionReader for OpenAiResponsesThreadProjectionReader { projection .items .extend(self.pending_external_tool_items(&submitted_run_id).await?); + let usage = self + .read_run_usage( + &request.projection_read, + &request.actor_scope, + &submitted_run_id, + &request.requested_model, + ) + .await; return Ok(OpenAiResponseProjection::new(response_object( request.public_id, request.mapping.created_at, request.requested_model, OpenAiResponseStatus::Completed, projection.items, + usage, ))); } if self @@ -954,12 +1003,21 @@ impl OpenAiResponsesProjectionReader for OpenAiResponsesThreadProjectionReader { projection .items .extend(self.pending_external_tool_items(&submitted_run_id).await?); + let usage = self + .read_run_usage( + &request.projection_read, + &request.actor_scope, + &submitted_run_id, + &request.requested_model, + ) + .await; return Ok(OpenAiResponseProjection::new(response_object( request.public_id, request.mapping.created_at, request.requested_model, OpenAiResponseStatus::Completed, projection.items, + usage, ))); } if let Some(status) = projected.status @@ -968,12 +1026,21 @@ impl OpenAiResponsesProjectionReader for OpenAiResponsesThreadProjectionReader { OpenAiResponseStatus::Failed | OpenAiResponseStatus::Cancelled ) { + let usage = self + .read_run_usage( + &request.projection_read, + &request.actor_scope, + &submitted_run_id, + &request.requested_model, + ) + .await; return Ok(OpenAiResponseProjection::new(response_object( request.public_id, request.mapping.created_at, request.requested_model, status, Vec::new(), + usage, ))); } tokio::time::sleep(self.poll_interval).await; @@ -1015,14 +1082,25 @@ impl OpenAiResponsesProjectionReader for OpenAiResponsesThreadProjectionReader { projection .items .extend(self.pending_external_tool_items(&submitted_run_id).await?); + let model = request + .requested_model + .clone() + .unwrap_or_else(|| "reborn".to_string()); + let usage = self + .read_run_usage( + &request.projection_read, + &request.actor_scope, + &submitted_run_id, + &model, + ) + .await; return Ok(response_object( request.public_id, request.mapping.created_at, - request - .requested_model - .unwrap_or_else(|| "reborn".to_string()), + model, OpenAiResponseStatus::Completed, projection.items, + usage, )); } let status = match ( @@ -1036,14 +1114,25 @@ impl OpenAiResponsesProjectionReader for OpenAiResponsesThreadProjectionReader { (false, Some(status)) => status, (false, None) => OpenAiResponseStatus::InProgress, }; + let model = request + .requested_model + .clone() + .unwrap_or_else(|| "reborn".to_string()); + let usage = self + .read_run_usage( + &request.projection_read, + &request.actor_scope, + &submitted_run_id, + &model, + ) + .await; Ok(response_object( request.public_id, request.mapping.created_at, - request - .requested_model - .unwrap_or_else(|| "reborn".to_string()), + model, status, projection.items, + usage, )) } } @@ -1388,6 +1477,7 @@ fn response_object( model: String, status: OpenAiResponseStatus, output: Vec, + usage: Option, ) -> OpenAiResponseObject { let error = if matches!(status, OpenAiResponseStatus::Failed) { Some(OpenAiResponseErrorObject::from_kind( @@ -1405,10 +1495,99 @@ fn response_object( output, error, incomplete_details: None, - usage: None, + usage, } } +/// Build the OpenAI-compatible `usage` object from a run's cumulative token +/// totals, pricing it (when the LLM cost table is compiled in) for the given +/// effective model. Token counts are always reported; `cost` is present only +/// under `root-llm-provider`. +fn response_usage_from_model_usage(usage: &LoopModelUsage, model: &str) -> OpenAiResponseUsage { + // OpenAI reports total input (including cache) as `input_tokens`, with the + // cached subset broken out under `input_tokens_details`. + let total_input = usage + .input_tokens + .saturating_add(usage.cache_read_input_tokens) + .saturating_add(usage.cache_creation_input_tokens); + OpenAiResponseUsage { + input_tokens: total_input, + output_tokens: usage.output_tokens, + total_tokens: total_input.saturating_add(usage.output_tokens), + input_tokens_details: (usage.cache_read_input_tokens > 0).then_some( + OpenAiResponseInputTokensDetails { + cached_tokens: usage.cache_read_input_tokens, + }, + ), + cost: response_cost_from_model_usage(usage, model), + } +} + +/// Price a run's token usage in USD for the given model. Fresh input and +/// cache-creation tokens bill at the input rate; cache-read tokens bill at the +/// provider's cache-read discount; output at the output rate. Unknown models +/// fall back to the cost table's default (≈GPT-4o), so a new paid model never +/// silently prices at zero. +#[cfg(feature = "root-llm-provider")] +fn response_cost_from_model_usage(usage: &LoopModelUsage, model: &str) -> Option { + use rust_decimal::Decimal; + + let (input_rate, output_rate) = + ironclaw_llm::costs::model_cost(model).unwrap_or_else(ironclaw_llm::costs::default_cost); + let discount = cache_read_discount_for_model(model); + let billable_input = Decimal::from( + usage + .input_tokens + .saturating_add(usage.cache_creation_input_tokens), + ); + let input_cost = billable_input * input_rate; + let cached_input_cost = if discount > Decimal::ONE { + Decimal::from(usage.cache_read_input_tokens) * input_rate / discount + } else { + Decimal::from(usage.cache_read_input_tokens) * input_rate + }; + let output_cost = Decimal::from(usage.output_tokens) * output_rate; + let total = input_cost + cached_input_cost + output_cost; + Some(OpenAiCompatCost { + input_cost_usd: format_usd(input_cost), + cached_input_cost_usd: format_usd(cached_input_cost), + output_cost_usd: format_usd(output_cost), + total_cost_usd: format_usd(total), + currency: OpenAiCompatCost::USD.to_string(), + }) +} + +#[cfg(not(feature = "root-llm-provider"))] +fn response_cost_from_model_usage( + _usage: &LoopModelUsage, + _model: &str, +) -> Option { + None +} + +/// Cache-read discount divisor by model family, mirroring the provider defaults +/// documented on `LlmProvider::cache_read_discount` (Anthropic 10× i.e. 90% off, +/// OpenAI 2× i.e. 50% off, others no discount). +#[cfg(feature = "root-llm-provider")] +fn cache_read_discount_for_model(model: &str) -> rust_decimal::Decimal { + use rust_decimal::Decimal; + let lower = model.to_ascii_lowercase(); + if lower.contains("claude") { + Decimal::from(10) + } else if lower.contains("gpt") || lower.starts_with("o1") || lower.starts_with("o3") { + Decimal::from(2) + } else { + Decimal::ONE + } +} + +/// Format a USD `Decimal` for the wire: trimmed of trailing zeros, never in +/// scientific notation. +#[cfg(feature = "root-llm-provider")] +fn format_usd(amount: rust_decimal::Decimal) -> String { + amount.normalize().to_string() +} + fn thread_scope_from_projection_read( projection_read: &ProjectionReadRequest, ) -> Result { diff --git a/crates/ironclaw_reborn_composition/src/llm_admin/openai_compat_serve/tests.rs b/crates/ironclaw_reborn_composition/src/llm_admin/openai_compat_serve/tests.rs index 2fefbe6379b..0b098278113 100644 --- a/crates/ironclaw_reborn_composition/src/llm_admin/openai_compat_serve/tests.rs +++ b/crates/ironclaw_reborn_composition/src/llm_admin/openai_compat_serve/tests.rs @@ -887,6 +887,7 @@ fn turn_run_state( resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref: None, @@ -1221,3 +1222,66 @@ fn model_entries_fall_back_to_default_model_when_no_active_selection() { assert_eq!(entries[0].id, "claude-opus-4"); assert_eq!(entries[0].owned_by.as_deref(), Some("anthropic")); } + +#[test] +fn response_usage_reports_total_input_including_cache_and_breaks_out_cached_tokens() { + let usage = LoopModelUsage { + input_tokens: 1_000, + output_tokens: 500, + cache_read_input_tokens: 2_000, + cache_creation_input_tokens: 0, + }; + let built = response_usage_from_model_usage(&usage, "gpt-4o"); + // OpenAI `input_tokens` is the total input including the cached subset. + assert_eq!(built.input_tokens, 3_000); + assert_eq!(built.output_tokens, 500); + assert_eq!(built.total_tokens, 3_500); + assert_eq!( + built + .input_tokens_details + .expect("cached detail") + .cached_tokens, + 2_000 + ); +} + +#[cfg(feature = "root-llm-provider")] +#[test] +fn response_cost_prices_input_output_and_discounts_cached_tokens() { + // gpt-4o rates: input 0.0000025/tok, output 0.00001/tok; OpenAI cache-read + // discount is 2x (50% off). + let usage = LoopModelUsage { + input_tokens: 1_000, + output_tokens: 500, + cache_read_input_tokens: 2_000, + cache_creation_input_tokens: 0, + }; + let cost = response_cost_from_model_usage(&usage, "gpt-4o").expect("cost under llm feature"); + assert_eq!(cost.currency, "USD"); + // input = 1000 * 0.0000025 = 0.0025 + assert_eq!(cost.input_cost_usd, "0.0025"); + // cached = 2000 * 0.0000025 / 2 = 0.0025 + assert_eq!(cost.cached_input_cost_usd, "0.0025"); + // output = 500 * 0.00001 = 0.005 + assert_eq!(cost.output_cost_usd, "0.005"); + // total = 0.0025 + 0.0025 + 0.005 = 0.01 + assert_eq!(cost.total_cost_usd, "0.01"); +} + +#[cfg(feature = "root-llm-provider")] +#[test] +fn response_cost_falls_back_to_default_rate_for_unknown_model() { + // Unknown models must not price at zero — the cost table default (~GPT-4o) + // applies so a new paid model still bills. + let usage = LoopModelUsage { + input_tokens: 1_000, + output_tokens: 0, + cache_read_input_tokens: 0, + cache_creation_input_tokens: 0, + }; + let cost = + response_cost_from_model_usage(&usage, "some-brand-new-model-9000").expect("cost present"); + // default input rate is 0.0000025 → 1000 * 0.0000025 = 0.0025 + assert_eq!(cost.input_cost_usd, "0.0025"); + assert_eq!(cost.cached_input_cost_usd, "0"); +} diff --git a/crates/ironclaw_reborn_composition/src/projection/tests.rs b/crates/ironclaw_reborn_composition/src/projection/tests.rs index 1894ca9f636..8b386796432 100644 --- a/crates/ironclaw_reborn_composition/src/projection/tests.rs +++ b/crates/ironclaw_reborn_composition/src/projection/tests.rs @@ -405,6 +405,7 @@ fn turn_run_state( resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: chrono::Utc::now(), checkpoint_id: None, gate_ref: Some(GateRef::new("gate:auth-required").unwrap()), diff --git a/crates/ironclaw_reborn_composition/src/slack/slack_delivery.rs b/crates/ironclaw_reborn_composition/src/slack/slack_delivery.rs index fd1e2c4ef6e..ce273e7cc7d 100644 --- a/crates/ironclaw_reborn_composition/src/slack/slack_delivery.rs +++ b/crates/ironclaw_reborn_composition/src/slack/slack_delivery.rs @@ -3382,6 +3382,7 @@ mod tests { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref: scripted.gate_ref, diff --git a/crates/ironclaw_reborn_composition/src/slack/slack_serve/e2e_tests.rs b/crates/ironclaw_reborn_composition/src/slack/slack_serve/e2e_tests.rs index d10cd117487..cddec2c4c80 100644 --- a/crates/ironclaw_reborn_composition/src/slack/slack_serve/e2e_tests.rs +++ b/crates/ironclaw_reborn_composition/src/slack/slack_serve/e2e_tests.rs @@ -2656,6 +2656,7 @@ fn turn_state( resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: chrono::Utc::now(), checkpoint_id: None, gate_ref, diff --git a/crates/ironclaw_reborn_composition/src/test_support/budget_gateway.rs b/crates/ironclaw_reborn_composition/src/test_support/budget_gateway.rs index 735c5779abb..bdc12c37532 100644 --- a/crates/ironclaw_reborn_composition/src/test_support/budget_gateway.rs +++ b/crates/ironclaw_reborn_composition/src/test_support/budget_gateway.rs @@ -63,6 +63,7 @@ impl ScriptedReply { HostManagedModelResponse::assistant_reply(self.text).with_usage(LoopModelUsage { input_tokens: self.input_tokens, output_tokens: self.output_tokens, + ..LoopModelUsage::default() }) } } diff --git a/crates/ironclaw_reborn_composition/tests/auth_lifecycle.rs b/crates/ironclaw_reborn_composition/tests/auth_lifecycle.rs index 9123df9a99a..92a9c877df1 100644 --- a/crates/ironclaw_reborn_composition/tests/auth_lifecycle.rs +++ b/crates/ironclaw_reborn_composition/tests/auth_lifecycle.rs @@ -198,6 +198,7 @@ impl TurnCoordinator for LifecycleTurnCoordinator { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref: Some(self.gate_ref.clone()), diff --git a/crates/ironclaw_reborn_openai_compat/src/chat.rs b/crates/ironclaw_reborn_openai_compat/src/chat.rs index 88e8c54a486..2ee9d0aaa81 100644 --- a/crates/ironclaw_reborn_openai_compat/src/chat.rs +++ b/crates/ironclaw_reborn_openai_compat/src/chat.rs @@ -167,9 +167,24 @@ pub enum OpenAiChatFinishReason { ContentFilter, } -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct OpenAiUsage { pub prompt_tokens: u32, pub completion_tokens: u32, pub total_tokens: u32, + /// Breakdown of the prompt tokens. OpenAI-standard shape; currently carries + /// the cached-token subset. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub prompt_tokens_details: Option, + /// IronClaw extension: computed USD cost for this completion. `None` when the + /// run reported no usage. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub cost: Option, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +pub struct OpenAiPromptTokensDetails { + /// Prompt tokens served from the provider's prompt cache (a subset of + /// `prompt_tokens`). + pub cached_tokens: u32, } diff --git a/crates/ironclaw_reborn_openai_compat/src/cost.rs b/crates/ironclaw_reborn_openai_compat/src/cost.rs new file mode 100644 index 00000000000..a91564ec152 --- /dev/null +++ b/crates/ironclaw_reborn_openai_compat/src/cost.rs @@ -0,0 +1,34 @@ +//! IronClaw cost extension for OpenAI-compatible usage objects. +//! +//! OpenAI's Chat Completions and Responses APIs report token `usage` but no +//! monetary cost. IronClaw adds a namespaced `cost` object so callers can see +//! the USD spend of a response without maintaining their own price table. +//! +//! Amounts are decimal strings (e.g. `"0.000042"`), not JSON floats: per-token +//! prices are tiny and float rounding would corrupt them. The route crate keeps +//! no `rust_decimal` dependency — host composition formats the `Decimal` prices +//! into these strings when it fills the usage object. + +use serde::{Deserialize, Serialize}; + +/// Computed USD cost for a completed response. Attached to the surface's usage +/// object as the `cost` field. `currency` is always [`OpenAiCompatCost::USD`]. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct OpenAiCompatCost { + /// Cost of the (uncached) input tokens. + pub input_cost_usd: String, + /// Cost of the cached-input tokens (a subset of the input tokens, priced at + /// the provider's cache-read discount). `"0"` when nothing was cached. + pub cached_input_cost_usd: String, + /// Cost of the output tokens. + pub output_cost_usd: String, + /// Sum of the three components above. + pub total_cost_usd: String, + /// Currency of every amount above. Always `"USD"`. + pub currency: String, +} + +impl OpenAiCompatCost { + /// The only currency IronClaw prices in. + pub const USD: &'static str = "USD"; +} diff --git a/crates/ironclaw_reborn_openai_compat/src/lib.rs b/crates/ironclaw_reborn_openai_compat/src/lib.rs index 2ace82ad48e..5acddb32324 100644 --- a/crates/ironclaw_reborn_openai_compat/src/lib.rs +++ b/crates/ironclaw_reborn_openai_compat/src/lib.rs @@ -16,6 +16,7 @@ mod chat; mod chat_workflow; #[cfg(feature = "openai-compat-beta")] mod content_parts; +mod cost; mod descriptors; mod error; #[cfg(feature = "openai-compat-beta")] @@ -49,7 +50,7 @@ pub use chat::{ OpenAiChatCompletionResponse, OpenAiChatDelta, OpenAiChatFinishReason, OpenAiChatFunction, OpenAiChatMessage, OpenAiChatMessageRole, OpenAiChatStreamChoice, OpenAiChatTool, OpenAiChatToolCall, OpenAiChatToolCallDelta, OpenAiChatToolCallFunction, - OpenAiChatToolCallFunctionDelta, OpenAiChatToolKind, OpenAiUsage, + OpenAiChatToolCallFunctionDelta, OpenAiChatToolKind, OpenAiPromptTokensDetails, OpenAiUsage, }; #[cfg(feature = "openai-compat-beta")] pub use chat_workflow::{ @@ -58,6 +59,7 @@ pub use chat_workflow::{ OpenAiChatCompletionsWorkflow, OpenAiChatModelOnlyTools, OpenAiCompatAuthenticatedCaller, OpenAiCompatInboundAttachmentSubmit, }; +pub use cost::OpenAiCompatCost; pub use descriptors::{ OPENAI_COMPAT_PATTERN_CHAT_COMPLETIONS, OPENAI_COMPAT_PATTERN_MODELS_API_LIST, OPENAI_COMPAT_PATTERN_MODELS_LIST, OPENAI_COMPAT_PATTERN_RESPONSES_API_CREATE, @@ -109,10 +111,10 @@ pub use refs_storage::RebornLibSqlOpenAiCompatRefStore; #[cfg(feature = "postgres")] pub use refs_storage::RebornPostgresOpenAiCompatRefStore; pub use responses::{ - OpenAiResponseErrorObject, OpenAiResponseObject, OpenAiResponseOutputItem, - OpenAiResponseOutputItemStatus, OpenAiResponseStatus, OpenAiResponseUsage, - OpenAiResponsesCreateRequest, OpenAiResponsesInput, OpenAiResponsesInputItem, - OpenAiResponsesMessageRole, + OpenAiResponseErrorObject, OpenAiResponseInputTokensDetails, OpenAiResponseObject, + OpenAiResponseOutputItem, OpenAiResponseOutputItemStatus, OpenAiResponseStatus, + OpenAiResponseUsage, OpenAiResponsesCreateRequest, OpenAiResponsesInput, + OpenAiResponsesInputItem, OpenAiResponsesMessageRole, }; #[cfg(feature = "openai-compat-beta")] pub use responses_workflow::{ diff --git a/crates/ironclaw_reborn_openai_compat/src/responses.rs b/crates/ironclaw_reborn_openai_compat/src/responses.rs index 1570e4defa1..6a0ddd80cf7 100644 --- a/crates/ironclaw_reborn_openai_compat/src/responses.rs +++ b/crates/ironclaw_reborn_openai_compat/src/responses.rs @@ -151,11 +151,26 @@ pub struct OpenAiResponseObject { pub usage: Option, } -#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct OpenAiResponseUsage { pub input_tokens: u32, pub output_tokens: u32, pub total_tokens: u32, + /// Breakdown of the input tokens. OpenAI-standard shape; currently carries + /// the cached-token subset. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub input_tokens_details: Option, + /// IronClaw extension: computed USD cost for this response. `None` when the + /// run reported no usage (e.g. a still-running or replay-stub turn). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub cost: Option, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +pub struct OpenAiResponseInputTokensDetails { + /// Input tokens served from the provider's prompt cache (a subset of + /// `input_tokens`). + pub cached_tokens: u32, } #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] diff --git a/crates/ironclaw_reborn_openai_compat/tests/chat_workflow_handlers_contract.rs b/crates/ironclaw_reborn_openai_compat/tests/chat_workflow_handlers_contract.rs index 51d1dd4ee0d..2c5967c5ff7 100644 --- a/crates/ironclaw_reborn_openai_compat/tests/chat_workflow_handlers_contract.rs +++ b/crates/ironclaw_reborn_openai_compat/tests/chat_workflow_handlers_contract.rs @@ -1051,6 +1051,8 @@ async fn model_only_tool_call_output_shape_is_preserved() { prompt_tokens: 3, completion_tokens: 5, total_tokens: 8, + prompt_tokens_details: None, + cost: None, }), effective_model: Some("gpt-reborn-effective".to_string()), internal_refs: Some( diff --git a/crates/ironclaw_reborn_openai_compat/tests/dto_contract.rs b/crates/ironclaw_reborn_openai_compat/tests/dto_contract.rs index 5521f4588d2..bfbc41330a9 100644 --- a/crates/ironclaw_reborn_openai_compat/tests/dto_contract.rs +++ b/crates/ironclaw_reborn_openai_compat/tests/dto_contract.rs @@ -262,6 +262,8 @@ fn response_dtos_serialize_openai_shapes() { input_tokens: 3, output_tokens: 5, total_tokens: 8, + input_tokens_details: None, + cost: None, }), }; let response_json = serde_json::to_value(response).expect("response json"); diff --git a/crates/ironclaw_reborn_openai_compat/tests/responses_workflow_handlers_contract.rs b/crates/ironclaw_reborn_openai_compat/tests/responses_workflow_handlers_contract.rs index 4f893b0ca86..7a952a611c5 100644 --- a/crates/ironclaw_reborn_openai_compat/tests/responses_workflow_handlers_contract.rs +++ b/crates/ironclaw_reborn_openai_compat/tests/responses_workflow_handlers_contract.rs @@ -1311,6 +1311,8 @@ fn completed_response(id: OpenAiResponseId, text: &str) -> OpenAiResponseObject input_tokens: 3, output_tokens: 5, total_tokens: 8, + input_tokens_details: None, + cost: None, }), } } diff --git a/crates/ironclaw_reborn_openai_compat/tests/streaming_handlers_contract.rs b/crates/ironclaw_reborn_openai_compat/tests/streaming_handlers_contract.rs index 77de5147d3b..28c0144e3ab 100644 --- a/crates/ironclaw_reborn_openai_compat/tests/streaming_handlers_contract.rs +++ b/crates/ironclaw_reborn_openai_compat/tests/streaming_handlers_contract.rs @@ -935,6 +935,8 @@ fn completed_response(public_id: OpenAiResponseId, text: &str) -> OpenAiResponse input_tokens: 1, output_tokens: text.len() as u32, total_tokens: 1 + text.len() as u32, + input_tokens_details: None, + cost: None, }), } } diff --git a/crates/ironclaw_runner/src/loop_exit_applier/tests/mod.rs b/crates/ironclaw_runner/src/loop_exit_applier/tests/mod.rs index 99b2fcb4bbb..4e7a821d182 100644 --- a/crates/ironclaw_runner/src/loop_exit_applier/tests/mod.rs +++ b/crates/ironclaw_runner/src/loop_exit_applier/tests/mod.rs @@ -72,7 +72,7 @@ async fn no_reply_completion_requires_profile_permission() { reply_message_refs: vec![], result_refs: vec![], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: test_exit_id(), }); @@ -98,7 +98,7 @@ async fn result_only_completion_uses_verified_result_refs_without_no_reply_permi reply_message_refs: vec![], result_refs: vec![LoopResultRef::new("result:tool-output").expect("valid")], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: test_exit_id(), }); @@ -320,7 +320,7 @@ async fn loop_exit_events_hide_raw_diagnostics() { let exit = LoopExit::Failed(LoopFailed { reason_kind: LoopFailureKind::ModelError, checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, exit_id: test_exit_id(), explanation_message_refs: Vec::new(), @@ -1352,7 +1352,7 @@ async fn thread_checkpoint_evidence_fails_closed_for_failure_evidence() { let failed = LoopFailed { reason_kind: LoopFailureKind::ModelError, checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, exit_id: test_exit_id(), explanation_message_refs: Vec::new(), @@ -1413,7 +1413,7 @@ async fn thread_checkpoint_evidence_verifies_failure_from_final_checkpoint_state let failed = LoopFailed { reason_kind: LoopFailureKind::ModelError, checkpoint_id: Some(checkpoint.checkpoint_id), - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, exit_id: test_exit_id(), explanation_message_refs: Vec::new(), @@ -1505,7 +1505,7 @@ async fn thread_checkpoint_evidence_rejects_unverified_failure_explanation_ref() let failed = LoopFailed { reason_kind: LoopFailureKind::ModelError, checkpoint_id: Some(checkpoint.checkpoint_id), - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, exit_id: test_exit_id(), explanation_message_refs: vec![ @@ -1571,7 +1571,7 @@ async fn loop_exit_applier_accepts_thread_checkpoint_failure_evidence() { let exit = LoopExit::Failed(LoopFailed { reason_kind: LoopFailureKind::ModelError, checkpoint_id: Some(checkpoint.checkpoint_id), - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, exit_id: test_exit_id(), explanation_message_refs: Vec::new(), @@ -1657,7 +1657,7 @@ async fn loop_exit_applier_accepts_run_scoped_failure_checkpoint_ref_and_rejects LoopExit::Failed(LoopFailed { reason_kind: LoopFailureKind::ModelError, checkpoint_id: Some(accepted_checkpoint.checkpoint_id), - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, exit_id: test_exit_id(), explanation_message_refs: Vec::new(), @@ -1680,7 +1680,7 @@ async fn loop_exit_applier_accepts_run_scoped_failure_checkpoint_ref_and_rejects LoopExit::Failed(LoopFailed { reason_kind: LoopFailureKind::ModelError, checkpoint_id: Some(rejected_checkpoint.checkpoint_id), - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, exit_id: test_exit_id(), explanation_message_refs: Vec::new(), @@ -1739,7 +1739,7 @@ async fn thread_checkpoint_evidence_rejects_mismatched_failure_checkpoint_state( let failed = LoopFailed { reason_kind: LoopFailureKind::ModelError, checkpoint_id: Some(checkpoint.checkpoint_id), - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, exit_id: test_exit_id(), explanation_message_refs: Vec::new(), diff --git a/crates/ironclaw_runner/src/loop_exit_applier/tests/support.rs b/crates/ironclaw_runner/src/loop_exit_applier/tests/support.rs index 114e9966821..ff34a370f37 100644 --- a/crates/ironclaw_runner/src/loop_exit_applier/tests/support.rs +++ b/crates/ironclaw_runner/src/loop_exit_applier/tests/support.rs @@ -148,6 +148,7 @@ pub(super) fn running_run_state( resolved_run_profile_id: ironclaw_turns::RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: chrono::Utc::now(), checkpoint_id: None, gate_ref: None, @@ -283,7 +284,7 @@ pub(super) fn completed_exit( reply_message_refs, result_refs: vec![], final_checkpoint_id, - usage_summary_ref: None, + model_usage: None, exit_id: test_exit_id(), }) } @@ -390,6 +391,7 @@ pub(super) fn claimed_run() -> ClaimedTurnRun { resolved_run_profile_id: ironclaw_turns::RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: chrono::Utc::now(), checkpoint_id: None, gate_ref: None, @@ -672,6 +674,7 @@ fn state_for_mapping( resolved_run_profile_id: ironclaw_turns::RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: chrono::Utc::now(), checkpoint_id: None, gate_ref, diff --git a/crates/ironclaw_runner/src/model_gateway.rs b/crates/ironclaw_runner/src/model_gateway.rs index 29d6b35d911..445cb86920f 100644 --- a/crates/ironclaw_runner/src/model_gateway.rs +++ b/crates/ironclaw_runner/src/model_gateway.rs @@ -1453,6 +1453,8 @@ async fn tool_response_to_host( .with_usage(LoopModelUsage { input_tokens: response.input_tokens, output_tokens: response.output_tokens, + cache_read_input_tokens: response.cache_read_input_tokens, + cache_creation_input_tokens: response.cache_creation_input_tokens, })); } let advertised_tool_names = capabilities @@ -1536,6 +1538,8 @@ async fn tool_response_to_host( .with_usage(LoopModelUsage { input_tokens: response.input_tokens, output_tokens: response.output_tokens, + cache_read_input_tokens: response.cache_read_input_tokens, + cache_creation_input_tokens: response.cache_creation_input_tokens, })); } @@ -1559,6 +1563,8 @@ async fn tool_response_to_host( .with_usage(LoopModelUsage { input_tokens: response.input_tokens, output_tokens: response.output_tokens, + cache_read_input_tokens: response.cache_read_input_tokens, + cache_creation_input_tokens: response.cache_creation_input_tokens, })) } FinishReason::Length => Err(HostManagedModelError::safe( @@ -1875,6 +1881,8 @@ fn response_to_host_reply( let usage = LoopModelUsage { input_tokens: response.input_tokens, output_tokens: response.output_tokens, + cache_read_input_tokens: response.cache_read_input_tokens, + cache_creation_input_tokens: response.cache_creation_input_tokens, }; match response.finish_reason { FinishReason::Stop => { diff --git a/crates/ironclaw_runner/src/subagent/await_edge/resolver.rs b/crates/ironclaw_runner/src/subagent/await_edge/resolver.rs index bf80aba1834..50494a84531 100644 --- a/crates/ironclaw_runner/src/subagent/await_edge/resolver.rs +++ b/crates/ironclaw_runner/src/subagent/await_edge/resolver.rs @@ -983,6 +983,7 @@ mod tests { status: TurnStatus::Completed, profile: ironclaw_turns::TurnRunProfile::from_resolved(resolved_run_profile), resolved_model_route: None, + model_usage: None, checkpoint_id: None, gate_ref: None, blocked_activity_id: None, diff --git a/crates/ironclaw_runner/src/text_loop_driver.rs b/crates/ironclaw_runner/src/text_loop_driver.rs index fb015b4acdd..f287d0c20fb 100644 --- a/crates/ironclaw_runner/src/text_loop_driver.rs +++ b/crates/ironclaw_runner/src/text_loop_driver.rs @@ -163,7 +163,7 @@ fn completed_final_reply( reply_message_refs: vec![reply_ref], result_refs: Vec::new(), final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id, }) } diff --git a/crates/ironclaw_runner/src/turn_run_executor.rs b/crates/ironclaw_runner/src/turn_run_executor.rs index 98ce7dd2b4a..783ecb34fe1 100644 --- a/crates/ironclaw_runner/src/turn_run_executor.rs +++ b/crates/ironclaw_runner/src/turn_run_executor.rs @@ -482,6 +482,7 @@ mod tests { resolved_run_profile_id: ironclaw_turns::RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: chrono::Utc::now(), checkpoint_id: None, gate_ref: None, @@ -654,6 +655,7 @@ mod tests { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: chrono::Utc::now(), checkpoint_id: None, gate_ref: None, @@ -751,7 +753,7 @@ mod tests { reply_message_refs: vec![LoopMessageRef::new("msg:test").expect("valid")], result_refs: vec![], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: LoopExitId::new("exit:test").expect("valid"), })) } @@ -766,7 +768,7 @@ mod tests { reply_message_refs: vec![LoopMessageRef::new("msg:test").expect("valid")], result_refs: vec![], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: LoopExitId::new("exit:test").expect("valid"), })) } diff --git a/crates/ironclaw_runner/src/turn_scheduler/tests.rs b/crates/ironclaw_runner/src/turn_scheduler/tests.rs index da09bed7ae2..67b0be658a4 100644 --- a/crates/ironclaw_runner/src/turn_scheduler/tests.rs +++ b/crates/ironclaw_runner/src/turn_scheduler/tests.rs @@ -404,6 +404,7 @@ async fn claimed_test_run(thread_id: &str) -> ClaimedTurnRun { resolved_run_profile_id: RunProfileId::interactive_default(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref: None, diff --git a/crates/ironclaw_runner/tests/concurrent_workers.rs b/crates/ironclaw_runner/tests/concurrent_workers.rs index d99fd960c45..ee2c596dac0 100644 --- a/crates/ironclaw_runner/tests/concurrent_workers.rs +++ b/crates/ironclaw_runner/tests/concurrent_workers.rs @@ -756,7 +756,7 @@ async fn scheduler_executor_applies_loop_exit_end_to_end() { Ok(LoopExit::Failed(LoopFailed { reason_kind: LoopFailureKind::DriverBug, checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, explanation_message_refs: Vec::new(), safe_summary: None, diff --git a/crates/ironclaw_runner/tests/hooks_integration.rs b/crates/ironclaw_runner/tests/hooks_integration.rs index e72a2edb916..f28941b60e6 100644 --- a/crates/ironclaw_runner/tests/hooks_integration.rs +++ b/crates/ironclaw_runner/tests/hooks_integration.rs @@ -1085,6 +1085,7 @@ impl Fixture { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref: None, diff --git a/crates/ironclaw_runner/tests/loop_driver_host.rs b/crates/ironclaw_runner/tests/loop_driver_host.rs index 550a50676ae..23037e33e06 100644 --- a/crates/ironclaw_runner/tests/loop_driver_host.rs +++ b/crates/ironclaw_runner/tests/loop_driver_host.rs @@ -8441,7 +8441,7 @@ impl AgentLoopDriver for ScriptCapabilityFinalReplyDriver { reply_message_refs: vec![reply_ref], result_refs: vec![], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: LoopExitId::new("exit:turn-runner-script-capability-e2e").unwrap(), })) } @@ -8566,7 +8566,7 @@ impl AgentLoopDriver for TextOnlyFinalReplyDriver { reply_message_refs: vec![reply_ref], result_refs: vec![], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: LoopExitId::new("exit:turn-runner-e2e").unwrap(), })) } @@ -8829,6 +8829,7 @@ impl HostFixture { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref: None, diff --git a/crates/ironclaw_runner/tests/loop_milestone_event_projection.rs b/crates/ironclaw_runner/tests/loop_milestone_event_projection.rs index d24e841335c..8c7f9983ec1 100644 --- a/crates/ironclaw_runner/tests/loop_milestone_event_projection.rs +++ b/crates/ironclaw_runner/tests/loop_milestone_event_projection.rs @@ -803,6 +803,7 @@ impl HostFixture { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: Utc::now(), checkpoint_id: None, gate_ref: None, diff --git a/crates/ironclaw_turns/src/events.rs b/crates/ironclaw_turns/src/events.rs index 44466da9d60..b7950e8d2ad 100644 --- a/crates/ironclaw_turns/src/events.rs +++ b/crates/ironclaw_turns/src/events.rs @@ -689,6 +689,7 @@ mod tests { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: chrono::Utc::now(), checkpoint_id: None, gate_ref: Some(GateRef::new("gate:auth-a").expect("gate ref")), @@ -728,6 +729,7 @@ mod tests { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: chrono::Utc::now(), checkpoint_id: None, gate_ref: None, diff --git a/crates/ironclaw_turns/src/filesystem_store/projection.rs b/crates/ironclaw_turns/src/filesystem_store/projection.rs index 1a3ab121392..608f89945cc 100644 --- a/crates/ironclaw_turns/src/filesystem_store/projection.rs +++ b/crates/ironclaw_turns/src/filesystem_store/projection.rs @@ -61,6 +61,7 @@ pub(super) fn run_state_from_record(run: TurnRunRecord, actor: TurnActor) -> Tur resolved_run_profile_id: run.profile.id, resolved_run_profile_version: run.profile.version, resolved_model_route: run.resolved_model_route, + model_usage: run.model_usage, received_at: run.received_at, checkpoint_id: run.checkpoint_id, gate_ref: run.gate_ref, diff --git a/crates/ironclaw_turns/src/filesystem_store/runner_lease.rs b/crates/ironclaw_turns/src/filesystem_store/runner_lease.rs index e46faabe81b..2415b65eafa 100644 --- a/crates/ironclaw_turns/src/filesystem_store/runner_lease.rs +++ b/crates/ironclaw_turns/src/filesystem_store/runner_lease.rs @@ -663,6 +663,7 @@ mod tests { status, profile, resolved_model_route: None, + model_usage: None, checkpoint_id: None, gate_ref: None, blocked_activity_id: None, diff --git a/crates/ironclaw_turns/src/ids.rs b/crates/ironclaw_turns/src/ids.rs index 2e74f64321e..2e698dc66a8 100644 --- a/crates/ironclaw_turns/src/ids.rs +++ b/crates/ironclaw_turns/src/ids.rs @@ -246,7 +246,6 @@ loop_ref!(LoopExitId, "loop_exit_id", "exit:"); loop_ref!(LoopMessageRef, "loop_message_ref", "msg:"); loop_ref!(LoopResultRef, "loop_result_ref", "result:"); loop_ref!(LoopGateRef, "loop_gate_ref", "gate:"); -loop_ref!(LoopUsageSummaryRef, "loop_usage_summary_ref", "usage:"); loop_ref!(LoopDiagnosticRef, "loop_diagnostic_ref", "diag:"); // GateRef and LoopGateRef carry the same validated `gate:` string by diff --git a/crates/ironclaw_turns/src/lib.rs b/crates/ironclaw_turns/src/lib.rs index e5eb80ac599..68fc6421c8b 100644 --- a/crates/ironclaw_turns/src/lib.rs +++ b/crates/ironclaw_turns/src/lib.rs @@ -64,9 +64,9 @@ pub use filesystem_store::{ }; pub use ids::{ AcceptedMessageRef, CapabilityActivityId, GateRef, IdempotencyKey, LoopDiagnosticRef, - LoopExitId, LoopGateRef, LoopMessageRef, LoopResultRef, LoopUsageSummaryRef, - ReplyTargetBindingRef, RunProfileId, RunProfileRequest, RunProfileVersion, SourceBindingRef, - TurnCheckpointId, TurnId, TurnLeaseToken, TurnRunId, TurnRunnerId, + LoopExitId, LoopGateRef, LoopMessageRef, LoopResultRef, ReplyTargetBindingRef, RunProfileId, + RunProfileRequest, RunProfileVersion, SourceBindingRef, TurnCheckpointId, TurnId, + TurnLeaseToken, TurnRunId, TurnRunnerId, }; pub use lifecycle::{ DefaultTurnLifecycleEventBus, LifecyclePublicationErrorPort, LifecyclePublishingTurnStateStore, diff --git a/crates/ironclaw_turns/src/loop_exit.rs b/crates/ironclaw_turns/src/loop_exit.rs index 15d2c18c9f7..be478f7ab62 100644 --- a/crates/ironclaw_turns/src/loop_exit.rs +++ b/crates/ironclaw_turns/src/loop_exit.rs @@ -6,8 +6,8 @@ use serde::{Deserialize, Serialize, de}; use crate::{ BlockedReason, CapabilityActivityId, GateRef, LoopDiagnosticRef, LoopExitId, LoopGateRef, - LoopMessageRef, LoopResultRef, LoopUsageSummaryRef, ResolvedRunProfile, SanitizedFailure, - TurnCheckpointId, TurnError, TurnId, TurnRunId, TurnRunState, TurnScope, + LoopMessageRef, LoopResultRef, ResolvedRunProfile, SanitizedFailure, TurnCheckpointId, + TurnError, TurnId, TurnRunId, TurnRunState, TurnScope, run_profile::{LoopCheckpointKind, LoopCheckpointStateRef}, runner::{ ApplyValidatedLoopExitRequest, ClaimedTurnRun, TurnRunTransitionPort, TurnRunnerOutcome, @@ -123,6 +123,10 @@ impl LoopExitApplier { exit: LoopExit, ) -> Result { let policy = self.derive_policy(claimed, &exit).await?; + // Capture the loop's reported usage before `validate` consumes the exit + // and collapses it to a coarse outcome; carry it so the terminal + // transition can persist it on the run record. + let model_usage = exit.reported_model_usage(); let decision = exit.validate(policy); self.transition_port .apply_validated_loop_exit(ApplyValidatedLoopExitRequest { @@ -130,6 +134,7 @@ impl LoopExitApplier { runner_id: claimed.runner_id, lease_token: claimed.lease_token, mapping: decision.mapping, + model_usage, }) .await } @@ -270,6 +275,16 @@ impl LoopExit { } } + /// The cumulative model usage a terminal exit reported, if any. Blocked and + /// cancelled exits carry none (usage rides completion/failure only). + fn reported_model_usage(&self) -> Option { + match self { + Self::Completed(exit) => exit.model_usage, + Self::Failed(exit) => exit.model_usage, + Self::Blocked(_) | Self::Cancelled(_) => None, + } + } + fn validate(self, policy: LoopExitValidationPolicy) -> LoopExitValidationDecision { let exit_id = self.exit_id().clone(); match self { @@ -315,7 +330,7 @@ impl LoopExit { Self::Failed(LoopFailed { reason_kind, checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, exit_id, explanation_message_refs: Vec::new(), @@ -333,7 +348,12 @@ pub struct LoopCompleted { #[serde(deserialize_with = "deserialize_bounded_unique_refs")] pub result_refs: Vec, pub final_checkpoint_id: Option, - pub usage_summary_ref: Option, + /// Cumulative provider-reported token usage the loop accumulated across its + /// model calls. Carried to the run record at the terminal transition so the + /// OpenAI-compatible surfaces can report `usage` and cost. `None` when the + /// loop saw no usage (replay stubs, providers without a usage object). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model_usage: Option, pub exit_id: LoopExitId, } @@ -428,7 +448,10 @@ pub enum LoopCancelledReasonKind { pub struct LoopFailed { pub reason_kind: LoopFailureKind, pub checkpoint_id: Option, - pub usage_summary_ref: Option, + /// Cumulative provider-reported token usage accumulated before the failure. + /// See [`LoopCompleted::model_usage`]. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model_usage: Option, pub diagnostic_ref: Option, pub exit_id: LoopExitId, #[serde( diff --git a/crates/ironclaw_turns/src/loop_exit/tests/mod.rs b/crates/ironclaw_turns/src/loop_exit/tests/mod.rs index cfb2d2b12b1..d6927c2aebe 100644 --- a/crates/ironclaw_turns/src/loop_exit/tests/mod.rs +++ b/crates/ironclaw_turns/src/loop_exit/tests/mod.rs @@ -132,7 +132,7 @@ fn completed_ask_user_exit_maps_to_trusted_completed_outcome_without_final_check reply_message_refs: vec![message_ref("msg:assistant-question")], result_refs: vec![], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: exit_id.clone(), }) .validate(LoopExitValidationPolicy { @@ -157,7 +157,7 @@ fn completed_exit_without_durable_refs_maps_to_protocol_failure_or_recovery() { reply_message_refs: vec![], result_refs: vec![], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: exit_id("exit:missing-refs"), }); @@ -207,7 +207,7 @@ fn completed_exit_requires_host_verified_completion_refs_before_trusted_mapping( reply_message_refs: vec![message_ref("msg:assistant-final")], result_refs: vec![], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: exit_id("exit:unverified-completion"), }); @@ -239,7 +239,7 @@ fn final_checkpoint_policy_rejects_terminal_exit_without_checkpoint() { reply_message_refs: vec![message_ref("msg:assistant-final")], result_refs: vec![], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: exit_id("exit:no-final-checkpoint-completed"), }), LoopExit::Cancelled(LoopCancelled { @@ -305,7 +305,7 @@ fn validation_policy_requires_final_checkpoint_only_when_configured() { reply_message_refs: vec![message_ref("msg:assistant-final")], result_refs: vec![], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: exit_id("exit:checkpoint-policy"), }) .validate(LoopExitValidationPolicy { @@ -459,7 +459,6 @@ fn loop_failed_legacy_payload_deserializes_and_empty_new_fields_serialize_to_leg let legacy = json!({ "reason_kind": "iteration_limit", "checkpoint_id": null, - "usage_summary_ref": null, "diagnostic_ref": null, "exit_id": "exit:legacy-failed" }); @@ -480,7 +479,7 @@ fn verified_failed_exit_carries_safe_summary() { let decision = LoopExit::Failed(LoopFailed { reason_kind: LoopFailureKind::ModelError, checkpoint_id: Some(final_checkpoint_id), - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, exit_id: exit_id("exit:verified-failed"), explanation_message_refs: vec![explanation_ref.clone()], @@ -504,7 +503,7 @@ fn verified_failed_exit_does_not_reuse_final_checkpoint_as_resume_checkpoint() { let decision = LoopExit::Failed(LoopFailed { reason_kind: LoopFailureKind::IterationLimit, checkpoint_id: Some(final_checkpoint_id), - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, exit_id: exit_id("exit:final-only-failed"), explanation_message_refs: Vec::new(), @@ -527,7 +526,7 @@ fn unverified_failed_exit_drops_explanation_refs_and_keeps_existing_violation_be let decision = LoopExit::Failed(LoopFailed { reason_kind: LoopFailureKind::ModelError, checkpoint_id: Some(TurnCheckpointId::new()), - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, exit_id: exit_id("exit:unverified-failed"), explanation_message_refs: vec![message_ref("msg:unverified-explanation")], @@ -554,7 +553,7 @@ fn strict_final_checkpoint_policy_trusts_failed_exit_only_after_verification() { let exit = LoopExit::Failed(LoopFailed { reason_kind: LoopFailureKind::IterationLimit, checkpoint_id: Some(checkpoint_id), - usage_summary_ref: None, + model_usage: None, diagnostic_ref: None, exit_id: exit_id("exit:strict-failed-checkpoint"), explanation_message_refs: Vec::new(), @@ -621,7 +620,6 @@ fn loop_exit_wire_shape_rejects_raw_payload_fields_and_recovery_required_variant "reply_message_refs": ["msg:assistant-final"], "result_refs": [], "final_checkpoint_id": null, - "usage_summary_ref": null, "exit_id": "exit:raw", "raw_reply_text": "secret prompt-adjacent content" } @@ -653,7 +651,6 @@ fn loop_exit_rejects_oversized_or_duplicate_ref_vectors() { "reply_message_refs": oversized_messages, "result_refs": [], "final_checkpoint_id": null, - "usage_summary_ref": null, "exit_id": "exit:oversized" } }); @@ -665,7 +662,6 @@ fn loop_exit_rejects_oversized_or_duplicate_ref_vectors() { "reply_message_refs": ["msg:dup", "msg:dup"], "result_refs": [], "final_checkpoint_id": null, - "usage_summary_ref": null, "exit_id": "exit:duplicates" } }); @@ -706,7 +702,7 @@ fn no_reply_with_empty_refs_requires_explicit_policy_permission() { reply_message_refs: vec![], result_refs: vec![], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: exit_id("exit:no-reply-empty"), }); @@ -740,7 +736,7 @@ fn no_reply_with_empty_refs_maps_to_completed_when_policy_allows_it() { reply_message_refs: vec![], result_refs: vec![], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: exit_id("exit:no-reply-allowed"), }) .validate(LoopExitValidationPolicy { @@ -764,7 +760,7 @@ fn delegated_result_with_result_refs_maps_to_trusted_completed() { reply_message_refs: vec![], result_refs: vec![result_ref("result:delegated-job-1")], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: exit_id("exit:delegated"), }) .validate(LoopExitValidationPolicy { @@ -788,7 +784,7 @@ fn result_only_with_result_refs_maps_to_trusted_completed() { reply_message_refs: vec![], result_refs: vec![result_ref("result:tool-output-1")], final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: exit_id("exit:result-only"), }) .validate(LoopExitValidationPolicy { @@ -854,7 +850,7 @@ fn completion_kind_must_match_durable_reference_shape() { reply_message_refs, result_refs, final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: exit_id("exit:mismatched-completion-kind"), }) .validate(policy); diff --git a/crates/ironclaw_turns/src/memory/mod.rs b/crates/ironclaw_turns/src/memory/mod.rs index dfbb6350d02..6bb168be3cc 100644 --- a/crates/ironclaw_turns/src/memory/mod.rs +++ b/crates/ironclaw_turns/src/memory/mod.rs @@ -307,6 +307,7 @@ struct RunRecord { status: RunStatusCell, profile: TurnRunProfile, resolved_model_route: Option, + model_usage: Option, accepted_message_ref: AcceptedMessageRef, source_binding_ref: SourceBindingRef, reply_target_binding_ref: ReplyTargetBindingRef, @@ -1184,6 +1185,7 @@ impl TurnStateStore for InMemoryTurnStateStore { status: RunStatusCell::new(TurnStatus::Queued), profile: profile.clone(), resolved_model_route: None, + model_usage: None, accepted_message_ref: request.accepted_message_ref.clone(), source_binding_ref: request.source_binding_ref.clone(), reply_target_binding_ref: request.reply_target_binding_ref.clone(), @@ -1632,6 +1634,7 @@ impl TurnSpawnTreeStateStore for InMemoryTurnStateStore { status: RunStatusCell::new(TurnStatus::Queued), profile: profile.clone(), resolved_model_route: None, + model_usage: None, accepted_message_ref: request.accepted_message_ref.clone(), source_binding_ref: request.source_binding_ref.clone(), reply_target_binding_ref: request.reply_target_binding_ref.clone(), @@ -2133,6 +2136,7 @@ impl TurnRunTransitionPort for InMemoryTurnStateStore { request.runner_id, request.lease_token, request.mapping, + request.model_usage, ) }; // A validated loop exit can either park a run on a gate or terminate one @@ -2197,6 +2201,7 @@ impl Inner { status: RunStatusCell::new(run.status), profile: run.profile, resolved_model_route: run.resolved_model_route, + model_usage: run.model_usage, accepted_message_ref: run.accepted_message_ref, source_binding_ref: run.source_binding_ref, reply_target_binding_ref: run.reply_target_binding_ref, @@ -2925,6 +2930,7 @@ impl Inner { status: RunStatusCell::new(TurnStatus::Queued), profile, resolved_model_route: None, + model_usage: None, accepted_message_ref, source_binding_ref: request.source_binding_ref.clone(), reply_target_binding_ref: request.reply_target_binding_ref.clone(), @@ -3146,8 +3152,9 @@ impl Inner { runner_id: crate::TurnRunnerId, lease_token: crate::TurnLeaseToken, mapping: LoopExitMapping, + model_usage: Option, ) -> Result { - let record = self.take_record(run_id)?; + let mut record = self.take_record(run_id)?; let result = (|| { if let Err(error) = ensure_active_lease(&record, runner_id, lease_token, Utc::now()) { return AppliedLoopTransition::Rejected { @@ -3155,6 +3162,14 @@ impl Inner { error, }; } + // Accumulate the exit's reported usage onto the run so a block/resume + // sequence sums each leg rather than overwriting. Best-effort + // telemetry; a run that reported no usage leaves the prior total. + if let Some(usage) = model_usage { + let mut total = record.model_usage.unwrap_or_default(); + total.add_assign(&usage); + record.model_usage = Some(total); + } match mapping { LoopExitMapping::RunnerOutcome(TurnRunnerOutcome::Completed) => { self.complete_claimed_record(record) @@ -3739,6 +3754,7 @@ impl RunRecord { status: self.status.get(), profile: self.profile.clone(), resolved_model_route: self.resolved_model_route.clone(), + model_usage: self.model_usage, checkpoint_id: self.checkpoint_id, gate_ref: self.gate_ref.clone(), blocked_activity_id: self.blocked_activity_id, @@ -3772,6 +3788,7 @@ impl RunRecord { resolved_run_profile_id: self.profile.id.clone(), resolved_run_profile_version: self.profile.version, resolved_model_route: self.resolved_model_route.clone(), + model_usage: self.model_usage, received_at: self.received_at, checkpoint_id: self.checkpoint_id, gate_ref: self.gate_ref.clone(), diff --git a/crates/ironclaw_turns/src/run_profile/host.rs b/crates/ironclaw_turns/src/run_profile/host.rs index 2985553fec8..3e2bd4ca4cf 100644 --- a/crates/ironclaw_turns/src/run_profile/host.rs +++ b/crates/ironclaw_turns/src/run_profile/host.rs @@ -1278,6 +1278,40 @@ pub struct LoopModelResponse { pub struct LoopModelUsage { pub input_tokens: u32, pub output_tokens: u32, + /// Tokens read from the provider's server-side prompt cache (e.g. Anthropic + /// cache reads). A subset of `input_tokens`, billed at a discount. Zero when + /// caching is unsupported or on a cache miss. + #[serde(default, skip_serializing_if = "is_zero_u32")] + pub cache_read_input_tokens: u32, + /// Tokens written to the provider's server-side prompt cache. Zero when + /// caching is unsupported or no new prefix was cached. + #[serde(default, skip_serializing_if = "is_zero_u32")] + pub cache_creation_input_tokens: u32, +} + +fn is_zero_u32(value: &u32) -> bool { + *value == 0 +} + +impl LoopModelUsage { + /// Accumulate another call's usage into this running per-run total. + pub fn add_assign(&mut self, other: &LoopModelUsage) { + self.input_tokens = self.input_tokens.saturating_add(other.input_tokens); + self.output_tokens = self.output_tokens.saturating_add(other.output_tokens); + self.cache_read_input_tokens = self + .cache_read_input_tokens + .saturating_add(other.cache_read_input_tokens); + self.cache_creation_input_tokens = self + .cache_creation_input_tokens + .saturating_add(other.cache_creation_input_tokens); + } + + /// Total billable tokens (input + output). Cache tokens are already counted + /// within `input_tokens` by every provider that reports them, so they are + /// not added again here. + pub fn total_tokens(&self) -> u32 { + self.input_tokens.saturating_add(self.output_tokens) + } } #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] diff --git a/crates/ironclaw_turns/src/runner.rs b/crates/ironclaw_turns/src/runner.rs index ccda4f3cde5..65d958fd832 100644 --- a/crates/ironclaw_turns/src/runner.rs +++ b/crates/ironclaw_turns/src/runner.rs @@ -110,6 +110,11 @@ pub struct ApplyValidatedLoopExitRequest { pub runner_id: TurnRunnerId, pub lease_token: TurnLeaseToken, pub mapping: LoopExitMapping, + /// Cumulative provider-reported token usage the loop accumulated for this + /// run, carried from the `LoopExit` so a terminal transition can persist it + /// on the run record. `None` when the exit reported no usage. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model_usage: Option, } #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] diff --git a/crates/ironclaw_turns/src/status.rs b/crates/ironclaw_turns/src/status.rs index 614a4917fd1..4849d835826 100644 --- a/crates/ironclaw_turns/src/status.rs +++ b/crates/ironclaw_turns/src/status.rs @@ -397,6 +397,12 @@ pub struct TurnRunState { pub resolved_run_profile_version: RunProfileVersion, #[serde(default, skip_serializing_if = "Option::is_none")] pub resolved_model_route: Option, + /// Cumulative provider-reported token usage for this run's model calls, + /// captured at loop exit. `None` for runs that reported no usage (replay + /// stubs) or that pre-date usage capture. Read by the OpenAI-compatible + /// Responses/Chat surfaces to report `usage` and cost. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model_usage: Option, pub received_at: TurnTimestamp, pub checkpoint_id: Option, pub gate_ref: Option, diff --git a/crates/ironclaw_turns/src/store.rs b/crates/ironclaw_turns/src/store.rs index d1a8ff18025..bcbcf1c9ff5 100644 --- a/crates/ironclaw_turns/src/store.rs +++ b/crates/ironclaw_turns/src/store.rs @@ -215,6 +215,11 @@ pub struct TurnRunRecord { pub profile: TurnRunProfile, #[serde(default, skip_serializing_if = "Option::is_none")] pub resolved_model_route: Option, + /// Cumulative provider-reported token usage for this run's model calls, + /// captured at loop exit. Rides the JSON-blob snapshot like + /// `resolved_model_route`; `None` when no usage was reported. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model_usage: Option, pub checkpoint_id: Option, pub gate_ref: Option, #[serde(default, skip_serializing_if = "Option::is_none")] @@ -587,6 +592,7 @@ mod tests { status: TurnStatus::Completed, profile, resolved_model_route: None, + model_usage: None, checkpoint_id: None, gate_ref: None, blocked_activity_id: None, diff --git a/crates/ironclaw_turns/tests/agent_loop_host_contract.rs b/crates/ironclaw_turns/tests/agent_loop_host_contract.rs index e1dfbecc618..56f50c2c676 100644 --- a/crates/ironclaw_turns/tests/agent_loop_host_contract.rs +++ b/crates/ironclaw_turns/tests/agent_loop_host_contract.rs @@ -2295,6 +2295,7 @@ async fn loop_prompt_bundle_public_serialization_hides_raw_content() { resolved_run_profile_id: host.context.resolved_run_profile.profile_id.clone(), resolved_run_profile_version: host.context.resolved_run_profile.profile_version, resolved_model_route: None, + model_usage: None, received_at: Utc.with_ymd_and_hms(2026, 5, 7, 12, 0, 0).unwrap(), checkpoint_id: None, gate_ref: None, @@ -2641,7 +2642,7 @@ impl AgentLoopDriver for ReplyDriver { reply_message_refs: vec![message_ref], result_refs: Vec::new(), final_checkpoint_id: None, - usage_summary_ref: None, + model_usage: None, exit_id: LoopExitId::new("exit:reply-driver").unwrap(), })) } @@ -4029,6 +4030,7 @@ async fn turn_run_state_product_context_defaults_to_none_when_missing_from_json( resolved_run_profile_id: context.resolved_run_profile.profile_id.clone(), resolved_run_profile_version: context.resolved_run_profile.profile_version, resolved_model_route: None, + model_usage: None, received_at: Utc.with_ymd_and_hms(2026, 6, 11, 21, 32, 0).unwrap(), checkpoint_id: None, gate_ref: None, @@ -4090,6 +4092,7 @@ async fn turn_run_state_resume_disposition_defaults_to_none_when_missing_from_js resolved_run_profile_id: context.resolved_run_profile.profile_id.clone(), resolved_run_profile_version: context.resolved_run_profile.profile_version, resolved_model_route: None, + model_usage: None, received_at: Utc.with_ymd_and_hms(2026, 6, 11, 21, 32, 0).unwrap(), checkpoint_id: None, gate_ref: None, diff --git a/crates/ironclaw_turns/tests/checkpoint_state_store_contract.rs b/crates/ironclaw_turns/tests/checkpoint_state_store_contract.rs index 6a042f153d2..78fe3131673 100644 --- a/crates/ironclaw_turns/tests/checkpoint_state_store_contract.rs +++ b/crates/ironclaw_turns/tests/checkpoint_state_store_contract.rs @@ -433,6 +433,7 @@ fn turn_run_state_actor_is_serde_backward_compatible() { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: fixed_time(), checkpoint_id: None, gate_ref: None, @@ -485,6 +486,7 @@ fn turn_checkpoint_public_status_does_not_expose_checkpoint_payload() { resolved_run_profile_id: RunProfileId::default_profile(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: fixed_time(), checkpoint_id: Some(checkpoint_id), gate_ref: Some(GateRef::new("gate-checkpoint-public").unwrap()), @@ -830,6 +832,7 @@ fn turn_persistence_snapshot_legacy_run_defaults_resume_disposition_to_none() { status: TurnStatus::Completed, profile: TurnRunProfile::from_resolved(resolved), resolved_model_route: None, + model_usage: None, checkpoint_id: None, gate_ref: None, blocked_activity_id: None, @@ -968,6 +971,7 @@ fn turn_persistence_snapshot_legacy_run_preserves_denied_resume_disposition() { status: TurnStatus::Completed, profile: TurnRunProfile::from_resolved(resolved), resolved_model_route: None, + model_usage: None, checkpoint_id: None, gate_ref: None, blocked_activity_id: None, diff --git a/crates/ironclaw_turns/tests/filesystem_turn_state_contract.rs b/crates/ironclaw_turns/tests/filesystem_turn_state_contract.rs index c15144cdde4..26d76f98b40 100644 --- a/crates/ironclaw_turns/tests/filesystem_turn_state_contract.rs +++ b/crates/ironclaw_turns/tests/filesystem_turn_state_contract.rs @@ -2572,6 +2572,7 @@ async fn filesystem_turn_state_row_store_validated_loop_exit_completion_remains_ .unwrap(); store .apply_validated_loop_exit(ApplyValidatedLoopExitRequest { + model_usage: None, run_id: first_run_id, runner_id: first_runner_id, lease_token: first_lease_token, @@ -2603,6 +2604,7 @@ async fn filesystem_turn_state_row_store_validated_loop_exit_completion_remains_ .unwrap(); store .apply_validated_loop_exit(ApplyValidatedLoopExitRequest { + model_usage: None, run_id: second_run_id, runner_id: second_runner_id, lease_token: second_lease_token, diff --git a/crates/ironclaw_turns/tests/per_inbound_type_concurrency_cap.rs b/crates/ironclaw_turns/tests/per_inbound_type_concurrency_cap.rs index bf8a5129d37..5dcc45c9054 100644 --- a/crates/ironclaw_turns/tests/per_inbound_type_concurrency_cap.rs +++ b/crates/ironclaw_turns/tests/per_inbound_type_concurrency_cap.rs @@ -420,6 +420,7 @@ async fn trigger_counter_decrements_via_apply_validated_loop_exit() { store .apply_validated_loop_exit(ApplyValidatedLoopExitRequest { + model_usage: None, run_id, runner_id, lease_token, @@ -463,6 +464,7 @@ async fn trigger_counter_decrements_via_apply_validated_loop_exit_cancelled() { // apply_validated_loop_exit Cancelled → cancel_or_fail_claimed_record → cancel_claimed_record. store .apply_validated_loop_exit(ApplyValidatedLoopExitRequest { + model_usage: None, run_id, runner_id, lease_token, diff --git a/crates/ironclaw_turns/tests/per_user_concurrency_cap.rs b/crates/ironclaw_turns/tests/per_user_concurrency_cap.rs index 9e2797d2cfd..013387dae2c 100644 --- a/crates/ironclaw_turns/tests/per_user_concurrency_cap.rs +++ b/crates/ironclaw_turns/tests/per_user_concurrency_cap.rs @@ -351,6 +351,7 @@ async fn running_counter_decrements_via_apply_validated_loop_exit_completed() { store .apply_validated_loop_exit(ApplyValidatedLoopExitRequest { + model_usage: None, run_id, runner_id, lease_token, @@ -389,6 +390,7 @@ async fn running_counter_decrements_via_apply_validated_loop_exit_cancelled() { // which calls cancel_claimed_record (CancelRequested → Cancelled). store .apply_validated_loop_exit(ApplyValidatedLoopExitRequest { + model_usage: None, run_id, runner_id, lease_token, diff --git a/crates/ironclaw_turns/tests/retry_failed_turn_store_contract.rs b/crates/ironclaw_turns/tests/retry_failed_turn_store_contract.rs index 7c39142ad41..8135b78f140 100644 --- a/crates/ironclaw_turns/tests/retry_failed_turn_store_contract.rs +++ b/crates/ironclaw_turns/tests/retry_failed_turn_store_contract.rs @@ -177,6 +177,7 @@ where { store .apply_validated_loop_exit(ApplyValidatedLoopExitRequest { + model_usage: None, run_id: claimed.state.run_id, runner_id: claimed.runner_id, lease_token: claimed.lease_token, @@ -197,6 +198,7 @@ where { store .apply_validated_loop_exit(ApplyValidatedLoopExitRequest { + model_usage: None, run_id: claimed.state.run_id, runner_id: claimed.runner_id, lease_token: claimed.lease_token, diff --git a/crates/ironclaw_turns/tests/turn_coordinator_contract.rs b/crates/ironclaw_turns/tests/turn_coordinator_contract.rs index 07c610e51ba..d2ad6215738 100644 --- a/crates/ironclaw_turns/tests/turn_coordinator_contract.rs +++ b/crates/ironclaw_turns/tests/turn_coordinator_contract.rs @@ -54,6 +54,7 @@ where P: TurnRunTransitionPort + ?Sized, { port.apply_validated_loop_exit(ApplyValidatedLoopExitRequest { + model_usage: None, run_id, runner_id, lease_token, @@ -7390,6 +7391,7 @@ impl TurnRunTransitionPort for AtomicLoopExitPort { resolved_run_profile_id: RunProfileId::new("default").unwrap(), resolved_run_profile_version: RunProfileVersion::new(1), resolved_model_route: None, + model_usage: None, received_at: received_at(), checkpoint_id: None, gate_ref: None, @@ -7953,6 +7955,7 @@ async fn cancel_on_legacy_recovery_required_run_reports_already_terminal() { // a RecoveryRequired mapping (the compat shim). let rr_state = store .apply_validated_loop_exit(ApplyValidatedLoopExitRequest { + model_usage: None, run_id, runner_id, lease_token, From 195b8e1c9bfc4a30d3654442f117d6865c605634 Mon Sep 17 00:00:00 2001 From: Illia Polosukhin Date: Sat, 11 Jul 2026 06:35:13 +0000 Subject: [PATCH 2/6] feat(reborn): route caller-requested model on OpenAI-compatible API (Phase 2) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Stacked on the usage/cost PR. Makes the OpenAI-compatible Responses/Chat `model` field actually route the turn's LLM call instead of being validated-and-echoed but ignored. Semantics: route if the provider can serve it, else silently fall back to the deployment's active model. The requested model is threaded from the request to the model call as an advisory route, honored at the provider boundary: - product_adapters: UserMessagePayload gains optional requested_model (wire-defaulted; with_requested_model filters empty). - openai_compat: chat + responses workflows set requested_model from request.model on the submitted payload (previously dropped). - product_workflow: AcceptedProductInboundTurn::submit carries it onto the new SubmitTurnRequest.requested_model. - ironclaw_turns: submit_turn records it as an advisory LoopModelRouteSnapshot (LoopModelRouteSnapshot::advisory — only model_id meaningful, is_advisory() distinguishes it, None for empty/invalid). Reuses the existing resolved_model_route field that already flows run -> loop context -> gateway, so no new run-state field. - ironclaw_runner: LlmProviderModelGateway::request_model_override prefers the request's route model_id over the profile default, falling back to the active model when absent (providers that honor per-request overrides serve it, others fall back). attach_model_route_snapshot passes an advisory snapshot through unvalidated when no route resolver is wired (default runtime); routed fail-closed hosts are unchanged. Child/subagent runs and idempotent replays carry no requested model. Known limitation: per-request routing only takes effect for providers that honor CompletionRequest.model (NEAR AI); RigAdapter-backed providers ignore it and fall back. Strict operator-configured-only validation would need the model catalog wired into the default runtime (follow-up). Tests: gateway honors requested route over profile default + falls back when absent (recording-provider seam); submit records advisory route from requested_model (and none when absent); advisory() construction/validation; UserMessagePayload requested_model serde round-trip + empty filtering. Co-Authored-By: Claude Opus 4.8 (1M context) --- crates/ironclaw_conversations/src/inbound.rs | 1 + .../tests/support/host_runtime_harness.rs | 1 + .../tests/turn_event_publisher_contract.rs | 1 + .../ironclaw_product_adapters/src/inbound.rs | 59 ++++++++++++++++- .../src/auth_continuation.rs | 1 + .../src/inbound_turn.rs | 8 +++ .../src/reborn_services.rs | 1 + .../tests/product_workflow_contract.rs | 1 + .../src/factory.rs | 2 + .../src/factory/auth_tests.rs | 3 + .../src/runtime.rs | 4 ++ .../src/runtime/tests/auth_interaction.rs | 1 + .../src/chat_workflow.rs | 3 +- .../src/responses_workflow.rs | 3 +- .../ironclaw_runner/src/loop_driver_host.rs | 11 +++- crates/ironclaw_runner/src/model_gateway.rs | 50 ++++++++++++--- .../src/subagent/await_edge/boot_recovery.rs | 1 + .../src/subagent/await_edge/resolver.rs | 1 + .../tests/concurrent_workers.rs | 2 + crates/ironclaw_runner/tests/llm_gateway.rs | 50 +++++++++++++++ .../ironclaw_runner/tests/loop_driver_host.rs | 4 ++ .../tests/turn_scheduler_contract.rs | 1 + crates/ironclaw_turns/src/memory/mod.rs | 8 ++- crates/ironclaw_turns/src/request.rs | 5 ++ crates/ironclaw_turns/src/run_profile/host.rs | 63 +++++++++++++++++++ .../tests/active_run_ref_state_contract.rs | 1 + .../tests/agent_loop_host_contract.rs | 1 + .../tests/filesystem_turn_state_contract.rs | 1 + .../tests/per_inbound_type_concurrency_cap.rs | 1 + .../tests/per_user_concurrency_cap.rs | 3 + .../tests/retry_failed_turn_store_contract.rs | 1 + .../tests/turn_coordinator_contract.rs | 46 ++++++++++++++ tests/integration/subagent_await_edge.rs | 2 + 33 files changed, 326 insertions(+), 15 deletions(-) diff --git a/crates/ironclaw_conversations/src/inbound.rs b/crates/ironclaw_conversations/src/inbound.rs index fc6038a6f15..77b64ee0ca2 100644 --- a/crates/ironclaw_conversations/src/inbound.rs +++ b/crates/ironclaw_conversations/src/inbound.rs @@ -243,6 +243,7 @@ where let turn_submission_result = self .turn_coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: resolution.turn_scope.clone(), actor: accepted_message.actor.clone(), accepted_message_ref: accepted_message.message_ref.clone(), diff --git a/crates/ironclaw_host_runtime/tests/support/host_runtime_harness.rs b/crates/ironclaw_host_runtime/tests/support/host_runtime_harness.rs index fa3579c7a27..e4d7e6212c5 100644 --- a/crates/ironclaw_host_runtime/tests/support/host_runtime_harness.rs +++ b/crates/ironclaw_host_runtime/tests/support/host_runtime_harness.rs @@ -2264,6 +2264,7 @@ pub(crate) fn http_without_body_then_operation_failed_wat() -> String { #[cfg(feature = "libsql")] pub(crate) fn submit_turn_request(thread: &str, idempotency_key: &str) -> SubmitTurnRequest { SubmitTurnRequest { + requested_model: None, scope: TurnScope::new( TenantId::new("tenant1").unwrap(), Some(AgentId::new("agent1").unwrap()), diff --git a/crates/ironclaw_loop_support/tests/turn_event_publisher_contract.rs b/crates/ironclaw_loop_support/tests/turn_event_publisher_contract.rs index 62634960ed0..e2fe2063a25 100644 --- a/crates/ironclaw_loop_support/tests/turn_event_publisher_contract.rs +++ b/crates/ironclaw_loop_support/tests/turn_event_publisher_contract.rs @@ -30,6 +30,7 @@ fn actor() -> TurnActor { fn submit_request(thread: &str, idempotency_key: &str) -> SubmitTurnRequest { SubmitTurnRequest { + requested_model: None, scope: scope(thread), actor: actor(), accepted_message_ref: AcceptedMessageRef::new(format!("message-{idempotency_key}")) diff --git a/crates/ironclaw_product_adapters/src/inbound.rs b/crates/ironclaw_product_adapters/src/inbound.rs index 107615a2a9f..154dc39d31d 100644 --- a/crates/ironclaw_product_adapters/src/inbound.rs +++ b/crates/ironclaw_product_adapters/src/inbound.rs @@ -15,6 +15,7 @@ use crate::outbound::ProjectionCursor; use crate::redaction::RedactedString; const USER_MESSAGE_TEXT_MAX_BYTES: usize = 64 * 1024; +const REQUESTED_MODEL_MAX_BYTES: usize = 256; const COMMAND_MAX_BYTES: usize = 256; const COMMAND_ARGUMENTS_MAX_BYTES: usize = 64 * 1024; const THREAD_HINT_MAX_BYTES: usize = 512; @@ -99,6 +100,13 @@ pub struct UserMessagePayload { pub text: String, pub attachments: Vec, pub trigger: ProductTriggerReason, + /// Caller-requested model for this turn (e.g. an OpenAI-compatible client's + /// `model` field). A model *hint*, not authority: the coordinator routes to + /// it only when the operator has it configured, otherwise it falls back to + /// the deployment's active model. `None` for surfaces that don't select a + /// model (chat UI, channels). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub requested_model: Option, } impl UserMessagePayload { @@ -111,13 +119,25 @@ impl UserMessagePayload { text: text.into(), attachments, trigger, + requested_model: None, }; payload.validate()?; Ok(payload) } + /// Attach a caller-requested model to this payload. See + /// [`UserMessagePayload::requested_model`]. + pub fn with_requested_model(mut self, requested_model: Option) -> Self { + self.requested_model = requested_model.filter(|model| !model.is_empty()); + self + } + pub fn validate(&self) -> Result<(), ProductAdapterError> { - validate_payload_string("user message text", &self.text, USER_MESSAGE_TEXT_MAX_BYTES) + validate_payload_string("user message text", &self.text, USER_MESSAGE_TEXT_MAX_BYTES)?; + if let Some(model) = &self.requested_model { + validate_payload_string("requested model", model, REQUESTED_MODEL_MAX_BYTES)?; + } + Ok(()) } } @@ -126,6 +146,8 @@ struct UserMessagePayloadWire { text: String, attachments: Vec, trigger: ProductTriggerReason, + #[serde(default)] + requested_model: Option, } impl<'de> Deserialize<'de> for UserMessagePayload { @@ -134,7 +156,9 @@ impl<'de> Deserialize<'de> for UserMessagePayload { D: Deserializer<'de>, { let wire = UserMessagePayloadWire::deserialize(deserializer)?; - Self::new(wire.text, wire.attachments, wire.trigger).map_err(serde::de::Error::custom) + Self::new(wire.text, wire.attachments, wire.trigger) + .map(|payload| payload.with_requested_model(wire.requested_model)) + .map_err(serde::de::Error::custom) } } @@ -882,6 +906,37 @@ mod tests { use crate::auth::AuthRequirement; use crate::external::{ExternalActorRef, ExternalConversationRef, ExternalEventId}; + #[test] + fn user_message_payload_round_trips_and_filters_requested_model() { + let with_model = UserMessagePayload::new("hi", vec![], ProductTriggerReason::DirectChat) + .unwrap() + .with_requested_model(Some("gpt-4o".to_string())); + assert_eq!(with_model.requested_model.as_deref(), Some("gpt-4o")); + // Round-trips over the wire (custom Deserialize via the wire struct). + let decoded: UserMessagePayload = + serde_json::from_str(&serde_json::to_string(&with_model).unwrap()).unwrap(); + assert_eq!(decoded.requested_model.as_deref(), Some("gpt-4o")); + + // Omitted → None, and not serialized when absent. + let without = + UserMessagePayload::new("hi", vec![], ProductTriggerReason::DirectChat).unwrap(); + assert!(without.requested_model.is_none()); + assert!( + !serde_json::to_string(&without) + .unwrap() + .contains("requested_model") + ); + + // An empty requested model is filtered to None. + assert!( + UserMessagePayload::new("hi", vec![], ProductTriggerReason::DirectChat) + .unwrap() + .with_requested_model(Some(String::new())) + .requested_model + .is_none() + ); + } + fn sample_context() -> TrustedInboundContext { let evidence = ProtocolAuthEvidence::test_verified( AuthRequirement::SharedSecretHeader { diff --git a/crates/ironclaw_product_workflow/src/auth_continuation.rs b/crates/ironclaw_product_workflow/src/auth_continuation.rs index 05d3a94cb31..55adad71bb9 100644 --- a/crates/ironclaw_product_workflow/src/auth_continuation.rs +++ b/crates/ironclaw_product_workflow/src/auth_continuation.rs @@ -795,6 +795,7 @@ mod tests { let actor = TurnActor::new(UserId::new("alice").unwrap()); let submit = coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: scope.clone(), actor: actor.clone(), accepted_message_ref: AcceptedMessageRef::new("message-auth-real").unwrap(), diff --git a/crates/ironclaw_product_workflow/src/inbound_turn.rs b/crates/ironclaw_product_workflow/src/inbound_turn.rs index 23055748768..5be923743df 100644 --- a/crates/ironclaw_product_workflow/src/inbound_turn.rs +++ b/crates/ironclaw_product_workflow/src/inbound_turn.rs @@ -491,6 +491,7 @@ where received_at: envelope.received_at(), adapter_id: prepared.adapter_id, surface_type: prepared.surface_type, + requested_model: payload.requested_model.clone(), })) .submit_or_replay(&self.thread_service, &self.turn_coordinator) .await @@ -654,6 +655,10 @@ impl ProductInboundTurnHandoff { received_at, adapter_id, surface_type, + // The requested model is not persisted in the message store, so an + // idempotent resubmission of an accepted message falls back to the + // deployment's active model rather than recovering the original hint. + requested_model: None, }, ))) } @@ -703,6 +708,7 @@ struct AcceptedProductInboundTurn { received_at: DateTime, adapter_id: ProductAdapterId, surface_type: TurnSurfaceType, + requested_model: Option, } impl AcceptedProductInboundTurn { @@ -725,6 +731,7 @@ impl AcceptedProductInboundTurn { received_at, adapter_id, surface_type, + requested_model, } = self; let turn_scope = TurnScope::new_with_owner( binding.tenant_id.clone(), @@ -779,6 +786,7 @@ impl AcceptedProductInboundTurn { source_binding_ref, reply_target_binding_ref, requested_run_profile: None, + requested_model, idempotency_key, received_at, requested_run_id: None, diff --git a/crates/ironclaw_product_workflow/src/reborn_services.rs b/crates/ironclaw_product_workflow/src/reborn_services.rs index d249afe3f4d..2c06db5a730 100644 --- a/crates/ironclaw_product_workflow/src/reborn_services.rs +++ b/crates/ironclaw_product_workflow/src/reborn_services.rs @@ -3755,6 +3755,7 @@ impl RebornServicesApi for RebornServices { )?; let product_context = ironclaw_product_context::resolve_web_ui(scope.product_owner(&actor)); let submit = SubmitTurnRequest { + requested_model: None, scope: scope.clone(), actor, accepted_message_ref: accepted_message_ref.clone(), diff --git a/crates/ironclaw_product_workflow/tests/product_workflow_contract.rs b/crates/ironclaw_product_workflow/tests/product_workflow_contract.rs index 2d643fc6151..4b645060132 100644 --- a/crates/ironclaw_product_workflow/tests/product_workflow_contract.rs +++ b/crates/ironclaw_product_workflow/tests/product_workflow_contract.rs @@ -2995,6 +2995,7 @@ async fn before_inbound_policy_path_probes_replay_once() { async fn before_inbound_policy_rewrite_revalidates_payload_before_turn_path() { let (workflow, inbound, ledger, policy) = build_workflow_with_policy(); policy.rewrite_user_message(UserMessagePayload { + requested_model: None, text: "a".repeat(64 * 1024 + 1), attachments: vec![], trigger: ProductTriggerReason::DirectChat, diff --git a/crates/ironclaw_reborn_composition/src/factory.rs b/crates/ironclaw_reborn_composition/src/factory.rs index 83c5c288e19..f2370b2de48 100644 --- a/crates/ironclaw_reborn_composition/src/factory.rs +++ b/crates/ironclaw_reborn_composition/src/factory.rs @@ -6572,6 +6572,7 @@ mod tests { Some(owner.clone()), ); let submit = ironclaw_turns::SubmitTurnRequest { + requested_model: None, scope, actor: ironclaw_turns::TurnActor::new(owner), accepted_message_ref: ironclaw_turns::AcceptedMessageRef::new("configured-message-ref") @@ -6695,6 +6696,7 @@ mod tests { Some(owner.clone()), ); let submit = ironclaw_turns::SubmitTurnRequest { + requested_model: None, scope, actor: ironclaw_turns::TurnActor::new(owner), accepted_message_ref: ironclaw_turns::AcceptedMessageRef::new("default-message-ref") diff --git a/crates/ironclaw_reborn_composition/src/factory/auth_tests.rs b/crates/ironclaw_reborn_composition/src/factory/auth_tests.rs index 48162c4c320..013ceacf663 100644 --- a/crates/ironclaw_reborn_composition/src/factory/auth_tests.rs +++ b/crates/ironclaw_reborn_composition/src/factory/auth_tests.rs @@ -193,6 +193,7 @@ async fn local_dev_oauth_turn_gate_callback_resumes_default_turn_coordinator() { let actor = TurnActor::new(UserId::new("alice").unwrap()); let submit = turn_coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: scope.clone(), actor: actor.clone(), accepted_message_ref: AcceptedMessageRef::new("message-auth-callback").unwrap(), @@ -761,6 +762,7 @@ async fn submit_and_block_provider_auth_run( ) -> TurnRunId { let submit = turn_coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: scope.clone(), actor, accepted_message_ref: AcceptedMessageRef::new(format!("message-fanout-{suffix}")) @@ -877,6 +879,7 @@ async fn submit_and_block_auth_run( ) -> ironclaw_turns::TurnRunId { let submit = turn_coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: scope.clone(), actor, accepted_message_ref: AcceptedMessageRef::new("message-auth-callback-2").unwrap(), diff --git a/crates/ironclaw_reborn_composition/src/runtime.rs b/crates/ironclaw_reborn_composition/src/runtime.rs index 68d378f53d6..793b58d7db9 100644 --- a/crates/ironclaw_reborn_composition/src/runtime.rs +++ b/crates/ironclaw_reborn_composition/src/runtime.rs @@ -2351,6 +2351,7 @@ impl RebornRuntime { let response = match self .turn_coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: scope.clone(), actor: TurnActor::new(self.actor_user_id.clone()), accepted_message_ref: accepted_message_ref.clone(), @@ -7727,6 +7728,7 @@ output_schema_ref = "schemas/write.output.json" let parent = runtime .turn_coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: parent_scope.clone(), actor: actor.clone(), accepted_message_ref: AcceptedMessageRef::new("msg:cancel-parent").unwrap(), @@ -10069,6 +10071,7 @@ output_schema_ref = "schemas/write.output.json" let submitted = runtime .turn_coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: scope.clone(), actor: actor.clone(), accepted_message_ref: AcceptedMessageRef::new("msg:audit").unwrap(), @@ -10650,6 +10653,7 @@ output_schema_ref = "schemas/write.output.json" let submitted_a = runtime .turn_coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: scope.clone(), actor: actor.clone(), accepted_message_ref: AcceptedMessageRef::new("msg:rejected-busy-a").unwrap(), diff --git a/crates/ironclaw_reborn_composition/src/runtime/tests/auth_interaction.rs b/crates/ironclaw_reborn_composition/src/runtime/tests/auth_interaction.rs index cf8f7312804..511dc706b82 100644 --- a/crates/ironclaw_reborn_composition/src/runtime/tests/auth_interaction.rs +++ b/crates/ironclaw_reborn_composition/src/runtime/tests/auth_interaction.rs @@ -199,6 +199,7 @@ async fn submit_and_block_auth_run( .turn_state .submit_turn( SubmitTurnRequest { + requested_model: None, scope: scope.clone(), actor, accepted_message_ref: AcceptedMessageRef::new("message-runtime-auth-read-model") diff --git a/crates/ironclaw_reborn_openai_compat/src/chat_workflow.rs b/crates/ironclaw_reborn_openai_compat/src/chat_workflow.rs index 7b206cc1400..35c9033cb52 100644 --- a/crates/ironclaw_reborn_openai_compat/src/chat_workflow.rs +++ b/crates/ironclaw_reborn_openai_compat/src/chat_workflow.rs @@ -795,7 +795,8 @@ fn chat_user_message_and_attachments( bytes: image.bytes, }) .collect(); - let payload = UserMessagePayload::new(text, vec![], ProductTriggerReason::DirectChat)?; + let payload = UserMessagePayload::new(text, vec![], ProductTriggerReason::DirectChat)? + .with_requested_model(Some(request.model.clone())); Ok((payload, attachments)) } diff --git a/crates/ironclaw_reborn_openai_compat/src/responses_workflow.rs b/crates/ironclaw_reborn_openai_compat/src/responses_workflow.rs index 56a03fbf559..5bb17f74bc4 100644 --- a/crates/ironclaw_reborn_openai_compat/src/responses_workflow.rs +++ b/crates/ironclaw_reborn_openai_compat/src/responses_workflow.rs @@ -1386,7 +1386,8 @@ fn responses_user_message_payload( responses_input_to_product_text(request)?, vec![], ProductTriggerReason::DirectChat, - )?) + )? + .with_requested_model(Some(request.model.clone()))) } fn responses_input_to_product_text( diff --git a/crates/ironclaw_runner/src/loop_driver_host.rs b/crates/ironclaw_runner/src/loop_driver_host.rs index d31323ca8b0..77c60303b8c 100644 --- a/crates/ironclaw_runner/src/loop_driver_host.rs +++ b/crates/ironclaw_runner/src/loop_driver_host.rs @@ -1968,9 +1968,14 @@ where .validate() .map_err(|reason| RebornLoopDriverHostError::InvalidRequest { reason })?; let Some(resolver) = &self.model_route_resolver else { - return Err(RebornLoopDriverHostError::InvalidRequest { - reason: "model route resolver is required for this host".to_string(), - }); + // No route resolver is wired (the default product runtime). The + // snapshot is a caller-requested model *hint* rather than an + // operator-approved route: pass it through unvalidated so the + // non-routed gateway can honor the model id when its provider + // supports per-request overrides and otherwise fall back to the + // active model. Routed hosts (resolver present) still validate + // and fail closed below. + return Ok(run_context); }; let slot = slot_for_model_profile(&run_context)?; let route = crate::model_routes::ModelRoute::new( diff --git a/crates/ironclaw_runner/src/model_gateway.rs b/crates/ironclaw_runner/src/model_gateway.rs index 445cb86920f..819c1db350b 100644 --- a/crates/ironclaw_runner/src/model_gateway.rs +++ b/crates/ironclaw_runner/src/model_gateway.rs @@ -391,7 +391,14 @@ where "model profile is not permitted", ) })?; - let model_override = request_model_override(route, self.provider.as_ref())?; + let model_override = request_model_override( + route, + self.provider.as_ref(), + request + .resolved_model_route + .as_ref() + .map(|snapshot| snapshot.model_id.as_str()), + )?; let model_profile_id = request.model_profile_id.clone(); let run_id = request.run_id; let turn_id = request.turn_id; @@ -426,7 +433,14 @@ where "model profile is not permitted", ) })?; - let model_override = request_model_override(route, self.provider.as_ref())?; + let model_override = request_model_override( + route, + self.provider.as_ref(), + request + .resolved_model_route + .as_ref() + .map(|snapshot| snapshot.model_id.as_str()), + )?; let model_profile_id = request.model_profile_id.clone(); let run_id = request.run_id; let turn_id = request.turn_id; @@ -461,7 +475,14 @@ where "model profile is not permitted", ) })?; - let model_override = request_model_override(route, self.provider.as_ref())?; + let model_override = request_model_override( + route, + self.provider.as_ref(), + request + .resolved_model_route + .as_ref() + .map(|snapshot| snapshot.model_id.as_str()), + )?; let model_profile_id = request.model_profile_id.clone(); let run_id = request.run_id; let turn_id = request.turn_id; @@ -501,7 +522,14 @@ where "model profile is not permitted", ) })?; - let model_override = request_model_override(route, self.provider.as_ref())?; + let model_override = request_model_override( + route, + self.provider.as_ref(), + request + .resolved_model_route + .as_ref() + .map(|snapshot| snapshot.model_id.as_str()), + )?; let model_profile_id = request.model_profile_id.clone(); let run_id = request.run_id; let turn_id = request.turn_id; @@ -979,14 +1007,22 @@ fn host_error_to_model_gateway_error(error: AgentLoopHostError) -> LoopModelGate fn request_model_override

( route: &LlmModelProfileRoute, provider: &P, + requested_model: Option<&str>, ) -> Result where P: LlmProvider + ?Sized, { - let model_override = route - .model_override - .as_deref() + // A per-run caller-requested model (an advisory route hint set at submit) + // takes precedence over the profile default. Providers that honor + // per-request overrides (e.g. NEAR AI) serve the requested model; providers + // that bake the model at construction ignore it and fall back to their + // active model — the "route if the provider can serve it, else fall back" + // behavior, decided at the provider boundary rather than a route allowlist. + let model_override = requested_model + .map(str::trim) + .filter(|model| !model.is_empty()) .map(str::to_string) + .or_else(|| route.model_override.as_deref().map(str::to_string)) .unwrap_or_else(|| provider.active_model_name()); let trimmed = model_override.trim(); if trimmed.is_empty() || trimmed.eq_ignore_ascii_case("default") { diff --git a/crates/ironclaw_runner/src/subagent/await_edge/boot_recovery.rs b/crates/ironclaw_runner/src/subagent/await_edge/boot_recovery.rs index 512de91edf8..5f93161fa40 100644 --- a/crates/ironclaw_runner/src/subagent/await_edge/boot_recovery.rs +++ b/crates/ironclaw_runner/src/subagent/await_edge/boot_recovery.rs @@ -731,6 +731,7 @@ mod tests { .. } = coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: parent_scope.clone(), actor: actor.clone(), accepted_message_ref: ironclaw_turns::AcceptedMessageRef::new( diff --git a/crates/ironclaw_runner/src/subagent/await_edge/resolver.rs b/crates/ironclaw_runner/src/subagent/await_edge/resolver.rs index 50494a84531..e97b9f90e7a 100644 --- a/crates/ironclaw_runner/src/subagent/await_edge/resolver.rs +++ b/crates/ironclaw_runner/src/subagent/await_edge/resolver.rs @@ -1375,6 +1375,7 @@ mod tests { ); let root_run_id = match coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: root_scope.clone(), actor: actor.clone(), accepted_message_ref: ironclaw_turns::AcceptedMessageRef::new("msg:tr-root") diff --git a/crates/ironclaw_runner/tests/concurrent_workers.rs b/crates/ironclaw_runner/tests/concurrent_workers.rs index ee2c596dac0..29d624c4e46 100644 --- a/crates/ironclaw_runner/tests/concurrent_workers.rs +++ b/crates/ironclaw_runner/tests/concurrent_workers.rs @@ -194,6 +194,7 @@ async fn submit_run_on_thread( let submit = turn_store .submit_turn( SubmitTurnRequest { + requested_model: None, scope: turn_scope, actor: TurnActor::new(user_id.clone()), accepted_message_ref: AcceptedMessageRef::new(format!( @@ -311,6 +312,7 @@ async fn submit_owned_run_on_thread( let submit = turn_store .submit_turn( SubmitTurnRequest { + requested_model: None, scope: turn_scope, actor: TurnActor::new(user_id.clone()), accepted_message_ref: AcceptedMessageRef::new(format!( diff --git a/crates/ironclaw_runner/tests/llm_gateway.rs b/crates/ironclaw_runner/tests/llm_gateway.rs index e234f305945..acc60905fc4 100644 --- a/crates/ironclaw_runner/tests/llm_gateway.rs +++ b/crates/ironclaw_runner/tests/llm_gateway.rs @@ -100,6 +100,56 @@ async fn gateway_calls_llm_provider_for_allowed_model_profile() { assert_eq!(requests[0].messages[1].content, "hello model"); } +#[tokio::test] +async fn gateway_honors_caller_requested_model_route_over_profile_default() { + let provider = Arc::new(RecordingLlmProvider::reply("assistant response")); + // Profile default resolves to "profile-default-model"; the caller's per-run + // requested-model route must take precedence. + let policy = LlmModelProfilePolicy::new().allow_model_profile( + interactive_model(), + Some("profile-default-model".to_string()), + ); + let gateway = LlmProviderModelGateway::with_provider_identity( + STATIC_PROVIDER_ID, + provider.clone(), + policy, + ); + + let request = model_request_with_route(interactive_model(), "requested", "caller-picked-model"); + gateway.stream_model(request).await.unwrap(); + + let requests = provider.requests.lock().unwrap(); + assert_eq!(requests.len(), 1); + assert_eq!( + requests[0].model.as_deref(), + Some("caller-picked-model"), + "the per-run requested model must override the profile default" + ); +} + +#[tokio::test] +async fn gateway_falls_back_to_profile_default_when_no_requested_route() { + let provider = Arc::new(RecordingLlmProvider::reply("assistant response")); + let policy = LlmModelProfilePolicy::new().allow_model_profile( + interactive_model(), + Some("profile-default-model".to_string()), + ); + let gateway = LlmProviderModelGateway::with_provider_identity( + STATIC_PROVIDER_ID, + provider.clone(), + policy, + ); + + // No resolved_model_route on the request → the profile default is used. + gateway + .stream_model(model_request(interactive_model())) + .await + .unwrap(); + + let requests = provider.requests.lock().unwrap(); + assert_eq!(requests[0].model.as_deref(), Some("profile-default-model")); +} + #[tokio::test] async fn gateway_stream_model_with_progress_uses_provider_streaming_and_sanitizes_updates() { let provider = Arc::new(StreamingRecordingLlmProvider::new( diff --git a/crates/ironclaw_runner/tests/loop_driver_host.rs b/crates/ironclaw_runner/tests/loop_driver_host.rs index 23037e33e06..a3a66014528 100644 --- a/crates/ironclaw_runner/tests/loop_driver_host.rs +++ b/crates/ironclaw_runner/tests/loop_driver_host.rs @@ -1985,6 +1985,7 @@ async fn turn_runner_worker_completes_after_libsql_turn_and_thread_services_reop let submit = turn_store .submit_turn( SubmitTurnRequest { + requested_model: None, scope: turn_scope.clone(), actor: TurnActor::new(user_id), accepted_message_ref: AcceptedMessageRef::new("accepted-libsql-restart") @@ -3589,6 +3590,7 @@ async fn default_planned_runtime_composes_no_profile_coordinator_and_profiled_ho let SubmitTurnResponse::Accepted { run_id, status, .. } = composition .coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: fixture.context.scope.clone(), actor: TurnActor::new(UserId::new("user-text-host").unwrap()), accepted_message_ref: AcceptedMessageRef::new("accepted-runtime-planned").unwrap(), @@ -3761,6 +3763,7 @@ async fn pre_minted_scheduler_wake_wiring_drives_scheduler_on_coordinator_submit let SubmitTurnResponse::Accepted { run_id, .. } = composition .coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: fixture.context.scope.clone(), actor: TurnActor::new(UserId::new("user-preminted-wake").unwrap()), accepted_message_ref: AcceptedMessageRef::new("accepted-preminted").unwrap(), @@ -8692,6 +8695,7 @@ async fn queue_fixture_turn( let submit = turn_store .submit_turn( SubmitTurnRequest { + requested_model: None, scope: fixture.context.scope.clone(), actor: TurnActor::new(UserId::new("user-text-host").unwrap()), accepted_message_ref: AcceptedMessageRef::new(format!( diff --git a/crates/ironclaw_runner/tests/turn_scheduler_contract.rs b/crates/ironclaw_runner/tests/turn_scheduler_contract.rs index 73acd3d2c24..f6707359343 100644 --- a/crates/ironclaw_runner/tests/turn_scheduler_contract.rs +++ b/crates/ironclaw_runner/tests/turn_scheduler_contract.rs @@ -2297,6 +2297,7 @@ where fn submit_turn_request(thread: &str, idempotency_key: &str) -> SubmitTurnRequest { SubmitTurnRequest { + requested_model: None, scope: scope(thread), actor: TurnActor::new(UserId::new("user1").unwrap()), accepted_message_ref: AcceptedMessageRef::new(format!("message-{thread}")).unwrap(), diff --git a/crates/ironclaw_turns/src/memory/mod.rs b/crates/ironclaw_turns/src/memory/mod.rs index 6bb168be3cc..2ba6afac9aa 100644 --- a/crates/ironclaw_turns/src/memory/mod.rs +++ b/crates/ironclaw_turns/src/memory/mod.rs @@ -1184,7 +1184,10 @@ impl TurnStateStore for InMemoryTurnStateStore { run_id, status: RunStatusCell::new(TurnStatus::Queued), profile: profile.clone(), - resolved_model_route: None, + resolved_model_route: request + .requested_model + .as_deref() + .and_then(crate::run_profile::LoopModelRouteSnapshot::advisory), model_usage: None, accepted_message_ref: request.accepted_message_ref.clone(), source_binding_ref: request.source_binding_ref.clone(), @@ -1386,6 +1389,7 @@ impl TurnSpawnTreeStateStore for InMemoryTurnStateStore { return response; } SubmitTurnRequest { + requested_model: None, scope: request.child_scope.clone(), actor: request.actor.clone(), accepted_message_ref: request.accepted_message_ref.clone(), @@ -4114,6 +4118,7 @@ mod tests { let response = store .submit_turn( SubmitTurnRequest { + requested_model: None, scope: scope.clone(), actor: TurnActor::new(UserId::new(format!("user-{index}")).unwrap()), accepted_message_ref: AcceptedMessageRef::new(format!("accepted-{index}")) @@ -4192,6 +4197,7 @@ mod tests { let response = store .submit_turn( SubmitTurnRequest { + requested_model: None, scope: scope.clone(), actor: TurnActor::new(UserId::new("user-lease-overlay").unwrap()), accepted_message_ref: AcceptedMessageRef::new("accepted-lease-overlay") diff --git a/crates/ironclaw_turns/src/request.rs b/crates/ironclaw_turns/src/request.rs index b4a7a4e7577..11142a08525 100644 --- a/crates/ironclaw_turns/src/request.rs +++ b/crates/ironclaw_turns/src/request.rs @@ -68,6 +68,11 @@ pub struct SubmitTurnRequest { pub source_binding_ref: SourceBindingRef, pub reply_target_binding_ref: ReplyTargetBindingRef, pub requested_run_profile: Option, + /// Caller-requested model for this turn. A hint the coordinator resolves to a + /// concrete per-run model route when the operator has it configured; when it + /// can't be resolved the run falls back to the deployment's active model. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub requested_model: Option, pub idempotency_key: IdempotencyKey, pub received_at: TurnTimestamp, #[serde(default, skip_serializing_if = "Option::is_none")] diff --git a/crates/ironclaw_turns/src/run_profile/host.rs b/crates/ironclaw_turns/src/run_profile/host.rs index 3e2bd4ca4cf..c64f6bb600a 100644 --- a/crates/ironclaw_turns/src/run_profile/host.rs +++ b/crates/ironclaw_turns/src/run_profile/host.rs @@ -518,6 +518,10 @@ fn origin_input_cursor_token() -> LoopInputCursorToken { LoopInputCursorToken("input-cursor:origin".to_string()) } +/// Placeholder component value marking a [`LoopModelRouteSnapshot`] as a +/// caller-requested advisory hint rather than an operator-resolved route. +const ADVISORY_MODEL_ROUTE_COMPONENT: &str = "requested"; + #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct LoopModelRouteSnapshot { pub provider_id: String, @@ -552,6 +556,38 @@ impl LoopModelRouteSnapshot { Ok(snapshot) } + /// Build an *advisory* route from a caller-requested model string. The + /// provider/config/auth components are placeholders (`"requested"`) — only + /// `model_id` carries meaning. Advisory routes exist so a caller (e.g. an + /// OpenAI-compatible client) can request a model without an operator-approved + /// route binding: the non-routed gateway honors the model id when its + /// provider supports per-request overrides and otherwise falls back to the + /// active model, while routed hosts still validate the route and fail closed. + /// Returns `None` when the model is empty or not a valid route component, so + /// the run falls back to the deployment's active model. + pub fn advisory(requested_model: &str) -> Option { + let model = requested_model.trim(); + if model.is_empty() { + return None; + } + Self::try_new( + ADVISORY_MODEL_ROUTE_COMPONENT, + model, + ADVISORY_MODEL_ROUTE_COMPONENT, + ADVISORY_MODEL_ROUTE_COMPONENT, + ) + .ok() + } + + /// Whether this route is a caller-requested advisory hint (see + /// [`LoopModelRouteSnapshot::advisory`]) rather than an operator-resolved + /// route. + pub fn is_advisory(&self) -> bool { + self.provider_id == ADVISORY_MODEL_ROUTE_COMPONENT + && self.config_version == ADVISORY_MODEL_ROUTE_COMPONENT + && self.auth_version == ADVISORY_MODEL_ROUTE_COMPONENT + } + pub fn validate(&self) -> Result<(), String> { validate_model_route_component_value("provider_id", &self.provider_id, 128, |character| { character.is_ascii_alphanumeric() || matches!(character, '_' | '-' | '.') @@ -2581,6 +2617,33 @@ fn unsupported_host_method(method: &'static str) -> AgentLoopHostError { mod tests { use super::*; + #[test] + fn advisory_model_route_carries_model_and_marks_itself_advisory() { + let route = LoopModelRouteSnapshot::advisory("gpt-4o").expect("valid model"); + assert_eq!(route.model_id, "gpt-4o"); + assert!(route.is_advisory()); + assert!(route.validate().is_ok()); + } + + #[test] + fn advisory_model_route_trims_and_rejects_empty_or_invalid_models() { + assert_eq!(LoopModelRouteSnapshot::advisory(" "), None); + assert_eq!(LoopModelRouteSnapshot::advisory(""), None); + // A model id with a space is not a valid route component → falls back. + assert_eq!(LoopModelRouteSnapshot::advisory("gpt 4o"), None); + // Surrounding whitespace is trimmed before validation. + assert_eq!( + LoopModelRouteSnapshot::advisory(" claude-opus-4-6 ").map(|route| route.model_id), + Some("claude-opus-4-6".to_string()) + ); + } + + #[test] + fn operator_resolved_route_is_not_advisory() { + let route = LoopModelRouteSnapshot::new("openai", "gpt-4o", "config:v1", "auth:v1"); + assert!(!route.is_advisory()); + } + struct DefinitionPort { definitions: Vec, } diff --git a/crates/ironclaw_turns/tests/active_run_ref_state_contract.rs b/crates/ironclaw_turns/tests/active_run_ref_state_contract.rs index 941b8092b59..fd102035be8 100644 --- a/crates/ironclaw_turns/tests/active_run_ref_state_contract.rs +++ b/crates/ironclaw_turns/tests/active_run_ref_state_contract.rs @@ -71,6 +71,7 @@ fn submit_request_for( idempotency_key: &str, ) -> ironclaw_turns::SubmitTurnRequest { ironclaw_turns::SubmitTurnRequest { + requested_model: None, scope, actor: turn_actor(), accepted_message_ref: AcceptedMessageRef::new(format!("message-{idempotency_key}")) diff --git a/crates/ironclaw_turns/tests/agent_loop_host_contract.rs b/crates/ironclaw_turns/tests/agent_loop_host_contract.rs index 56f50c2c676..75924caa589 100644 --- a/crates/ironclaw_turns/tests/agent_loop_host_contract.rs +++ b/crates/ironclaw_turns/tests/agent_loop_host_contract.rs @@ -3306,6 +3306,7 @@ async fn claimed_run_context() -> LoopRunContext { let coordinator = DefaultTurnCoordinator::new(store.clone()); let response = coordinator .submit_turn(SubmitTurnRequest { + requested_model: None, scope: scope.clone(), actor: TurnActor::new(UserId::new("user-loop").unwrap()), accepted_message_ref: AcceptedMessageRef::new("message-loop-host").unwrap(), diff --git a/crates/ironclaw_turns/tests/filesystem_turn_state_contract.rs b/crates/ironclaw_turns/tests/filesystem_turn_state_contract.rs index 26d76f98b40..ae80536e4e1 100644 --- a/crates/ironclaw_turns/tests/filesystem_turn_state_contract.rs +++ b/crates/ironclaw_turns/tests/filesystem_turn_state_contract.rs @@ -1181,6 +1181,7 @@ fn turn_actor() -> TurnActor { fn submit_request_for(scope: TurnScope, idempotency_key: &str) -> SubmitTurnRequest { SubmitTurnRequest { + requested_model: None, scope, actor: turn_actor(), accepted_message_ref: AcceptedMessageRef::new(format!("message-{idempotency_key}")) diff --git a/crates/ironclaw_turns/tests/per_inbound_type_concurrency_cap.rs b/crates/ironclaw_turns/tests/per_inbound_type_concurrency_cap.rs index 5dcc45c9054..32421684481 100644 --- a/crates/ironclaw_turns/tests/per_inbound_type_concurrency_cap.rs +++ b/crates/ironclaw_turns/tests/per_inbound_type_concurrency_cap.rs @@ -96,6 +96,7 @@ fn submit_request( ) -> SubmitTurnRequest { let owner = scope.explicit_owner_user_id().unwrap().clone(); SubmitTurnRequest { + requested_model: None, actor: actor_for(&owner), accepted_message_ref: AcceptedMessageRef::new(format!("message-{key}")).unwrap(), source_binding_ref: SourceBindingRef::new("source-web").unwrap(), diff --git a/crates/ironclaw_turns/tests/per_user_concurrency_cap.rs b/crates/ironclaw_turns/tests/per_user_concurrency_cap.rs index 013387dae2c..3fe2e9568bc 100644 --- a/crates/ironclaw_turns/tests/per_user_concurrency_cap.rs +++ b/crates/ironclaw_turns/tests/per_user_concurrency_cap.rs @@ -57,6 +57,7 @@ fn actor_for(user: &UserId) -> TurnActor { fn submit_request_for(scope: TurnScope, key: &str) -> SubmitTurnRequest { let actor = actor_for(scope.explicit_owner_user_id().unwrap()); SubmitTurnRequest { + requested_model: None, actor, accepted_message_ref: AcceptedMessageRef::new(format!("message-{key}")).unwrap(), source_binding_ref: SourceBindingRef::new("source-web").unwrap(), @@ -583,6 +584,7 @@ async fn ownerless_runs_are_not_counted_against_cap() { ); let make_req = |scope: TurnScope, key: &'static str| SubmitTurnRequest { + requested_model: None, scope, actor: actor.clone(), accepted_message_ref: AcceptedMessageRef::new(format!("msg-{key}")).unwrap(), @@ -683,6 +685,7 @@ async fn actor_fallback_runs_are_capped_under_actor_user_id() { ); let make_req = |scope: TurnScope, actor: TurnActor, key: &'static str| SubmitTurnRequest { + requested_model: None, scope, actor, accepted_message_ref: AcceptedMessageRef::new(format!("msg-{key}")).unwrap(), diff --git a/crates/ironclaw_turns/tests/retry_failed_turn_store_contract.rs b/crates/ironclaw_turns/tests/retry_failed_turn_store_contract.rs index 8135b78f140..fec839272ae 100644 --- a/crates/ironclaw_turns/tests/retry_failed_turn_store_contract.rs +++ b/crates/ironclaw_turns/tests/retry_failed_turn_store_contract.rs @@ -85,6 +85,7 @@ fn actor() -> TurnActor { fn submit_request(thread: &str, idempotency_key: &str) -> SubmitTurnRequest { SubmitTurnRequest { + requested_model: None, scope: scope(thread), actor: actor(), accepted_message_ref: AcceptedMessageRef::new(format!("message-{thread}")).unwrap(), diff --git a/crates/ironclaw_turns/tests/turn_coordinator_contract.rs b/crates/ironclaw_turns/tests/turn_coordinator_contract.rs index d2ad6215738..1e20cc83a6b 100644 --- a/crates/ironclaw_turns/tests/turn_coordinator_contract.rs +++ b/crates/ironclaw_turns/tests/turn_coordinator_contract.rs @@ -419,6 +419,51 @@ async fn prepare_turn_mints_ids_without_side_effects_and_submit_binds_requested_ assert!(matches!(err, TurnError::Conflict { .. })); } +#[tokio::test] +async fn submit_turn_records_advisory_model_route_from_requested_model() { + let (coordinator, _store) = coordinator(); + let mut request = submit_request("thread-model-select", "idem-model-select"); + request.requested_model = Some("gpt-4o".to_string()); + + let run_id = accepted_run_id(&coordinator.submit_turn(request).await.unwrap()); + + let state = coordinator + .get_run_state(GetRunStateRequest { + scope: scope("thread-model-select"), + run_id, + }) + .await + .unwrap(); + let route = state + .resolved_model_route + .expect("advisory model route recorded from requested_model"); + assert_eq!(route.model_id, "gpt-4o"); + assert!( + route.is_advisory(), + "a caller-requested model is an advisory route, not an operator-resolved one" + ); +} + +#[tokio::test] +async fn submit_turn_without_requested_model_records_no_route() { + let (coordinator, _store) = coordinator(); + let run_id = accepted_run_id( + &coordinator + .submit_turn(submit_request("thread-no-model", "idem-no-model")) + .await + .unwrap(), + ); + + let state = coordinator + .get_run_state(GetRunStateRequest { + scope: scope("thread-no-model"), + run_id, + }) + .await + .unwrap(); + assert!(state.resolved_model_route.is_none()); +} + #[tokio::test] async fn children_of_get_run_record_and_tree_reservation_are_scope_checked() { let (coordinator, store) = coordinator(); @@ -6920,6 +6965,7 @@ async fn complete_queued_run(store: &InMemoryTurnStateStore, run_id: TurnRunId, fn submit_request(thread: &str, idempotency_key: &str) -> SubmitTurnRequest { SubmitTurnRequest { + requested_model: None, scope: scope(thread), actor: actor(), accepted_message_ref: AcceptedMessageRef::new(format!("message-{thread}")).unwrap(), diff --git a/tests/integration/subagent_await_edge.rs b/tests/integration/subagent_await_edge.rs index 0185f97826f..472ae4e2494 100644 --- a/tests/integration/subagent_await_edge.rs +++ b/tests/integration/subagent_await_edge.rs @@ -410,6 +410,7 @@ async fn rollback_deleted_edge_is_reconstructed_so_the_parent_still_gets_the_res // state a real parent is in while its blocking-mode child runs. let submitted = coordinator .submit_turn(ironclaw_turns::SubmitTurnRequest { + requested_model: None, scope: parent_scope.clone(), actor: actor.clone(), accepted_message_ref: ironclaw_turns::AcceptedMessageRef::new("msg:parent-rollback") @@ -705,6 +706,7 @@ async fn mixed_status_batch_group_reports_each_members_own_status_and_reason() { // 1. Submit and block the parent on a shared dependent-run gate. let submitted = coordinator .submit_turn(ironclaw_turns::SubmitTurnRequest { + requested_model: None, scope: parent_scope.clone(), actor: actor.clone(), accepted_message_ref: ironclaw_turns::AcceptedMessageRef::new("msg:parent-mixed-batch") From 670fd368d2b876f1d6d557ebc483e34e0bb84feb Mon Sep 17 00:00:00 2001 From: Illia Polosukhin Date: Tue, 14 Jul 2026 04:44:40 +0000 Subject: [PATCH 3/6] refactor(reborn): drop dead LoopModelRouteSnapshot::is_advisory predicate MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The is_advisory() method had zero production consumers: the model gateway reads snapshot.model_id uniformly regardless of advisory-ness, and loop_driver_host gates on route-resolver presence rather than the advisory flag. The predicate (and the operator_resolved_route_is_not_advisory test that existed only to exercise it) dressed the three "requested" sentinel placeholder components up as a first-class concept nothing acts on. Keep advisory() — the actual reuse seam that stores a caller-requested model hint in resolved_model_route — and have the remaining tests assert on model_id, the only component that carries meaning. Addresses thermo-nuclear code-quality review of PR #5985. Co-Authored-By: Claude Opus 4.8 (1M context) --- crates/ironclaw_turns/src/run_profile/host.rs | 18 +----------------- .../tests/turn_coordinator_contract.rs | 4 ---- 2 files changed, 1 insertion(+), 21 deletions(-) diff --git a/crates/ironclaw_turns/src/run_profile/host.rs b/crates/ironclaw_turns/src/run_profile/host.rs index c64f6bb600a..e53cefaf1cb 100644 --- a/crates/ironclaw_turns/src/run_profile/host.rs +++ b/crates/ironclaw_turns/src/run_profile/host.rs @@ -579,15 +579,6 @@ impl LoopModelRouteSnapshot { .ok() } - /// Whether this route is a caller-requested advisory hint (see - /// [`LoopModelRouteSnapshot::advisory`]) rather than an operator-resolved - /// route. - pub fn is_advisory(&self) -> bool { - self.provider_id == ADVISORY_MODEL_ROUTE_COMPONENT - && self.config_version == ADVISORY_MODEL_ROUTE_COMPONENT - && self.auth_version == ADVISORY_MODEL_ROUTE_COMPONENT - } - pub fn validate(&self) -> Result<(), String> { validate_model_route_component_value("provider_id", &self.provider_id, 128, |character| { character.is_ascii_alphanumeric() || matches!(character, '_' | '-' | '.') @@ -2618,10 +2609,9 @@ mod tests { use super::*; #[test] - fn advisory_model_route_carries_model_and_marks_itself_advisory() { + fn advisory_model_route_carries_requested_model_and_validates() { let route = LoopModelRouteSnapshot::advisory("gpt-4o").expect("valid model"); assert_eq!(route.model_id, "gpt-4o"); - assert!(route.is_advisory()); assert!(route.validate().is_ok()); } @@ -2638,12 +2628,6 @@ mod tests { ); } - #[test] - fn operator_resolved_route_is_not_advisory() { - let route = LoopModelRouteSnapshot::new("openai", "gpt-4o", "config:v1", "auth:v1"); - assert!(!route.is_advisory()); - } - struct DefinitionPort { definitions: Vec, } diff --git a/crates/ironclaw_turns/tests/turn_coordinator_contract.rs b/crates/ironclaw_turns/tests/turn_coordinator_contract.rs index 1e20cc83a6b..f4032d82d18 100644 --- a/crates/ironclaw_turns/tests/turn_coordinator_contract.rs +++ b/crates/ironclaw_turns/tests/turn_coordinator_contract.rs @@ -438,10 +438,6 @@ async fn submit_turn_records_advisory_model_route_from_requested_model() { .resolved_model_route .expect("advisory model route recorded from requested_model"); assert_eq!(route.model_id, "gpt-4o"); - assert!( - route.is_advisory(), - "a caller-requested model is an advisory route, not an operator-resolved one" - ); } #[tokio::test] From 65d02c0ffaa5f4c3b47000d9b21b112934f14814 Mon Sep 17 00:00:00 2001 From: Illia Polosukhin Date: Tue, 14 Jul 2026 05:13:39 +0000 Subject: [PATCH 4/6] fix(reborn): distinguish advisory vs operator route on resolver-less host MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Merging main surfaced a semantic conflict between two independently-correct changes: - #5985 made a resolver-less host (the default product runtime) pass a persisted model-route snapshot through unvalidated, so an OpenAI-compatible caller's requested model is honored by the non-routed gateway. - main independently hardened the same host to FAIL CLOSED when a model-route snapshot is present but no resolver is wired (an operator route we cannot validate is a misconfiguration), with tests locking that behavior. The auto-merge collapsed these into an unconditional pass-through, which broke `text_only_host_factory_rejects_persisted_model_route_snapshot_without_resolver`. Both intentions are correct and coexist by discriminating the snapshot kind — exactly what LoopModelRouteSnapshot::is_advisory() expresses: - advisory snapshot (caller-requested hint) + no resolver -> pass through - operator route + no resolver -> fail closed ("resolver required") This reverses the earlier "drop dead is_advisory()" commit on this branch: the predicate looked unused in #5985 in isolation, but main's stricter guard makes it load-bearing. is_advisory() and its unit tests are restored, the guard now branches on it, and a caller-path regression test (`text_only_host_factory_passes_advisory_model_route_snapshot_without_resolver`) pins the pass-through side that main's reject test does not cover. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../ironclaw_runner/src/loop_driver_host.rs | 17 +++++++--- .../ironclaw_runner/tests/loop_driver_host.rs | 33 +++++++++++++++++++ crates/ironclaw_turns/src/run_profile/host.rs | 20 ++++++++++- .../tests/turn_coordinator_contract.rs | 4 +++ tools/ironclaw_stress/src/user_turn.rs | 3 ++ 5 files changed, 71 insertions(+), 6 deletions(-) diff --git a/crates/ironclaw_runner/src/loop_driver_host.rs b/crates/ironclaw_runner/src/loop_driver_host.rs index 3268fddab48..e17cdf965b2 100644 --- a/crates/ironclaw_runner/src/loop_driver_host.rs +++ b/crates/ironclaw_runner/src/loop_driver_host.rs @@ -1968,14 +1968,21 @@ where .validate() .map_err(|reason| RebornLoopDriverHostError::InvalidRequest { reason })?; let Some(resolver) = &self.model_route_resolver else { - // No route resolver is wired (the default product runtime). The - // snapshot is a caller-requested model *hint* rather than an + // No route resolver is wired (the default product runtime). An + // *advisory* snapshot is a caller-requested model hint, not an // operator-approved route: pass it through unvalidated so the // non-routed gateway can honor the model id when its provider // supports per-request overrides and otherwise fall back to the - // active model. Routed hosts (resolver present) still validate - // and fail closed below. - return Ok(run_context); + // active model. A non-advisory (operator) route persisted on a + // resolver-less host is a misconfiguration we cannot validate, + // so fail closed. Routed hosts (resolver present) validate every + // route below. + if snapshot.is_advisory() { + return Ok(run_context); + } + return Err(RebornLoopDriverHostError::InvalidRequest { + reason: "model route resolver is required for this host".to_string(), + }); }; let slot = slot_for_model_profile(&run_context)?; let route = crate::model_routes::ModelRoute::new( diff --git a/crates/ironclaw_runner/tests/loop_driver_host.rs b/crates/ironclaw_runner/tests/loop_driver_host.rs index d36ac3d857a..4d8da234105 100644 --- a/crates/ironclaw_runner/tests/loop_driver_host.rs +++ b/crates/ironclaw_runner/tests/loop_driver_host.rs @@ -4582,6 +4582,39 @@ async fn text_only_host_factory_rejects_persisted_model_route_snapshot_without_r ); } +#[tokio::test] +async fn text_only_host_factory_passes_advisory_model_route_snapshot_without_resolver() { + // A caller-requested model reaches the default (resolver-less) product + // runtime as an *advisory* snapshot. Unlike an operator route, it must pass + // through unvalidated so the non-routed gateway can honor the requested + // model id, rather than failing closed the way an operator route does. + let fixture = HostFixture::new("thread-host-advisory-no-resolver", "hello routed host").await; + let advisory_snapshot = + LoopModelRouteSnapshot::advisory("gpt-4o").expect("valid advisory model"); + let context = fixture + .context + .clone() + .with_resolved_model_route(advisory_snapshot.clone()); + let mut claimed = fixture.claimed.clone(); + claimed.state.resolved_model_route = Some(advisory_snapshot.clone()); + + let host = fixture + .factory() + .build_text_only_host(RebornLoopDriverHostRequest { + claimed_run: claimed, + loop_run_context: context, + }) + .await + .expect("advisory snapshot passes through a resolver-less host"); + + let host_dyn: &(dyn AgentLoopDriverHost + Send + Sync) = &host; + assert_eq!( + host_dyn.run_context().resolved_model_route, + Some(advisory_snapshot), + "the advisory model id must survive on the run context for the gateway to honor it" + ); +} + #[tokio::test] async fn text_only_host_factory_rejects_persisted_model_route_snapshot_denied_by_policy() { let fixture = HostFixture::new("thread-host-model-route-denied", "hello routed host").await; diff --git a/crates/ironclaw_turns/src/run_profile/host.rs b/crates/ironclaw_turns/src/run_profile/host.rs index 577270cf6b9..72bb4547c97 100644 --- a/crates/ironclaw_turns/src/run_profile/host.rs +++ b/crates/ironclaw_turns/src/run_profile/host.rs @@ -579,6 +579,17 @@ impl LoopModelRouteSnapshot { .ok() } + /// Whether this route is a caller-requested advisory hint (see + /// [`LoopModelRouteSnapshot::advisory`]) rather than an operator-resolved + /// route. A non-routed host passes an advisory snapshot through unvalidated + /// but fails closed on an operator route it cannot validate without a + /// resolver. + pub fn is_advisory(&self) -> bool { + self.provider_id == ADVISORY_MODEL_ROUTE_COMPONENT + && self.config_version == ADVISORY_MODEL_ROUTE_COMPONENT + && self.auth_version == ADVISORY_MODEL_ROUTE_COMPONENT + } + pub fn validate(&self) -> Result<(), String> { validate_model_route_component_value("provider_id", &self.provider_id, 128, |character| { character.is_ascii_alphanumeric() || matches!(character, '_' | '-' | '.') @@ -2619,12 +2630,19 @@ mod tests { use super::*; #[test] - fn advisory_model_route_carries_requested_model_and_validates() { + fn advisory_model_route_carries_model_and_marks_itself_advisory() { let route = LoopModelRouteSnapshot::advisory("gpt-4o").expect("valid model"); assert_eq!(route.model_id, "gpt-4o"); + assert!(route.is_advisory()); assert!(route.validate().is_ok()); } + #[test] + fn operator_resolved_route_is_not_advisory() { + let route = LoopModelRouteSnapshot::new("openai", "gpt-4o", "config:v1", "auth:v1"); + assert!(!route.is_advisory()); + } + #[test] fn advisory_model_route_trims_and_rejects_empty_or_invalid_models() { assert_eq!(LoopModelRouteSnapshot::advisory(" "), None); diff --git a/crates/ironclaw_turns/tests/turn_coordinator_contract.rs b/crates/ironclaw_turns/tests/turn_coordinator_contract.rs index ae57afec00b..2900c266e9f 100644 --- a/crates/ironclaw_turns/tests/turn_coordinator_contract.rs +++ b/crates/ironclaw_turns/tests/turn_coordinator_contract.rs @@ -489,6 +489,10 @@ async fn submit_turn_records_advisory_model_route_from_requested_model() { .resolved_model_route .expect("advisory model route recorded from requested_model"); assert_eq!(route.model_id, "gpt-4o"); + assert!( + route.is_advisory(), + "a caller-requested model is an advisory route, not an operator-resolved one" + ); } #[tokio::test] diff --git a/tools/ironclaw_stress/src/user_turn.rs b/tools/ironclaw_stress/src/user_turn.rs index 7ea17c4717d..932581424dc 100644 --- a/tools/ironclaw_stress/src/user_turn.rs +++ b/tools/ironclaw_stress/src/user_turn.rs @@ -767,6 +767,7 @@ where reply_target_binding_ref: ReplyTargetBindingRef::new(reply_target) .map_err(|error| OperationFailure::invalid_request("prefill_submit", error))?, requested_run_profile: None, + requested_model: None, idempotency_key: IdempotencyKey::new(format!( "ironclaw-stress-prefill:{operation_ref}" )) @@ -946,6 +947,7 @@ where reply_target_binding_ref: ReplyTargetBindingRef::new(reply_target) .map_err(|error| OperationFailure::invalid_request("submit_turn", error))?, requested_run_profile: None, + requested_model: None, idempotency_key: IdempotencyKey::new(format!( "ironclaw-stress:{operation_ref}" )) @@ -1062,6 +1064,7 @@ where reply_target_binding_ref: ReplyTargetBindingRef::new(reply_target) .map_err(|error| OperationFailure::invalid_request("submit_turn", error))?, requested_run_profile: None, + requested_model: None, idempotency_key: IdempotencyKey::new(format!( "ironclaw-stress:{operation_ref}" )) From 2e36e17f4e47b0b415cc98cc7e8289199a5cd369 Mon Sep 17 00:00:00 2001 From: Illia Polosukhin Date: Tue, 14 Jul 2026 05:31:54 +0000 Subject: [PATCH 5/6] fix(reborn): don't forward the "default" model alias as a routing hint MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Responses/Chat OpenAI-compat surfaces forwarded request.model verbatim as a per-run requested-model hint. But "default" is the server's alias for "use the active model" (the models listing advertises it), not a concrete model id. Forwarding it created an advisory route with model_id "default", which request_model_override rejects as non-concrete (PolicyDenied) — failing every run whose client sent model="default", including 6 legacy Responses API E2E scenarios (status "failed" instead of "completed"). Map the wire model through model_validation::requested_model_hint before forwarding: the "default" sentinel (and, defensively, empty) yields None so the run falls back to normal resolution — the active model on the non-routed gateway, the resolver's default route on routed hosts — while a concrete model name is still forwarded. Keeps the model_gateway "default"-is-not-concrete guard intact for genuine route/active-model misconfiguration. Regression: model_validation unit tests for the sentinel/whitespace/concrete cases; the legacy Responses API E2E suite exercises the full path. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../src/chat_workflow.rs | 4 +- .../src/model_validation.rs | 39 +++++++++++++++++++ .../src/responses_workflow.rs | 4 +- 3 files changed, 45 insertions(+), 2 deletions(-) diff --git a/crates/ironclaw_reborn_openai_compat/src/chat_workflow.rs b/crates/ironclaw_reborn_openai_compat/src/chat_workflow.rs index 35c9033cb52..a2ac3a98c0c 100644 --- a/crates/ironclaw_reborn_openai_compat/src/chat_workflow.rs +++ b/crates/ironclaw_reborn_openai_compat/src/chat_workflow.rs @@ -796,7 +796,9 @@ fn chat_user_message_and_attachments( }) .collect(); let payload = UserMessagePayload::new(text, vec![], ProductTriggerReason::DirectChat)? - .with_requested_model(Some(request.model.clone())); + .with_requested_model(crate::model_validation::requested_model_hint( + &request.model, + )); Ok((payload, attachments)) } diff --git a/crates/ironclaw_reborn_openai_compat/src/model_validation.rs b/crates/ironclaw_reborn_openai_compat/src/model_validation.rs index cc002bec3ca..ae6781fd67c 100644 --- a/crates/ironclaw_reborn_openai_compat/src/model_validation.rs +++ b/crates/ironclaw_reborn_openai_compat/src/model_validation.rs @@ -12,6 +12,27 @@ use crate::OpenAiCompatHttpError; /// Maximum accepted `model` string length, in bytes. pub(crate) const MAX_MODEL_NAME_BYTES: usize = 256; +/// The OpenAI-compatible alias every client may send to mean "use the server's +/// active/default model" rather than naming a concrete one. The models listing +/// advertises it, so it is not a routable model id. +const DEFAULT_MODEL_ALIAS: &str = "default"; + +/// Map a validated client `model` string to an optional caller-requested model +/// *hint* for turn routing. +/// +/// Returns `None` for the [`DEFAULT_MODEL_ALIAS`] sentinel (and defensively for +/// empty), so a client asking for the server default does not pin an advisory +/// route to the non-routable `"default"` id — which the model gateway rejects as +/// non-concrete and which would fail route resolution on routed hosts. A +/// concrete model name is forwarded as `Some`. +pub(crate) fn requested_model_hint(model: &str) -> Option { + let trimmed = model.trim(); + if trimmed.is_empty() || trimmed.eq_ignore_ascii_case(DEFAULT_MODEL_ALIAS) { + return None; + } + Some(trimmed.to_string()) +} + /// Validate the client-supplied `model` string before it is carried as a /// projection/policy hint. /// @@ -82,4 +103,22 @@ mod tests { let at_cap = "m".repeat(MAX_MODEL_NAME_BYTES); assert!(validate_model_name(&at_cap).is_ok()); } + + #[test] + fn requested_model_hint_drops_default_sentinel() { + assert_eq!(requested_model_hint("default"), None); + assert_eq!(requested_model_hint("DEFAULT"), None); + assert_eq!(requested_model_hint("Default"), None); + assert_eq!(requested_model_hint(""), None); + assert_eq!(requested_model_hint(" "), None); + } + + #[test] + fn requested_model_hint_forwards_concrete_model() { + assert_eq!(requested_model_hint("gpt-4o"), Some("gpt-4o".to_string())); + assert_eq!( + requested_model_hint("anthropic/claude-opus-4"), + Some("anthropic/claude-opus-4".to_string()) + ); + } } diff --git a/crates/ironclaw_reborn_openai_compat/src/responses_workflow.rs b/crates/ironclaw_reborn_openai_compat/src/responses_workflow.rs index 5bb17f74bc4..03436f05685 100644 --- a/crates/ironclaw_reborn_openai_compat/src/responses_workflow.rs +++ b/crates/ironclaw_reborn_openai_compat/src/responses_workflow.rs @@ -1387,7 +1387,9 @@ fn responses_user_message_payload( vec![], ProductTriggerReason::DirectChat, )? - .with_requested_model(Some(request.model.clone()))) + .with_requested_model(crate::model_validation::requested_model_hint( + &request.model, + ))) } fn responses_input_to_product_text( From f5aff5dc7e640027b6e3d39679120019ba1a100e Mon Sep 17 00:00:00 2001 From: Illia Polosukhin Date: Tue, 14 Jul 2026 05:46:01 +0000 Subject: [PATCH 6/6] fix(reborn): bound requested_model on every UserMessagePayload construction MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CodeRabbit review flagged that UserMessagePayload::with_requested_model attaches the model hint AFTER new() ran validate() (with requested_model still None), so the 256-byte REQUESTED_MODEL_MAX_BYTES bound was bypassed. The crate contract is explicit: "Validated DTOs must validate both constructors and serde deserialization." - The custom Deserialize (untrusted wire path) now re-validates the assembled payload, so a wire-supplied requested_model is bounded like every other ingress field. This is the real bypass — an unbounded model string could otherwise deserialize and flow to persistence/dispatch. - The Responses and Chat OpenAI-compat builder call sites now validate the assembled payload before submitting (defense in depth; request.model is already capped at parse by validate_model_name, but the type invariant must hold locally). Regression: user_message_payload_bounds_requested_model_on_every_path asserts an over-limit hint is rejected via both the builder+validate path and deserialization, and that a hint at the cap is accepted. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../ironclaw_product_adapters/src/inbound.rs | 45 ++++++++++++++++++- .../src/chat_workflow.rs | 3 ++ .../src/responses_workflow.rs | 8 +++- 3 files changed, 52 insertions(+), 4 deletions(-) diff --git a/crates/ironclaw_product_adapters/src/inbound.rs b/crates/ironclaw_product_adapters/src/inbound.rs index 154dc39d31d..83ff7e69af5 100644 --- a/crates/ironclaw_product_adapters/src/inbound.rs +++ b/crates/ironclaw_product_adapters/src/inbound.rs @@ -156,9 +156,14 @@ impl<'de> Deserialize<'de> for UserMessagePayload { D: Deserializer<'de>, { let wire = UserMessagePayloadWire::deserialize(deserializer)?; - Self::new(wire.text, wire.attachments, wire.trigger) + let payload = Self::new(wire.text, wire.attachments, wire.trigger) .map(|payload| payload.with_requested_model(wire.requested_model)) - .map_err(serde::de::Error::custom) + .map_err(serde::de::Error::custom)?; + // `new` validated the payload while `requested_model` was still `None`; + // re-validate the assembled value so the wire-supplied model hint is + // bounded like every other ingress field (bypass flagged in PR review). + payload.validate().map_err(serde::de::Error::custom)?; + Ok(payload) } } @@ -937,6 +942,42 @@ mod tests { ); } + #[test] + fn user_message_payload_bounds_requested_model_on_every_path() { + let over_limit = "m".repeat(REQUESTED_MODEL_MAX_BYTES + 1); + + // Explicit validation after the builder rejects an over-long hint. + let built = UserMessagePayload::new("hi", vec![], ProductTriggerReason::DirectChat) + .unwrap() + .with_requested_model(Some(over_limit.clone())); + assert!(built.validate().is_err()); + + // Deserialization must not smuggle an unbounded hint past validation: + // the wire path attaches `requested_model` after `new`, so it re-validates. + let wire = serde_json::json!({ + "text": "hi", + "attachments": [], + "trigger": "direct_chat", + "requested_model": over_limit, + }) + .to_string(); + let decoded: Result = serde_json::from_str(&wire); + assert!( + decoded.is_err(), + "an over-long requested_model must be rejected during deserialization" + ); + + // A hint at the cap is accepted on both paths. + let at_cap = "m".repeat(REQUESTED_MODEL_MAX_BYTES); + assert!( + UserMessagePayload::new("hi", vec![], ProductTriggerReason::DirectChat) + .unwrap() + .with_requested_model(Some(at_cap)) + .validate() + .is_ok() + ); + } + fn sample_context() -> TrustedInboundContext { let evidence = ProtocolAuthEvidence::test_verified( AuthRequirement::SharedSecretHeader { diff --git a/crates/ironclaw_reborn_openai_compat/src/chat_workflow.rs b/crates/ironclaw_reborn_openai_compat/src/chat_workflow.rs index a2ac3a98c0c..1c667327d41 100644 --- a/crates/ironclaw_reborn_openai_compat/src/chat_workflow.rs +++ b/crates/ironclaw_reborn_openai_compat/src/chat_workflow.rs @@ -799,6 +799,9 @@ fn chat_user_message_and_attachments( .with_requested_model(crate::model_validation::requested_model_hint( &request.model, )); + // The builder attaches the model hint after `new`'s validation, so bound the + // assembled payload before it is submitted. + payload.validate()?; Ok((payload, attachments)) } diff --git a/crates/ironclaw_reborn_openai_compat/src/responses_workflow.rs b/crates/ironclaw_reborn_openai_compat/src/responses_workflow.rs index 03436f05685..afb08b6f6b3 100644 --- a/crates/ironclaw_reborn_openai_compat/src/responses_workflow.rs +++ b/crates/ironclaw_reborn_openai_compat/src/responses_workflow.rs @@ -1382,14 +1382,18 @@ fn validate_temperature(temperature: Option) -> Result<(), OpenAiCompatHttp fn responses_user_message_payload( request: &OpenAiResponsesCreateRequest, ) -> Result { - Ok(UserMessagePayload::new( + let payload = UserMessagePayload::new( responses_input_to_product_text(request)?, vec![], ProductTriggerReason::DirectChat, )? .with_requested_model(crate::model_validation::requested_model_hint( &request.model, - ))) + )); + // The builder attaches the model hint after `new`'s validation, so bound the + // assembled payload before it is submitted. + payload.validate()?; + Ok(payload) } fn responses_input_to_product_text(