Skip to content
Closed
4 changes: 4 additions & 0 deletions crates/ironclaw_agent_loop/src/executor.rs
Original file line number Diff line number Diff line change
Expand Up @@ -108,6 +108,10 @@ pub enum AgentLoopExecutorError {
safe_summary: LoopSafeSummary,
reason_kind: Option<AgentLoopHostErrorReasonKind>,
diagnostic_ref: Option<LoopDiagnosticRef>,
/// Secret-scrubbed model-visible raw cause carried from the host error
/// so the runner/explainer can surface the real fault instead of a
/// generic category. See [`AgentLoopHostError::detail`].
detail: Option<String>,
},
#[error("planner returned a contract violation: {detail}")]
PlannerContract { detail: &'static str },
Expand Down
87 changes: 75 additions & 12 deletions crates/ironclaw_agent_loop/src/executor/capability_helpers.rs
Original file line number Diff line number Diff line change
Expand Up @@ -312,19 +312,42 @@ pub(super) fn model_visible_capability_failure_observation(
Some(CapabilityFailureDetail::InvalidInput { issues }) => {
invalid_input_observation(bounded_input_issues(issues))
}
_ => ModelVisibleToolObservation {
schema_version:
ironclaw_turns::run_profile::MODEL_VISIBLE_TOOL_OBSERVATION_SCHEMA_VERSION,
status: ToolObservationStatus::Error,
summary: model_visible_failure_summary(&failure.error_kind),
detail: ToolObservationDetail::GenericFailure {
failure_kind: failure.error_kind.clone(),
},
artifacts: Vec::new(),
recovery: Some(generic_failure_recovery(&failure.error_kind)),
trust: ObservationTrust::UntrustedToolOutput,
},
detail => {
let diagnostic = match detail {
Some(CapabilityFailureDetail::Diagnostic { text }) => {
Some(bounded_diagnostic_detail(text))
}
_ => None,
};
ModelVisibleToolObservation {
schema_version:
ironclaw_turns::run_profile::MODEL_VISIBLE_TOOL_OBSERVATION_SCHEMA_VERSION,
status: ToolObservationStatus::Error,
summary: model_visible_failure_summary(&failure.error_kind),
detail: ToolObservationDetail::GenericFailure {
failure_kind: failure.error_kind.clone(),
detail: diagnostic,
},
artifacts: Vec::new(),
recovery: Some(generic_failure_recovery(&failure.error_kind)),
trust: ObservationTrust::UntrustedToolOutput,
}
}
}
}

/// Bound a free-text diagnostic to the model-visible detail cap, truncating on
/// a UTF-8 boundary. The diagnostic is already secret-scrubbed by the producer.
fn bounded_diagnostic_detail(value: &str) -> String {
const MAX: usize = ironclaw_turns::run_profile::MODEL_OBSERVATION_DETAIL_MAX_BYTES;
if value.len() <= MAX {
return value.to_string();
}
let mut end = MAX;
while end > 0 && !value.is_char_boundary(end) {
end -= 1;
}
value[..end].to_string() // safety: `end` reduced to a valid UTF-8 boundary.
}

fn model_visible_failure_summary(error_kind: &CapabilityFailureKind) -> String {
Expand Down Expand Up @@ -636,6 +659,46 @@ mod tests {
assert_eq!(state.result_refs.len(), 3);
}

#[test]
fn generic_failure_observation_carries_diagnostic_detail_to_model() {
let path = "missing input_schema_ref at /system/extensions/google-calendar/schemas/google-calendar/list_calendars.input.v1.json";
let failure = CapabilityFailure {
error_kind: CapabilityFailureKind::MissingRuntime,
safe_summary: "capability invocation failed".to_string(),
detail: Some(CapabilityFailureDetail::Diagnostic {
text: path.to_string(),
}),
};

let observation = model_visible_capability_failure_observation(&failure);
observation
.validate()
.expect("observation with path-bearing diagnostic must validate");

let ToolObservationDetail::GenericFailure { detail, .. } = &observation.detail else {
panic!("expected a generic failure detail");
};
assert_eq!(
detail.as_deref(),
Some(path),
"the raw path-bearing cause must reach the model-visible observation"
);
}

#[test]
fn generic_failure_observation_without_diagnostic_has_no_detail() {
let failure = CapabilityFailure {
error_kind: CapabilityFailureKind::Backend,
safe_summary: "capability invocation failed".to_string(),
detail: None,
};
let observation = model_visible_capability_failure_observation(&failure);
let ToolObservationDetail::GenericFailure { detail, .. } = &observation.detail else {
panic!("expected a generic failure detail");
};
assert_eq!(detail.as_deref(), None);
}

#[test]
fn pending_auth_resume_candidate_carries_non_empty_effective_capability_ids() {
use crate::state::PendingAuthResume;
Expand Down
2 changes: 2 additions & 0 deletions crates/ironclaw_agent_loop/src/executor/checkpoint.rs
Original file line number Diff line number Diff line change
Expand Up @@ -213,13 +213,15 @@ fn checkpoint_host_error(
| AgentLoopHostErrorKind::Internal
| AgentLoopHostErrorKind::BudgetAccountingFailed
) {
let detail = error.detail.clone();
return AgentLoopExecutorError::HostUnavailableWithDiagnostics {
stage: HostStage::Checkpoint,
kind: error.kind,
safe_summary: LoopSafeSummary::new(error.safe_summary)
.unwrap_or_else(|_| LoopSafeSummary::model_gateway_failed()),
reason_kind: error.reason_kind,
diagnostic_ref: error.diagnostic_ref,
detail,
};
}
AgentLoopExecutorError::CheckpointFailed { stage: kind }
Expand Down
2 changes: 2 additions & 0 deletions crates/ironclaw_agent_loop/src/executor/model.rs
Original file line number Diff line number Diff line change
Expand Up @@ -129,13 +129,15 @@ impl ExecutorStage<ModelInput> for ModelStage {
return budget_approval_blocked_exit(ctx, state, gate_ref).await;
}
let Some(class) = model_error_class(&error) else {
let detail = error.detail.clone();
return Err(AgentLoopExecutorError::HostUnavailableWithDiagnostics {
stage: HostStage::Model,
kind: error.kind,
safe_summary: LoopSafeSummary::new(error.safe_summary)
.unwrap_or_else(|_| LoopSafeSummary::model_gateway_failed()),
reason_kind: error.reason_kind,
diagnostic_ref: error.diagnostic_ref,
detail,
});
};
if !recorded_failure {
Expand Down
27 changes: 27 additions & 0 deletions crates/ironclaw_agent_loop/src/executor/tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1091,6 +1091,7 @@ async fn model_budget_approval_required_without_gate_ref_fails_diagnostics_not_r
safe_summary: LoopSafeSummary::new("budget approval required").expect("safe"),
reason_kind: None,
diagnostic_ref: None,
detail: None,
}
);
assert_eq!(host.model_requests().len(), 1);
Expand Down Expand Up @@ -2779,10 +2780,36 @@ async fn model_unrecoverable_host_error_preserves_sanitized_diagnostics() {
safe_summary: LoopSafeSummary::new("model credentials are unavailable").expect("safe"),
reason_kind: None,
diagnostic_ref: Some(LoopDiagnosticRef::new("diag:model-credentials").expect("valid")),
detail: None,
}
);
}

#[tokio::test]
async fn model_unrecoverable_host_error_carries_detail_to_executor_error() {
let host = MockHost::new(Vec::new()).with_model_errors(vec![
AgentLoopHostError::new(
AgentLoopHostErrorKind::CredentialUnavailable,
"model credentials are unavailable",
)
.with_detail("HTTP 404 model not found"),
]);
let executor = CanonicalAgentLoopExecutor;
let state = LoopExecutionState::initial_for_run(host.run_context());

let error = executor
.execute_family(&crate::families::default(), &host, state)
.await
.expect_err("unrecoverable model errors stop before a loop exit");

match error {
AgentLoopExecutorError::HostUnavailableWithDiagnostics { detail, .. } => {
assert_eq!(detail.as_deref(), Some("HTTP 404 model not found"));
}
other => panic!("expected HostUnavailableWithDiagnostics, got {other:?}"),
}
}

#[tokio::test]
async fn capability_abort_finalizes_explanation_and_failed_exit_refs_partial_first() {
let script = ScenarioScript {
Expand Down
2 changes: 2 additions & 0 deletions crates/ironclaw_agent_loop/src/executor/tests/cancellation.rs
Original file line number Diff line number Diff line change
Expand Up @@ -414,6 +414,7 @@ async fn cancellation_after_pending_input_ack_permissive_profile_propagates_chec
safe_summary: LoopSafeSummary::new("scripted checkpoint payload failure").unwrap(),
reason_kind: None,
diagnostic_ref: None,
detail: None,
}
);
}
Expand Down Expand Up @@ -701,6 +702,7 @@ async fn cancellation_checkpoint_payload_unavailable_propagates_for_permissive_p
safe_summary: LoopSafeSummary::new("scripted checkpoint payload failure").unwrap(),
reason_kind: None,
diagnostic_ref: None,
detail: None,
}
);
assert!(host.checkpoint_kinds().is_empty());
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -260,6 +260,7 @@ pub(super) fn blocked_event_with(
}),
sanitized_reason: Some("approval_required".to_string()),
retryable: None,
detail: None,
}
}

Expand All @@ -281,6 +282,7 @@ pub(super) fn lifecycle_event(
blocked_gate: None,
sanitized_reason: None,
retryable: None,
detail: None,
}
}

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -552,7 +552,12 @@ async fn first_party_missing_handler_fails_closed_without_side_effect_handler()
panic!("expected missing first-party handler to fail closed, got {outcome:?}");
};
assert_eq!(failure.capability_id, capability_id());
assert_eq!(failure.kind, RuntimeFailureKind::Backend);
// A capability the model named that has no registered handler is a
// model-fixable request error (UndeclaredCapability -> InvalidInput ->
// model-visible tool error), not an infra Backend fault — it can never
// resolve by retrying, so it must surface to the model immediately. See the
// From<DispatchFailureKind> mapping in production.rs.
assert_eq!(failure.kind, RuntimeFailureKind::InvalidInput);
assert_eq!(
failure.message.as_deref(),
Some("dispatch failed: UndeclaredCapability")
Expand Down
92 changes: 86 additions & 6 deletions crates/ironclaw_loop_support/src/capability_port.rs
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ use ironclaw_turns::{
LoopProcessRef, LoopRunContext, LoopSafeSummary, ProcessHandleSummary, ProviderToolCall,
ProviderToolCallCapabilityIds, ProviderToolCallReplay, ProviderToolDefinition,
RegisterProviderToolCallRequest, VisibleCapabilityRequest, VisibleCapabilitySurface,
sanitize_model_visible_text,
},
};
use serde_json::Value;
Expand Down Expand Up @@ -1653,9 +1654,10 @@ impl LoopCapabilityPort for HostRuntimeLoopCapabilityPort {
if error.error.kind == AgentLoopHostErrorKind::InvalidInvocation
&& is_provider_tool_call_input_ref(effective_input_ref) =>
{
let host_error = *error.error;
let result = Ok(CapabilityOutcome::Failed(CapabilityFailure {
error_kind: CapabilityFailureKind::InvalidInput,
safe_summary: error.error.safe_summary,
safe_summary: host_error.safe_summary,
detail: error.detail,
}));
guard.commit();
Expand All @@ -1666,7 +1668,7 @@ impl LoopCapabilityPort for HostRuntimeLoopCapabilityPort {
)?;
return result;
}
Err(error) => return Err(error.error),
Err(error) => return Err(*error.error),
};
let runtime_input =
match host_runtime_input_for_capability(&request.capability_id, input) {
Expand Down Expand Up @@ -2526,12 +2528,30 @@ fn runtime_failure_to_loop(
&failure,
"capability invocation failed",
),
detail: None,
detail: runtime_failure_diagnostic_detail(&failure),
}))
}
}
}

/// Build a model-visible, secret-scrubbed diagnostic from a runtime failure's
/// raw message when the failure has no structured detail. This preserves the
/// real cause (paths, schema refs, codes) that the strict safe-summary
/// validator drops — only secret VALUES are redacted.
fn runtime_failure_diagnostic_detail(
failure: &RuntimeCapabilityFailure,
) -> Option<CapabilityFailureDetail> {
if failure.detail.is_some() {
return None;
}
let raw = failure.safe_summary()?;
let scrubbed = sanitize_model_visible_text(raw);
if scrubbed.trim().is_empty() {
return None;
}
Some(CapabilityFailureDetail::Diagnostic { text: scrubbed })
}

fn runtime_model_visible_failure_to_loop(
failure: RuntimeCapabilityFailure,
) -> Result<CapabilityOutcome, AgentLoopHostError> {
Expand All @@ -2545,10 +2565,16 @@ fn runtime_model_visible_failure_to_loop(
}));
}

let error_kind = model_visible_runtime_failure_kind_to_loop(failure.kind)?;
let safe_summary = runtime_failure_safe_summary(&failure, "capability invocation failed");
let detail = match runtime_failure_detail_to_loop(failure.detail.clone()) {
Some(structured) => Some(structured),
None => runtime_failure_diagnostic_detail(&failure),
};
Ok(CapabilityOutcome::Failed(CapabilityFailure {
error_kind: model_visible_runtime_failure_kind_to_loop(failure.kind)?,
safe_summary: runtime_failure_safe_summary(&failure, "capability invocation failed"),
detail: runtime_failure_detail_to_loop(failure.detail),
error_kind,
safe_summary,
detail,
}))
}

Expand Down Expand Up @@ -3145,6 +3171,60 @@ mod tests {
));
}

#[test]
fn runtime_failure_carries_path_bearing_cause_into_model_visible_diagnostic() {
// Anchor: a host-runtime capability failure whose reason contains a path
// (rejected by the strict safe-summary validator) must NOT be collapsed
// to the generic fallback — the real cause reaches the model via detail.
let capability_id =
CapabilityId::new("google-calendar.list_calendars").expect("valid capability id");
let path = "missing input_schema_ref at /system/extensions/google-calendar/schemas/google-calendar/list_calendars.input.v1.json";
let outcome = runtime_failure_to_loop(RuntimeCapabilityFailure::new(
capability_id,
RuntimeFailureKind::MissingRuntime,
Some(path.to_string()),
))
.expect("convert host runtime failure");

let CapabilityOutcome::Failed(failure) = outcome else {
panic!("expected a model-visible Failed outcome");
};
// The summary stays generic (the path tripped the strict validator) ...
assert_eq!(failure.safe_summary, "capability invocation failed");
// ... but the raw path-bearing cause now rides the diagnostic detail.
let Some(CapabilityFailureDetail::Diagnostic { text }) = failure.detail else {
panic!("expected a diagnostic detail carrying the raw cause");
};
assert_eq!(text, path, "the path string must reach the model intact");
}

#[test]
fn runtime_failure_diagnostic_redacts_secret_values() {
let capability_id = CapabilityId::new("demo.echo").expect("valid capability id");
let reason = "auth failed using sk-LIVEsecretvalue while reaching provider";
let outcome = runtime_failure_to_loop(RuntimeCapabilityFailure::new(
capability_id,
RuntimeFailureKind::MissingRuntime,
Some(reason.to_string()),
))
.expect("convert host runtime failure");

let CapabilityOutcome::Failed(failure) = outcome else {
panic!("expected a model-visible Failed outcome");
};
let Some(CapabilityFailureDetail::Diagnostic { text }) = failure.detail else {
panic!("expected a diagnostic detail");
};
assert!(
!text.contains("sk-LIVEsecretvalue"),
"secret value must be redacted from the model-visible detail: {text}"
);
assert!(
text.contains("[redacted]"),
"redaction marker should be present: {text}"
);
}

#[test]
fn runtime_failure_to_loop_routes_retryable_failures_to_retry_classes() {
let capability_id = CapabilityId::new("demo.echo").expect("valid capability id");
Expand Down
Loading
Loading