Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -628,7 +628,13 @@ fn reborn_contracts_crates_carry_a_checked_size_ceiling() {
// CapabilitySurfacePolicy and capability-id scope algebra are neutral
// host declarations; enforcement remains in host_runtime/loop_host.
("ironclaw_host_api", 18_784),
("ironclaw_loop_contracts", 14_479),
// Raised 14_479 -> 14_530 by #7166 §5 (queryable rollout metrics): the
// growth is the `disclosure_metrics` vocabulary — a numbers-only
// per-model-call measurement DTO, its catalog-size bucket enum, and the
// milestone record that carries them. Declarations only; the counters
// are computed in ironclaw_loop_host and projected in
// ironclaw_turn_runner / ironclaw_event_projections.
("ironclaw_loop_contracts", 14_530),
// Raised 15_685 -> 15_758 by #7220 (operator inspector API): the growth
// is bounded, output-only read-view descriptors. Capture, retention,
// authorization, and transport behavior remain in their owning
Expand Down
210 changes: 210 additions & 0 deletions crates/contracts/ironclaw_loop_contracts/src/disclosure_metrics.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,210 @@
//! Per-model-call progressive-tool-disclosure measurements.
//!
//! Progressive tool disclosure (`tool_search` / `tool_describe` / `tool_call`)
//! already computes every number an operator needs to judge the rollout —
//! catalog size, advertised size, estimated schema tokens, how often the model
//! searched, whether a search came back empty, which rank it eventually picked,
//! how many deferred tools were promoted, how many recoverable failures the
//! bridge absorbed, and how many calls aimed outside the disclosed surface.
//! Until now those numbers only reached ephemeral `debug!` shadow logs.
//!
//! This module defines the neutral, closed-vocabulary carrier for those
//! measurements. It is deliberately numbers-only: no tool names, no queries, no
//! schemas, no descriptions. That is what lets the record cross into the
//! durable runtime event log, which is redaction-bound.
//!
//! The counters are cumulative over the disclosure port's lifetime (one run),
//! so a per-call delta is a subtraction between consecutive records and a
//! per-run total is the last record. Cumulative was chosen over per-call deltas
//! because a dropped or best-effort-skipped record then loses precision, not
//! correctness.

use serde::{Deserialize, Serialize};

/// Coarse catalog-size bucket, the third rollout query dimension alongside
/// model and run profile.
///
/// Buckets are anchored on the disclosure caps that gate deferral
/// (`DisclosureCaps::default().max_tools == 32`): `AtOrBelowCaps` is the
/// "stays direct, pays no discovery round trip" cohort the issue calls out
/// separately from the wide catalogs deferral exists for.
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum CatalogSizeBucket {
/// No authorized tools at all (no-tool chat).
Empty,
/// 1–8 tools.
Tiny,
/// 9–32 tools — at or below the default disclosure caps.
AtOrBelowCaps,
/// 33–96 tools.
Wide,
/// More than 96 tools.
VeryWide,
}

impl CatalogSizeBucket {
/// Classify a full (authorized, policy-effective) tool count.
pub const fn from_tool_count(full_tool_count: u32) -> Self {
match full_tool_count {
0 => Self::Empty,
1..=8 => Self::Tiny,
9..=32 => Self::AtOrBelowCaps,
33..=96 => Self::Wide,
_ => Self::VeryWide,
}
}

/// Stable closed-vocabulary label for durable events and queries.
pub const fn as_str(self) -> &'static str {
match self {
Self::Empty => "empty",
Self::Tiny => "tiny",
Self::AtOrBelowCaps => "at_or_below_caps",
Self::Wide => "wide",
Self::VeryWide => "very_wide",
}
}

/// Parse a label produced by [`Self::as_str`]. Unknown labels are rejected
/// rather than silently bucketed, so a future bucket cannot masquerade as
/// an existing cohort in a rollout comparison.
pub fn from_str_label(label: &str) -> Option<Self> {
match label {
"empty" => Some(Self::Empty),
"tiny" => Some(Self::Tiny),
"at_or_below_caps" => Some(Self::AtOrBelowCaps),
"wide" => Some(Self::Wide),
"very_wide" => Some(Self::VeryWide),
_ => None,
}
}
}

/// Disclosure measurements observed at one model call.
///
/// Every field is already computed inside the disclosure port; this type only
/// carries them. See the module docs for the cumulative-counter contract.
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
pub struct ToolDisclosureCallMetrics {
/// Whether this call actually deferred (bridged) rather than advertising
/// the full surface flat. Disclosure can be wired but inert below caps.
pub deferred: bool,
/// Authorized, policy-effective tool count before narrowing.
pub full_tool_count: u32,
/// Tool count actually advertised to the provider on this call.
pub advertised_tool_count: u32,
/// Estimated schema tokens for the full authorized surface.
pub full_schema_tokens: u32,
/// Estimated schema tokens actually advertised on this call.
pub advertised_schema_tokens: u32,
/// Cumulative `tool_search` invocations.
pub tool_search_count: u32,
/// Cumulative `tool_search` invocations that returned zero results.
pub empty_search_count: u32,
/// 1-based rank, within its originating `tool_search` result list, of the
/// most recently selected (invoked or promoted) deferred tool. `None` when
/// nothing ranked has been selected yet.
pub selected_result_rank: Option<u32>,
/// Cumulative earned promotions of a deferred tool to the flat surface.
pub promotions: u32,
/// Cumulative recoverable outcomes the bridge returned instead of ending
/// the run — describe-first schema returns and recoverable bridge failures.
pub recoveries: u32,
/// Cumulative attempts to search, describe, or call a tool that is not on
/// the authorized disclosed surface.
pub outside_surface_attempts: u32,
}

impl ToolDisclosureCallMetrics {
/// Catalog-size bucket for this call's full authorized surface.
pub const fn catalog_size_bucket(&self) -> CatalogSizeBucket {
CatalogSizeBucket::from_tool_count(self.full_tool_count)
}

/// Schema-token reduction achieved on this call, in percent, or `None`
/// when the full surface estimated zero tokens (nothing to reduce).
pub fn schema_token_reduction_pct(&self) -> Option<f64> {
if self.full_schema_tokens == 0 {
return None;
}
Some(
100.0
* (1.0
- (f64::from(self.advertised_schema_tokens)
/ f64::from(self.full_schema_tokens))),
)
}
}

#[cfg(test)]
mod tests {
use super::*;

#[test]
fn catalog_buckets_split_on_the_disclosure_cap_boundary() {
// The cap boundary is the load-bearing one: 32 is the last catalog that
// stays direct, 33 is the first that pays for discovery. A rollout
// comparison that blurs those two cohorts cannot answer "did small
// catalogs regress".
assert_eq!(
CatalogSizeBucket::from_tool_count(0),
CatalogSizeBucket::Empty
);
assert_eq!(
CatalogSizeBucket::from_tool_count(8),
CatalogSizeBucket::Tiny
);
assert_eq!(
CatalogSizeBucket::from_tool_count(32),
CatalogSizeBucket::AtOrBelowCaps
);
assert_eq!(
CatalogSizeBucket::from_tool_count(33),
CatalogSizeBucket::Wide
);
assert_eq!(
CatalogSizeBucket::from_tool_count(97),
CatalogSizeBucket::VeryWide
);
}

#[test]
fn bucket_labels_round_trip_and_reject_unknown_cohorts() {
for bucket in [
CatalogSizeBucket::Empty,
CatalogSizeBucket::Tiny,
CatalogSizeBucket::AtOrBelowCaps,
CatalogSizeBucket::Wide,
CatalogSizeBucket::VeryWide,
] {
assert_eq!(
CatalogSizeBucket::from_str_label(bucket.as_str()),
Some(bucket)
);
}
assert_eq!(CatalogSizeBucket::from_str_label("huge"), None);
}

#[test]
fn reduction_is_none_when_there_was_nothing_to_reduce() {
let metrics = ToolDisclosureCallMetrics::default();
assert_eq!(metrics.schema_token_reduction_pct(), None);
}

#[test]
fn reduction_reports_the_share_of_full_schema_tokens_avoided() {
let metrics = ToolDisclosureCallMetrics {
full_schema_tokens: 1_000,
advertised_schema_tokens: 163,
..ToolDisclosureCallMetrics::default()
};
let reduction = metrics
.schema_token_reduction_pct()
.expect("nonzero full surface reports a reduction");
assert!(
(reduction - 83.7).abs() < 1e-9,
"unexpected reduction {reduction}"
);
}
}
13 changes: 13 additions & 0 deletions crates/contracts/ironclaw_loop_contracts/src/host/capability.rs
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@ use serde::{Deserialize, Deserializer, Serialize};
pub use ironclaw_host_api::capability::CapabilityDescriptionTrust;

use crate::content_digest::ContentDigest;
use crate::disclosure_metrics::ToolDisclosureCallMetrics;
use crate::model_observation::{CapabilityFailureDetail, ModelVisibleToolObservation};
use ironclaw_host_api::turn::{CapabilityActivityId, LoopResultRef};

Expand Down Expand Up @@ -548,6 +549,18 @@ pub trait LoopCapabilityPort: Send + Sync {
Ok(Vec::new())
}

/// Progressive-tool-disclosure measurements for the surface this port is
/// currently presenting, or `None` when disclosure is not in play.
///
/// Read-only observation of numbers the port already computed — never a
/// recomputation and never an authority. Any port that wraps another MUST
/// delegate this: a decorator that silently keeps the default `None`
/// erases the rollout evidence for every run that goes through it, and
/// does so without any error to notice.
fn tool_disclosure_metrics(&self) -> Option<ToolDisclosureCallMetrics> {
None
Comment on lines +552 to +561

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🗄️ Data Integrity & Integration | 🟠 Major | 🏗️ Heavy lift

Record metrics for the final capability surface.

The contract requires unchanged delegation, but these decorators change the represented surface. The durable record can report incorrect advertised counts and schema tokens. The policy and subagent decorators can also report an incorrect full count and catalog bucket.

  • crates/contracts/ironclaw_loop_contracts/src/host/capability.rs#L552-L561: require transparent decorators to delegate, but require surface-transforming decorators to produce metrics for their transformed surface.
  • crates/loop/ironclaw_loop_host/src/capability_surface_filter.rs#L78-L85: preserve the inner full-surface values, but derive advertised values from the model-visible filtered surface.
  • crates/loop/ironclaw_loop_host/src/capability_surface_filter.rs#L184-L191: derive metrics after the policy-resolved filter applies.
  • crates/loop/ironclaw_loop_host/src/subagent_spawn_port.rs#L1161-L1168: include spawn_subagent in the relevant full and advertised measurements.

Add a caller-level test through the built model host. Test a filtered surface and a subagent-decorated surface. Assert the durable metrics payload. This follows the Test through the caller invariant.

📍 Affects 3 files
  • crates/contracts/ironclaw_loop_contracts/src/host/capability.rs#L552-L561 (this comment)
  • crates/loop/ironclaw_loop_host/src/capability_surface_filter.rs#L78-L85
  • crates/loop/ironclaw_loop_host/src/capability_surface_filter.rs#L184-L191
  • crates/loop/ironclaw_loop_host/src/subagent_spawn_port.rs#L1161-L1168
🤖 Prompt for AI Agents
Verify each finding against current code. Fix only still-valid issues, skip the
rest with a brief reason, keep changes minimal, and validate.

In `@crates/contracts/ironclaw_loop_contracts/src/host/capability.rs` around lines
552 - 561, Update tool_disclosure_metrics and its decorators so metrics describe
the final capability surface: transparent decorators must delegate unchanged,
while capability_surface_filter.rs lines 78-85 and 184-191 must preserve inner
full-surface values and derive advertised values after filtering and policy
resolution; subagent_spawn_port.rs lines 1161-1168 must include spawn_subagent
in full and advertised measurements. Add caller-level tests through the built
model host covering filtered and subagent-decorated surfaces and asserting the
durable metrics payload.

Sources: Coding guidelines, Path instructions

}

fn provider_tool_call_capability_ids(
&self,
tool_call: &ProviderToolCall,
Expand Down
6 changes: 4 additions & 2 deletions crates/contracts/ironclaw_loop_contracts/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@ mod checkpoint_payload;
mod compaction;
mod content_digest;
mod context_budget;
mod disclosure_metrics;
mod driver;
mod host;
mod instruction_bundle;
Expand Down Expand Up @@ -54,6 +55,7 @@ pub use compaction::{
};
pub use content_digest::{ContentDigest, ContentDigestError, normalize_for_hash};
pub use context_budget::PromptContextTokenBudget;
pub use disclosure_metrics::{CatalogSizeBucket, ToolDisclosureCallMetrics};
pub use driver::{
AgentLoopDriver, AgentLoopDriverDescriptor, AgentLoopDriverError, AgentLoopDriverResumeRequest,
AgentLoopDriverRunRequest,
Expand Down Expand Up @@ -102,8 +104,8 @@ pub use memory_context::{
pub use milestones::{
HookDecisionSummary, HookMilestoneSink, InMemoryHookMilestoneSink,
InMemoryLoopHostMilestoneSink, LoopHostMilestone, LoopHostMilestoneEmitter,
LoopHostMilestoneKind, LoopHostMilestoneSink, PromptSkillContextMetadata,
RunScopedHookMilestoneSink,
LoopHostMilestoneKind, LoopHostMilestoneSink, ModelCallMetricsRecord,
PromptSkillContextMetadata, RunScopedHookMilestoneSink,
};
pub use model::{
LoopModelBudgetAccountant, LoopModelGateway, LoopModelGatewayError, LoopModelGatewayRequest,
Expand Down
49 changes: 49 additions & 0 deletions crates/contracts/ironclaw_loop_contracts/src/milestones.rs
Original file line number Diff line number Diff line change
Expand Up @@ -13,15 +13,47 @@ use ironclaw_host_api::turn::{
TurnId, TurnRunId, TurnScope,
};

use super::host::LoopModelUsage;
use super::host::{
AgentLoopHostError, AgentLoopHostErrorKind, BatchPolicyKind, CapabilitySurfaceVersion,
LoopCheckpointKind, LoopDriverNoteKind, LoopGateKind, LoopPromptBundleRef, LoopRecoveryClass,
LoopRecoveryDisposition, LoopRecoveryStage, LoopRunContext, LoopSafeSummary, PromptMode,
};
use super::refs::{LoopDriverId, ModelProfileId};
use super::{CompactionInitiator, SkillTrustLevel, SystemInferenceTaskId};
use crate::disclosure_metrics::ToolDisclosureCallMetrics;
use crate::{LoopCompletionKind, LoopFailureKind};

/// Cost, latency, and tool-disclosure measurements for one completed model
/// call.
///
/// Everything here is either a number or a closed-vocabulary label, which is
/// what allows a durable-event adapter to project it without a redaction
/// review per field.
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct ModelCallMetricsRecord {
/// Zero-based agent-loop iteration this call belongs to.
pub iteration: u32,
/// Requested run/model profile — the "run profile" rollout dimension.
pub requested_model: ModelProfileId,
/// Concrete provider model that served the call, when the gateway
/// reported one. Diagnostic evidence, never a routing input.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub effective_model: Option<String>,
/// Index into the ordered fallback chain. Nonzero means the call ran on a
/// fallback route rather than the primary — the observable retry signal.
pub fallback_index: u32,
/// `None` on success; the failure classification otherwise.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub failure_kind: Option<AgentLoopHostErrorKind>,
pub duration_ms: u64,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub usage: Option<LoopModelUsage>,
/// Absent when tool disclosure was not in play for this call.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub disclosure: Option<ToolDisclosureCallMetrics>,
}

#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct LoopHostMilestone {
pub scope: TurnScope,
Expand Down Expand Up @@ -96,6 +128,14 @@ pub enum LoopHostMilestoneKind {
ModelFailed {
reason_kind: AgentLoopHostErrorKind,
},
/// Measurements for one completed model call, success or failure.
///
/// Separate from `ModelCompleted`/`ModelFailed`, which are lifecycle
/// transitions the driver reacts to. This carries no authority and drives
/// no decision: it exists so the rollout numbers survive the process.
ModelCallMetricsRecorded {
record: ModelCallMetricsRecord,
},
CapabilityInvoked {
activity_id: CapabilityActivityId,
capability_id: CapabilityId,
Expand Down Expand Up @@ -282,6 +322,7 @@ impl LoopHostMilestoneKind {
Self::CompactionCompleted { .. } => "compaction_completed",
Self::CompactionFailed { .. } => "compaction_failed",
Self::CompactionLeakDetected { .. } => "compaction_leak_detected",
Self::ModelCallMetricsRecorded { .. } => "model_call_metrics_recorded",
Self::AssistantReplyFinalized { .. } => "assistant_reply_finalized",
Self::Blocked { .. } => "blocked",
Self::Completed { .. } => "completed",
Expand Down Expand Up @@ -505,6 +546,14 @@ where
.await
}

pub async fn model_call_metrics_recorded(
&self,
record: ModelCallMetricsRecord,
) -> Result<(), AgentLoopHostError> {
self.publish(LoopHostMilestoneKind::ModelCallMetricsRecorded { record })
.await
}

pub async fn capability_invoked(
&self,
activity_id: CapabilityActivityId,
Expand Down
10 changes: 5 additions & 5 deletions crates/events/ironclaw_event_log/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -48,11 +48,11 @@ pub use in_memory::{
InMemoryAuditSink, InMemoryDurableAuditLog, InMemoryDurableEventLog, InMemoryEventSink,
};
pub use runtime_event::{
RuntimeEvent, RuntimeEventId, RuntimeEventKind, UNCLASSIFIED_ERROR_KIND,
UNCLASSIFIED_HOOK_LABEL, deserialize_trusted_runtime_event,
runtime_event_from_trusted_json_slice, runtime_event_from_trusted_json_str,
sanitize_error_kind, sanitize_error_summary, sanitize_hook_id, sanitize_hook_label,
sanitize_recovery_label,
ModelCallMetrics, ModelCallOutcome, RuntimeEvent, RuntimeEventId, RuntimeEventKind,
ToolDisclosureMetrics, UNCLASSIFIED_ERROR_KIND, UNCLASSIFIED_HOOK_LABEL,
deserialize_trusted_runtime_event, runtime_event_from_trusted_json_slice,
runtime_event_from_trusted_json_str, sanitize_error_kind, sanitize_error_summary,
sanitize_hook_id, sanitize_hook_label, sanitize_model_label, sanitize_recovery_label,
};
pub use security_audit::{
InMemorySecurityAuditSink, NoopSecurityAuditSink, SecurityAuditEvent, SecurityAuditSink,
Expand Down
Loading