Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -92,3 +92,10 @@ tests/fixtures/llm_traces/live/*.log
# scripts/build-test-tools.sh; only sources/manifests/schemas/prompts are tracked.
test-tools/*.zip
test-tools/*/wasm-src/target/

# node_modules must never be committed — vendored JS deps belong to the package
# manager + lockfile, not git. A bare `node_modules/` (no leading slash) matches
# a directory of that name at ANY depth, so this catches every frontend
# (webui_v2, e2e, tooling, …) and stops the recurring accidental adds
# (#6298/#6305).
node_modules/
111 changes: 59 additions & 52 deletions crates/ironclaw_reborn_composition/src/factory.rs
Original file line number Diff line number Diff line change
Expand Up @@ -178,20 +178,9 @@ use ironclaw_triggers::{
use ironclaw_trust::{AdminConfig, AdminEntry, HostTrustAssignment, HostTrustPolicy};
#[cfg(feature = "test-support")]
use ironclaw_trust::{AuthorityCeiling, EffectiveTrustClass, TrustDecision, TrustProvenance};
#[cfg(all(
feature = "inmemory-turn-state",
any(feature = "libsql", feature = "postgres")
))]
use ironclaw_turns::FilesystemTurnStateBlockPersistence;
#[cfg(any(feature = "libsql", feature = "postgres"))]
use ironclaw_turns::FilesystemTurnStateStoreKind;
#[cfg(any(feature = "libsql", feature = "postgres"))]
use ironclaw_turns::InMemoryRunProfileResolver;
#[cfg(any(
feature = "inmemory-turn-state",
not(any(feature = "libsql", feature = "postgres"))
))]
use ironclaw_turns::InMemoryTurnStateStore;
use ironclaw_turns::{
CheckpointStateStore, DefaultTurnCoordinator, ExternalToolCatalog, InMemoryExternalToolCatalog,
LoopCheckpointStore,
Expand Down Expand Up @@ -257,19 +246,17 @@ impl ironclaw_network::NetworkHttpEgress for TestNetworkHttpEgress {
}
}

// The in-memory turn-state authority wins whenever `inmemory-turn-state` is on
// (the runtime-wedge fix), and is also the only option in pure-memory builds.
// Otherwise the durable per-user filesystem row store is used.
#[cfg(all(
not(feature = "inmemory-turn-state"),
any(feature = "libsql", feature = "postgres")
))]
// One turn-state store, backend-injected — the production
// `FilesystemTurnStateStoreKind<F>` (row layout) every deployment uses, never a
// bespoke `InMemoryTurnStateStore` standalone authority (arch-simplification
// §4.3). The `inmemory-turn-state` feature no longer selects the store TYPE, only
// the durability POLICY: it flips this store to `WriteBehind` at the build arm.
// The no-durable-features build backs it with `InMemoryBackend` directly
// (volatile, `LocalOnly`), matching the sibling run-state/approval/lease stores.
#[cfg(any(feature = "libsql", feature = "postgres"))]
pub(crate) type ComposedTurnStateStore = FilesystemTurnStateStoreKind<CompositeRootFilesystem>;
#[cfg(any(
feature = "inmemory-turn-state",
not(any(feature = "libsql", feature = "postgres"))
))]
pub(crate) type ComposedTurnStateStore = InMemoryTurnStateStore;
#[cfg(not(any(feature = "libsql", feature = "postgres")))]
pub(crate) type ComposedTurnStateStore = FilesystemTurnStateStoreKind<InMemoryBackend>;

#[cfg(any(feature = "libsql", feature = "postgres"))]
type ComposedResourceGovernor = FilesystemResourceGovernor<CompositeRootFilesystem>;
Expand Down Expand Up @@ -2474,35 +2461,46 @@ async fn build_local_runtime_store_graph(
let persistent_approval_policies = Arc::new(FilesystemPersistentApprovalPolicyStore::new(
Arc::clone(&scoped_filesystem),
));
// Runtime-wedge fix: with `inmemory-turn-state`, the whole local-dev/hosted
// runtime family (incl. hosted-single-tenant-volume) coordinates turn state
// in one in-process authority — no per-user `state.json` CAS livelock.
// Otherwise the durable filesystem row store is used. `ComposedTurnStateStore`
// resolves to the matching concrete type via the same feature cfg, so both
// arms satisfy every downstream consumer (coordinator, `LoopCheckpointStore`,
// `RuntimeTurnStateStore`) with no trait-object plumbing.
// #6263 Step 4 — the `inmemory-turn-state` profile no longer stands up the raw
// in-memory authority + block-persistence snapshot. EVERY deployment now
// composes the durable filesystem ROW store (typed journal/delta rows + a hot
// in-process snapshot cache). `ComposedTurnStateStore` is
// `FilesystemTurnStateStoreKind<F>` unconditionally, so the feature no longer
// selects the store TYPE. The row store is crash-recoverable (rehydrates from
// its own rows on boot) and, unlike the old blob store, has NO per-user
// `state.json` CAS livelock (journal/row model, not whole-snapshot CAS) — so it
// is strictly more durable than the former in-memory authority (which lost
// in-flight turns on crash and persisted only the gate-blocked set + a graceful
// shutdown snapshot). Existing deployments migrate automatically: their on-disk
// block-persistence snapshot at `/turns/state.json` is imported as the row
// store's first delta on an empty-rows boot
// (`FilesystemTurnStateRowStore::migrate_legacy_blob_if_needed` reads the SAME
// path/format the block-persistence sink wrote), so no gate-parked/approval turn
// is lost on first boot after the flip.
//
// POLICY: ships at the `WriteThrough` default (every transition synchronously
// durable). The async `WriteBehind` policy that this flip was intended to select
// under `inmemory-turn-state` is DELIBERATELY NOT wired here yet: its durable
// query paths (`FilesystemTurnStateRowStore::get_run_state` et al. read
// materialized rows, not the hot cache — see
// `filesystem_store/row_store/traits.rs`), so a just-submitted run whose row is
// still async-materializing reads back `ScopeNotFound`. That breaks the
// runtime's read-after-submit (`send_user_message` → `wait_for_terminal` →
// `get_run_state`) on EVERY turn (proven: `budget_approval_e2e` fails under
// `WriteBehind`, passes under `WriteThrough`; store-tier submit→get_run_state
// returns `ScopeNotFound` single-threaded). `WriteBehind` must make its query
// paths cache-aware — validated by the §11.4 reference-model suite — before it
// can be selected here. The feature arm below is the seam that flip lands on.
#[cfg(feature = "inmemory-turn-state")]
let turn_state = {
// Persist-on-block: the in-memory authority owns turn state in-process
// (no per-user `state.json` CAS livelock) but is otherwise volatile.
// Attach a durable sink that snapshots only when the gate-blocked set
// changes, and rehydrate from the last such snapshot on startup so a
// deploy never silently drops a turn parked on approval/auth.
let block_persistence = Arc::new(FilesystemTurnStateBlockPersistence::new(Arc::clone(
&turn_state_filesystem,
)));
// Fail loud on a real recovery fault. `load()` returns an empty snapshot
// for the normal missing-snapshot case (fresh volume), so an `Err` here is
// a genuine read/deserialization/integrity failure — falling back to an
// empty store would silently drop persisted blocked approvals/auth and
// defeat the recovery guarantee. Surface it as a build error instead
// (`RebornBuildError: Turn`), so an operator sees the failure rather than
// losing gate-parked turns.
let restored = block_persistence.load().await?;
let store =
InMemoryTurnStateStore::from_persistence_snapshot(restored, turn_state_store_limits)?;
Arc::new(store.with_block_persistence(block_persistence))
};
let turn_state = Arc::new(
production_turn_state_store(Arc::clone(&turn_state_filesystem), turn_state_store_limits)
// TODO(#6263 Step 4): `TurnStateDurabilityPolicy::WriteBehind` here once
// the row store's durable-read query paths are cache-aware (they
// currently return `ScopeNotFound` for an async-materializing run,
// breaking runtime read-after-submit). Pinned to `WriteThrough` so the
// profile is on the durable row store safely, with no regression.
.with_durability_policy(ironclaw_turns::TurnStateDurabilityPolicy::WriteThrough),
);
#[cfg(not(feature = "inmemory-turn-state"))]
let turn_state = Arc::new(production_turn_state_store(
Arc::clone(&turn_state_filesystem),
Expand Down Expand Up @@ -2690,7 +2688,16 @@ async fn build_local_runtime_store_graph(
let persistent_approval_policies = Arc::new(FilesystemPersistentApprovalPolicyStore::new(
Arc::clone(&approvals_filesystem),
));
let turn_state = Arc::new(InMemoryTurnStateStore::with_limits(turn_state_store_limits));
// Turn state runs the production `FilesystemTurnStateStoreKind` row store over
// a dedicated volatile `InMemoryBackend` (§4.3) at the `WriteThrough` default —
// no bespoke `InMemoryTurnStateStore` standalone authority. Matches the sibling
// run-state/approval stores in this build (volatile, `LocalOnly`); the row
// store persists every transition synchronously, so this build needs no
// shutdown drain.
let turn_state = Arc::new(
FilesystemTurnStateStoreKind::row(crate::wrap_scoped(Arc::new(InMemoryBackend::new())))
.with_limits(turn_state_store_limits),
);
// §4.3: checkpoint payloads run the production `FilesystemCheckpointStateStore`
// over a dedicated volatile `InMemoryBackend`, and checkpoint metadata lives in
// the turn-state store — the same `LoopCheckpointStore` wiring the durable
Expand Down
44 changes: 31 additions & 13 deletions crates/ironclaw_reborn_composition/src/runtime.rs
Original file line number Diff line number Diff line change
Expand Up @@ -638,11 +638,18 @@ pub(crate) const TELEGRAM_TRIGGER_POST_SUBMIT_HOOK_KEY: &str = "telegram-host-be
pub struct RebornRuntime {
services: RebornServices,
turn_coordinator: Arc<dyn TurnCoordinator>,
/// Concrete in-memory turn-state authority, kept so graceful `shutdown` can
/// flush the full snapshot durably (recovering in-flight turns on the next
/// restart, not just gate-blocked ones). `None` when no local runtime is
/// wired (e.g. production-parts launches); the durable filesystem store
/// already persists every transition, so it needs no shutdown flush.
/// Turn-state row store, kept so graceful `shutdown` can drain the
/// `WriteBehind` durable tail (awaiting the acks of non-critical transitions
/// that committed at memory speed) so a planned restart recovers in-flight
/// turns, not just the synchronously-durable gate-park/terminal ones. `None`
/// when no local runtime is wired (e.g. production-parts launches).
///
/// This is the graceful-restart seam for `WriteBehind`. The `inmemory-turn-state`
/// profile currently ships the row store at the `WriteThrough` default (see the
/// factory build arm — `WriteBehind` is blocked on the row store's
/// non-cache-aware query paths), so `drain()` is a no-op today: `WriteThrough`
/// persists every transition synchronously and buffers nothing. The wiring stays
/// so the seam is live the moment the arm selects `WriteBehind`.
#[cfg(feature = "inmemory-turn-state")]
turn_state_flush: Option<Arc<ComposedTurnStateStore>>,
turn_tree_store: Arc<dyn TurnSpawnTreeStateStore>,
Expand Down Expand Up @@ -2546,14 +2553,25 @@ impl RebornRuntime {
projection.shutdown().await;
}
// Everything that mutates turn state (trigger poller, credential-refresh
// worker, scheduler/runner) is now stopped, so the in-memory authority is
// quiescent. Flush its full snapshot durably so a planned restart recovers
// in-flight turns, not just gate-blocked ones. No-op unless a durable sink
// is attached; the durable filesystem store persists every transition and
// needs no shutdown flush (hence this is only wired under the feature).
// worker, scheduler/runner) is now stopped, so the row store is quiescent.
// Drain the `WriteBehind` durable tail — awaiting the acks of non-critical
// transitions that committed at memory speed — so a planned restart
// recovers in-flight turns, not just the synchronously-durable
// gate-park/terminal ones. The `inmemory-turn-state` profile currently ships
// the row store at the `WriteThrough` default (its tail is always empty), so
// this drain is a no-op today; it is the live seam for when the build arm
// selects `WriteBehind`. Best-effort: a drain failure means the flusher
// latched degraded mid-shutdown; log it so the operator sees the un-drained
// tail rather than failing the clean exit path.
#[cfg(feature = "inmemory-turn-state")]
if let Some(turn_state) = &self.turn_state_flush {
turn_state.flush().await;
if let Some(turn_state) = &self.turn_state_flush
&& let Err(error) = turn_state.drain().await
{
tracing::warn!(
%error,
"turn-state WriteBehind drain failed during graceful shutdown; the un-acked \
non-critical tail may not be durable on restart"
);
}
Comment on lines 2566 to 2575

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

📐 Maintainability & Code Quality | 🟡 Minor | ⚡ Quick win

warn! on the shutdown-drain path contradicts the REPL/TUI logging rule.

RebornRuntime::shutdown is REPL/TUI-reachable, and this same impl block already chose debug! over warn! for exactly this reason (wait_for_terminal_or_gate, ~Line 2745: "debug! not warn! per the logging rule — this runtime is REPL/TUI-reachable"). A degraded-drain diagnostic is internal, not intentionally-rendered user status, so prefer debug! here for consistency.

As per path instructions: "REPL/TUI logging: info!/warn! corrupt the terminal UI — internal diagnostics use debug!".

Proposed change
-            tracing::warn!(
+            tracing::debug!(
                 %error,
                 "turn-state WriteBehind drain failed during graceful shutdown; the un-acked \
                  non-critical tail may not be durable on restart"
             );
📝 Committable suggestion

‼️ IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
#[cfg(feature = "inmemory-turn-state")]
if let Some(turn_state) = &self.turn_state_flush {
turn_state.flush().await;
if let Some(turn_state) = &self.turn_state_flush
&& let Err(error) = turn_state.drain().await
{
tracing::warn!(
%error,
"turn-state WriteBehind drain failed during graceful shutdown; the un-acked \
non-critical tail may not be durable on restart"
);
}
#[cfg(feature = "inmemory-turn-state")]
if let Some(turn_state) = &self.turn_state_flush
&& let Err(error) = turn_state.drain().await
{
tracing::debug!(
%error,
"turn-state WriteBehind drain failed during graceful shutdown; the un-acked \
non-critical tail may not be durable on restart"
);
}
🤖 Prompt for AI Agents
Verify each finding against current code. Fix only still-valid issues, skip the
rest with a brief reason, keep changes minimal, and validate.

In `@crates/ironclaw_reborn_composition/src/runtime.rs` around lines 2566 - 2575,
In RebornRuntime::shutdown, change the turn_state_flush drain failure log from
warn! to debug!, preserving the existing %error field and message. Keep the
diagnostic internal and consistent with the nearby wait_for_terminal_or_gate
logging rule for REPL/TUI-reachable runtime paths.

Source: Path instructions

Ok(())
}
Expand Down Expand Up @@ -4067,7 +4085,7 @@ pub async fn build_reborn_runtime(
)
});

// Concrete in-memory store handle for the graceful-shutdown flush (see the
// Row-store handle for the graceful-shutdown `WriteBehind` drain (see the
// field doc). `local_runtime` is `Option<&…>` (`Copy`), so mapping it here
// doesn't disturb its later use.
#[cfg(feature = "inmemory-turn-state")]
Expand Down
98 changes: 98 additions & 0 deletions crates/ironclaw_reborn_composition/tests/runtime.rs
Original file line number Diff line number Diff line change
Expand Up @@ -218,6 +218,104 @@ async fn stub_gateway_send_cancels_recovery_required_and_releases_conversation()
runtime.shutdown().await.unwrap();
}

/// Minimal completing model gateway: every model call returns a plain assistant
/// reply, so a turn reaches `TurnStatus::Completed` without needing a real LLM.
#[cfg(feature = "inmemory-turn-state")]
#[derive(Default)]
struct AlwaysReplyGateway;

#[cfg(feature = "inmemory-turn-state")]
#[async_trait]
impl HostManagedModelGateway for AlwaysReplyGateway {
async fn stream_model(
&self,
_request: HostManagedModelRequest,
) -> Result<HostManagedModelResponse, HostManagedModelError> {
Ok(HostManagedModelResponse::assistant_reply(
"done".to_string(),
))
}

async fn stream_model_with_capabilities(
&self,
_request: HostManagedModelRequest,
_capabilities: Arc<dyn LoopCapabilityPort>,
) -> Result<HostManagedModelResponse, HostManagedModelError> {
Ok(HostManagedModelResponse::assistant_reply(
"done".to_string(),
))
}
}

/// #6263 Step 4 — production-flip wiring at the composition seam. With
/// `inmemory-turn-state` on, `build_reborn_runtime` composes the durable turn-state
/// ROW store (`factory.rs`), replacing the former in-memory authority +
/// block-persistence snapshot. This drives a real turn end to end over that store
/// (submit → claim → terminal, through the production runtime), then gracefully
/// `shutdown()`s — which routes through `RebornRuntime::shutdown →
/// FilesystemTurnStateStoreKind::drain`. The profile ships the row store at the
/// `WriteThrough` default (WriteBehind is blocked on the row store's non-cache-aware
/// query paths — see the factory arm), so the shutdown drain is a no-op here; the
/// test locks that composing the flipped store, serving a real turn over it, and
/// draining on shutdown all succeed without error/hang/panic.
///
/// Deeper durability is pinned one tier down, over the raw store where
/// scope/backend are controlled precisely: terminal/gate-park recovery across a
/// store reopen and the drain-flushes-the-tail contract in
/// `ironclaw_turns::row_store_crash_consistency` (incl.
/// `write_behind_drain_flushes_the_async_tail_for_graceful_restart`), and the
/// block-persistence→row migration in
/// `filesystem_turn_state_contract::filesystem_turn_state_row_store_migrates_block_persistence_gate_park_snapshot`.
#[cfg(feature = "inmemory-turn-state")]
#[tokio::test]
async fn inmemory_turn_state_row_store_serves_turn_and_drains_on_shutdown() {
let _guard = runtime_composition_test_guard().await;
let root = tempfile::tempdir().unwrap();
let input = RebornRuntimeInput::from_services(
RebornBuildInput::local_dev("wb-durable-owner", root.path().join("local-dev"))
.with_runtime_policy(local_dev_runtime_policy()),
)
.with_identity(RebornRuntimeIdentity {
tenant_id: "wb-durable-tenant".to_string(),
agent_id: "wb-durable-agent".to_string(),
source_binding_id: "wb-durable-source".to_string(),
reply_target_binding_id: "wb-durable-reply".to_string(),
})
.with_runner_settings(
TurnRunnerSettings::default()
.set_heartbeat_interval(Duration::from_secs(60))
.set_poll_interval(Duration::from_secs(60)),
)
.with_model_gateway_override(Arc::new(AlwaysReplyGateway));

// Compose the durable row store via the production build path and drive a real
// turn to Completed over it: proves the flipped store serves the full
// submit → claim → terminal transition set through the production runtime.
let runtime = build_reborn_runtime(input).await.unwrap();
let conversation = runtime.new_conversation().await.unwrap();
let reply = tokio::time::timeout(
SEND_USER_MESSAGE_TIMEOUT,
runtime.send_user_message(&conversation, "durable please"),
)
.await
.unwrap()
.unwrap();
assert_eq!(
reply.status,
TurnStatus::Completed,
"turn must complete over the WriteBehind store, got {:?} ({:?})",
reply.status,
reply.failure_category
);

// Graceful shutdown drains the WriteBehind tail through
// `FilesystemTurnStateStoreKind::drain`; a broken drain wiring surfaces here.
Comment on lines +303 to +312

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

📐 Maintainability & Code Quality | 🔵 Trivial | 💤 Low value

Assertion message and comment say "WriteBehind" but this store ships WriteThrough.

The test's own doc comment states the profile ships the row store at the WriteThrough default and the drain is a no-op here, yet the assert message ("turn must complete over the WriteBehind store") and the shutdown comment ("drains the WriteBehind tail") claim otherwise. Align the wording with the actual policy so a future reader isn't misled about what is exercised.

🤖 Prompt for AI Agents
Verify each finding against current code. Fix only still-valid issues, skip the
rest with a brief reason, keep changes minimal, and validate.

In `@crates/ironclaw_reborn_composition/tests/runtime.rs` around lines 303 - 312,
Update the assertion message and shutdown comment near the turn completion check
to refer to the store’s WriteThrough policy, not WriteBehind. Describe shutdown
as using the WriteThrough default and its no-op drain, while preserving the
existing assertion and behavior.

runtime
.shutdown()
.await
.expect("graceful shutdown drains the WriteBehind tail without error");
}

#[tokio::test]
async fn send_user_message_with_cancellation_cancels_submitted_run() {
let _guard = runtime_composition_test_guard().await;
Expand Down
27 changes: 26 additions & 1 deletion crates/ironclaw_turns/src/filesystem_store.rs
Original file line number Diff line number Diff line change
Expand Up @@ -76,7 +76,7 @@ use io::{deserialize_snapshot, fs_error, snapshot_entry, snapshot_path};
use profile_resolver::PreResolvedRunProfileResolver;
use runner_lease::{RunnerLeaseMemory, RunnerLeaseOverlay, RunnerLeaseRecord, RunnerLeaseStore};

pub use row_store::FilesystemTurnStateRowStore;
pub use row_store::{FilesystemTurnStateRowStore, TurnStateDurabilityPolicy};

#[cfg(test)]
mod tests;
Expand Down Expand Up @@ -668,6 +668,31 @@ where
}
}

/// Select the durable-commit mode. Forwards to the row store's
/// [`with_durability_policy`](FilesystemTurnStateRowStore::with_durability_policy);
/// the blob layout is `WriteThrough`-only (it persists every transition
/// synchronously via whole-snapshot CAS and buffers nothing), so the policy
/// is a no-op there and the store is returned unchanged.
pub fn with_durability_policy(self, durability_policy: TurnStateDurabilityPolicy) -> Self {
match self {
Self::Blob(store) => Self::Blob(store),
Self::Row(store) => {
Self::Row(Box::new((*store).with_durability_policy(durability_policy)))
}
}
}

/// Flush any pending `WriteBehind` durable tail so a planned shutdown leaves
/// nothing un-durable. Forwards to the row store's
/// [`drain`](FilesystemTurnStateRowStore::drain); the blob layout buffers
/// nothing (`WriteThrough`-only), so draining it is a no-op.
pub async fn drain(&self) -> Result<(), TurnError> {
match self {
Self::Blob(_) => Ok(()),
Self::Row(store) => store.drain().await,
}
}

pub async fn persistence_snapshot(&self) -> Result<TurnPersistenceSnapshot, TurnError> {
match self {
Self::Blob(store) => store.persistence_snapshot().await,
Expand Down
Loading
Loading