From 54ab40880d2d1163b748d58dbd51546a052b2c97 Mon Sep 17 00:00:00 2001 From: Henry Park Date: Sat, 4 Jul 2026 11:32:47 -0700 Subject: [PATCH 1/5] =?UTF-8?q?test(reborn):=20integration-suite=20restruc?= =?UTF-8?q?ture=20=E2=80=94=20tests/integration/=20home,=20framework=20dis?= =?UTF-8?q?entanglement,=20single-run=20coverage=20(#5633)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * test(reborn): move roadmap integration suite to tests/integration/ (skeleton + renames) Commit 1/8 of the integration-suite restructure. Byte-pure moves: tests/support/reborn/ -> tests/integration/support/, 27 reborn_integration_*.rs bins -> tests/integration/.rs, 7 reborn_group_* dirs -> tests/integration/group_/. Cargo [[test]] names stay identical; only paths move (27 entries added, 7 retargeted). Only content edits: the #[path] mount lines each move forces (flat bins, group mains, 33 parity/QA bins) — wiring folded into this commit so every commit builds green. Co-Authored-By: Claude Fable 5 * test(reborn): split harness.rs into harness/{mod,recorder,options,assembly} + doubles/, extract binary-E2E family to tests/support/reborn_parity_qa/ Commit 2/8. Symbol-preserving split of the 5530-line harness.rs at its final home: recorder.rs (capability recorder), options.rs (HostRuntimeHarnessOptions, verbatim), assembly.rs (local_dev_* runtime/filesystem/policy/mount helpers), doubles/ (14 files, one per substituted production port, rustdoc header names the production seam), residual core in harness/mod.rs with pub(crate) use re-exports preserving every surviving reborn_support::harness:: path. RebornBinaryE2EHarness + SubmittedTurn + RebornHarnessSharedStorage + HarnessLoopExitEvidencePort + assert_milestone_order + trace_tool_call_response and model_replay.rs leave to tests/support/reborn_parity_qa/ (their consumers are exclusively parity/QA); the 33 parity/QA bins mount the new parity_qa_support tree and repoint only those imports. Only content changes beyond the moves: use/mod lines, visibility bumps the new module boundaries force, and per-file dead_code allows replacing the old blanket allow. Co-Authored-By: Claude Fable 5 * test(reborn): collapse harness tools-constructor table into ToolsProfile + harness/profiles/ domains Commit 3/8 — the design commit. HostRuntimeHarnessOptions gains Default; new ToolsProfile value type (options.rs) captures the shared new_with_options seven-arg shape plus the four observed post-construct steps (network-policy override, provider-trust override, asset copy, auto-approve default), applied in the constructors' existing order by the one shared ToolsProfile::build() path. Every tools constructor leaves harness/mod.rs for a per-domain profiles/.rs file (16 domains): ToolsProfile rows become _profile() + a thin build wrapper; bespoke constructors (qa_smoke, core_builtin, github issue trio, mock_mcp, web_access) move verbatim. The 4-deep core_builtin suffix-variant chain folds into one CoreBuiltinOptions; the four variants are deleted. project_tools_with_fault_injection (new since the plan's table) rides in profiles/project.rs. Thick group pairs (live_approvals, live_auth_and_approval, profile_tools, outbound_target_tools) adopt ToolsProfile::build_group_capability_with_base over GroupBaseData with auto-approve disables kept explicit at call sites; capability_backend::install() now selects via the same profile fns, deleting its duplicate constructor-selection table. The six group-builder runtime setters move to group_options.rs (private child module of group.rs, group_constructors precedent). Constructor doc comments carried verbatim. Flip-checked: falsifying the trigger profile's capability set turns reborn_group_triggers red for the right reason. Co-Authored-By: Claude Fable 5 * test(reborn): finish parity/QA disentanglement — qa/delivery/network support to reborn_parity_qa/, drop dead approval.rs Commit 4/8. git-mv qa_trace.rs, qa_scenarios.rs, delivery.rs, network.rs from tests/integration/support/ to tests/support/reborn_parity_qa/ (their consumers are exclusively parity/QA bins) and rewrite those consumers' imports (qa_recorded_behavior, qa_smoke_scenarios_e2e, qa_doc_grounding, qa_web_fetch, qa_channel_delivery, outbound_reply_target parity, support_unit_tests). Delete approval.rs — zero consumers anywhere (pub use GateRef + an unused alias). classify-test-scope.sh's reborn path arm now matches tests/integration/* and tests/support/reborn_parity_qa/* instead of the removed tests/support/reborn/*; fixture updated. New tests/support/reborn_parity_qa/CLAUDE.md documents the tier split and the one-way import direction (parity/QA imports FROM tests/integration/support/, never the reverse — direction grep clean). Co-Authored-By: Claude Fable 5 * ci(reborn): single-run integration-tier coverage lanes + suite-boundary guard Commit 6/8. Kills the run-twice shape: the 34 tests/integration/ suites now run ONCE, instrumented, in a 5-lane reborn-integration-coverage matrix (4 modulo partitions of the 27 flat bins + 1 group lane), features unified to libsql. Each lane produces its lcov from one combined 'cargo llvm-cov --workspace --features libsql test --test ...' invocation — the split --no-report + 'report --lcov' shape silently drops all crates/ironclaw_* files because the standalone report subcommand has no --workspace flag (verified empirically). New coverage-report job merges lane lcovs (checked-in Python merger; sums DA per file:line, recomputes LF/LH, filters to crates/ironclaw_*), applies tests/integration/ coverage-exemptions.toml (seeded empty; every entry needs reason + issue link), and renders a per-crate table to the job summary + a sticky PR comment. Informational only — pass/fail gating rides the instrumented lanes; coverage-report failures warn, never red, the roll-up. reborn-coverage.yml (the duplicate compile+run) is deleted. Parity/QA root-partition lanes stay uninstrumented — the harness-only coverage boundary is enforced by job topology. Instrumented lanes use a dedicated reborn-integration-cov cache key. Discovery scripts retarget to tests/integration/ (group and int-tier enumeration; root partitions now match only the parity/QA bins by construction). New scripts/ci/check-test-suite-boundaries.sh (invoked next to classify-test-scope.sh) enforces the one-way support-tree direction, the parity/QA mount rule, and that tests/support/reborn/ never reappears. Co-Authored-By: Claude Fable 5 * docs(reborn): repoint living docs to tests/integration/ layout Commit 7/8. tests/integration/support/CLAUDE.md moves to tests/integration/CLAUDE.md with surgical fixes for the new reality (harness/ split + profiles/ + doubles/ file map, new mount-line boilerplate for flat bins and group mains, [[test]] naming convention, binary-E2E family pointer to tests/support/reborn_parity_qa/CLAUDE.md). Root CLAUDE.md spec table + testing pointers, the ironclaw-reborn-testing and reborn-feature skills, and the two living docs/reborn/ pages repoint the same way. Historical dated plans/specs under docs/superpowers/ deliberately untouched. Co-Authored-By: Claude Fable 5 * test(reborn): comment de-bloat across tests/integration/ Commit 8/8. Comment-only pass over all 126 tests/integration/ files (zero code changes — verified by non-comment diff-line scan). Deleted: narration/play-by-play, migration/wave/lane provenance, restated signatures, duplicated profile/wrapper doc blocks, history essays. Kept, compressed to 1-2 lines: why-pins, mutation-verified notes, doubles seam contracts, product decisions — invariant + issue/PR ref; every C-*/E-*/T0-* seam id and DEFERRED/COVERED pointer retained (regex-checked per lane). ~1,800 comment lines removed net. Co-Authored-By: Claude Fable 5 * test(reborn): apply post-PR review findings — explicit ToolsProfile user_id, const dedup, stale-path sweep + guard Review follow-up (thermo-nuclear + code-review passes on #5633): - ToolsProfile::new now takes user_id explicitly — the service_label-seeded placeholder was a silently-valid wrong value if a profile ever forgot to override it; all 15 profile call sites pass their fixed domain user id. - TEST_CAPABILITY_ID/TEST_CAPABILITY_SURFACE_VERSION deduplicated: binary_e2e now imports the doubles-tree constants instead of carrying drift-prone copies. - Stale tests/support/reborn/ references swept from tests/** and Cargo.toml (snapshot source headers included); check-test-suite-boundaries.sh gains check #4 failing on any reappearance of the retired path under tests/. (Doc-comment stragglers inside crates/ are deliberately left for a separate docs-only PR — production files stay outside this PR's diff surface.) Co-Authored-By: Claude Fable 5 * test(reborn): address #5633 review findings — script hardening, doc repoints, harness re-encapsulation CI scripts: lcov crate regex accepts relative paths ((?:^|/)crates/) in merge+summary; exemption module paths must start with crates/ (suffix match could over-exempt; +A9 regression case); classify-test-scope now marks the coverage/boundary helper scripts as Reborn scope so helper-only PRs can't fast-pass the pipeline they drive (+6 regression cases). Docs/comments: stale pointers to the removed tests/integration/support/ CLAUDE.md now target tests/integration/CLAUDE.md; group-main boilerplate doc shows the real ../support/mod.rs mount; harness/mod.rs header rewritten for HostRuntimeCapabilityHarness (binary-E2E contract moved); doubles headers cite each substituted symbol's actual production file; parity-bin comment and missing #[allow(dead_code)] mount attr fixed. Structure (review-accepted): build_group_capability_with_base moved from ToolsProfile into group_constructors.rs — profile layer no longer imports group::GroupBaseData, and GroupBaseData/canonical_subject_user return to module-private. Capability-port assembly moved into harness-owned create_recording_capability_port; HostRuntimeHarnessCapabilityPortFactory is a thin trait adapter and 13 harness fields + 2 methods narrowed from pub(crate) to module-private. Coverage: live_shell_uses_local_process_port pins the ShellMode::Live caller path — real LocalHostProcessPort output surfaces in the tool result while the inert recording port stays untouched. Co-Authored-By: Claude Fable 5 --------- Co-authored-by: Claude Fable 5 --- .../skills/ironclaw-reborn-testing/SKILL.md | 8 +- .../references/exemplar-tests.md | 6 +- .claude/skills/reborn-feature/SKILL.md | 2 +- .github/workflows/reborn-coverage.yml | 153 - .github/workflows/reborn-tests.yml | 196 + CLAUDE.md | 5 +- Cargo.toml | 121 +- .../2026-06-08-subagent-durability-spec.md | 2 +- docs/reborn/engine-v2-to-reborn-parity.md | 14 +- scripts/ci/check-test-suite-boundaries.sh | 129 + scripts/ci/classify-test-scope.sh | 5 +- scripts/ci/reborn-coverage-comment.sh | 20 +- scripts/ci/reborn-coverage-int-tier-tests.sh | 22 +- scripts/ci/reborn-coverage-lane-run.sh | 129 + scripts/ci/reborn-coverage-merge-lcov.sh | 105 + scripts/ci/reborn-coverage-summary.sh | 250 +- scripts/ci/run-reborn-group-tests.sh | 30 +- scripts/ci/test-classify-test-scope.sh | 50 +- scripts/ci/test-reborn-coverage.sh | 491 +- .../{support/reborn => integration}/CLAUDE.md | 65 +- .../attach.rs} | 88 +- .../auth_failure.rs} | 115 +- .../auth_gate.rs} | 151 +- .../backend_matrix.rs} | 32 +- .../budget.rs} | 30 +- .../cancel.rs} | 89 +- .../comm_context.rs} | 29 +- tests/integration/coverage-exemptions.toml | 31 + .../durable.rs} | 3 +- .../golden_payload.rs} | 124 +- .../greeting.rs} | 28 +- tests/integration/group_approvals/main.rs | 141 + ..._approval_request_persists_after_reopen.rs | 20 +- ...io_approve_always_persists_cross_thread.rs | 19 +- .../scenario_ask_each_time_resumes_once.rs | 41 +- .../scenario_concurrent_dual_gate_resume.rs | 118 + .../scenario_failure_category_demasked.rs | 69 + .../scenario_gate_ref_edge_cases.rs | 15 +- .../scenario_gate_then_approve.rs | 17 +- .../scenario_gate_then_deny.rs | 14 +- .../group_extensions}/main.rs | 38 +- ...nario_activate_then_active_cross_thread.rs | 121 + ...nario_install_then_visible_cross_thread.rs | 30 +- ...stall_unknown_extension_id_fails_safely.rs | 47 + ...cenario_remove_then_absent_cross_thread.rs | 47 +- tests/integration/group_journeys/main.rs | 112 + .../scenario_auth_deny_then_retry_journey.rs | 83 + .../scenario_auth_then_approval_journey.rs | 53 +- .../scenario_interactive_approval_journey.rs | 31 +- .../scenario_multi_actor_gate_isolation.rs | 39 +- .../group_memory}/main.rs | 15 +- .../scenario_memory_search_finds_seeded.rs | 11 +- ...scenario_memory_tree_reflects_structure.rs | 5 +- .../scenario_write_then_read_cross_thread.rs | 3 +- .../group_multiuser}/main.rs | 4 +- ...io_auto_approve_isolation_across_actors.rs | 41 +- ...scenario_memory_isolation_across_actors.rs | 44 +- .../scenario_two_actors_own_threads.rs | 12 +- .../group_skills}/main.rs | 4 +- .../scenario_install_list_remove.rs | 0 .../group_triggers}/main.rs | 74 +- .../scenario_trigger_persists_after_reopen.rs | 6 +- .../scenario_trigger_self_create_denied.rs | 112 + .../scenario_triggered_chained_gate.rs | 31 +- .../scenario_triggered_gate.rs | 41 +- .../scenario_verbs_lifecycle.rs | 27 +- .../hooks.rs} | 60 +- .../http_matcher.rs} | 16 +- .../mcp.rs} | 38 +- .../oauth_connect.rs} | 0 .../oauth_refresh.rs} | 25 +- .../outbound_target.rs} | 54 +- .../process_port.rs} | 77 +- .../profile.rs} | 51 +- .../project_create.rs} | 89 +- .../safety.rs} | 3 +- .../secret_injection.rs} | 20 +- .../secrets.rs} | 44 +- .../skill_activate.rs} | 68 +- .../support}/assertions.rs | 164 +- .../reborn => integration/support}/builder.rs | 229 +- .../support}/capability_backend.rs | 15 +- .../support}/comm_context.rs | 21 +- .../reborn => integration/support}/config.rs | 0 .../doubles/empty_identity_context_source.rs | 18 + ...xed_runtime_credential_account_resolver.rs | 28 + .../doubles/github_harness_authorizer.rs | 60 + .../harness_capability_port_factory.rs | 24 + ...runtime_harness_capability_port_factory.rs | 27 + tests/integration/support/doubles/mod.rs | 37 + .../recording_approval_request_store.rs | 77 + .../recording_capability_result_writer.rs | 60 + .../recording_delegating_capability_port.rs | 62 + .../support/doubles/recording_host_runtime.rs | 126 + .../doubles/recording_network_http_egress.rs | 74 + .../doubles/recording_runtime_http_egress.rs | 115 + .../doubles/recording_test_capability_port.rs | 259 + ...tic_capability_surface_profile_resolver.rs | 19 + .../support/doubles/static_secret_store.rs | 112 + .../support}/extension_surface.rs | 0 .../support}/filesystem.rs | 15 +- .../reborn => integration/support}/github.rs | 7 +- tests/integration/support/golden.rs | 123 + .../reborn => integration/support}/group.rs | 294 +- .../support}/group_constructors.rs | 213 +- tests/integration/support/group_options.rs | 89 + tests/integration/support/harness/assembly.rs | 462 ++ tests/integration/support/harness/mod.rs | 1339 ++++ tests/integration/support/harness/options.rs | 295 + .../support/harness/profiles/attachment.rs | 36 + .../support/harness/profiles/coding_read.rs | 36 + .../support/harness/profiles/core_builtin.rs | 225 + .../support/harness/profiles/extension.rs | 139 + .../support/harness/profiles/file.rs | 63 + .../support/harness/profiles/github.rs | 181 + .../support/harness/profiles/mock_mcp.rs | 98 + .../support/harness/profiles/mod.rs | 16 + .../support/harness/profiles/outbound.rs | 52 + .../support/harness/profiles/process.rs | 35 + .../support/harness/profiles/profile.rs | 38 + .../support/harness/profiles/project.rs | 77 + .../support/harness/profiles/qa_smoke.rs | 123 + .../support/harness/profiles/skill.rs | 90 + .../support/harness/profiles/trace_commons.rs | 52 + .../support/harness/profiles/trigger.rs | 38 + .../support/harness/profiles/web_access.rs | 72 + tests/integration/support/harness/recorder.rs | 143 + .../support}/harness_mcp.rs | 27 +- .../support}/harness_web_access.rs | 52 +- .../reborn => integration/support}/hooks.rs | 58 +- .../support}/http_matcher.rs | 82 +- .../reborn => integration/support}/mod.rs | 8 +- .../support}/oauth_flow.rs | 23 +- .../support}/outbound_preferences.rs | 47 +- .../reborn => integration/support}/process.rs | 27 +- .../support}/product_workflow.rs | 0 .../support}/project_service_fault.rs | 57 +- tests/integration/support/reply.rs | 97 + .../support}/scope_gateway.rs | 94 +- .../support}/scripted_provider.rs | 89 +- .../support}/session_thread.rs | 91 +- .../support}/test_adapter.rs | 0 .../support}/triggered_submit.rs | 162 +- .../tool_call.rs} | 108 +- .../tracecap.rs} | 44 +- .../triggered_submit.rs} | 65 +- .../web_access.rs} | 87 +- ...ter_installation_scope_isolation_parity.rs | 13 +- tests/reborn_agent_scope_isolation_parity.rs | 13 +- tests/reborn_approval_traces_parity.rs | 13 +- ...direct_chat_user_scope_isolation_parity.rs | 13 +- tests/reborn_group_approvals/main.rs | 182 - .../scenario_concurrent_dual_gate_resume.rs | 155 - .../scenario_failure_category_demasked.rs | 102 - ...nario_activate_then_active_cross_thread.rs | 155 - ...stall_unknown_extension_id_fails_safely.rs | 58 - tests/reborn_group_journeys/main.rs | 145 - .../scenario_auth_deny_then_retry_journey.rs | 106 - .../scenario_trigger_self_create_denied.rs | 153 - ...orn_http_network_scope_isolation_parity.rs | 9 +- ...identity_project_scope_isolation_parity.rs | 15 +- ..._identity_prompt_scope_isolation_parity.rs | 10 +- ..._identity_tenant_scope_isolation_parity.rs | 15 +- tests/reborn_minimal_dispatch_parity.rs | 7 +- ...und_reply_target_scope_isolation_parity.rs | 10 +- .../reborn_project_scope_isolation_parity.rs | 13 +- tests/reborn_qa_channel_delivery.rs | 17 +- tests/reborn_qa_connect_flows.rs | 13 +- tests/reborn_qa_doc_grounding.rs | 17 +- tests/reborn_qa_recorded_behavior.rs | 19 +- tests/reborn_qa_routines.rs | 13 +- tests/reborn_qa_smoke_scenarios_e2e.rs | 16 +- tests/reborn_qa_web_fetch.rs | 17 +- tests/reborn_recorded_trace_parity.rs | 15 +- tests/reborn_response_order_parity.rs | 7 +- tests/reborn_subagent_spawn_e2e.rs | 15 +- ...n_tenant_binding_scope_isolation_parity.rs | 13 +- .../reborn_thread_binding_isolation_parity.rs | 10 +- tests/reborn_tool_param_coercion_parity.rs | 11 +- .../reborn_trace_coding_read_tools_parity.rs | 9 +- .../reborn_trace_core_builtin_tools_parity.rs | 9 +- tests/reborn_trace_error_path_parity.rs | 9 +- tests/reborn_trace_file_tools_parity.rs | 9 +- .../reborn_trace_first_party_tool_coverage.rs | 9 +- ...reborn_trace_wasm_github_fixture_parity.rs | 9 +- ...born_turn_state_lock_free_submit_parity.rs | 13 +- ...orn_wrong_scope_access_isolation_parity.rs | 10 +- .../golden_payload__context_surfacing.snap | 2 +- .../golden_payload__gated_turn_approve.snap | 2 +- tests/snapshots/golden_payload__greeting.snap | 2 +- .../golden_payload__image_attachment.snap | 2 +- .../snapshots/golden_payload__multi_turn.snap | 2 +- .../golden_payload__parallel_tool_calls.snap | 2 +- .../snapshots/golden_payload__tool_call.snap | 2 +- tests/support/reborn/approval.rs | 21 - tests/support/reborn/golden.rs | 181 - tests/support/reborn/harness.rs | 5530 ----------------- tests/support/reborn/reply.rs | 119 - tests/support/reborn_parity_qa/CLAUDE.md | 56 + tests/support/reborn_parity_qa/binary_e2e.rs | 1465 +++++ .../{reborn => reborn_parity_qa}/delivery.rs | 0 tests/support/reborn_parity_qa/mod.rs | 14 + .../model_replay.rs | 0 .../{reborn => reborn_parity_qa}/network.rs | 0 .../qa_scenarios.rs | 0 .../{reborn => reborn_parity_qa}/qa_trace.rs | 0 tests/support_unit_tests.rs | 16 +- 207 files changed, 10347 insertions(+), 10165 deletions(-) delete mode 100644 .github/workflows/reborn-coverage.yml create mode 100755 scripts/ci/check-test-suite-boundaries.sh create mode 100755 scripts/ci/reborn-coverage-lane-run.sh create mode 100755 scripts/ci/reborn-coverage-merge-lcov.sh rename tests/{support/reborn => integration}/CLAUDE.md (91%) rename tests/{reborn_integration_attach.rs => integration/attach.rs} (60%) rename tests/{reborn_integration_auth_failure.rs => integration/auth_failure.rs} (67%) rename tests/{reborn_integration_auth_gate.rs => integration/auth_gate.rs} (50%) rename tests/{reborn_integration_backend_matrix.rs => integration/backend_matrix.rs} (64%) rename tests/{reborn_integration_budget.rs => integration/budget.rs} (58%) rename tests/{reborn_integration_cancel.rs => integration/cancel.rs} (64%) rename tests/{reborn_integration_comm_context.rs => integration/comm_context.rs} (70%) create mode 100644 tests/integration/coverage-exemptions.toml rename tests/{reborn_integration_durable.rs => integration/durable.rs} (96%) rename tests/{reborn_integration_golden_payload.rs => integration/golden_payload.rs} (55%) rename tests/{reborn_integration_greeting.rs => integration/greeting.rs} (60%) create mode 100644 tests/integration/group_approvals/main.rs rename tests/{reborn_group_approvals => integration/group_approvals}/scenario_approval_request_persists_after_reopen.rs (71%) rename tests/{reborn_group_approvals => integration/group_approvals}/scenario_approve_always_persists_cross_thread.rs (65%) rename tests/{reborn_group_approvals => integration/group_approvals}/scenario_ask_each_time_resumes_once.rs (54%) create mode 100644 tests/integration/group_approvals/scenario_concurrent_dual_gate_resume.rs create mode 100644 tests/integration/group_approvals/scenario_failure_category_demasked.rs rename tests/{reborn_group_approvals => integration/group_approvals}/scenario_gate_ref_edge_cases.rs (83%) rename tests/{reborn_group_approvals => integration/group_approvals}/scenario_gate_then_approve.rs (71%) rename tests/{reborn_group_approvals => integration/group_approvals}/scenario_gate_then_deny.rs (65%) rename tests/{reborn_group_extensions => integration/group_extensions}/main.rs (52%) create mode 100644 tests/integration/group_extensions/scenario_activate_then_active_cross_thread.rs rename tests/{reborn_group_extensions => integration/group_extensions}/scenario_install_then_visible_cross_thread.rs (55%) create mode 100644 tests/integration/group_extensions/scenario_install_unknown_extension_id_fails_safely.rs rename tests/{reborn_group_extensions => integration/group_extensions}/scenario_remove_then_absent_cross_thread.rs (52%) create mode 100644 tests/integration/group_journeys/main.rs create mode 100644 tests/integration/group_journeys/scenario_auth_deny_then_retry_journey.rs rename tests/{reborn_group_journeys => integration/group_journeys}/scenario_auth_then_approval_journey.rs (56%) rename tests/{reborn_group_journeys => integration/group_journeys}/scenario_interactive_approval_journey.rs (67%) rename tests/{reborn_group_journeys => integration/group_journeys}/scenario_multi_actor_gate_isolation.rs (79%) rename tests/{reborn_group_memory => integration/group_memory}/main.rs (79%) rename tests/{reborn_group_memory => integration/group_memory}/scenario_memory_search_finds_seeded.rs (85%) rename tests/{reborn_group_memory => integration/group_memory}/scenario_memory_tree_reflects_structure.rs (93%) rename tests/{reborn_group_memory => integration/group_memory}/scenario_write_then_read_cross_thread.rs (94%) rename tests/{reborn_group_multiuser => integration/group_multiuser}/main.rs (98%) rename tests/{reborn_group_multiuser => integration/group_multiuser}/scenario_auto_approve_isolation_across_actors.rs (68%) rename tests/{reborn_group_multiuser => integration/group_multiuser}/scenario_memory_isolation_across_actors.rs (62%) rename tests/{reborn_group_multiuser => integration/group_multiuser}/scenario_two_actors_own_threads.rs (87%) rename tests/{reborn_group_skills => integration/group_skills}/main.rs (96%) rename tests/{reborn_group_skills => integration/group_skills}/scenario_install_list_remove.rs (100%) rename tests/{reborn_group_triggers => integration/group_triggers}/main.rs (55%) rename tests/{reborn_group_triggers => integration/group_triggers}/scenario_trigger_persists_after_reopen.rs (90%) create mode 100644 tests/integration/group_triggers/scenario_trigger_self_create_denied.rs rename tests/{reborn_group_triggers => integration/group_triggers}/scenario_triggered_chained_gate.rs (76%) rename tests/{reborn_group_triggers => integration/group_triggers}/scenario_triggered_gate.rs (80%) rename tests/{reborn_group_triggers => integration/group_triggers}/scenario_verbs_lifecycle.rs (82%) rename tests/{reborn_integration_hooks.rs => integration/hooks.rs} (64%) rename tests/{reborn_integration_http_matcher.rs => integration/http_matcher.rs} (96%) rename tests/{reborn_integration_mcp.rs => integration/mcp.rs} (86%) rename tests/{reborn_integration_oauth_connect.rs => integration/oauth_connect.rs} (100%) rename tests/{reborn_integration_oauth_refresh.rs => integration/oauth_refresh.rs} (83%) rename tests/{reborn_integration_outbound_target.rs => integration/outbound_target.rs} (83%) rename tests/{reborn_integration_process_port.rs => integration/process_port.rs} (57%) rename tests/{reborn_integration_profile.rs => integration/profile.rs} (64%) rename tests/{reborn_integration_project_create.rs => integration/project_create.rs} (57%) rename tests/{reborn_integration_safety.rs => integration/safety.rs} (97%) rename tests/{reborn_integration_secret_injection.rs => integration/secret_injection.rs} (79%) rename tests/{reborn_integration_secrets.rs => integration/secrets.rs} (80%) rename tests/{reborn_integration_skill_activate.rs => integration/skill_activate.rs} (77%) rename tests/{support/reborn => integration/support}/assertions.rs (82%) rename tests/{support/reborn => integration/support}/builder.rs (85%) rename tests/{support/reborn => integration/support}/capability_backend.rs (91%) rename tests/{support/reborn => integration/support}/comm_context.rs (75%) rename tests/{support/reborn => integration/support}/config.rs (100%) create mode 100644 tests/integration/support/doubles/empty_identity_context_source.rs create mode 100644 tests/integration/support/doubles/fixed_runtime_credential_account_resolver.rs create mode 100644 tests/integration/support/doubles/github_harness_authorizer.rs create mode 100644 tests/integration/support/doubles/harness_capability_port_factory.rs create mode 100644 tests/integration/support/doubles/host_runtime_harness_capability_port_factory.rs create mode 100644 tests/integration/support/doubles/mod.rs create mode 100644 tests/integration/support/doubles/recording_approval_request_store.rs create mode 100644 tests/integration/support/doubles/recording_capability_result_writer.rs create mode 100644 tests/integration/support/doubles/recording_delegating_capability_port.rs create mode 100644 tests/integration/support/doubles/recording_host_runtime.rs create mode 100644 tests/integration/support/doubles/recording_network_http_egress.rs create mode 100644 tests/integration/support/doubles/recording_runtime_http_egress.rs create mode 100644 tests/integration/support/doubles/recording_test_capability_port.rs create mode 100644 tests/integration/support/doubles/static_capability_surface_profile_resolver.rs create mode 100644 tests/integration/support/doubles/static_secret_store.rs rename tests/{support/reborn => integration/support}/extension_surface.rs (100%) rename tests/{support/reborn => integration/support}/filesystem.rs (87%) rename tests/{support/reborn => integration/support}/github.rs (90%) create mode 100644 tests/integration/support/golden.rs rename tests/{support/reborn => integration/support}/group.rs (77%) rename tests/{support/reborn => integration/support}/group_constructors.rs (68%) create mode 100644 tests/integration/support/group_options.rs create mode 100644 tests/integration/support/harness/assembly.rs create mode 100644 tests/integration/support/harness/mod.rs create mode 100644 tests/integration/support/harness/options.rs create mode 100644 tests/integration/support/harness/profiles/attachment.rs create mode 100644 tests/integration/support/harness/profiles/coding_read.rs create mode 100644 tests/integration/support/harness/profiles/core_builtin.rs create mode 100644 tests/integration/support/harness/profiles/extension.rs create mode 100644 tests/integration/support/harness/profiles/file.rs create mode 100644 tests/integration/support/harness/profiles/github.rs create mode 100644 tests/integration/support/harness/profiles/mock_mcp.rs create mode 100644 tests/integration/support/harness/profiles/mod.rs create mode 100644 tests/integration/support/harness/profiles/outbound.rs create mode 100644 tests/integration/support/harness/profiles/process.rs create mode 100644 tests/integration/support/harness/profiles/profile.rs create mode 100644 tests/integration/support/harness/profiles/project.rs create mode 100644 tests/integration/support/harness/profiles/qa_smoke.rs create mode 100644 tests/integration/support/harness/profiles/skill.rs create mode 100644 tests/integration/support/harness/profiles/trace_commons.rs create mode 100644 tests/integration/support/harness/profiles/trigger.rs create mode 100644 tests/integration/support/harness/profiles/web_access.rs create mode 100644 tests/integration/support/harness/recorder.rs rename tests/{support/reborn => integration/support}/harness_mcp.rs (93%) rename tests/{support/reborn => integration/support}/harness_web_access.rs (70%) rename tests/{support/reborn => integration/support}/hooks.rs (65%) rename tests/{support/reborn => integration/support}/http_matcher.rs (53%) rename tests/{support/reborn => integration/support}/mod.rs (80%) rename tests/{support/reborn => integration/support}/oauth_flow.rs (82%) rename tests/{support/reborn => integration/support}/outbound_preferences.rs (74%) rename tests/{support/reborn => integration/support}/process.rs (73%) rename tests/{support/reborn => integration/support}/product_workflow.rs (100%) rename tests/{support/reborn => integration/support}/project_service_fault.rs (53%) create mode 100644 tests/integration/support/reply.rs rename tests/{support/reborn => integration/support}/scope_gateway.rs (61%) rename tests/{support/reborn => integration/support}/scripted_provider.rs (63%) rename tests/{support/reborn => integration/support}/session_thread.rs (56%) rename tests/{support/reborn => integration/support}/test_adapter.rs (100%) rename tests/{support/reborn => integration/support}/triggered_submit.rs (59%) rename tests/{reborn_integration_tool_call.rs => integration/tool_call.rs} (52%) rename tests/{reborn_integration_tracecap.rs => integration/tracecap.rs} (56%) rename tests/{reborn_integration_triggered_submit.rs => integration/triggered_submit.rs} (57%) rename tests/{reborn_integration_web_access.rs => integration/web_access.rs} (74%) delete mode 100644 tests/reborn_group_approvals/main.rs delete mode 100644 tests/reborn_group_approvals/scenario_concurrent_dual_gate_resume.rs delete mode 100644 tests/reborn_group_approvals/scenario_failure_category_demasked.rs delete mode 100644 tests/reborn_group_extensions/scenario_activate_then_active_cross_thread.rs delete mode 100644 tests/reborn_group_extensions/scenario_install_unknown_extension_id_fails_safely.rs delete mode 100644 tests/reborn_group_journeys/main.rs delete mode 100644 tests/reborn_group_journeys/scenario_auth_deny_then_retry_journey.rs delete mode 100644 tests/reborn_group_triggers/scenario_trigger_self_create_denied.rs delete mode 100644 tests/support/reborn/approval.rs delete mode 100644 tests/support/reborn/golden.rs delete mode 100644 tests/support/reborn/harness.rs delete mode 100644 tests/support/reborn/reply.rs create mode 100644 tests/support/reborn_parity_qa/CLAUDE.md create mode 100644 tests/support/reborn_parity_qa/binary_e2e.rs rename tests/support/{reborn => reborn_parity_qa}/delivery.rs (100%) create mode 100644 tests/support/reborn_parity_qa/mod.rs rename tests/support/{reborn => reborn_parity_qa}/model_replay.rs (100%) rename tests/support/{reborn => reborn_parity_qa}/network.rs (100%) rename tests/support/{reborn => reborn_parity_qa}/qa_scenarios.rs (100%) rename tests/support/{reborn => reborn_parity_qa}/qa_trace.rs (100%) diff --git a/.claude/skills/ironclaw-reborn-testing/SKILL.md b/.claude/skills/ironclaw-reborn-testing/SKILL.md index a91b1f1dd55..d9309108d94 100644 --- a/.claude/skills/ironclaw-reborn-testing/SKILL.md +++ b/.claude/skills/ironclaw-reborn-testing/SKILL.md @@ -1,17 +1,17 @@ --- name: ironclaw-reborn-testing -description: Use when adding or reviewing tests for Reborn behavior — choosing a test tier, covering a bug fix, testing model/tool-choice behavior, touching tests/support/reborn or tests/fixtures/llm_traces, or when a test needs Postgres, Docker, or a live LLM. +description: Use when adding or reviewing tests for Reborn behavior — choosing a test tier, covering a bug fix, testing model/tool-choice behavior, touching tests/integration or tests/fixtures/llm_traces, or when a test needs Postgres, Docker, or a live LLM. --- # Reborn Testing -Pick the tier first; everything else follows. The repo's tier knowledge lives in `tests/support/reborn/CLAUDE.md` (396 lines — read it before writing harness tests); this skill is the decision layer plus the traps. +Pick the tier first; everything else follows. The repo's tier knowledge lives in `tests/integration/CLAUDE.md` (read it before writing harness tests); this skill is the decision layer plus the traps. ## Tier decision tree 1. **Pure logic, no gated side effect** → unit test in the crate (`mod tests` / crate `tests/`). -2. **A helper gates a side effect (HTTP, DB write, egress body, approval, dispatch)** → you also need a caller-path test driving the real entry point (`*_handler`, facade method, adapter, coordinator). Helper-only is insufficient — `.claude/rules/testing.md` has the bug catalog. Gold standard: `tests/reborn_group_approvals/scenario_gate_then_approve.rs` asserts the approved write **exists on disk**. -3. **Whole-turn Reborn behavior (submit → runner → loop → reply), deterministic** → the in-process scripted-model harness (`tests/support/reborn/`, run as `cargo test --test reborn_`, zero setup, offline). **Mock only at the vendor-SDK seam** (`TraceLlm`): the real `ironclaw_llm` decorator chain (retry/failover/circuit-breaker) must execute. Mocking at the gateway seam skips it — that's the gateway-seam replay tier's job (`RebornBinaryE2EHarness` / `RebornTraceReplayModelGateway`), not yours by default. +2. **A helper gates a side effect (HTTP, DB write, egress body, approval, dispatch)** → you also need a caller-path test driving the real entry point (`*_handler`, facade method, adapter, coordinator). Helper-only is insufficient — `.claude/rules/testing.md` has the bug catalog. Gold standard: `tests/integration/group_approvals/scenario_gate_then_approve.rs` asserts the approved write **exists on disk**. +3. **Whole-turn Reborn behavior (submit → runner → loop → reply), deterministic** → the in-process scripted-model harness (`tests/integration/`, run as `cargo test --test reborn_integration_`, zero setup, offline). **Mock only at the vendor-SDK seam** (`TraceLlm`): the real `ironclaw_llm` decorator chain (retry/failover/circuit-breaker) must execute. Mocking at the gateway seam skips it — that's the gateway-seam replay tier's job (`RebornBinaryE2EHarness` / `RebornTraceReplayModelGateway`), not yours by default. 4. **Model tool-choice / request-shape is the behavior under test** → recorded QA fixtures (`tests/fixtures/llm_traces/reborn_qa/` + `tests/reborn_qa_recorded_behavior.rs`: ignored live recorder → hermetic contract assertions → hermetic replay). Fixtures must pass `scripts/ci/check-reborn-qa-fixtures.sh` (secret/PII scrub). Never commit unscrubbed traces. 5. **Browser-visible** → `tests/e2e/` Playwright (`reborn_v2_*` fixtures for WebChat v2). **Live LLM** → `#[ignore]` canary tier; supplemental only, never the PR gate. diff --git a/.claude/skills/ironclaw-reborn-testing/references/exemplar-tests.md b/.claude/skills/ironclaw-reborn-testing/references/exemplar-tests.md index df3a2591768..dae316b51ef 100644 --- a/.claude/skills/ironclaw-reborn-testing/references/exemplar-tests.md +++ b/.claude/skills/ironclaw-reborn-testing/references/exemplar-tests.md @@ -12,13 +12,13 @@ Living companions to the tier tree in `../SKILL.md`. Each exemplar is a real in- ## 1. Side-effect proof at the caller -`tests/reborn_group_approvals/scenario_gate_then_approve.rs` — drives a scripted `builtin.write_file` through the **real** stack: first-party runtime → `PermissionMode::Ask` → `TurnStatus::BlockedApproval` → real `ApprovalResolver::approve_dispatch` (lease issued) → `coordinator.resume_turn` → `Completed` — and then **asserts the file exists on disk**. The assertion target is the side effect itself, not a mock's call count. When your change gates a side effect, this is the shape: drive the public entry point, assert the world changed. Re-verify: `ls tests/reborn_group_approvals/`. +`tests/integration/group_approvals/scenario_gate_then_approve.rs` — drives a scripted `builtin.write_file` through the **real** stack: first-party runtime → `PermissionMode::Ask` → `TurnStatus::BlockedApproval` → real `ApprovalResolver::approve_dispatch` (lease issued) → `coordinator.resume_turn` → `Completed` — and then **asserts the file exists on disk**. The assertion target is the side effect itself, not a mock's call count. When your change gates a side effect, this is the shape: drive the public entry point, assert the world changed. Re-verify: `ls tests/integration/group_approvals/`. ## 2. The scripted-model harness seam -The in-process harness (`tests/support/reborn/`, spec in its `CLAUDE.md`) fakes exactly one thing: the vendor SDK at the bottom (`TraceLlm`). Everything else — product workflow, coordinator, scheduler, agent loop, the real `ironclaw_llm` retry/failover/circuit-breaker chain — executes for real, and assertions read *persisted state* (filesystem, thread history), never internals. +The in-process harness (code in `tests/integration/support/`, spec in `tests/integration/CLAUDE.md`) fakes exactly one thing: the vendor SDK at the bottom (`TraceLlm`). Everything else — product workflow, coordinator, scheduler, agent loop, the real `ironclaw_llm` retry/failover/circuit-breaker chain — executes for real, and assertions read *persisted state* (filesystem, thread history), never internals. -- **Right**: mock at the vendor-SDK seam; assert from durable state; `cargo test --test reborn_` runs offline with zero setup. +- **Right**: mock at the vendor-SDK seam; assert from durable state; `cargo test --test reborn_integration_` runs offline with zero setup. - **Wrong**: mocking at the gateway seam (skips the whole `ironclaw_llm` chain — that's the separate binary-replay tier's job); hand-building `TraceStep`s; asserting on internal structs. ## 3. The declare/enforce contract pair diff --git a/.claude/skills/reborn-feature/SKILL.md b/.claude/skills/reborn-feature/SKILL.md index e9e8879e8a2..5929bc0f541 100644 --- a/.claude/skills/reborn-feature/SKILL.md +++ b/.claude/skills/reborn-feature/SKILL.md @@ -87,7 +87,7 @@ For a feature with N endpoints, expect to touch (in dependency order): | HTTP | `ironclaw_webui_v2` | route constants + pattern + `*_descriptor()` (use `read_policy`/`mutation_policy`) + add to `webui_v2_routes()`; thin handler over `state.services()`; mount in `router.rs`; **update `tests/webui_v2_descriptors_contract.rs`** (it locks the table) | | Wiring | `ironclaw_reborn_composition` + `ironclaw_reborn_cli` | thread inputs through `RebornRuntimeInput`/`RebornRuntime`; attach in `build_webui_services`; pass from `serve.rs` | | Frontend | `ironclaw_webui_v2_static` | call endpoints via `apiFetch` in `static/js/pages/*/lib/*-api.js`; consume in hooks. No build step — `node --check .js` to syntax-check | -| Tests | `tests/support/reborn/` + crate tests | for whole-turn behavior, add a scripted-model harness case (see `tests/support/reborn/CLAUDE.md` — mock only at the vendor-SDK seam); facade changes extend `crates/ironclaw_product_workflow/tests/reborn_services_contract.rs`, and handler changes extend `crates/ironclaw_webui_v2/tests/webui_v2_handlers_contract.rs` | +| Tests | `tests/integration/` + crate tests | for whole-turn behavior, add a scripted-model harness case (see `tests/integration/CLAUDE.md` — mock only at the vendor-SDK seam); facade changes extend `crates/ironclaw_product_workflow/tests/reborn_services_contract.rs`, and handler changes extend `crates/ironclaw_webui_v2/tests/webui_v2_handlers_contract.rs` | ## Boundary rules (the guardrails that will reject your PR) diff --git a/.github/workflows/reborn-coverage.yml b/.github/workflows/reborn-coverage.yml deleted file mode 100644 index f8f51b89009..00000000000 --- a/.github/workflows/reborn-coverage.yml +++ /dev/null @@ -1,153 +0,0 @@ -# Reborn integration-tier coverage -# -# Runs cargo-llvm-cov over the Reborn in-process integration-tier test binaries -# (tests/reborn_integration_*.rs + tests/reborn_group_*/) and reports a -# Reborn-crate-scoped line-coverage percentage in the PR's job summary. -# -# This is the measurable signal behind the Reborn backend's "100% int-tier -# coverage" goal (task T0-COV): it yields a per-PR percentage and a per-crate -# hole list. It is INFORMATIONAL only — it -# does not gate the PR on a coverage threshold. -# -# Scope notes: -# - Features: default (postgres + libsql + html-to-markdown + tui), matching -# how scripts/ci/run-reborn-root-partition.sh runs these same suites. No live -# Postgres is needed; the suites skip Postgres backends when DATABASE_URL is -# unset (backend_matrix stays libsql-only per the roadmap). -# - The % is filtered to the Reborn crate families (crates/ironclaw_reborn*, -# ironclaw_product*, ironclaw_architecture, the v2 channel adapters, -# ironclaw_webui_v2*) via scripts/ci/reborn-coverage-summary.sh. - -name: Reborn Coverage - -on: - pull_request: - branches: [main] - push: - branches: [main] - workflow_dispatch: - -permissions: - contents: read - # The sticky coverage comment upserts a PR comment via the issues API. - pull-requests: write - -concurrency: - # PR number (not head_ref) for pull_request: two fork PRs can share a branch - # name (main/fix/…), and head_ref grouping + cancel-in-progress would let one - # cancel the other's run. number is null off-PR, so fall back to github.ref. - group: reborn-coverage-${{ github.event_name }}-${{ github.event.pull_request.number || github.ref }} - cancel-in-progress: true - -env: - # Tests must never touch the real OS keychain (see reborn-tests.yml). - IRONCLAW_DISABLE_OS_KEYCHAIN: "1" - CARGO_PROFILE_DEV_DEBUG: 0 - CARGO_PROFILE_TEST_DEBUG: 0 - # Keep instrumented build/runtime disk and time bounded. - CARGO_INCREMENTAL: "0" - -jobs: - reborn-coverage: - name: Reborn Coverage - runs-on: ubuntu-latest - # Generous backstop, not a tight bound: a cold-cache instrumented build of - # the Reborn closure is heavy (no mold), so this only catches a runaway/hung - # run — it sits well above any real build yet halves GitHub's 360-minute - # default, which would otherwise tie up a runner for hours. - timeout-minutes: 180 - env: - # Reborn suites must not reach a live LLM (mirrors reborn-tests.yml). - ANTHROPIC_API_KEY: "" - LLM_BACKEND: "" - LLM_USE_CODEX_AUTH: "false" - OLLAMA_BASE_URL: "" - OPENAI_API_KEY: "" - steps: - - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6 - with: - persist-credentials: false - - - name: Free runner disk space - run: | - df -h / - sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /usr/local/.ghcup /usr/local/share/boost /opt/hostedtoolcache/CodeQL || true - docker system prune -af || true - df -h / - - - uses: dtolnay/rust-toolchain@29eef336d9b2848a0b548edc03f92a220660cdb8 # stable - with: - components: llvm-tools-preview - - - uses: Swatinem/rust-cache@e18b497796c12c097a38f9edb9d0641fb99eee32 # v2 - with: - key: reborn-coverage - # Instrumented target dirs are multi-GB. Only save on main so the ~10 GB - # repo cache LRU holds ONE shared, main-seeded instrumented cache that - # every PR restores — instead of each PR writing a branch-scoped copy - # (invisible to other PRs) that evicts the useful seed. - save-if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/main' }} - - - name: Setup OVH sccache - uses: ./.github/actions/setup-sccache-dist - with: - scheduler-url: ${{ vars.SCCACHE_DIST_SCHEDULER_URL }} - auth-token: ${{ secrets.SCCACHE_DIST_AUTH_TOKEN }} - cache-ssh-host: ${{ vars.SCCACHE_CACHE_SSH_HOST }} - cache-ssh-user: ${{ vars.SCCACHE_CACHE_SSH_USER }} - cache-ssh-port: ${{ vars.SCCACHE_CACHE_SSH_PORT }} - cache-ssh-private-key: ${{ secrets.SCCACHE_CACHE_SSH_PRIVATE_KEY }} - cache-ssh-known-hosts: ${{ secrets.SCCACHE_CACHE_SSH_KNOWN_HOSTS }} - redis-password: ${{ secrets.SCCACHE_REDIS_PASSWORD }} - - - name: Install cargo-llvm-cov - uses: taiki-e/install-action@62b0f2dec647a8e604c6a0fda0e38530180dce20 # cargo-llvm-cov - - - name: Generate Reborn integration-tier coverage - run: | - set -euo pipefail - # Process substitution hides the discovery script's exit code, so - # guard the array explicitly (mirrors run-reborn-root-partition.sh). - mapfile -t test_args < <(scripts/ci/reborn-coverage-int-tier-tests.sh) - if [ "${#test_args[@]}" -eq 0 ]; then - echo "No Reborn integration-tier test arguments discovered" >&2 - exit 1 - fi - echo "Running coverage over Reborn int-tier binaries:" - printf ' %s %s\n' "${test_args[@]}" - - cargo llvm-cov clean --workspace - # Combined run+report in ONE invocation: --workspace scopes the report - # to every member crate (incl. the linked-but-not-root Reborn crates), - # which the standalone `report` subcommand cannot do. The named root - # `--test` targets still run only in the root package. Default features, - # matching scripts/ci/run-reborn-root-partition.sh. - cargo llvm-cov --workspace "${test_args[@]}" \ - --json --output-path reborn-llvm-cov.json -- --nocapture - - - name: Render Reborn coverage summary - run: | - set -euo pipefail - scripts/ci/reborn-coverage-summary.sh reborn-llvm-cov.json | tee -a "$GITHUB_STEP_SUMMARY" - - # Surface the same %/hole-list as a sticky PR comment so it lives in the - # conversation, not just the job summary. PR-only, and non-fatal by design: - # fork PRs get a read-only GITHUB_TOKEN and the comment API 403s, which must - # never red the check. This is visibility, never a gate. - - name: Post sticky coverage comment - if: github.event_name == 'pull_request' - continue-on-error: true - env: - GH_TOKEN: ${{ github.token }} - PR_NUMBER: ${{ github.event.pull_request.number }} - run: scripts/ci/reborn-coverage-comment.sh reborn-llvm-cov.json - - # The JSON export is the full per-file detail behind the summary table — - # the "real hole list" — kept as an artifact for offline inspection. - - name: Upload coverage report - if: always() - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 - with: - name: reborn-coverage - path: reborn-llvm-cov.json - if-no-files-found: warn diff --git a/.github/workflows/reborn-tests.yml b/.github/workflows/reborn-tests.yml index 6f0dd08ab38..86187162ec0 100644 --- a/.github/workflows/reborn-tests.yml +++ b/.github/workflows/reborn-tests.yml @@ -82,6 +82,11 @@ jobs: printf '%s\n' "$CHANGED_FILES" | scripts/ci/classify-test-scope.sh >> "$GITHUB_OUTPUT" + # Guard the tests/integration/ <-> tests/support/reborn_parity_qa/ + # one-way boundary on every PR/merge_group run, in the same step as + # the scope classifier above (both are cheap, checkout-only checks). + scripts/ci/check-test-suite-boundaries.sh + package-matrix: name: Discover Reborn package crates needs: changes @@ -448,6 +453,177 @@ jobs: - name: Run Reborn group tests run: scripts/ci/run-reborn-group-tests.sh + reborn-integration-coverage: + name: Reborn integration coverage (${{ matrix.lane }}) + needs: changes + if: needs.changes.outputs.docs_only != 'true' && needs.changes.outputs.has_reborn_tests == 'true' + runs-on: ubuntu-latest + # This is the ONLY execution of the 27 flat reborn_integration_* suites + # (there is no separate uninstrumented pass/fail run of them — `cargo + # llvm-cov ... test` has the same pass/fail semantics as `cargo test`, just + # instrumented) plus one more, instrumented, run of the 7 reborn_group_* + # suites already covered uninstrumented by reborn-group-tests above. + # Generous backstop: a cold-cache instrumented build of the Reborn closure + # is heavier than a plain build (see the retired reborn-coverage.yml's + # 180m budget for the same closure, run as a single job); each of these 5 + # lanes only carries a slice of the 34 int-tier suites, so 120m is ample. + timeout-minutes: 120 + env: + ANTHROPIC_API_KEY: "" + LLM_BACKEND: "" + LLM_USE_CODEX_AUTH: "false" + OLLAMA_BASE_URL: "" + OPENAI_API_KEY: "" + CARGO_INCREMENTAL: "0" + strategy: + fail-fast: false + matrix: + lane: ["0", "1", "2", "3", "groups"] + steps: + - name: Checkout repository + uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6 + with: + ref: ${{ inputs.ref || github.sha }} + persist-credentials: false + + - name: Free disk space + run: | + df -h / + sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /usr/local/.ghcup /usr/local/share/boost /opt/hostedtoolcache/CodeQL || true + docker system prune -af || true + df -h / + + - name: Install Rust + uses: dtolnay/rust-toolchain@29eef336d9b2848a0b548edc03f92a220660cdb8 # stable + with: + components: llvm-tools-preview + + - name: Install mold and clang + run: | + scripts/ci/install-ci-apt-packages.sh clang mold + test -x /usr/bin/mold + + - name: Restore Rust cache + uses: Swatinem/rust-cache@e18b497796c12c097a38f9edb9d0641fb99eee32 # v2 + with: + # ONE shared key across all 5 instrumented lanes: they all + # instrument the same workspace build, so a per-lane key would just + # mint 5 near-identical multi-GB entries that crowd the ~10 GB repo + # cache LRU. A dedicated key (never the plain `reborn-tests-*` keys + # above) keeps an instrumented (llvm-cov) build from ever evicting + # or being evicted by a plain one — the artifacts are incompatible. + shared-key: reborn-integration-cov + # Keep saves to protected-branch and merge_group runs so the ~10 GB + # repo cache LRU is seeded by shared states, not arbitrary PR branches. + save-if: ${{ (github.event_name == 'push' && github.ref == 'refs/heads/main') || github.event_name == 'merge_group' }} + + - name: Setup OVH sccache cache + uses: ./.github/actions/setup-sccache-dist + with: + cache-ssh-host: ${{ vars.SCCACHE_CACHE_SSH_HOST }} + cache-ssh-user: ${{ vars.SCCACHE_CACHE_SSH_USER }} + cache-ssh-port: ${{ vars.SCCACHE_CACHE_SSH_PORT }} + cache-ssh-private-key: ${{ secrets.SCCACHE_CACHE_SSH_PRIVATE_KEY }} + cache-ssh-known-hosts: ${{ secrets.SCCACHE_CACHE_SSH_KNOWN_HOSTS }} + redis-password: ${{ secrets.SCCACHE_REDIS_PASSWORD }} + + - name: Install cargo-llvm-cov + uses: taiki-e/install-action@62b0f2dec647a8e604c6a0fda0e38530180dce20 # cargo-llvm-cov + + - name: Run this lane's Reborn integration-tier suites (instrumented) and produce its lcov + env: + REBORN_COV_LANE_MODE: ${{ matrix.lane == 'groups' && 'group' || 'flat-partition' }} + REBORN_COV_LANE_PARTITIONS: 4 + REBORN_COV_LANE_INDEX: ${{ matrix.lane }} + REBORN_COV_LANE_TEST_TIMEOUT: 45m + run: | + # profraw-only (not --workspace): the shared reborn-integration-cov + # cache key is restored by whichever of the 5 matrix legs saved last + # (same shared-key-across-a-matrix pattern as crate-tests' single + # reborn-tests-crates key), so a stale *.profraw left over from a + # different lane's previous run could otherwise leak into this + # lane's report. Clearing just the profraw data (not the compiled + # instrumented binaries) keeps the build-cache speed win while + # guaranteeing this lane's lcov reflects only its own test run. + cargo llvm-cov clean --profraw-only + scripts/ci/reborn-coverage-lane-run.sh part-${{ matrix.lane }}.lcov + + - name: Upload lane lcov artifact + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 + with: + name: reborn-integration-cov-part-${{ matrix.lane }} + path: part-${{ matrix.lane }}.lcov + if-no-files-found: error + + coverage-report: + name: Reborn integration-tier coverage report + needs: [changes, reborn-integration-coverage] + if: needs.changes.outputs.docs_only != 'true' && needs.changes.outputs.has_reborn_tests == 'true' + runs-on: ubuntu-latest + timeout-minutes: 15 + permissions: + contents: read + # The sticky coverage comment upserts a PR comment via the issues API. + pull-requests: write + steps: + - name: Checkout repository + uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6 + with: + ref: ${{ inputs.ref || github.sha }} + persist-credentials: false + + - name: Download lane lcov artifacts + uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4 + with: + pattern: reborn-integration-cov-part-* + path: coverage-parts + merge-multiple: true + + - name: Merge lane lcov reports + run: | + set -euo pipefail + shopt -s nullglob + parts=(coverage-parts/*.lcov) + if [ "${#parts[@]}" -eq 0 ]; then + echo "No Reborn integration coverage lane artifacts found" >&2 + exit 1 + fi + scripts/ci/reborn-coverage-merge-lcov.sh reborn-integration-merged.lcov "${parts[@]}" + + - name: Render Reborn integration-tier coverage summary + run: | + set -euo pipefail + scripts/ci/reborn-coverage-summary.sh \ + reborn-integration-merged.lcov \ + tests/integration/coverage-exemptions.toml \ + | tee -a "$GITHUB_STEP_SUMMARY" + + # Surface the same %/hole-list as a sticky PR comment so it lives in the + # conversation, not just the job summary. PR-only, and non-fatal by + # design: fork PRs get a read-only GITHUB_TOKEN and the comment API + # 403s, which must never red the check. This is visibility, never a gate + # (mirrors the retired reborn-coverage.yml workflow's comment step). + - name: Post sticky coverage comment + if: github.event_name == 'pull_request' + continue-on-error: true + env: + GH_TOKEN: ${{ github.token }} + PR_NUMBER: ${{ github.event.pull_request.number }} + run: > + scripts/ci/reborn-coverage-comment.sh + reborn-integration-merged.lcov + tests/integration/coverage-exemptions.toml + + # The merged lcov is the full per-file detail behind the summary table — + # the "real hole list" — kept as an artifact for offline inspection. + - name: Upload merged coverage report + if: always() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 + with: + name: reborn-integration-coverage-merged + path: reborn-integration-merged.lcov + if-no-files-found: warn + webui-v2-js-tests: name: Reborn WebUI v2 JS tests needs: changes @@ -538,6 +714,8 @@ jobs: - crate-tests - root-reborn-parity-tests - reborn-group-tests + - reborn-integration-coverage + - coverage-report - webui-v2-js-tests - qa-recorded-fixtures steps: @@ -578,6 +756,24 @@ jobs: exit 1 fi + # This is the ONLY execution of the 27 flat reborn_integration_* + # suites (see the reborn-integration-coverage job), so it gates like + # every other test job above — unlike coverage-report below, this is + # real pass/fail, not the coverage percentage. + if [[ "${{ needs.reborn-integration-coverage.result }}" != "success" ]]; then + echo "One or more Reborn integration-tier coverage lanes failed: ${{ needs.reborn-integration-coverage.result }}" + exit 1 + fi + + # coverage-report only merges lane lcov files and renders/posts the + # informational summary — it never re-runs tests, so a failure here + # is a reporting-pipeline bug, not a test regression. Warn instead of + # failing the roll-up: the coverage signal has no ratchet/threshold + # yet, and the underlying tests are already gated above. + if [[ "${{ needs.coverage-report.result }}" != "success" ]]; then + echo "::warning::Reborn integration-tier coverage report job did not succeed: ${{ needs.coverage-report.result }} (informational only, not gating)" + fi + if [[ "${{ needs.webui-v2-js-tests.result }}" != "success" ]]; then echo "Reborn WebUI v2 JS tests failed: ${{ needs.webui-v2-js-tests.result }}" exit 1 diff --git a/CLAUDE.md b/CLAUDE.md index 1ca05f40d90..d2c7784b380 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -61,7 +61,7 @@ Two rules are non-negotiable for **all** tests: Where to look: hard rules (tiers, test-through-the-caller, regression-with-every-fix) in `.claude/rules/testing.md`; **Reborn -integration tests** authoring guide in `tests/support/reborn/CLAUDE.md`; +integration tests** authoring guide in `tests/integration/CLAUDE.md`; Python/Playwright suite in `tests/e2e/CLAUDE.md`. ## Code Style @@ -285,7 +285,8 @@ When modifying a module with a spec, read the spec first. Code follows spec; spe | `src/workspace/` | `src/workspace/README.md` | | `crates/ironclaw_reborn_webui_ingress/` | `crates/ironclaw_reborn_webui_ingress/CLAUDE.md` | | `crates/ironclaw_reborn_identity/` | `crates/ironclaw_reborn_identity/CONTRACT.md` | -| `tests/support/reborn/` | `tests/support/reborn/CLAUDE.md` | +| `tests/integration/` | `tests/integration/CLAUDE.md` | +| `tests/support/reborn_parity_qa/` | `tests/support/reborn_parity_qa/CLAUDE.md` | | `tests/e2e/` | `tests/e2e/CLAUDE.md` | ## Job State Machine diff --git a/Cargo.toml b/Cargo.toml index 3f0b5ca27a7..7bc34f65a19 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -282,7 +282,7 @@ ironclaw_filesystem = { path = "crates/ironclaw_filesystem", version = "0.1.0" } ironclaw_mcp = { path = "crates/ironclaw_mcp", version = "0.1.0" } ironclaw_first_party_extensions = { path = "crates/ironclaw_first_party_extensions", version = "0.1.0" } # Recording hook doubles + `HookDispatcherBuilderFactory` builders for the -# C-HOOKS / E-HOOK-INFRA int-tier coverage (tests/support/reborn/hooks.rs). +# C-HOOKS / E-HOOK-INFRA int-tier coverage (tests/integration/support/hooks.rs). ironclaw_hooks = { path = "crates/ironclaw_hooks", version = "0.1.0" } ironclaw_host_runtime = { path = "crates/ironclaw_host_runtime", version = "0.1.0" } ironclaw_network = { path = "crates/ironclaw_network", version = "0.1.0" } @@ -363,36 +363,141 @@ pr7-ready = [] [[test]] name = "reborn_group_approvals" -path = "tests/reborn_group_approvals/main.rs" +path = "tests/integration/group_approvals/main.rs" [[test]] name = "reborn_group_memory" -path = "tests/reborn_group_memory/main.rs" +path = "tests/integration/group_memory/main.rs" [[test]] name = "reborn_group_extensions" -path = "tests/reborn_group_extensions/main.rs" +path = "tests/integration/group_extensions/main.rs" [[test]] name = "reborn_group_triggers" -path = "tests/reborn_group_triggers/main.rs" +path = "tests/integration/group_triggers/main.rs" [[test]] name = "reborn_group_multiuser" -path = "tests/reborn_group_multiuser/main.rs" +path = "tests/integration/group_multiuser/main.rs" [[test]] name = "reborn_group_skills" -path = "tests/reborn_group_skills/main.rs" +path = "tests/integration/group_skills/main.rs" [[test]] name = "reborn_group_journeys" -path = "tests/reborn_group_journeys/main.rs" +path = "tests/integration/group_journeys/main.rs" [[test]] name = "reborn_integration_oauth_refresh" +path = "tests/integration/oauth_refresh.rs" required-features = ["libsql"] +[[test]] +name = "reborn_integration_attach" +path = "tests/integration/attach.rs" + +[[test]] +name = "reborn_integration_auth_failure" +path = "tests/integration/auth_failure.rs" + +[[test]] +name = "reborn_integration_auth_gate" +path = "tests/integration/auth_gate.rs" + +[[test]] +name = "reborn_integration_backend_matrix" +path = "tests/integration/backend_matrix.rs" + +[[test]] +name = "reborn_integration_budget" +path = "tests/integration/budget.rs" + +[[test]] +name = "reborn_integration_cancel" +path = "tests/integration/cancel.rs" + +[[test]] +name = "reborn_integration_comm_context" +path = "tests/integration/comm_context.rs" + +[[test]] +name = "reborn_integration_durable" +path = "tests/integration/durable.rs" + +[[test]] +name = "reborn_integration_golden_payload" +path = "tests/integration/golden_payload.rs" + +[[test]] +name = "reborn_integration_greeting" +path = "tests/integration/greeting.rs" + +[[test]] +name = "reborn_integration_hooks" +path = "tests/integration/hooks.rs" + +[[test]] +name = "reborn_integration_http_matcher" +path = "tests/integration/http_matcher.rs" + +[[test]] +name = "reborn_integration_mcp" +path = "tests/integration/mcp.rs" + +[[test]] +name = "reborn_integration_oauth_connect" +path = "tests/integration/oauth_connect.rs" + +[[test]] +name = "reborn_integration_outbound_target" +path = "tests/integration/outbound_target.rs" + +[[test]] +name = "reborn_integration_process_port" +path = "tests/integration/process_port.rs" + +[[test]] +name = "reborn_integration_profile" +path = "tests/integration/profile.rs" + +[[test]] +name = "reborn_integration_project_create" +path = "tests/integration/project_create.rs" + +[[test]] +name = "reborn_integration_safety" +path = "tests/integration/safety.rs" + +[[test]] +name = "reborn_integration_secret_injection" +path = "tests/integration/secret_injection.rs" + +[[test]] +name = "reborn_integration_secrets" +path = "tests/integration/secrets.rs" + +[[test]] +name = "reborn_integration_skill_activate" +path = "tests/integration/skill_activate.rs" + +[[test]] +name = "reborn_integration_tool_call" +path = "tests/integration/tool_call.rs" + +[[test]] +name = "reborn_integration_tracecap" +path = "tests/integration/tracecap.rs" + +[[test]] +name = "reborn_integration_triggered_submit" +path = "tests/integration/triggered_submit.rs" + +[[test]] +name = "reborn_integration_web_access" +path = "tests/integration/web_access.rs" + [[test]] name = "e2e_thread_scheduling" required-features = ["libsql", "integration"] diff --git a/docs/reborn/2026-06-08-subagent-durability-spec.md b/docs/reborn/2026-06-08-subagent-durability-spec.md index 16f938a65a0..9e1d5498c05 100644 --- a/docs/reborn/2026-06-08-subagent-durability-spec.md +++ b/docs/reborn/2026-06-08-subagent-durability-spec.md @@ -1831,7 +1831,7 @@ Existing parity harness in this repo: - `crates/ironclaw_hooks_parity/tests/parity_matrix.rs` — hooks-tier behavioral parity matrix. - `tests/reborn_wrong_scope_access_isolation_parity.rs` — cross-scope isolation parity at integration test tier. -- `tests/support_unit_tests.rs` and `tests/support/reborn/product_workflow.rs` — `RebornProductWorkflowHarness` / `FilesystemIdempotencyLedger` parity helpers with `filesystem_temp` + `filesystem_shared_backend` constructors. +- `tests/support_unit_tests.rs` and `tests/integration/support/product_workflow.rs` — `RebornProductWorkflowHarness` / `FilesystemIdempotencyLedger` parity helpers with `filesystem_temp` + `filesystem_shared_backend` constructors. Subagent store parity tests do **not** belong in `ironclaw_hooks_parity` (hooks-specific contract). Correct location: diff --git a/docs/reborn/engine-v2-to-reborn-parity.md b/docs/reborn/engine-v2-to-reborn-parity.md index 75c6d72810f..7a01259296f 100644 --- a/docs/reborn/engine-v2-to-reborn-parity.md +++ b/docs/reborn/engine-v2-to-reborn-parity.md @@ -52,19 +52,19 @@ evidence; **Partial** = an equivalent exists but with a scoped delta noted; | **Capability** — unit of effect (actions + knowledge + policies), replaces Tool/Skill/Hook/Extension; `types/capability.rs`, `capability/registry.rs` | Host-mediated capability invocation: descriptor lookup, trust-aware authorization, dispatch, spawn, obligations | `contracts/capabilities.md`, `contracts/capability-access.md`, `contracts/dispatcher.md` | `ironclaw_capabilities`, `ironclaw_authorization`, `ironclaw_dispatcher` | `tests/reborn_minimal_dispatch_parity.rs`, `crates/ironclaw_dispatcher/tests/vertical_slice_contract.rs`, `crates/ironclaw_host_runtime/tests/reborn_invoke_vertical_slice.rs` | Covered | | **Capability leases** — scoped/time-limited/use-limited grants; `capability/lease.rs::LeaseManager`, `LeasePlanner` | Grant- and lease-backed dispatch/spawn gates; async lease stores in the control plane | `contracts/capability-access.md`, `contracts/approvals.md` | `ironclaw_authorization`, `ironclaw_approvals` | `tests/reborn_agent_scope_isolation_parity.rs`, `tests/reborn_wrong_scope_access_isolation_parity.rs`, `crates/ironclaw_reborn_composition/tests/budget_e2e.rs` | Covered | | **PolicyEngine** — deterministic effect-level allow/deny/approve (`Deny > RequireApproval > Allow`) + provenance taint; `capability/policy.rs`, `types/provenance.rs` | `CapabilityDispatchAuthorizer` returning `Decision::Allow{obligations}/Deny/RequireApproval`; trust-class enforcement and taint at the kernel boundary | `contracts/capability-access.md`, `contracts/kernel-boundary.md`, `contracts/trust-boundary-hardening.md` | `ironclaw_authorization`, `ironclaw_trust` | `tests/reborn_agent_scope_isolation_parity.rs`, `tests/reborn_tenant_binding_scope_isolation_parity.rs`, `crates/ironclaw_reborn_composition/src/factory/local_dev_host_tests/approval_gates.rs` | Covered | -| **EffectType** taxonomy — `ReadLocal/ReadExternal/WriteLocal/WriteExternal/CredentialedNetwork/Compute/Financial`; `types/capability.rs` | Effect surface decomposed into typed resource/network/secrets/filesystem contracts feeding the authorizer + `ResourceEstimate` | `contracts/resources.md`, `contracts/network.md`, `contracts/secrets.md`, `contracts/filesystem.md` | `ironclaw_resources`, `ironclaw_network`, `ironclaw_secrets`, `ironclaw_filesystem` | `tests/reborn_http_network_scope_isolation_parity.rs`, `crates/ironclaw_reborn/tests/secrets.rs`, `tests/reborn_integration_secrets.rs` | Covered | -| **MemoryDoc** — durable knowledge (Summary/Lesson/Skill/Issue/Spec/Note); `types/memory.rs`, `memory/store.rs::MemoryStore`, `RetrievalEngine` | Provider-neutral `MemoryService` contract + native provider (chunking/indexing/embeddings/search, prompt-context assembly) | `contracts/memory.md`, `contracts/memory-profiles.md`, `contracts/storage-placement.md` | `ironclaw_memory`, `ironclaw_memory_native` | `tests/reborn_group_memory/` (suite), `tests/reborn_qa_doc_grounding.rs` | Covered (note 1) | +| **EffectType** taxonomy — `ReadLocal/ReadExternal/WriteLocal/WriteExternal/CredentialedNetwork/Compute/Financial`; `types/capability.rs` | Effect surface decomposed into typed resource/network/secrets/filesystem contracts feeding the authorizer + `ResourceEstimate` | `contracts/resources.md`, `contracts/network.md`, `contracts/secrets.md`, `contracts/filesystem.md` | `ironclaw_resources`, `ironclaw_network`, `ironclaw_secrets`, `ironclaw_filesystem` | `tests/reborn_http_network_scope_isolation_parity.rs`, `crates/ironclaw_reborn/tests/secrets.rs`, `tests/integration/secrets.rs` | Covered | +| **MemoryDoc** — durable knowledge (Summary/Lesson/Skill/Issue/Spec/Note); `types/memory.rs`, `memory/store.rs::MemoryStore`, `RetrievalEngine` | Provider-neutral `MemoryService` contract + native provider (chunking/indexing/embeddings/search, prompt-context assembly) | `contracts/memory.md`, `contracts/memory-profiles.md`, `contracts/storage-placement.md` | `ironclaw_memory`, `ironclaw_memory_native` | `tests/integration/group_memory/` (suite), `tests/reborn_qa_doc_grounding.rs` | Covered (note 1) | | **Project** — unit of context (scopes memory/threads/missions); `types/project.rs` | First-class `ProjectRecord` + membership ACL (`Owner>Editor>Viewer`) over the scoped filesystem substrate | `contracts/storage-placement.md`; plan `docs/plans/2026-06-17-reborn-projects.md` | `ironclaw_projects` | `crates/ironclaw_projects/tests/repository_contract.rs`, `tests/reborn_project_scope_isolation_parity.rs`, `tests/reborn_identity_project_scope_isolation_parity.rs` | Covered | | **Missions** — long-running goals that spawn threads on cadence; `runtime/mission.rs::MissionManager`, budget/rate gates | Scheduled trigger intake → synthetic inbound turn on the normal turn pipeline; budgets via authorization/approvals | `contracts/triggers.md`, `contracts/approvals.md` | `ironclaw_triggers` | `tests/reborn_qa_routines.rs`, `crates/ironclaw_reborn_composition/tests/trigger_poller_e2e.rs`, `crates/ironclaw_reborn_composition/tests/trigger_webui_timeline_e2e.rs`, `crates/ironclaw_reborn_composition/tests/budget_approval_e2e.rs` | Partial (note 2) | | **Learning missions** — error diagnosis, skill repair, skill extraction, conversation insights (`MissionManager::ensure_learning_missions`) | Skill distillation/refinement pipeline (extract + repair) over trace input | plan `docs/plans/2026-06-16-reborn-skill-evolution.md`; `contracts/skills-extension.md` | `ironclaw_skill_learning` | `crates/ironclaw_skill_learning/src/lib.rs` (`distill_skill_runs_inference_then_validates`, `parses_a_valid_skill_and_extracts_the_name`, `parse_refinement_accepts_a_refined_skill`) | Partial (note 2) | -| **Gates / Approvals** — `gate/` (`ExecutionGate`, `GatePipeline`, `LeaseGate`, `GateResolution`, `ResumeKind`), auth/approval resume | Durable approval requests resolved into bounded scoped leases; typed gate/resume with exact invocation identity; deny-continue flow | `contracts/approvals.md`, `contracts/capability-access.md`, `contracts/run-state.md`; plan `docs/plans/2026-06-15-reborn-approval-deny-continue.md` | `ironclaw_approvals`, `ironclaw_run_state` | `tests/reborn_approval_traces_parity.rs`, `tests/reborn_integration_auth_failure.rs`, `crates/ironclaw_reborn_composition/tests/budget_approval_e2e.rs`, `crates/ironclaw_reborn_composition/src/factory/local_dev_host_tests/approval_gates.rs` | Covered (note 3) | -| **CodeAct / Tier 1** — embedded Python via Monty (RLM): context-as-variables, `llm_query()` recursive subagent, compact output metadata; `executor/scripting.rs` | Two parts: (a) native script/software execution lane (`RuntimeKind::Script`); (b) CodeAct is an *allowed* pluggable parent loop family, and recursive subagents exist as `spawn_subagent` | `contracts/scripts.md`, `contracts/agent-loop-protocol.md` (CodeAct as parent protocol) | `ironclaw_scripts`, `ironclaw_agent_loop`, `ironclaw_process_sandbox` | `tests/reborn_integration_process_port.rs`, `tests/reborn_subagent_spawn_e2e.rs`, `crates/ironclaw_reborn_composition/tests/subagent_runtime_wiring.rs` | Partial (note 4) | +| **Gates / Approvals** — `gate/` (`ExecutionGate`, `GatePipeline`, `LeaseGate`, `GateResolution`, `ResumeKind`), auth/approval resume | Durable approval requests resolved into bounded scoped leases; typed gate/resume with exact invocation identity; deny-continue flow | `contracts/approvals.md`, `contracts/capability-access.md`, `contracts/run-state.md`; plan `docs/plans/2026-06-15-reborn-approval-deny-continue.md` | `ironclaw_approvals`, `ironclaw_run_state` | `tests/reborn_approval_traces_parity.rs`, `tests/integration/auth_failure.rs`, `crates/ironclaw_reborn_composition/tests/budget_approval_e2e.rs`, `crates/ironclaw_reborn_composition/src/factory/local_dev_host_tests/approval_gates.rs` | Covered (note 3) | +| **CodeAct / Tier 1** — embedded Python via Monty (RLM): context-as-variables, `llm_query()` recursive subagent, compact output metadata; `executor/scripting.rs` | Two parts: (a) native script/software execution lane (`RuntimeKind::Script`); (b) CodeAct is an *allowed* pluggable parent loop family, and recursive subagents exist as `spawn_subagent` | `contracts/scripts.md`, `contracts/agent-loop-protocol.md` (CodeAct as parent protocol) | `ironclaw_scripts`, `ironclaw_agent_loop`, `ironclaw_process_sandbox` | `tests/integration/process_port.rs`, `tests/reborn_subagent_spawn_e2e.rs`, `crates/ironclaw_reborn_composition/tests/subagent_runtime_wiring.rs` | Partial (note 4) | | **Self-modify** — prompt overlays / orchestrator patches applied by the self-improvement mission; skill versioning/rollback (`memory/skill_tracker.rs::SkillTracker`) | Versioned skill evolution (distill/refine with validation); overlay/orchestrator self-patching intentionally not carried over (Reborn has no Monty orchestrator to patch) | plan `docs/plans/2026-06-16-reborn-skill-evolution.md`; `contracts/skills-extension.md` | `ironclaw_skill_learning`, `ironclaw_skills` | `crates/ironclaw_skill_learning/src/lib.rs` (refine/distill tests) | Partial (note 4) | -| **Per-project Docker sandbox** — filesystem/shell tools routed through a per-project container (`SANDBOX_ENABLED`; `crates/Dockerfile.sandbox`) | Typed process plans, backend-neutral sandbox backends, hardened Docker command construction, fail-closed network-host validation, timeout/cancel cleanup | `contracts/processes.md`, `contracts/scripts.md`; design `docs/reborn/2026-05-26-docker-process-sandbox-mvp.md` | `ironclaw_process_sandbox`, `ironclaw_processes`, `ironclaw_wasm_sandbox_core` | `tests/reborn_integration_process_port.rs`, `crates/ironclaw_reborn_composition/tests/production_runtime_automations.rs` | Covered (note 5) | +| **Per-project Docker sandbox** — filesystem/shell tools routed through a per-project container (`SANDBOX_ENABLED`; `crates/Dockerfile.sandbox`) | Typed process plans, backend-neutral sandbox backends, hardened Docker command construction, fail-closed network-host validation, timeout/cancel cleanup | `contracts/processes.md`, `contracts/scripts.md`; design `docs/reborn/2026-05-26-docker-process-sandbox-mvp.md` | `ironclaw_process_sandbox`, `ironclaw_processes`, `ironclaw_wasm_sandbox_core` | `tests/integration/process_port.rs`, `crates/ironclaw_reborn_composition/tests/production_runtime_automations.rs` | Covered (note 5) | | **OpenAI-compatible Responses API** — engine v2's OpenAI-compatible ingress | Contract-first, ProductWorkflow-backed Chat Completions + Responses (create/retrieve/cancel), idempotency/opaque-ref, projection-backed SSE streaming | `contracts/openai-compatible-api.md` | `ironclaw_reborn_openai_compat`, `ironclaw_reborn_openai_compat_storage` | `crates/ironclaw_reborn_openai_compat/tests/responses_workflow_handlers_contract.rs`, `.../chat_workflow_handlers_contract.rs`, `.../streaming_handlers_contract.rs`, `.../error_contract.rs`, `crates/ironclaw_reborn_openai_compat_storage/tests/ref_store_contract.rs` | Covered (note 6) | -| **Skills** — trusted/installed skills, activation criteria, `skill_*` tools | First-party in-process skills extension: portable `SKILL.md` bundles; kernel owns trust/visibility/leases/context injection; catalog-first model-selected activation | `contracts/skills-extension.md` | `ironclaw_skills`, `ironclaw_skill_learning` | `tests/reborn_group_extensions/` (suite), `crates/ironclaw_reborn_cli/tests/smoke.rs` | Covered | +| **Skills** — trusted/installed skills, activation criteria, `skill_*` tools | First-party in-process skills extension: portable `SKILL.md` bundles; kernel owns trust/visibility/leases/context injection; catalog-first model-selected activation | `contracts/skills-extension.md` | `ironclaw_skills`, `ironclaw_skill_learning` | `tests/integration/group_extensions/` (suite), `crates/ironclaw_reborn_cli/tests/smoke.rs` | Covered | | **Hooks** — lifecycle hooks (6 points); `Hook` trait | Host-mediated hook execution with multi-backend persistence and an adversarial parity oracle | (hooks are exercised through `contracts/extensions.md` + capability dispatch) | `ironclaw_hooks`, `ironclaw_hooks_libsql`, `ironclaw_hooks_postgres`, `ironclaw_hooks_parity` | `crates/ironclaw_reborn/tests/hooks_integration.rs`, `crates/ironclaw_hooks_parity/tests/parity_matrix.rs`, `crates/ironclaw_hooks_parity/tests/multi_host_adversarial.rs`, `crates/ironclaw_reborn_composition/tests/third_party_hook_projection.rs` | Covered | -| **Extensions** — installed extension/channel lifecycle | Host-mediated extension registry + first-party extension ports; product adapters (Telegram/Slack v2, GSuite, MCP-hosted) | `contracts/extensions.md`, `contracts/product-adapters.md`, `contracts/mcp.md` | `ironclaw_extensions`, `ironclaw_first_party_extensions`, `ironclaw_product_adapters`, `ironclaw_mcp` | `tests/reborn_adapter_installation_scope_isolation_parity.rs`, `tests/reborn_integration_mcp.rs`, `crates/ironclaw_reborn_cli/tests/extension.rs`, `crates/ironclaw_reborn_composition/tests/gsuite.rs` | Covered | +| **Extensions** — installed extension/channel lifecycle | Host-mediated extension registry + first-party extension ports; product adapters (Telegram/Slack v2, GSuite, MCP-hosted) | `contracts/extensions.md`, `contracts/product-adapters.md`, `contracts/mcp.md` | `ironclaw_extensions`, `ironclaw_first_party_extensions`, `ironclaw_product_adapters`, `ironclaw_mcp` | `tests/reborn_adapter_installation_scope_isolation_parity.rs`, `tests/integration/mcp.rs`, `crates/ironclaw_reborn_cli/tests/extension.rs`, `crates/ironclaw_reborn_composition/tests/gsuite.rs` | Covered | | **Effect execution / tool dispatch** — `traits/effect.rs::EffectExecutor`, `ThreadExecutionContext` | Composition-only `RuntimeDispatcher::dispatch_json` selecting a `RuntimeAdapter` per `RuntimeKind`, normalized results, fail-closed | `contracts/dispatcher.md`, `contracts/capabilities.md` | `ironclaw_dispatcher`, `ironclaw_capabilities` | `tests/reborn_minimal_dispatch_parity.rs`, `tests/reborn_tool_param_coercion_parity.rs`, `tests/reborn_trace_core_builtin_tools_parity.rs`, `tests/reborn_trace_file_tools_parity.rs`, `tests/reborn_trace_wasm_github_fixture_parity.rs` | Covered | | **Events + Projections** — `ThreadEvent`, `EventKind` (18 variants), event sourcing from day one; `types/event.rs` | Explicit boundary between realtime delivery, durable audit/history, transcript milestones, and derived projections; durable event store | `contracts/events.md`, `contracts/events-projections.md` | `ironclaw_events`, `ironclaw_event_projections`, `ironclaw_event_streams`, `ironclaw_reborn_event_store` | `crates/ironclaw_reborn_event_store/tests/durable_event_store_contract.rs`, `.../filesystem_event_log_contract.rs`, `.../coalescing_sink_contract.rs`, `crates/ironclaw_reborn/tests/loop_milestone_event_projection.rs` | Covered | | **`LlmBackend` trait** — `complete(messages, actions, config)`; host wraps `LlmProvider` | Host-managed model requests / tool-capable model gateway; provider-safe tool projection (dotted `CapabilityId` ↔ provider names) | `contracts/host-api.md`, `contracts/runtime-workflows.md` | `ironclaw_llm`, `ironclaw_reborn` (model routes/gateway) | `crates/ironclaw_reborn/tests/model_routes.rs`, `crates/ironclaw_reborn/tests/llm_gateway.rs` | Covered | diff --git a/scripts/ci/check-test-suite-boundaries.sh b/scripts/ci/check-test-suite-boundaries.sh new file mode 100755 index 00000000000..69e09dd2472 --- /dev/null +++ b/scripts/ci/check-test-suite-boundaries.sh @@ -0,0 +1,129 @@ +#!/usr/bin/env bash +# +# Guard the tests/integration/ (coverage-bearing, roadmap-integration) vs +# tests/reborn_*.rs + tests/support/reborn_parity_qa/ (parity/QA, NOT +# coverage-bearing) suite boundary established by the restructure that moved +# the roadmap integration suite to tests/integration/ (see +# docs/superpowers/specs/2026-06-26-reborn-integration-test-framework-design.md +# and the git history under tests/integration/). +# +# The direction is one-way: tests/reborn_*.rs parity/QA bins MAY reuse +# tests/integration/support/ (the roadmap harness), but tests/integration/ +# suites must NEVER depend back on the parity/QA support tree — that would +# silently pull QA-only fixtures/harness weight into the suites this repo's +# coverage report is scoped to, and would resurrect exactly the coupling the +# restructure split apart. +# +# Checks: +# 1. Direction guard — no file under tests/integration/ mentions the +# parity/QA support tree (by its local module alias `parity_qa_support` +# or its mount path `support/reborn_parity_qa/`). +# 2. Partition guard — every tests/reborn_*.rs bin that references one of +# the 6 parity/QA modules (binary_e2e, model_replay, qa_trace, +# qa_scenarios, delivery, network — see tests/support/reborn_parity_qa/ +# mod.rs) via its `parity_qa_support::` alias must declare the +# `#[path = "support/reborn_parity_qa/mod.rs"]` mount; and no file under +# tests/integration/ may declare that mount (redundant with #1's path +# check, kept as an explicit, separately-named assertion per the mount +# itself rather than the module names). +# 3. Regression guard — tests/support/reborn/ (the pre-restructure location, +# superseded by tests/support/reborn_parity_qa/) must not reappear. +# +# Exits non-zero with one message per violation (all checks run before +# exiting, so a single invocation reports every violation found, not just the +# first). + +set -euo pipefail + +repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +cd "${repo_root}" + +violations=0 + +fail() { + printf 'BOUNDARY VIOLATION: %s\n' "$1" >&2 + violations=$((violations + 1)) +} + +# --------------------------------------------------------------------------- +# 1. Direction guard: tests/integration/ must never reference the parity/QA +# support tree, by module alias or by mount path. +# --------------------------------------------------------------------------- + +if [ -d tests/integration ]; then + mapfile -t direction_hits < <( + grep -rl -E 'parity_qa_support|reborn_parity_qa' tests/integration/ 2>/dev/null | LC_ALL=C sort + ) + if [ "${#direction_hits[@]}" -gt 0 ]; then + fail "tests/integration/ must not depend on the parity/QA support tree, but found references in:" + printf ' %s\n' "${direction_hits[@]}" >&2 + fi +fi + +# --------------------------------------------------------------------------- +# 2. Partition guard: every tests/reborn_*.rs bin that uses one of the 6 +# parity/QA modules must declare the parity_qa_support mount; no file +# under tests/integration/ may declare that mount. +# --------------------------------------------------------------------------- + +parity_qa_modules=(binary_e2e model_replay qa_trace qa_scenarios delivery network) +mount_pattern='#\[path = "support/reborn_parity_qa/mod\.rs"\]' + +mapfile -t root_test_files < <(find tests -maxdepth 1 -type f -name 'reborn_*.rs' | LC_ALL=C sort) + +for file in "${root_test_files[@]}"; do + uses_parity_qa_module=false + for module in "${parity_qa_modules[@]}"; do + if grep -qE "parity_qa_support::${module}\b" "${file}"; then + uses_parity_qa_module=true + break + fi + done + + if [ "${uses_parity_qa_module}" = true ] && ! grep -qE "${mount_pattern}" "${file}"; then + fail "${file} references a parity_qa_support module but does not declare" \ + "the #[path = \"support/reborn_parity_qa/mod.rs\"] mount" + fi +done + +if [ -d tests/integration ]; then + mapfile -t mount_in_integration < <( + grep -rlE "${mount_pattern}" tests/integration/ 2>/dev/null | LC_ALL=C sort + ) + if [ "${#mount_in_integration[@]}" -gt 0 ]; then + fail "tests/integration/ must not declare the parity/QA support mount, but found it in:" + printf ' %s\n' "${mount_in_integration[@]}" >&2 + fi +fi + +# --------------------------------------------------------------------------- +# 3. Regression guard: the pre-restructure tests/support/reborn/ dir must not +# reappear (superseded by tests/support/reborn_parity_qa/). +# --------------------------------------------------------------------------- + +if [ -d tests/support/reborn ]; then + fail "tests/support/reborn/ has reappeared; the parity/QA support tree now lives at tests/support/reborn_parity_qa/" +fi + +# --------------------------------------------------------------------------- +# 4. Stale-reference guard: no file under tests/ may cite the retired +# tests/support/reborn/ path (comments included -- stale pointers mislead +# readers and tools; the live homes are tests/integration/support/ and +# tests/support/reborn_parity_qa/). +# --------------------------------------------------------------------------- + +mapfile -t stale_refs < <( + grep -rl 'tests/support/reborn/' tests/ 2>/dev/null | LC_ALL=C sort +) +if [ "${#stale_refs[@]}" -gt 0 ]; then + fail "stale 'tests/support/reborn/' references (retired path) found in:" + printf ' %s\n' "${stale_refs[@]}" >&2 +fi + +if [ "${violations}" -gt 0 ]; then + printf '\n%s test-suite boundary violation(s) found\n' "${violations}" >&2 + exit 1 +fi + +echo "Reborn test-suite boundaries OK: tests/integration/ <-> tests/support/reborn_parity_qa/ direction holds." +exit 0 diff --git a/scripts/ci/classify-test-scope.sh b/scripts/ci/classify-test-scope.sh index 62a3fa49057..98dc13dbb30 100755 --- a/scripts/ci/classify-test-scope.sh +++ b/scripts/ci/classify-test-scope.sh @@ -66,7 +66,7 @@ is_shared_test_path() { is_reborn_test_path() { local path="$1" case "$path" in - docs/reborn/*|scripts/reborn-e2e-rust.sh|scripts/ci/run-reborn-root-partition.sh|scripts/ci/run-reborn-group-tests.sh|tests/reborn_*|tests/support/reborn/*|tests/fixtures/llm_traces/reborn_qa/*|tests/e2e/scenarios/test_reborn_*) + docs/reborn/*|scripts/reborn-e2e-rust.sh|scripts/ci/run-reborn-root-partition.sh|scripts/ci/run-reborn-group-tests.sh|tests/reborn_*|tests/integration/*|tests/support/reborn_parity_qa/*|tests/fixtures/llm_traces/reborn_qa/*|tests/e2e/scenarios/test_reborn_*) return 0 ;; crates/ironclaw_architecture/*) @@ -84,6 +84,9 @@ is_reborn_test_path() { crates/ironclaw_conversations/*|crates/ironclaw_outbound/*|crates/ironclaw_triggers/*) return 0 ;; + scripts/ci/reborn-coverage-*.sh|scripts/ci/test-reborn-coverage.sh|scripts/ci/check-test-suite-boundaries.sh|scripts/ci/classify-test-scope.sh|scripts/ci/test-classify-test-scope.sh) + return 0 + ;; *) return 1 ;; diff --git a/scripts/ci/reborn-coverage-comment.sh b/scripts/ci/reborn-coverage-comment.sh index 8f4d4468720..2288b512608 100755 --- a/scripts/ci/reborn-coverage-comment.sh +++ b/scripts/ci/reborn-coverage-comment.sh @@ -17,16 +17,22 @@ # informational too — the "target: 0" is the roadmap goal, not a check that # can fail. # -# Usage: reborn-coverage-comment.sh +# Usage: reborn-coverage-comment.sh # # Requires env: GH_TOKEN (for gh), GITHUB_REPOSITORY, PR_NUMBER. set -euo pipefail -json_path="${1:?usage: reborn-coverage-comment.sh }" +lcov_path="${1:?usage: reborn-coverage-comment.sh }" +exemptions_path="${2:?usage: reborn-coverage-comment.sh }" -if [ ! -f "${json_path}" ]; then - echo "coverage JSON not found: ${json_path}" >&2 +if [ ! -f "${lcov_path}" ]; then + echo "coverage lcov file not found: ${lcov_path}" >&2 + exit 1 +fi + +if [ ! -f "${exemptions_path}" ]; then + echo "coverage exemptions manifest not found: ${exemptions_path}" >&2 exit 1 fi @@ -40,9 +46,9 @@ summary_sh="${script_dir}/reborn-coverage-summary.sh" marker='' # Reuse the canonical summary renderer for the body and the breadth holes — no -# duplicated %/table/aggregation jq lives here. -summary_body="$("${summary_sh}" "${json_path}")" -mapfile -t zero_crates < <("${summary_sh}" --zero-crates "${json_path}") +# duplicated %/table/aggregation logic lives here. +summary_body="$("${summary_sh}" "${lcov_path}" "${exemptions_path}")" +mapfile -t zero_crates < <("${summary_sh}" --zero-crates "${lcov_path}" "${exemptions_path}") callout="" if [ "${#zero_crates[@]}" -gt 0 ]; then diff --git a/scripts/ci/reborn-coverage-int-tier-tests.sh b/scripts/ci/reborn-coverage-int-tier-tests.sh index 55d161e2ca0..f32b4091520 100755 --- a/scripts/ci/reborn-coverage-int-tier-tests.sh +++ b/scripts/ci/reborn-coverage-int-tier-tests.sh @@ -3,12 +3,17 @@ # Print the cargo `--test ` arguments for the Reborn in-process # integration-tier test binaries, one per line. # -# Integration-tier (task T0-COV) is the set of in-process suites: -# - tests/reborn_integration_*.rs (single-file root test binaries) -# - tests/reborn_group_*/ ([[test]] binaries; name == dir name) +# Integration-tier (task T0-COV) is the set of in-process suites under +# tests/integration/ (post-restructure home of the roadmap integration suite; +# see docs/superpowers/specs/2026-06-26-reborn-integration-test-framework-design.md): +# - tests/integration/.rs (flat [[test]] binaries; Cargo `name` is +# reborn_integration_) +# - tests/integration/group_/ ([[test]] binaries; Cargo `name` is +# reborn_group_) # # Discovery is dynamic so coverage automatically picks up new int-tier suites -# as they land (mirrors scripts/ci/run-reborn-root-partition.sh). +# as they land (mirrors scripts/ci/run-reborn-group-tests.sh's dir->name +# rewrite and scripts/ci/run-reborn-root-partition.sh's overall shape). set -euo pipefail @@ -17,10 +22,11 @@ cd "${repo_root}" mapfile -t names < <( { - find tests -maxdepth 1 -type f -name 'reborn_integration_*.rs' \ - | sed -E 's#^tests/##; s#\.rs$##' - find tests -maxdepth 1 -type d -name 'reborn_group_*' \ - | sed -E 's#^tests/##' + find tests/integration -maxdepth 1 -type f -name '*.rs' \ + | sed -E 's#^tests/integration/#reborn_integration_#; s#\.rs$##' + find tests/integration -mindepth 1 -maxdepth 1 -type d -name 'group_*' \ + -exec sh -c 'test -f "$1/main.rs"' _ {} ';' -print \ + | sed -E 's#^tests/integration/group_#reborn_group_#' } | LC_ALL=C sort -u ) diff --git a/scripts/ci/reborn-coverage-lane-run.sh b/scripts/ci/reborn-coverage-lane-run.sh new file mode 100755 index 00000000000..243bbf06d06 --- /dev/null +++ b/scripts/ci/reborn-coverage-lane-run.sh @@ -0,0 +1,129 @@ +#!/usr/bin/env bash +# +# Run one instrumented Reborn integration-tier coverage lane and produce that +# lane's lcov tracefile. +# +# The 34 tests/integration/ suites (see reborn-coverage-int-tier-tests.sh for +# the canonical enumeration) are split across 5 lanes in +# .github/workflows/reborn-tests.yml's `reborn-integration-coverage` matrix +# job: 4 modulo-partitions of the 27 flat `reborn_integration_*` suites, plus +# one dedicated lane for all 7 `reborn_group_*` suites. +# +# This script is the SINGLE execution of the 34 int-tier suites for pass/fail +# purposes too: `cargo llvm-cov ... test` has the same pass/fail semantics as +# `cargo test`, so there is no separate uninstrumented run of the 27 flat +# suites. (The 7 group suites also still have their own uninstrumented +# `reborn-group-tests` job via run-reborn-group-tests.sh, which stays as the +# fast low-contention pass/fail signal for that suite; this lane additionally +# runs them once more, instrumented, for coverage.) +# +# All of this lane's assigned suites run in ONE `cargo llvm-cov ... test` +# invocation, with one repeated `--test ` per suite, `--workspace` so +# the report covers every linked workspace crate (not just the root +# package), and `--lcov --output-path` attached directly to that same +# invocation. This mirrors the retired reborn-coverage.yml workflow's +# working `cargo llvm-cov --workspace "${test_args[@]}" --json ...` shape — +# deliberately NOT split into a `--no-report test` pass followed by a +# separate `cargo llvm-cov report` call, because the standalone `report` +# subcommand has no `--workspace`/`-p` flag of its own (confirmed via `cargo +# llvm-cov report --help`) and empirically defaults to reporting only the +# current/root package, silently dropping every crates/ironclaw_* file. The +# combined single-invocation form is the only one observed to include the +# other workspace crates' coverage. +# +# Reuses reborn-coverage-int-tier-tests.sh as the single source of truth for +# suite discovery/naming (the tests/integration/ -> reborn_integration_*/ +# reborn_group_* rewrite rules), so this script never re-derives that mapping. +# +# Modes (REBORN_COV_LANE_MODE): +# flat-partition Modulo-partitions the 27 reborn_integration_* suites +# across REBORN_COV_LANE_PARTITIONS lanes; REBORN_COV_LANE_INDEX +# (0-based) selects this lane's slice — mirrors +# scripts/ci/run-reborn-root-partition.sh's partitioning. +# group Runs all reborn_group_* suites (7 total) — the one +# dedicated group coverage lane. +# +# Usage: REBORN_COV_LANE_MODE=... [other env] reborn-coverage-lane-run.sh + +set -euo pipefail + +output_lcov="${1:?usage: reborn-coverage-lane-run.sh }" + +script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +mode="${REBORN_COV_LANE_MODE:?REBORN_COV_LANE_MODE must be set to 'flat-partition' or 'group'}" +test_timeout="${REBORN_COV_LANE_TEST_TIMEOUT:-45m}" + +# reborn-coverage-int-tier-tests.sh prints alternating "--test"/"" +# lines; keep only the name lines (every 2nd line, portable awk — no GNU-only +# `sed -n 2~2p`). +mapfile -t all_names < <("${script_dir}/reborn-coverage-int-tier-tests.sh" | awk 'NR % 2 == 0') + +if [ "${#all_names[@]}" -eq 0 ]; then + echo "No Reborn integration-tier test binaries discovered" >&2 + exit 1 +fi + +selected_names=() + +case "${mode}" in + flat-partition) + partition_count="${REBORN_COV_LANE_PARTITIONS:?REBORN_COV_LANE_PARTITIONS must be set for flat-partition mode}" + partition_index="${REBORN_COV_LANE_INDEX:?REBORN_COV_LANE_INDEX must be set for flat-partition mode}" + + if ! [[ "${partition_count}" =~ ^[0-9]+$ ]] || [ "${partition_count}" -lt 1 ]; then + echo "REBORN_COV_LANE_PARTITIONS must be a positive integer; got '${partition_count}'" >&2 + exit 1 + fi + partition_count_int=$((10#${partition_count})) + + if ! [[ "${partition_index}" =~ ^[0-9]+$ ]]; then + echo "REBORN_COV_LANE_INDEX must be an integer in [0, ${partition_count_int}); got '${partition_index}'" >&2 + exit 1 + fi + partition_index_int=$((10#${partition_index})) + + if [ "${partition_index_int}" -ge "${partition_count_int}" ]; then + echo "REBORN_COV_LANE_INDEX must be an integer in [0, ${partition_count}); got '${partition_index}'" >&2 + exit 1 + fi + + mapfile -t flat_names < <(printf '%s\n' "${all_names[@]}" | grep '^reborn_integration_' | LC_ALL=C sort) + + for index in "${!flat_names[@]}"; do + if (( index % partition_count_int != partition_index_int )); then + continue + fi + selected_names+=("${flat_names[$index]}") + done + ;; + group) + mapfile -t selected_names < <(printf '%s\n' "${all_names[@]}" | grep '^reborn_group_' | LC_ALL=C sort) + ;; + *) + echo "Unknown REBORN_COV_LANE_MODE: ${mode} (expected 'flat-partition' or 'group')" >&2 + exit 1 + ;; +esac + +if [ "${#selected_names[@]}" -eq 0 ]; then + # Empty partitions are valid when the matrix has more partitions than tests + # or when the sorted test list leaves a sparse tail for this partition + # (mirrors run-reborn-root-partition.sh). Write an empty tracefile so the + # caller's `cargo llvm-cov report`-less contract (this script always + # produces output_lcov) holds even in the empty case. + echo "No Reborn integration-tier suites assigned to this coverage lane (mode=${mode}); passing by design" + : > "${output_lcov}" + exit 0 +fi + +test_args=() +for test_name in "${selected_names[@]}"; do + test_args+=(--test "${test_name}") +done + +echo "::group::cargo llvm-cov --workspace --features libsql test ${test_args[*]}" +timeout --signal=INT --kill-after=30s "${test_timeout}" \ + cargo llvm-cov --workspace --features libsql test "${test_args[@]}" \ + --lcov --output-path "${output_lcov}" \ + -- --nocapture +echo "::endgroup::" diff --git a/scripts/ci/reborn-coverage-merge-lcov.sh b/scripts/ci/reborn-coverage-merge-lcov.sh new file mode 100755 index 00000000000..902049f31c7 --- /dev/null +++ b/scripts/ci/reborn-coverage-merge-lcov.sh @@ -0,0 +1,105 @@ +#!/usr/bin/env bash +# +# Merge N per-lane lcov tracefiles from the Reborn integration-tier coverage +# lanes (.github/workflows/reborn-tests.yml's `reborn-integration-coverage` +# matrix — 4 flat partitions + 1 group lane, each producing its lcov via one +# combined `cargo llvm-cov ... test --lcov` invocation) into one deterministic tracefile, filtered to `crates/ironclaw_*` +# source files. +# +# Every lane instruments the SAME workspace build, so the same source file can +# appear in more than one lane's tracefile with different per-line hit counts +# (each lane only exercised its own subset of test binaries against that +# shared file). A naive concatenation would leave duplicate `SF:` blocks for +# that file instead of one true picture of "how many of the 5 lanes hit this +# line" — so this script sums DA: hit counts per (file, line) across all +# inputs and recomputes LF/LH from the merged counts, rather than trusting any +# single lane's LF/LH. +# +# Deliberately a small Python merger, not the `lcov` CLI (`lcov -a ... -o`): +# avoids an extra apt-get dependency in the coverage-report job (see +# scripts/ci/install-ci-apt-packages.sh for that pattern, used elsewhere) and +# keeps this fully testable locally without installing anything. +# +# Usage: +# reborn-coverage-merge-lcov.sh [input2.lcov ...] + +set -euo pipefail + +if [ "$#" -lt 2 ]; then + echo "usage: $0 [input2.lcov ...]" >&2 + exit 2 +fi + +output_path="$1" +shift + +for input_path in "$@"; do + if [ ! -f "${input_path}" ]; then + echo "input lcov file not found: ${input_path}" >&2 + exit 1 + fi +done + +python3 - "${output_path}" "$@" <<'PY' +import re +import sys + +output_path = sys.argv[1] +input_paths = sys.argv[2:] + +# filename -> { line_number: hit_count } +files: dict[str, dict[int, int]] = {} + +# Only source files under a crates/ironclaw_* directory are kept — this is +# the "all 71 crates" scope for the Reborn integration-tier coverage report, +# a superset of the historical Reborn-family-only allowlist (the int-tier +# suites now exercise the whole workspace closure, not just the Reborn crate +# families). +crate_re = re.compile(r"(?:^|/)crates/(ironclaw_[A-Za-z0-9_]+)/") + +for input_path in input_paths: + current_file = None + keep_current = False + with open(input_path, "r", encoding="utf-8") as fh: + for raw_line in fh: + line = raw_line.rstrip("\n") + if line.startswith("SF:"): + current_file = line[len("SF:"):] + keep_current = bool(crate_re.search(current_file)) + if keep_current: + files.setdefault(current_file, {}) + elif line.startswith("DA:") and keep_current and current_file is not None: + rest = line[len("DA:"):] + parts = rest.split(",") + line_no = int(parts[0]) + hit_count = int(parts[1]) + bucket = files[current_file] + bucket[line_no] = bucket.get(line_no, 0) + hit_count + elif line == "end_of_record": + current_file = None + keep_current = False + # LF:/LH:/other record kinds (FN, BRDA, ...) are ignored on input — + # this report only needs line coverage, and LF/LH are recomputed + # from the merged DA: counts below so a single lane's summary + # never gets trusted as the merged truth. + +if not files: + # A run with zero matching files is a legitimate (if unlikely) outcome — + # write an empty tracefile rather than erroring, mirroring how the + # downstream summary renderer treats "no data" as its own reportable case + # rather than a script failure. + open(output_path, "w", encoding="utf-8").close() + sys.exit(0) + +with open(output_path, "w", encoding="utf-8") as out: + for filename in sorted(files): + lines = files[filename] + out.write(f"SF:{filename}\n") + for line_no in sorted(lines): + out.write(f"DA:{line_no},{lines[line_no]}\n") + lines_found = len(lines) + lines_hit = sum(1 for count in lines.values() if count > 0) + out.write(f"LF:{lines_found}\n") + out.write(f"LH:{lines_hit}\n") + out.write("end_of_record\n") +PY diff --git a/scripts/ci/reborn-coverage-summary.sh b/scripts/ci/reborn-coverage-summary.sh index 807dd092ee8..00a49da3880 100755 --- a/scripts/ci/reborn-coverage-summary.sh +++ b/scripts/ci/reborn-coverage-summary.sh @@ -1,26 +1,30 @@ #!/usr/bin/env bash # -# Render a Reborn-scoped coverage summary from a cargo-llvm-cov JSON export. +# Render a Reborn-scoped coverage summary from a merged, crate-filtered lcov +# tracefile (see scripts/ci/reborn-coverage-merge-lcov.sh, which merges the 5 +# per-lane `cargo llvm-cov report --lcov` outputs from the +# `reborn-integration-coverage` matrix in .github/workflows/reborn-tests.yml +# and filters to crates/ironclaw_* source files). # -# cargo-llvm-cov instruments the whole workspace build, so a run over the -# Reborn integration-tier test binaries produces coverage for every linked -# crate. This script filters that export down to the Reborn crate families and -# computes an aggregate line-coverage percentage plus a per-crate breakdown -# (the "hole list" — which Reborn crates are least covered). +# The lcov input already covers every crate the int-tier suites link (all +# workspace `ironclaw_*` crates, not a Reborn-only allowlist) — this script's +# job is purely per-crate aggregation, exemption handling, and rendering. # # Usage: -# reborn-coverage-summary.sh +# reborn-coverage-summary.sh # Writes a GitHub-flavored Markdown report to stdout. The caller redirects # it into "$GITHUB_STEP_SUMMARY" (or anywhere else). -# reborn-coverage-summary.sh --zero-crates +# reborn-coverage-summary.sh --zero-crates # Writes, one per line, the Reborn crates that have instrumented lines but -# zero of them covered (the "breadth" holes). Same crate filter and -# aggregation as the report — this script is the single owner of that -# computation so reborn-coverage-comment.sh never has to recompute it. +# zero of them covered (the "breadth" holes). Same crate aggregation as +# the report — this script is the single owner of that computation so +# reborn-coverage-comment.sh never has to recompute it. # -# The Reborn crate families mirror the package allowlist in -# .github/workflows/reborn-tests.yml: ironclaw_reborn*, ironclaw_product*, -# ironclaw_architecture, the v2 channel adapters, and ironclaw_webui_v2*. +# Exemptions (tests/integration/coverage-exemptions.toml, schema documented +# there): each [[exemption]] `module` path is dropped from the per-crate +# covered/total accounting entirely (it neither helps nor hurts a crate's +# percentage) and listed, with its reason + issue, in its own report section. +# An exemptions file with zero entries is valid (nothing is exempted yet). set -euo pipefail @@ -30,76 +34,156 @@ if [ "${1:-}" = "--zero-crates" ]; then shift fi -json_path="${1:?usage: reborn-coverage-summary.sh [--zero-crates] }" +lcov_path="${1:?usage: reborn-coverage-summary.sh [--zero-crates] }" +exemptions_path="${2:?usage: reborn-coverage-summary.sh [--zero-crates] }" -if [ ! -f "${json_path}" ]; then - echo "coverage JSON not found: ${json_path}" >&2 +if [ ! -f "${lcov_path}" ]; then + echo "coverage lcov file not found: ${lcov_path}" >&2 exit 1 fi -# Matches the absolute filenames llvm-cov emits, e.g. -# /work/ironclaw/crates/ironclaw_reborn/src/runtime.rs -# -# Mirrors the Reborn crate allowlist in .github/workflows/reborn-tests.yml: -# prefix-match for the reborn_*/product_*/webui_v2_* families, exact-match -# (no trailing name chars before the "/") for the four single crates. The -# trailing "/" anchors the crate-name boundary in all cases. -reborn_regex='/crates/(ironclaw_(reborn|product|webui_v2)[a-z0-9_]*|ironclaw_architecture|ironclaw_slack_v2_adapter|ironclaw_telegram_v2_adapter|ironclaw_wasm_product_adapters)/' - -jq -r --arg re "${reborn_regex}" --arg mode "${mode}" ' - # Round to 2 decimal places. - def round2: . * 100 | round / 100; - - # All instrumented files that belong to a Reborn crate family. llvm-cov nests - # files under data[]; iterate every dataset (a single `cargo llvm-cov --json` - # emits one, but the export format permits several) so no coverage is dropped. - # The trailing "?"s swallow a missing/empty data or files, routing no-coverage - # runs through the $total == 0 branch below instead of erroring. - [ (.data[]?.files[]?) - | select(.filename | test($re)) - | { crate: (.filename | capture("/crates/(?ironclaw_[a-z0-9_]+)/").c), - covered: .summary.lines.covered, - count: .summary.lines.count } - ] as $files - - | ($files | map(.count) | add // 0) as $total - | ($files | map(.covered) | add // 0) as $hit - | (if $total > 0 then ($hit / $total * 100) else 0 end) as $pct - - | ($files - | group_by(.crate) - | map({ crate: .[0].crate, - covered: (map(.covered) | add // 0), - count: (map(.count) | add // 0) }) - | map(. + { pct: (if .count > 0 then (.covered / .count * 100) else 0 end) }) - | sort_by(.pct) - ) as $byCrate - - # --zero-crates: just the breadth holes (crates instrumented but 0% covered), - # one crate name per line, sorted. The Markdown report is the default mode. - | if $mode == "zero-crates" - then ( $byCrate[] | select(.count > 0 and .covered == 0) | .crate ) - else - "## Reborn integration-tier coverage", - "", - (if $total > 0 - then "**Line coverage (Reborn crates): \($pct | round2)%** — \($hit) / \($total) lines" - else "**No Reborn crate coverage data found** (0 instrumented lines matched the Reborn crate filter)." - end), - "", - (if $total > 0 - then ( "
Per-crate breakdown (\($byCrate | length) crates, lowest-covered first)", - "", - "| Crate | Line % | Covered / Total |", - "|---|---:|---:|", - ( $byCrate[] - | "| `\(.crate)` | \(.pct | round2)% | \(.covered) / \(.count) |" ), - "", - "
", - "", - "_This signal is informational: coverage never gates the PR — not the percentage, not the per-crate holes, not the 0-coverage callout._" - ) - else empty - end) - end -' "${json_path}" +if [ ! -f "${exemptions_path}" ]; then + echo "coverage exemptions manifest not found: ${exemptions_path}" >&2 + exit 1 +fi + +python3 - "${mode}" "${lcov_path}" "${exemptions_path}" <<'PY' +import re +import sys +import tomllib + +mode, lcov_path, exemptions_path = sys.argv[1], sys.argv[2], sys.argv[3] + +crate_re = re.compile(r"(?:^|/)crates/(ironclaw_[A-Za-z0-9_]+)/") + + +def round2(value: float) -> str: + rounded = round(value, 2) + # Match jq's `round` behavior for whole numbers (no trailing ".0"). + if rounded == int(rounded): + return str(int(rounded)) + return str(rounded) + + +# --------------------------------------------------------------------------- +# Parse the exemptions manifest. +# --------------------------------------------------------------------------- + +with open(exemptions_path, "rb") as fh: + manifest = tomllib.load(fh) + +exemptions = manifest.get("exemption", []) +exempt_modules: set[str] = set() +for entry in exemptions: + module = entry.get("module") + if not module: + print(f"malformed exemption entry (missing 'module'): {entry}", file=sys.stderr) + sys.exit(1) + if not entry.get("reason"): + print(f"exemption for '{module}' is missing 'reason'", file=sys.stderr) + sys.exit(1) + if not entry.get("issue"): + print(f"exemption for '{module}' is missing 'issue'", file=sys.stderr) + sys.exit(1) + if not module.startswith("crates/"): + print(f"exemption module path '{module}' must be repo-relative and start with 'crates/'", file=sys.stderr) + sys.exit(1) + exempt_modules.add(module) + +# --------------------------------------------------------------------------- +# Parse the lcov tracefile: per-file (covered, total) from LH:/LF:, per crate. +# --------------------------------------------------------------------------- + +by_crate: dict[str, dict[str, int]] = {} +total = 0 +hit = 0 + +current_file = None +current_covered = None +current_count = None + +with open(lcov_path, "r", encoding="utf-8") as fh: + for raw_line in fh: + line = raw_line.rstrip("\n") + if line.startswith("SF:"): + current_file = line[len("SF:"):] + current_covered = None + current_count = None + elif line.startswith("LF:"): + current_count = int(line[len("LF:"):]) + elif line.startswith("LH:"): + current_covered = int(line[len("LH:"):]) + elif line == "end_of_record": + if current_file is not None and current_covered is not None and current_count is not None: + # Exempted files are skipped entirely: they never enter the + # per-crate or aggregate accounting (neither help nor hurt). + is_exempt = any(current_file.endswith("/" + m) or current_file == m for m in exempt_modules) + if not is_exempt: + match = crate_re.search(current_file) + if match: + crate = match.group(1) + bucket = by_crate.setdefault(crate, {"covered": 0, "count": 0}) + bucket["covered"] += current_covered + bucket["count"] += current_count + total += current_count + hit += current_covered + current_file = None + current_covered = None + current_count = None + +pct = (hit / total * 100) if total > 0 else 0.0 + +rows = [] +for crate, counts in by_crate.items(): + count = counts["count"] + covered = counts["covered"] + crate_pct = (covered / count * 100) if count > 0 else 0.0 + rows.append({"crate": crate, "covered": covered, "count": count, "pct": crate_pct}) +rows.sort(key=lambda r: r["pct"]) + +if mode == "zero-crates": + for row in rows: + if row["count"] > 0 and row["covered"] == 0: + print(row["crate"]) + sys.exit(0) + +# --------------------------- report mode ----------------------------------- + +lines = ["## Reborn integration-tier coverage", ""] + +if total > 0: + lines.append(f"**Line coverage (Reborn crates): {round2(pct)}%** — {hit} / {total} lines") +else: + lines.append("**No Reborn crate coverage data found** (0 instrumented lines matched the Reborn crate filter).") +lines.append("") + +if total > 0: + lines.append(f"
Per-crate breakdown ({len(rows)} crates, lowest-covered first)") + lines.append("") + lines.append("| Crate | Line % | Covered / Total |") + lines.append("|---|---:|---:|") + for row in rows: + lines.append(f"| `{row['crate']}` | {round2(row['pct'])}% | {row['covered']} / {row['count']} |") + lines.append("") + lines.append("
") + lines.append("") + lines.append( + "_This signal is informational: coverage never gates the PR — not the percentage, " + "not the per-crate holes, not the 0-coverage callout._" + ) + +lines.append("") +lines.append(f"
Exemptions ({len(exemptions)} file(s) excluded from the accounting above)") +lines.append("") +if exemptions: + lines.append("| Module | Reason | Issue |") + lines.append("|---|---|---|") + for entry in sorted(exemptions, key=lambda e: e["module"]): + lines.append(f"| `{entry['module']}` | {entry['reason']} | {entry['issue']} |") +else: + lines.append("_No exemptions configured._") +lines.append("") +lines.append("
") + +print("\n".join(lines)) +PY diff --git a/scripts/ci/run-reborn-group-tests.sh b/scripts/ci/run-reborn-group-tests.sh index d6b2c048af7..d8911bba549 100755 --- a/scripts/ci/run-reborn-group-tests.sh +++ b/scripts/ci/run-reborn-group-tests.sh @@ -2,27 +2,29 @@ set -euo pipefail # Reborn shared-persistence "group" suites are subdirectory `[[test]]` binaries -# (tests/reborn_group_*/main.rs). Unlike the single-file reborn_*.rs root tests, -# each spins up one runtime and drives several tenants' shared libsql-backed -# stores across threads, so they run in a dedicated low-contention job instead -# of the modulo-partitioned root runner. `--features libsql` is explicit so the -# group binaries exercise the libsql-backed shared store independently of any -# future change to the default feature set. +# (tests/integration/group_*/main.rs, `[[test]]` name reborn_group_). Unlike +# the single-file reborn_integration_*.rs suites, each spins up one runtime and +# drives several tenants' shared libsql-backed stores across threads, so they +# run in a dedicated low-contention job instead of the modulo-partitioned +# integration runner. `--features libsql` is explicit so the group binaries +# exercise the libsql-backed shared store independently of any future change to +# the default feature set. test_timeout="${REBORN_GROUP_TEST_TIMEOUT:-28m}" -# The `[[test]]` `name` field equals the directory basename, so emit the -# directory path and let the `s#^tests/##` rewrite turn it into the Cargo test -# name (e.g. tests/reborn_group_memory -> reborn_group_memory). The `sh -c` -# predicate skips a half-scaffolded group dir (no main.rs yet) — returning false -# from `-exec` just filters that dir, it does NOT make `find` exit non-zero, so -# no `|| true` is needed; genuine `find` failures stay visible under `set -e`. +# The directory basename is `group_`; the `[[test]]` `name` field is +# `reborn_group_` (see Cargo.toml) — the two differ by the `reborn_` prefix, +# so rewrite it explicitly rather than assuming dir basename == test name (e.g. +# tests/integration/group_memory -> reborn_group_memory). The `sh -c` predicate +# skips a half-scaffolded group dir (no main.rs yet) — returning false from +# `-exec` just filters that dir, it does NOT make `find` exit non-zero, so no +# `|| true` is needed; genuine `find` failures stay visible under `set -e`. # `sh -c` (not a bare `{}/main.rs`) avoids POSIX implementation-defined `{}` # substring substitution so discovery is portable across GNU/BSD find. mapfile -t test_names < <( - find tests -mindepth 1 -maxdepth 1 -type d -name 'reborn_group_*' \ + find tests/integration -mindepth 1 -maxdepth 1 -type d -name 'group_*' \ -exec sh -c 'test -f "$1/main.rs"' _ {} ';' -print \ - | sed -E 's#^tests/##' \ + | sed -E 's#^tests/integration/group_#reborn_group_#' \ | LC_ALL=C sort ) diff --git a/scripts/ci/test-classify-test-scope.sh b/scripts/ci/test-classify-test-scope.sh index 70c96ab28b0..2351fd6b8d6 100755 --- a/scripts/ci/test-classify-test-scope.sh +++ b/scripts/ci/test-classify-test-scope.sh @@ -107,7 +107,7 @@ has_reborn_tests=true" assert_scope \ "reborn root tests and support" \ "tests/reborn_qa_smoke_scenarios_e2e.rs -tests/support/reborn/harness.rs +tests/integration/support/harness/mod.rs tests/e2e/scenarios/test_reborn_gateway_smoke.py" \ "docs_only=false has_core_code=true @@ -249,3 +249,51 @@ assert_scope_no_trailing_newline \ has_core_code=true has_legacy_tests=false has_reborn_tests=true" + +assert_scope \ + "reborn coverage lane-run script" \ + "scripts/ci/reborn-coverage-lane-run.sh" \ + "docs_only=false +has_core_code=true +has_legacy_tests=false +has_reborn_tests=true" + +assert_scope \ + "reborn coverage merge-lcov script" \ + "scripts/ci/reborn-coverage-merge-lcov.sh" \ + "docs_only=false +has_core_code=true +has_legacy_tests=false +has_reborn_tests=true" + +assert_scope \ + "reborn coverage summary script" \ + "scripts/ci/reborn-coverage-summary.sh" \ + "docs_only=false +has_core_code=true +has_legacy_tests=false +has_reborn_tests=true" + +assert_scope \ + "reborn coverage regression suite" \ + "scripts/ci/test-reborn-coverage.sh" \ + "docs_only=false +has_core_code=true +has_legacy_tests=false +has_reborn_tests=true" + +assert_scope \ + "test suite boundaries checker script" \ + "scripts/ci/check-test-suite-boundaries.sh" \ + "docs_only=false +has_core_code=true +has_legacy_tests=false +has_reborn_tests=true" + +assert_scope \ + "test-classify-test-scope script is itself reborn-scoped" \ + "scripts/ci/test-classify-test-scope.sh" \ + "docs_only=false +has_core_code=true +has_legacy_tests=true +has_reborn_tests=true" diff --git a/scripts/ci/test-reborn-coverage.sh b/scripts/ci/test-reborn-coverage.sh index bd29c5c4168..0098700e115 100755 --- a/scripts/ci/test-reborn-coverage.sh +++ b/scripts/ci/test-reborn-coverage.sh @@ -1,6 +1,7 @@ #!/usr/bin/env bash # -# Regression tests for the three Reborn coverage CI helpers: +# Regression tests for the Reborn coverage CI helpers: +# - reborn-coverage-merge-lcov.sh (per-lane lcov merge + crate filter) # - reborn-coverage-summary.sh (report + --zero-crates modes) # - reborn-coverage-comment.sh (sticky PR comment upsert via `gh api`) # - reborn-coverage-int-tier-tests.sh (int-tier suite discovery) @@ -10,9 +11,17 @@ # fixtures in a mktemp dir, and reports PASS/FAIL per case. Unlike that # precedent (which exits on the first failure), this suite runs every case # and prints a final summary, exiting non-zero only if something failed — -# with three scripts and ~20 cases, seeing the full picture in one run beats +# with four scripts and ~25 cases, seeing the full picture in one run beats # stopping at the first mismatch. # +# reborn-coverage-summary.sh and reborn-coverage-comment.sh consume a merged, +# crate-filtered lcov tracefile (scripts/ci/reborn-coverage-merge-lcov.sh) plus +# the exemptions manifest (tests/integration/coverage-exemptions.toml schema), +# not a cargo-llvm-cov JSON export — the coverage-report job in +# .github/workflows/reborn-tests.yml downloads 5 per-lane lcov artifacts, +# merges+filters them, then renders. Fixtures below build lcov tracefiles and +# exemptions TOML by hand rather than shelling out to cargo-llvm-cov. +# # reborn-coverage-comment.sh shells out to `gh api`. It is exercised here # against a fake `gh` (a fixture script placed first on PATH) that emulates # `gh api --paginate --jq ''` by running the given jq filter @@ -21,13 +30,15 @@ # # reborn-coverage-int-tier-tests.sh derives its repo root from its own path # (`$(dirname BASH_SOURCE)/../..`) and `cd`s there, so it cannot simply be -# pointed at a fixture tree via an argument. Each case here copies the real -# script into a temp tree's scripts/ci/ and builds a tests/ subtree next to -# it, so the copy's own repo-root resolution lands on the temp tree. +# pointed at a fixture tree via an argument. Each case copies the real +# script into a temp tree's scripts/ci/ and builds a tests/integration/ +# subtree next to it, so the copy's own repo-root resolution lands on the +# temp tree. set -euo pipefail script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +merge_sh="${script_dir}/reborn-coverage-merge-lcov.sh" summary_sh="${script_dir}/reborn-coverage-summary.sh" comment_sh="${script_dir}/reborn-coverage-comment.sh" int_tier_sh="${script_dir}/reborn-coverage-int-tier-tests.sh" @@ -38,6 +49,13 @@ trap 'rm -rf "${tmp_root}"' EXIT fixtures_dir="${tmp_root}/fixtures" mkdir -p "${fixtures_dir}" +# Empty-but-valid exemptions manifest reused by cases that don't care about +# exemption behavior specifically. +empty_exemptions="${fixtures_dir}/empty-exemptions.toml" +cat > "${empty_exemptions}" <<'TOML' +# No entries. +TOML + PASS_COUNT=0 FAIL_COUNT=0 @@ -131,167 +149,238 @@ capture() { } # --------------------------------------------------------------------------- -# A. reborn-coverage-summary.sh (default report mode) +# M. reborn-coverage-merge-lcov.sh # --------------------------------------------------------------------------- -cat > "${fixtures_dir}/a1_mixed.json" <<'JSON' -{ - "data": [ - { - "files": [ - { "filename": "/work/ironclaw/crates/ironclaw_reborn/src/runtime.rs", "summary": { "lines": { "covered": 80, "count": 100 } } }, - { "filename": "/work/ironclaw/crates/ironclaw_product_workflow/src/lib.rs", "summary": { "lines": { "covered": 50, "count": 50 } } }, - { "filename": "/work/ironclaw/crates/ironclaw_engine/src/lib.rs", "summary": { "lines": { "covered": 10, "count": 10 } } } - ] - } - ] -} -JSON +cat > "${fixtures_dir}/m1_part0.lcov" <<'EOF' +SF:/work/ironclaw/crates/ironclaw_reborn/src/runtime.rs +DA:1,1 +DA:2,0 +DA:3,1 +LF:3 +LH:2 +end_of_record +SF:/work/ironclaw/src/main.rs +DA:1,1 +LF:1 +LH:1 +end_of_record +EOF + +cat > "${fixtures_dir}/m1_part1.lcov" <<'EOF' +SF:/work/ironclaw/crates/ironclaw_reborn/src/runtime.rs +DA:1,0 +DA:2,1 +DA:3,0 +LF:3 +LH:1 +end_of_record +SF:/work/ironclaw/crates/ironclaw_product_workflow/src/lib.rs +DA:1,1 +DA:2,1 +LF:2 +LH:2 +end_of_record +EOF + +capture "${merge_sh}" "${tmp_root}/m1_merged.lcov" "${fixtures_dir}/m1_part0.lcov" "${fixtures_dir}/m1_part1.lcov" +assert_exit_code "M1: merge exits 0" 0 "${CAP_RC}" + +m1_merged_body="$(cat "${tmp_root}/m1_merged.lcov")" +assert_contains "M1: merged output keeps crates/ironclaw_reborn" "${m1_merged_body}" "SF:/work/ironclaw/crates/ironclaw_reborn/src/runtime.rs" +assert_not_contains "M1: merged output drops non-crates/ src/main.rs" "${m1_merged_body}" "src/main.rs" +assert_contains "M1: merged output keeps crates/ironclaw_product_workflow" "${m1_merged_body}" "ironclaw_product_workflow" +assert_contains "M1: per-line DA counts are SUMMED across lanes (line1: 1+0=1)" "${m1_merged_body}" "DA:1,1" +assert_contains "M1: per-line DA counts are SUMMED across lanes (line2: 0+1=1)" "${m1_merged_body}" "DA:2,1" +assert_contains "M1: per-line DA counts are SUMMED across lanes (line3: 1+0=1)" "${m1_merged_body}" "DA:3,1" +assert_contains "M1: LH recomputed from merged counts, not trusted from either lane (all 3 lines now covered)" \ + "${m1_merged_body}" "$(printf 'LF:3\nLH:3')" + +# M2: missing input -> non-zero exit, no output file written over a bad arg. +capture "${merge_sh}" "${tmp_root}/m2_merged.lcov" "${fixtures_dir}/does_not_exist.lcov" +assert_exit_code "M2: merge exits non-zero for missing input" 1 "${CAP_RC}" +assert_contains "M2: merge reports missing input file" "${CAP_ERR}" "input lcov file not found" + +# M3: single input with zero matching (crates/ironclaw_*) files -> empty output, exit 0. +cat > "${fixtures_dir}/m3_no_match.lcov" <<'EOF' +SF:/work/ironclaw/src/other.rs +DA:1,1 +LF:1 +LH:1 +end_of_record +EOF +capture "${merge_sh}" "${tmp_root}/m3_merged.lcov" "${fixtures_dir}/m3_no_match.lcov" +assert_exit_code "M3: merge exits 0 when nothing matches the crate filter" 0 "${CAP_RC}" +assert_eq "M3: merge writes an empty tracefile when nothing matches" "" "$(cat "${tmp_root}/m3_merged.lcov")" -capture "${summary_sh}" "${fixtures_dir}/a1_mixed.json" -assert_exit_code "A1: summary exits 0 for mixed Reborn/non-Reborn fixture" 0 "${CAP_RC}" -assert_not_contains "A1: non-Reborn crate excluded from output" "${CAP_OUT}" "ironclaw_engine" +# --------------------------------------------------------------------------- +# A. reborn-coverage-summary.sh (default report mode) +# --------------------------------------------------------------------------- + +cat > "${fixtures_dir}/a1_mixed.lcov" <<'EOF' +SF:/work/ironclaw/crates/ironclaw_reborn/src/runtime.rs +DA:1,1 +LF:100 +LH:80 +end_of_record +SF:/work/ironclaw/crates/ironclaw_product_workflow/src/lib.rs +LF:50 +LH:50 +end_of_record +EOF + +capture "${summary_sh}" "${fixtures_dir}/a1_mixed.lcov" "${empty_exemptions}" +assert_exit_code "A1: summary exits 0 for a mixed-crate fixture" 0 "${CAP_RC}" assert_contains "A1: aggregate matches hand-computed 86.67% (130/150)" "${CAP_OUT}" \ '**Line coverage (Reborn crates): 86.67%** — 130 / 150 lines' assert_contains "A1: table includes ironclaw_reborn row" "${CAP_OUT}" "| \`ironclaw_reborn\` | 80% | 80 / 100 |" assert_contains "A1: table includes ironclaw_product_workflow row" "${CAP_OUT}" \ "| \`ironclaw_product_workflow\` | 100% | 50 / 50 |" -cat > "${fixtures_dir}/a2_non_reborn_only.json" <<'JSON' -{ - "data": [ - { "files": [ { "filename": "/work/ironclaw/crates/ironclaw_engine/src/lib.rs", "summary": { "lines": { "covered": 10, "count": 10 } } } ] } - ] -} -JSON - -capture "${summary_sh}" "${fixtures_dir}/a2_non_reborn_only.json" -assert_exit_code "A2: summary exits 0 when only non-Reborn crates present" 0 "${CAP_RC}" -assert_contains "A2: prints no-data message when no Reborn files match" "${CAP_OUT}" \ +# A2: no data at all -> exit 0, "no data" message. +: > "${fixtures_dir}/a2_empty.lcov" +capture "${summary_sh}" "${fixtures_dir}/a2_empty.lcov" "${empty_exemptions}" +assert_exit_code "A2: summary exits 0 for an empty lcov file" 0 "${CAP_RC}" +assert_contains "A2: prints no-data message when the lcov file is empty" "${CAP_OUT}" \ "No Reborn crate coverage data found" -printf '{"data":[]}' > "${fixtures_dir}/a3_empty_data.json" -printf '{}' > "${fixtures_dir}/a3_absent_data.json" - -capture "${summary_sh}" "${fixtures_dir}/a3_empty_data.json" -assert_exit_code 'A3: {"data":[]} exits 0' 0 "${CAP_RC}" -assert_contains 'A3: {"data":[]} prints no-data message' "${CAP_OUT}" "No Reborn crate coverage data found" - -capture "${summary_sh}" "${fixtures_dir}/a3_absent_data.json" -assert_exit_code 'A3: {} exits 0' 0 "${CAP_RC}" -assert_contains 'A3: {} prints no-data message' "${CAP_OUT}" "No Reborn crate coverage data found" - -cat > "${fixtures_dir}/a4_multi_dataset.json" <<'JSON' -{ - "data": [ - { "files": [ { "filename": "/work/ironclaw/crates/ironclaw_reborn/src/a.rs", "summary": { "lines": { "covered": 10, "count": 20 } } } ] }, - { "files": [ { "filename": "/work/ironclaw/crates/ironclaw_reborn_cli/src/main.rs", "summary": { "lines": { "covered": 5, "count": 5 } } } ] } - ] -} -JSON - -capture "${summary_sh}" "${fixtures_dir}/a4_multi_dataset.json" -assert_exit_code "A4: multi-dataset summary exits 0" 0 "${CAP_RC}" -assert_contains "A4: aggregate counts files from BOTH data[] entries (15/25)" "${CAP_OUT}" \ - '**Line coverage (Reborn crates): 60%** — 15 / 25 lines' -assert_contains "A4: includes crate from data[0]" "${CAP_OUT}" "\`ironclaw_reborn\`" -assert_contains "A4: includes crate from data[1]" "${CAP_OUT}" "\`ironclaw_reborn_cli\`" - -cat > "${fixtures_dir}/a5_zero_sorted.json" <<'JSON' -{ - "data": [ - { - "files": [ - { "filename": "/work/ironclaw/crates/ironclaw_reborn_zero/src/a.rs", "summary": { "lines": { "covered": 0, "count": 10 } } }, - { "filename": "/work/ironclaw/crates/ironclaw_reborn_half/src/a.rs", "summary": { "lines": { "covered": 5, "count": 10 } } } - ] - } - ] -} -JSON - -capture "${summary_sh}" "${fixtures_dir}/a5_zero_sorted.json" +# A3: missing lcov file -> non-zero exit + not-found error on stderr. +capture "${summary_sh}" "${fixtures_dir}/does_not_exist.lcov" "${empty_exemptions}" +assert_exit_code "A3: summary exits non-zero for missing coverage lcov" 1 "${CAP_RC}" +assert_contains "A3: summary reports missing coverage lcov" "${CAP_ERR}" "coverage lcov file not found" + +# A4: missing exemptions manifest -> non-zero exit + not-found error. +capture "${summary_sh}" "${fixtures_dir}/a1_mixed.lcov" "${fixtures_dir}/does_not_exist_exemptions.toml" +assert_exit_code "A4: summary exits non-zero for missing exemptions manifest" 1 "${CAP_RC}" +assert_contains "A4: summary reports missing exemptions manifest" "${CAP_ERR}" "coverage exemptions manifest not found" + +# A5: zero-covered-crate fixture, sorted lowest-covered-first. +cat > "${fixtures_dir}/a5_zero_sorted.lcov" <<'EOF' +SF:/work/ironclaw/crates/ironclaw_reborn_zero/src/a.rs +LF:10 +LH:0 +end_of_record +SF:/work/ironclaw/crates/ironclaw_reborn_half/src/a.rs +LF:10 +LH:5 +end_of_record +EOF +capture "${summary_sh}" "${fixtures_dir}/a5_zero_sorted.lcov" "${empty_exemptions}" assert_exit_code "A5: zero-covered-crate fixture summary exits 0" 0 "${CAP_RC}" assert_contains "A5: zero-covered crate row shows 0%" "${CAP_OUT}" "| \`ironclaw_reborn_zero\` | 0% | 0 / 10 |" assert_line_before "A5: zero-covered crate sorted to top (lowest-covered first)" "${CAP_OUT}" \ "\`ironclaw_reborn_zero\`" "\`ironclaw_reborn_half\`" -# A6: allowlist boundary — two exact-match single crates, one family-prefix -# crate, and a lookalike (ironclaw_architecture_extra) that must be dropped: -# it is not one of the four exact-match crates and does not start with a -# reborn/product/webui_v2 family prefix. -cat > "${fixtures_dir}/a6_allowlist_boundary.json" <<'JSON' -{ - "data": [ - { - "files": [ - { "filename": "/work/ironclaw/crates/ironclaw_architecture/src/a.rs", "summary": { "lines": { "covered": 5, "count": 10 } } }, - { "filename": "/work/ironclaw/crates/ironclaw_slack_v2_adapter/src/a.rs", "summary": { "lines": { "covered": 3, "count": 10 } } }, - { "filename": "/work/ironclaw/crates/ironclaw_reborn_config/src/a.rs", "summary": { "lines": { "covered": 8, "count": 10 } } }, - { "filename": "/work/ironclaw/crates/ironclaw_architecture_extra/src/a.rs", "summary": { "lines": { "covered": 0, "count": 999 } } } - ] - } - ] -} -JSON +# A6: the crate filter now covers ALL crates/ironclaw_* (all workspace +# crates the int-tier suites link — a superset of the historical Reborn-only +# allowlist), but files outside crates/ entirely (or under a different +# top-level crates-like dir) are still excluded from the aggregate. +cat > "${fixtures_dir}/a6_boundary.lcov" <<'EOF' +SF:/work/ironclaw/crates/ironclaw_engine/src/a.rs +LF:10 +LH:5 +end_of_record +SF:/work/ironclaw/src/main.rs +LF:999 +LH:0 +end_of_record +EOF +capture "${summary_sh}" "${fixtures_dir}/a6_boundary.lcov" "${empty_exemptions}" +assert_exit_code "A6: boundary fixture summary exits 0" 0 "${CAP_RC}" +assert_contains "A6: any crates/ironclaw_* crate is included (not just the old Reborn allowlist)" "${CAP_OUT}" \ + "| \`ironclaw_engine\` | 50% | 5 / 10 |" +assert_not_contains "A6: non-crates/ file excluded from the table" "${CAP_OUT}" "main.rs" +assert_contains "A6: aggregate drops the non-crates file's 999 lines (5/10, not 5/1009)" "${CAP_OUT}" \ + '**Line coverage (Reborn crates): 50%** — 5 / 10 lines' + +# A7: exemptions manifest excludes a file from the accounting entirely and +# lists it in the report's own Exemptions section. +cat > "${fixtures_dir}/a7_exemptions.toml" <<'TOML' +[[exemption]] +module = "crates/ironclaw_engine/src/a.rs" +reason = "generated code, not exercisable by int-tier tests" +issue = "https://github.com/nearai/ironclaw/issues/1" +TOML + +capture "${summary_sh}" "${fixtures_dir}/a6_boundary.lcov" "${fixtures_dir}/a7_exemptions.toml" +assert_exit_code "A7: summary with an exemption exits 0" 0 "${CAP_RC}" +# The exempted crate's file path still legitimately appears in the report's +# own Exemptions section below, so assert against the per-crate table ROW +# specifically (not "the whole output"), matching the A6/A5 row-shaped checks. +assert_not_contains "A7: exempted crate dropped from the per-crate table" "${CAP_OUT}" "| \`ironclaw_engine\` |" +assert_contains "A7: exempted file listed in its own Exemptions section" "${CAP_OUT}" \ + "\`crates/ironclaw_engine/src/a.rs\`" +assert_contains "A7: exemption reason rendered" "${CAP_OUT}" "generated code, not exercisable by int-tier tests" +assert_contains "A7: exemption issue link rendered" "${CAP_OUT}" "https://github.com/nearai/ironclaw/issues/1" +assert_contains "A7: aggregate becomes 'no data' once the only crate is fully exempted" "${CAP_OUT}" \ + "No Reborn crate coverage data found" -capture "${summary_sh}" "${fixtures_dir}/a6_allowlist_boundary.json" -assert_exit_code "A6: allowlist boundary fixture summary exits 0" 0 "${CAP_RC}" -assert_contains "A6: table includes exact-match ironclaw_architecture row" "${CAP_OUT}" \ - "| \`ironclaw_architecture\` | 50% | 5 / 10 |" -assert_contains "A6: table includes exact-match ironclaw_slack_v2_adapter row" "${CAP_OUT}" \ - "| \`ironclaw_slack_v2_adapter\` | 30% | 3 / 10 |" -assert_contains "A6: table includes family-prefix ironclaw_reborn_config row" "${CAP_OUT}" \ - "| \`ironclaw_reborn_config\` | 80% | 8 / 10 |" -assert_not_contains "A6: lookalike ironclaw_architecture_extra excluded from table" "${CAP_OUT}" \ - "ironclaw_architecture_extra" -assert_contains "A6: aggregate drops lookalike's 999 lines (16/30, not 16/1029)" "${CAP_OUT}" \ - '**Line coverage (Reborn crates): 53.33%** — 16 / 30 lines' - -# A7: missing coverage JSON -> non-zero exit + not-found error on stderr. -capture "${summary_sh}" "${fixtures_dir}/does_not_exist.json" -assert_exit_code "A7: summary exits non-zero for missing coverage JSON" 1 "${CAP_RC}" -assert_contains "A7: summary reports missing coverage JSON" "${CAP_ERR}" "coverage JSON not found" +# A8: malformed exemption (missing reason/issue) -> summary refuses to render. +cat > "${fixtures_dir}/a8_malformed_exemptions.toml" <<'TOML' +[[exemption]] +module = "crates/ironclaw_engine/src/a.rs" +reason = "missing issue link" +TOML +capture "${summary_sh}" "${fixtures_dir}/a6_boundary.lcov" "${fixtures_dir}/a8_malformed_exemptions.toml" +assert_exit_code "A8: malformed exemption (no issue) exits non-zero" 1 "${CAP_RC}" +assert_contains "A8: malformed exemption reports the missing issue" "${CAP_ERR}" "missing 'issue'" + +# A9: a non-repo-relative exemption module path (missing the "crates/" prefix) +# would otherwise match every same-named file across all crates via the +# endswith("/" + m) check in the summary's own accounting — reject it at +# parse time instead of silently over-exempting. +cat > "${fixtures_dir}/a9_bare_module_exemptions.toml" <<'TOML' +[[exemption]] +module = "a.rs" +reason = "bare filename, not repo-relative" +issue = "https://github.com/nearai/ironclaw/issues/1" +TOML +capture "${summary_sh}" "${fixtures_dir}/a6_boundary.lcov" "${fixtures_dir}/a9_bare_module_exemptions.toml" +assert_exit_code "A9: non-crates/-prefixed exemption module exits non-zero" 1 "${CAP_RC}" +assert_contains "A9: non-crates/-prefixed exemption module reports the validation error" "${CAP_ERR}" \ + "must be repo-relative and start with 'crates/'" # --------------------------------------------------------------------------- # B. reborn-coverage-summary.sh --zero-crates # --------------------------------------------------------------------------- -cat > "${fixtures_dir}/b1_mixed_zero.json" <<'JSON' -{ - "data": [ - { - "files": [ - { "filename": "/work/ironclaw/crates/ironclaw_reborn_zero_a/src/a.rs", "summary": { "lines": { "covered": 0, "count": 5 } } }, - { "filename": "/work/ironclaw/crates/ironclaw_reborn_zero_b/src/a.rs", "summary": { "lines": { "covered": 0, "count": 3 } } }, - { "filename": "/work/ironclaw/crates/ironclaw_reborn_partial/src/a.rs", "summary": { "lines": { "covered": 4, "count": 10 } } }, - { "filename": "/work/ironclaw/crates/ironclaw_reborn_full/src/a.rs", "summary": { "lines": { "covered": 10, "count": 10 } } } - ] - } - ] -} -JSON - -capture "${summary_sh}" --zero-crates "${fixtures_dir}/b1_mixed_zero.json" +cat > "${fixtures_dir}/b1_mixed_zero.lcov" <<'EOF' +SF:/work/ironclaw/crates/ironclaw_reborn_zero_a/src/a.rs +LF:5 +LH:0 +end_of_record +SF:/work/ironclaw/crates/ironclaw_reborn_zero_b/src/a.rs +LF:3 +LH:0 +end_of_record +SF:/work/ironclaw/crates/ironclaw_reborn_partial/src/a.rs +LF:10 +LH:4 +end_of_record +SF:/work/ironclaw/crates/ironclaw_reborn_full/src/a.rs +LF:10 +LH:10 +end_of_record +EOF + +capture "${summary_sh}" --zero-crates "${fixtures_dir}/b1_mixed_zero.lcov" "${empty_exemptions}" assert_exit_code "B1: --zero-crates exits 0" 0 "${CAP_RC}" assert_eq "B1: --zero-crates emits exactly the 2 zero-covered crate names" \ "$(printf 'ironclaw_reborn_zero_a\nironclaw_reborn_zero_b')" "${CAP_OUT}" -cat > "${fixtures_dir}/b2_all_covered.json" <<'JSON' -{ - "data": [ - { "files": [ { "filename": "/work/ironclaw/crates/ironclaw_reborn_full/src/a.rs", "summary": { "lines": { "covered": 10, "count": 10 } } } ] } - ] -} -JSON - -capture "${summary_sh}" --zero-crates "${fixtures_dir}/b2_all_covered.json" +cat > "${fixtures_dir}/b2_all_covered.lcov" <<'EOF' +SF:/work/ironclaw/crates/ironclaw_reborn_full/src/a.rs +LF:10 +LH:10 +end_of_record +EOF +capture "${summary_sh}" --zero-crates "${fixtures_dir}/b2_all_covered.lcov" "${empty_exemptions}" assert_exit_code "B2: all-covered fixture --zero-crates exits 0" 0 "${CAP_RC}" assert_eq "B2: all-covered fixture --zero-crates emits nothing" "" "${CAP_OUT}" -capture "${summary_sh}" --zero-crates "${fixtures_dir}/a3_empty_data.json" -assert_exit_code "B3: empty data --zero-crates exits 0" 0 "${CAP_RC}" -assert_eq "B3: empty data --zero-crates emits nothing" "" "${CAP_OUT}" +capture "${summary_sh}" --zero-crates "${fixtures_dir}/a2_empty.lcov" "${empty_exemptions}" +assert_exit_code "B3: empty lcov --zero-crates exits 0" 0 "${CAP_RC}" +assert_eq "B3: empty lcov --zero-crates emits nothing" "" "${CAP_OUT}" # --------------------------------------------------------------------------- # C. reborn-coverage-comment.sh (sticky PR comment upsert via a fake `gh`) @@ -376,21 +465,19 @@ echo '{}' GHEOF chmod +x "${gh_bin_dir}/gh" -cat > "${fixtures_dir}/c_basic_coverage.json" <<'JSON' -{ - "data": [ - { "files": [ { "filename": "/work/ironclaw/crates/ironclaw_reborn/src/a.rs", "summary": { "lines": { "covered": 8, "count": 10 } } } ] } - ] -} -JSON +cat > "${fixtures_dir}/c_basic_coverage.lcov" <<'EOF' +SF:/work/ironclaw/crates/ironclaw_reborn/src/a.rs +LF:10 +LH:8 +end_of_record +EOF -cat > "${fixtures_dir}/c_zero_coverage.json" <<'JSON' -{ - "data": [ - { "files": [ { "filename": "/work/ironclaw/crates/ironclaw_reborn_zero/src/a.rs", "summary": { "lines": { "covered": 0, "count": 5 } } } ] } - ] -} -JSON +cat > "${fixtures_dir}/c_zero_coverage.lcov" <<'EOF' +SF:/work/ironclaw/crates/ironclaw_reborn_zero/src/a.rs +LF:5 +LH:0 +end_of_record +EOF cat > "${fixtures_dir}/c1_comments_empty.json" <<'JSON' [] @@ -417,7 +504,7 @@ capture env \ PATH="${gh_bin_dir}:${PATH}" \ FAKE_GH_COMMENTS_JSON="${fixtures_dir}/c1_comments_empty.json" \ FAKE_GH_LOG="${c1_log}" \ - "${comment_sh}" "${fixtures_dir}/c_basic_coverage.json" + "${comment_sh}" "${fixtures_dir}/c_basic_coverage.lcov" "${empty_exemptions}" assert_exit_code "C1: comment script exits 0 (no existing sticky)" 0 "${CAP_RC}" if [ -f "${c1_log}" ]; then @@ -440,7 +527,7 @@ capture env \ PATH="${gh_bin_dir}:${PATH}" \ FAKE_GH_COMMENTS_JSON="${fixtures_dir}/c2_comments_with_sticky.json" \ FAKE_GH_LOG="${c2_log}" \ - "${comment_sh}" "${fixtures_dir}/c_basic_coverage.json" + "${comment_sh}" "${fixtures_dir}/c_basic_coverage.lcov" "${empty_exemptions}" assert_exit_code "C2: comment script exits 0 (existing sticky)" 0 "${CAP_RC}" if [ -f "${c2_log}" ]; then @@ -460,7 +547,7 @@ capture env \ PATH="${gh_bin_dir}:${PATH}" \ FAKE_GH_COMMENTS_JSON="${fixtures_dir}/c1_comments_empty.json" \ FAKE_GH_LOG="${c3_log}" \ - "${comment_sh}" "${fixtures_dir}/c_zero_coverage.json" + "${comment_sh}" "${fixtures_dir}/c_zero_coverage.lcov" "${empty_exemptions}" assert_exit_code "C3: comment script exits 0 (zero-covered crate present)" 0 "${CAP_RC}" if [ -f "${c3_log}" ]; then @@ -478,7 +565,7 @@ fi capture env -u GH_TOKEN \ GITHUB_REPOSITORY="${gh_repo}" \ PR_NUMBER="${gh_pr}" \ - "${comment_sh}" "${fixtures_dir}/c_basic_coverage.json" + "${comment_sh}" "${fixtures_dir}/c_basic_coverage.lcov" "${empty_exemptions}" assert_exit_code "C4: GH_TOKEN unset exits non-zero" 1 "${CAP_RC}" assert_contains "C4: GH_TOKEN unset reports the missing-var guard" "${CAP_ERR}" "GH_TOKEN must be set" @@ -486,7 +573,7 @@ assert_contains "C4: GH_TOKEN unset reports the missing-var guard" "${CAP_ERR}" capture env -u PR_NUMBER \ GH_TOKEN="fake-token" \ GITHUB_REPOSITORY="${gh_repo}" \ - "${comment_sh}" "${fixtures_dir}/c_basic_coverage.json" + "${comment_sh}" "${fixtures_dir}/c_basic_coverage.lcov" "${empty_exemptions}" assert_exit_code "C5: PR_NUMBER unset exits non-zero" 1 "${CAP_RC}" assert_contains "C5: PR_NUMBER unset reports the missing-var guard" "${CAP_ERR}" "PR_NUMBER must be set" @@ -494,11 +581,11 @@ assert_contains "C5: PR_NUMBER unset reports the missing-var guard" "${CAP_ERR}" capture env -u GITHUB_REPOSITORY \ GH_TOKEN="fake-token" \ PR_NUMBER="${gh_pr}" \ - "${comment_sh}" "${fixtures_dir}/c_basic_coverage.json" + "${comment_sh}" "${fixtures_dir}/c_basic_coverage.lcov" "${empty_exemptions}" assert_exit_code "C6: GITHUB_REPOSITORY unset exits non-zero" 1 "${CAP_RC}" assert_contains "C6: GITHUB_REPOSITORY unset reports the missing-var guard" "${CAP_ERR}" "GITHUB_REPOSITORY must be set" -# C7: missing coverage JSON -> the "if [ ! -f ]" guard at the top of +# C7: missing coverage lcov -> the "if [ ! -f ]" guard at the top of # comment.sh fires before GH_TOKEN/GITHUB_REPOSITORY/PR_NUMBER are even # consulted, so no `gh` call — and therefore no mutation — is ever recorded. c7_log="${tmp_root}/c7-gh.log" @@ -509,26 +596,16 @@ capture env \ PATH="${gh_bin_dir}:${PATH}" \ FAKE_GH_COMMENTS_JSON="${fixtures_dir}/c1_comments_empty.json" \ FAKE_GH_LOG="${c7_log}" \ - "${comment_sh}" "${fixtures_dir}/does_not_exist.json" -assert_exit_code "C7: comment script exits non-zero for missing coverage JSON" 1 "${CAP_RC}" -assert_contains "C7: comment script reports missing coverage JSON" "${CAP_ERR}" "coverage JSON not found" + "${comment_sh}" "${fixtures_dir}/does_not_exist.lcov" "${empty_exemptions}" +assert_exit_code "C7: comment script exits non-zero for missing coverage lcov" 1 "${CAP_RC}" +assert_contains "C7: comment script reports missing coverage lcov" "${CAP_ERR}" "coverage lcov file not found" if [ -f "${c7_log}" ]; then report_fail "C7: fake gh did not record a mutation (guard fires before gh use)" else report_pass "C7: fake gh did not record a mutation (guard fires before gh use)" fi -# C8: existing-but-malformed coverage JSON aborts before any gh mutation. -# -# Distinct path from C7: this fixture file exists, so it passes comment.sh's -# "[ -f ]" guard and proceeds to call reborn-coverage-summary.sh to render -# the body — and that render fails on jq's parse error under `set -e`, -# before any POST/PATCH is ever issued. Pins the "render-before-mutate" -# ordering. -cat > "${fixtures_dir}/c8_malformed.json" <<'JSON' -{ this is not valid json -JSON - +# C8: missing exemptions manifest -> same "fires before gh use" guard shape. c8_log="${tmp_root}/c8-gh.log" capture env \ GH_TOKEN="fake-token" \ @@ -537,17 +614,13 @@ capture env \ PATH="${gh_bin_dir}:${PATH}" \ FAKE_GH_COMMENTS_JSON="${fixtures_dir}/c1_comments_empty.json" \ FAKE_GH_LOG="${c8_log}" \ - "${comment_sh}" "${fixtures_dir}/c8_malformed.json" -# Non-zero, not an exact code: jq's parse-error exit differs across versions. -if [ "${CAP_RC}" -ne 0 ]; then - report_pass "C8: comment script exits non-zero on malformed coverage JSON" -else - report_fail "C8: comment script exits non-zero on malformed coverage JSON (got 0)" -fi + "${comment_sh}" "${fixtures_dir}/c_basic_coverage.lcov" "${fixtures_dir}/does_not_exist_exemptions.toml" +assert_exit_code "C8: comment script exits non-zero for missing exemptions manifest" 1 "${CAP_RC}" +assert_contains "C8: comment script reports missing exemptions manifest" "${CAP_ERR}" "coverage exemptions manifest not found" if [ -f "${c8_log}" ]; then - report_fail "C8: fake gh did not record a mutation (render fails before gh use)" + report_fail "C8: fake gh did not record a mutation (guard fires before gh use)" else - report_pass "C8: fake gh did not record a mutation (render fails before gh use)" + report_pass "C8: fake gh did not record a mutation (guard fires before gh use)" fi # --------------------------------------------------------------------------- @@ -556,54 +629,80 @@ fi # # The script derives its repo root from its own path and `cd`s there, so # each case copies it into a fresh temp tree's scripts/ci/ and builds a -# tests/ subtree alongside it, then invokes the copy. +# tests/integration/ subtree alongside it, then invokes the copy. setup_int_tier_case() { local case_dir="$1" - mkdir -p "${case_dir}/scripts/ci" "${case_dir}/tests" + mkdir -p "${case_dir}/scripts/ci" "${case_dir}/tests/integration" cp "${int_tier_sh}" "${case_dir}/scripts/ci/reborn-coverage-int-tier-tests.sh" chmod +x "${case_dir}/scripts/ci/reborn-coverage-int-tier-tests.sh" } -# D1: empty tests/ -> non-zero exit + discovery error. +# D1: empty tests/integration/ -> non-zero exit + discovery error. d1="${tmp_root}/d1" setup_int_tier_case "${d1}" capture "${d1}/scripts/ci/reborn-coverage-int-tier-tests.sh" -assert_exit_code "D1: empty tests/ exits non-zero" 1 "${CAP_RC}" -assert_contains "D1: empty tests/ prints the discovery error" "${CAP_ERR}" \ +assert_exit_code "D1: empty tests/integration/ exits non-zero" 1 "${CAP_RC}" +assert_contains "D1: empty tests/integration/ prints the discovery error" "${CAP_ERR}" \ "No Reborn integration-tier test binaries discovered" -# D2: one tests/reborn_integration_foo.rs -> --test / reborn_integration_foo. +# D2: one tests/integration/foo.rs -> --test / reborn_integration_foo. d2="${tmp_root}/d2" setup_int_tier_case "${d2}" -: > "${d2}/tests/reborn_integration_foo.rs" +: > "${d2}/tests/integration/foo.rs" capture "${d2}/scripts/ci/reborn-coverage-int-tier-tests.sh" -assert_exit_code "D2: single integration file exits 0" 0 "${CAP_RC}" -assert_eq "D2: single integration file emits its --test pair" \ +assert_exit_code "D2: single flat integration file exits 0" 0 "${CAP_RC}" +assert_eq "D2: single flat integration file emits its --test pair" \ "$(printf -- '--test\nreborn_integration_foo')" "${CAP_OUT}" -# D3: one tests/reborn_group_bar/ -> --test / reborn_group_bar. +# D3: one tests/integration/group_bar/main.rs -> --test / reborn_group_bar. d3="${tmp_root}/d3" setup_int_tier_case "${d3}" -mkdir -p "${d3}/tests/reborn_group_bar" +mkdir -p "${d3}/tests/integration/group_bar" +: > "${d3}/tests/integration/group_bar/main.rs" capture "${d3}/scripts/ci/reborn-coverage-int-tier-tests.sh" assert_exit_code "D3: single group dir exits 0" 0 "${CAP_RC}" -assert_eq "D3: single group dir emits its --test pair" \ +assert_eq "D3: single group dir emits its --test pair, dir->name rewrite applied" \ + "$(printf -- '--test\nreborn_group_bar')" "${CAP_OUT}" + +# D3b: a half-scaffolded group dir (no main.rs yet) is skipped, not errored. +d3b="${tmp_root}/d3b" +setup_int_tier_case "${d3b}" +mkdir -p "${d3b}/tests/integration/group_bar" "${d3b}/tests/integration/group_incomplete" +: > "${d3b}/tests/integration/group_bar/main.rs" +capture "${d3b}/scripts/ci/reborn-coverage-int-tier-tests.sh" +assert_exit_code "D3b: half-scaffolded group dir does not error the whole discovery" 0 "${CAP_RC}" +assert_eq "D3b: half-scaffolded group dir (no main.rs) is skipped" \ "$(printf -- '--test\nreborn_group_bar')" "${CAP_OUT}" # D4: multiple files + dirs, created out of alphabetical order -> sorted, # deduped output. Group dirs ('g') sort before integration files ('i'). d4="${tmp_root}/d4" setup_int_tier_case "${d4}" -: > "${d4}/tests/reborn_integration_zeta.rs" -: > "${d4}/tests/reborn_integration_alpha.rs" -mkdir -p "${d4}/tests/reborn_group_omega" "${d4}/tests/reborn_group_beta" +: > "${d4}/tests/integration/zeta.rs" +: > "${d4}/tests/integration/alpha.rs" +mkdir -p "${d4}/tests/integration/group_omega" "${d4}/tests/integration/group_beta" +: > "${d4}/tests/integration/group_omega/main.rs" +: > "${d4}/tests/integration/group_beta/main.rs" capture "${d4}/scripts/ci/reborn-coverage-int-tier-tests.sh" assert_exit_code "D4: multiple suites exits 0" 0 "${CAP_RC}" assert_eq "D4: multiple suites sorted+deduped in expected order" \ "$(printf -- '--test\nreborn_group_beta\n--test\nreborn_group_omega\n--test\nreborn_integration_alpha\n--test\nreborn_integration_zeta')" \ "${CAP_OUT}" +# D5: a `support/` subdirectory alongside the flat files must not be +# mistaken for a suite (no main.rs, and doesn't match the `group_*` name +# pattern either) — mirrors the real tests/integration/support/ harness tree. +d5="${tmp_root}/d5" +setup_int_tier_case "${d5}" +: > "${d5}/tests/integration/only.rs" +mkdir -p "${d5}/tests/integration/support" +: > "${d5}/tests/integration/support/mod.rs" +capture "${d5}/scripts/ci/reborn-coverage-int-tier-tests.sh" +assert_exit_code "D5: support/ dir alongside flat suites exits 0" 0 "${CAP_RC}" +assert_eq "D5: support/ dir is not discovered as a suite" \ + "$(printf -- '--test\nreborn_integration_only')" "${CAP_OUT}" + # --------------------------------------------------------------------------- # Summary # --------------------------------------------------------------------------- diff --git a/tests/support/reborn/CLAUDE.md b/tests/integration/CLAUDE.md similarity index 91% rename from tests/support/reborn/CLAUDE.md rename to tests/integration/CLAUDE.md index d94692fd81b..cc88c27ca02 100644 --- a/tests/support/reborn/CLAUDE.md +++ b/tests/integration/CLAUDE.md @@ -96,16 +96,33 @@ So a two-turn thread where both turns raise and resolve a gate needs 4 entries (`submit_turn_until_blocked` / `approve_gate` / `deny_gate` / `enable_auto_approve`), and the `pub(super)` capture accessors (`captured_egress_requests` / `captured_capability_results` / `captured_system_prompts`) the assertion file reads. -- `harness_mcp.rs` — the mock-MCP scaffolding extracted from `harness.rs`: +- `harness_mcp.rs` — the mock-MCP scaffolding extracted from the harness: `LoopbackMcpRuntimeHttpEgress` (the real-HTTP loopback egress), the `LoopbackMcpRuntime` type alias + `build_loopback_mcp_runtime` factory, `mock_mcp_extension_package`, `local_dev_host_runtime_with_registry_egress_and_mcp`, and the MCP trust/network policies. `HostRuntimeCapabilityHarness::mock_mcp_tools` - stays in `harness.rs` (it is a full `Self {..}` constructor co-located with its - sibling constructors and would otherwise force every private field of the central - harness struct to widen); it delegates the MCP wiring to the `pub(super)` factories - in `harness_mcp.rs`. `harness.rs` remains large (a further `harness_auth.rs` - split is tracked in the coverage roadmap). + lives in `harness/profiles/mock_mcp.rs` (part of the `ToolsProfile` split below); + it delegates the MCP wiring to the `pub(super)` factories in `harness_mcp.rs`. +- `harness/` — the `HostRuntimeCapabilityHarness` split (was a single `harness.rs`): + `harness/mod.rs` (the central struct, `new_with_options`, and constructors), + `harness/assembly.rs` (composition/wiring helpers), `harness/options.rs` + (`HostRuntimeHarnessOptions`, `ToolsProfile` + `build()`), `harness/recorder.rs` + (`HarnessCapabilityRecorder`, `RecordedCapabilityResult`), and + `harness/profiles/.rs` — one file per capability domain (`attachment`, + `coding_read`, `core_builtin`, `extension`, `file`, `github`, `mock_mcp`, + `outbound`, `process`, `profile`, `project`, `qa_smoke`, `skill`, + `trace_commons`, `trigger`, `web_access`) — each returning a `ToolsProfile` via + a constructor like `profiles::file::file_tools_requiring_approval()` or + `profiles::core_builtin::core_builtin_tools(CoreBuiltinOptions)`. +- `doubles/` — one file per production-port test double substituted into the + harness (`RecordingTestCapabilityPort`, `RecordingHostRuntime`, + `RecordingRuntimeHttpEgress`, `RecordingNetworkHttpEgress`, + `RecordingApprovalRequestStore`, `RecordingCapabilityResultWriter`, + `GithubHarnessAuthorizer`, `StaticSecretStore`, + `StaticCapabilitySurfaceProfileResolver`, + `FixedRuntimeCredentialAccountResolver`, `EmptyIdentityContextSource`, + `HarnessCapabilityPortFactory`, `HostRuntimeHarnessCapabilityPortFactory`), + re-exported from `doubles/mod.rs`. - `group_constructors.rs` — the per-capability `RebornIntegrationGroup` / `RebornIntegrationGroupBuilder` preset constructors (`live_approvals`, `builtin_tools`, `extension_lifecycle`, `skill_management_tools`, etc.), a @@ -127,9 +144,19 @@ So a two-turn thread where both turns raise and resolve a gate needs 4 entries model-prompt assertion `assert_system_prompt_contains` (reads the scripted `TraceLlm`'s captured requests via `captured_system_prompts`, not the egress log). -- Tests live as flat `tests/reborn_*.rs` (Cargo requires top-level test files). - -Module paths: each `tests/reborn_*.rs` declares both `#[path = "support/reborn/mod.rs"] mod reborn_support;` and `mod support;`, then `use reborn_support::builder::RebornIntegrationHarness;` / `use reborn_support::reply::RebornScriptedReply;`. Inside the support tree, siblings reference each other via `super::` and `trace_llm` via `crate::support::trace_llm` (there is no `crate::support::reborn` path). Copy the includes from `tests/reborn_integration_greeting.rs`. +- Tests live as flat `tests/integration/.rs` bins (Cargo requires + top-level-per-bin test files), each registered as its own `[[test]]` in the + workspace `Cargo.toml` with `name = "reborn_integration_"`. + +Module paths: each `tests/integration/.rs` declares both +`#[path = "support/mod.rs"] mod reborn_support;` and +`#[path = "../support/mod.rs"] mod support;`, then +`use reborn_support::builder::RebornIntegrationHarness;` / +`use reborn_support::reply::RebornScriptedReply;`. Inside the support tree, +siblings reference each other via `super::` and `trace_llm` via +`crate::support::trace_llm` (the top-level `tests/support/` tree, mounted +separately from `tests/integration/support/` via the second `#[path]`). Copy +the includes from `tests/integration/greeting.rs`. Design rationale: see git history. @@ -195,7 +222,7 @@ of `assert_system_prompt_contains`, which reads only System-role model requests); `assert_conversation_history_role_contains(kind, needle)` restricts it to one `MessageKind`. Contracts (class discrimination, fail-loud decode) match the full-history siblings — see their doc-comments. Demo + regression: -`tests/reborn_integration_http_matcher.rs::multi_turn_baseline_sliced_history_assertions`. +`tests/integration/http_matcher.rs::multi_turn_baseline_sliced_history_assertions`. ### Keyed HTTP responses @@ -339,7 +366,7 @@ On a harness built from a `live_approvals` group: ### Test-support crate accessors -`HostRuntimeCapabilityHarness` (in `harness.rs`, gated on `#[cfg(feature = "test-support")]`) exposes: +`HostRuntimeCapabilityHarness` (in `harness/mod.rs`, gated on `#[cfg(feature = "test-support")]`) exposes: - `extension_installation_store_for_test()` — returns the `Option>` wired into the local-dev extension management port; mirrors the production installation store for test read-back assertions. Returns `None` when the local runtime has no extension management wired. @@ -371,14 +398,14 @@ the shared runtime** — either: model can exercise. A scenario that submits + asserts in one thread belongs in a flat -`tests/reborn_integration_*.rs` test as always. +`tests/integration/.rs` test as always. ### Group test binary layout -Group tests live in subdirectories under `tests/`: +Group tests live in subdirectories under `tests/integration/`: ``` -tests/reborn_group_approvals/ +tests/integration/group_approvals/ main.rs # one #[tokio::test], drives scenarios in order scenario_gate_then_resolve.rs # pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> scenario_approve_always_persists.rs @@ -390,14 +417,14 @@ Cargo discovers multi-file integration test binaries via `[[test]]` entries in ```toml [[test]] name = "reborn_group_approvals" -path = "tests/reborn_group_approvals/main.rs" +path = "tests/integration/group_approvals/main.rs" ``` ### `main.rs` boilerplate (required) ```rust -#[allow(dead_code)] #[path = "../support/reborn/mod.rs"] mod reborn_support; -#[allow(dead_code)] #[path = "../support/mod.rs"] mod support; +#[allow(dead_code)] #[path = "../support/mod.rs"] mod reborn_support; +#[allow(dead_code)] #[path = "../../support/mod.rs"] mod support; mod scenario_gate_then_resolve; mod scenario_approve_always_persists; @@ -470,7 +497,7 @@ On a `live_auth_and_approval()` group thread: - `deny_auth_gate(run_id, &gate_ref)` — works on both auth-gate groups. Multi-turn journey chaining (gate -> resolve -> next turn on ONE -conversation) is pinned by `tests/reborn_group_journeys/`; per-turn resume +conversation) is pinned by `tests/integration/group_journeys/`; per-turn resume idempotency keys are `(run_id, gate_ref)`-scoped so one run can resume through an approval gate and a subsequent auth gate without replay collision. @@ -485,7 +512,7 @@ shared runtime resolves each turn's thread by the run's own owner per-op `/threads` mount), so two actors' threads coexist over one coordinator with their history isolated under separate `/tenants//users//threads` subtrees. Driving test: -`tests/reborn_group_multiuser/`. +`tests/integration/group_multiuser/`. ### Key accessors on `RebornIntegrationGroup` diff --git a/tests/reborn_integration_attach.rs b/tests/integration/attach.rs similarity index 60% rename from tests/reborn_integration_attach.rs rename to tests/integration/attach.rs index bbdf50c47bb..a0cb3a3c88b 100644 --- a/tests/reborn_integration_attach.rs +++ b/tests/integration/attach.rs @@ -1,26 +1,23 @@ -//! C-ATTACH: `attachment_read_port` int-tier coverage (rev-3 Tier-2, A1 audit). +//! C-ATTACH: `attachment_read_port` int-tier coverage. //! //! Production wires `attachment_read_port` from the local-dev workspace -//! filesystem (`ProjectScopedAttachmentReader` over `local_runtime.workspace_filesystem`, +//! filesystem (`ProjectScopedAttachmentReader`, //! `crates/ironclaw_reborn_composition/src/runtime.rs:3328-3334`) so the loop -//! model port can read landed attachment bytes back and the gateway -//! (`crates/ironclaw_reborn/src/model_gateway.rs::convert_messages`) builds a -//! `ContentPart::ImageUrl` multimodal part for a vision-capable model. That seam -//! was never exercised by any Reborn test: `DefaultPlannedRuntimeParts.attachment_read_port` -//! was `None` everywhere, so every image attachment silently degraded to the -//! textual `` pointer regardless of the model's vision capability. +//! model port reads landed attachment bytes back for the gateway +//! (`convert_messages`) to build a `ContentPart::ImageUrl` for vision-capable +//! models. Regression: this port was `None` everywhere pre-fix, so images +//! silently degraded to the textual `` pointer. //! -//! This test wires the read port + the real `InboundAttachmentLander` (same -//! production `ProjectScopedAttachmentLander`) via `RebornIntegrationGroup::attachment_tools()`, -//! lands an image through the real `submit_inbound_with_attachments` production -//! entry point, routes the thread through a vision-pattern model id -//! (`.with_model_override`), and asserts the model-visible request actually -//! carried the image as a `data:` URL content part. +//! Wires the read port + real `InboundAttachmentLander` via +//! `RebornIntegrationGroup::attachment_tools()`, lands an image, routes +//! through a vision-pattern model id, and asserts the model request carried +//! the image as a `data:` URL. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::builder::RebornIntegrationHarness; @@ -61,12 +58,9 @@ async fn landed_image_attachment_reaches_the_model_as_a_multimodal_part() { .expect("landed image bytes reached the model intact as a multimodal content part"); } -/// Negative control: without `.with_model_override()` the thread -/// keeps the harness's default scripted model id, which is not a vision-pattern -/// match. `convert_messages` then drops the image part (text-only fallback), so -/// no multimodal content part should reach the model even though the -/// attachment landed and the read port is wired. Proves the model-override knob -/// (not just the read port) is load-bearing for this assertion. +/// Negative control: without a vision-model override, `convert_messages` +/// drops the image part (text-only fallback) — proves the override knob, not +/// just the read port, is load-bearing. #[tokio::test] async fn non_vision_model_does_not_receive_a_multimodal_image_part() { let group = RebornIntegrationGroup::attachment_tools() @@ -100,13 +94,9 @@ async fn non_vision_model_does_not_receive_a_multimodal_image_part() { ); } -/// Negative control: a plain harness (not built via -/// `RebornIntegrationGroup::attachment_tools()`) has no `InboundAttachmentLander` -/// wired, so `submit_turn_with_image_attachment`'s explicit no-lander guard must -/// fire before any turn is submitted. Proves that fail-fast path stays load-bearing -/// — without this, a regression that removed or weakened the guard would not be -/// caught, since every other test in this file only exercises harnesses with a -/// lander wired. +/// Negative control: a plain harness with no `InboundAttachmentLander` wired +/// must fail fast via the no-lander guard before any turn is submitted — +/// every other test in this file wires a lander. #[tokio::test] async fn submit_with_image_attachment_fails_fast_without_a_lander() { let harness = RebornIntegrationHarness::test_default() @@ -130,15 +120,11 @@ async fn submit_with_image_attachment_fails_fast_without_a_lander() { ); } -/// W4-ATTACH-VARIANTS: a `text/plain` attachment (`AttachmentKind::Document`, -/// per `ironclaw_common::AttachmentKind::from_mime_type`) is landed through the -/// real `InboundAttachmentLander`, its text extracted by -/// `land_inbound_attachments` -> `extract_document_text`, and rendered into the -/// `` block `ironclaw_threads::attachment_context::augment_model_content` -/// appends to the user message — the textual (non-multimodal) attachment path -/// no `RebornIntegrationHarness` test exercised before this file only covered -/// images. Reads the captured model request (not just tool output), proving -/// the extracted text reached the model. +/// W4-ATTACH-VARIANTS: a `text/plain` attachment (`AttachmentKind::Document`) +/// is landed and text-extracted via `land_inbound_attachments`, and its +/// content is verified in the captured model request (not just tool output) +/// — the textual (non-multimodal) attachment path, uncovered by the +/// image-only tests above. #[tokio::test] async fn doc_attachment_reaches_the_model_with_extracted_text() { const MARKER: &str = "ZAFFRE-DOCUMENT-MARKER-771"; @@ -168,19 +154,15 @@ async fn doc_attachment_reaches_the_model_with_extracted_text() { .assert_model_request_contains(MARKER) .await .expect("extracted document text reached the model"); - // `assert_model_request_contains` matches against the JSON-serialized - // captured request, so a literal `"` inside the rendered `` - // XML is escaped to `\"` in that serialization — the needle must match the + // Serialized capture escapes `"` to `\"`; the needle must match the // escaped form. harness .assert_model_request_contains("type=\\\"document\\\"") .await .expect("the attachment block tags the attachment as a document, not an image"); - // Non-vacuity guard: a marker that was never written must be ABSENT, so - // `assert_model_request_contains` is proven to discriminate rather than - // pass unconditionally (e.g. on a captured-request stringification bug - // that always matches). + // Non-vacuity guard: an unwritten marker must be ABSENT, proving the + // assertion discriminates rather than passing unconditionally. if harness .assert_model_request_contains("UNWRITTEN-MARKER-999") .await @@ -190,13 +172,10 @@ async fn doc_attachment_reaches_the_model_with_extracted_text() { } } -/// W4-ATTACH-VARIANTS: two attachments landed on a SINGLE turn both reach the -/// model in one captured request — proves `submit_turn_with_attachments` -/// carries N attachments through the real -/// `DefaultProductWorkflow::submit_inbound_with_attachments` entry point (every -/// prior attachment test in this file submitted exactly one), and that -/// `land_inbound_attachments`/`augment_model_content` render every landed -/// attachment into the `` block, not just the first. +/// W4-ATTACH-VARIANTS: two attachments landed in one turn both reach the +/// model in one captured request — proves N-attachment support through +/// `submit_inbound_with_attachments`, not just the single-attachment path +/// exercised above. #[tokio::test] async fn multiple_attachments_in_one_turn_all_reach_the_model() { const MARKER_ONE: &str = "TOPAZ-MARKER-ALPHA"; @@ -238,11 +217,8 @@ async fn multiple_attachments_in_one_turn_all_reach_the_model() { .assert_model_request_contains(MARKER_TWO) .await .expect("second attachment's extracted text reached the model"); - // Both attachments are indexed distinctly in the rendered block, so both - // ordinal markers must be present — proves two DISTINCT attachment blocks - // were rendered, not one attachment whose body happens to contain both - // marker strings. (See the escaping note above: needles match the - // JSON-serialized, quote-escaped form.) + // Both ordinal markers prove two DISTINCT attachment blocks were + // rendered, not one block whose body contains both strings. harness .assert_model_request_contains("index=\\\"1\\\"") .await diff --git a/tests/reborn_integration_auth_failure.rs b/tests/integration/auth_failure.rs similarity index 67% rename from tests/reborn_integration_auth_failure.rs rename to tests/integration/auth_failure.rs index 0aeb65d6665..1fb24648030 100644 --- a/tests/reborn_integration_auth_failure.rs +++ b/tests/integration/auth_failure.rs @@ -1,39 +1,25 @@ //! Reborn integration-test framework — auth/credential-failure coverage. //! -//! Two arms: +//! (a) `revoked_account_reads_back_revoked` — pure-store `update_status(.., +//! Revoked)` durable read-back; no refresh sweep involved. +//! (b) `invalid_grant_sweep_marks_account_revoked` — a scripted +//! `invalid_grant` 400 during a credential-refresh sweep (`sweep_once` -> +//! `refresh_account` -> `refresh_token` -> scripted egress) marks the account +//! `Revoked`. +//! (negative guard) `normal_sweep_does_not_mark_account_revoked` — same sweep +//! with a `200` egress leaves the account `Configured`, proving (b)'s +//! `Revoked` comes from `invalid_grant`, not the sweep itself. //! -//! **(a) `revoked_account_reads_back_revoked`** — pure-store: create a -//! `Configured` credential account via the standard OAuth connect flow, call -//! `CredentialAccountService::update_status(.., Revoked)`, and assert the -//! durable read-back carries `Revoked`. This arm exercises the status-mutation -//! path without touching the refresh sweep. +//! DEFERRED: live-401 reactive re-auth arm — needs a credentialed capability +//! backend test double, not yet available. //! -//! **(b) `invalid_grant_sweep_marks_account_revoked`** — end-to-end refresh -//! failure: an idle Google OAuth account receives a scripted `invalid_grant` -//! 400 response from the token endpoint during a credential-refresh sweep, and -//! the durable record is subsequently `Revoked`. The sweep path is -//! `sweep_once` → `ProviderBackedCredentialAccountService::refresh_account` → -//! `HostOAuthProviderClient::refresh_token` → scripted HTTP egress. -//! -//! **(negative guard) `normal_sweep_does_not_mark_account_revoked`** — the -//! same sweep flow with a normal `200` egress must leave the account status as -//! `Configured`, proving that `Revoked` in arm (b) is caused by the -//! `invalid_grant` error, not by the sweep machinery itself. -//! -//! **Deferred — live-401 re-auth arm**: reactive re-auth after a credentialed -//! capability backend returns HTTP 401 requires a credentialed capability -//! backend stub and is out of scope here. Track as a follow-up once a -//! `CapabilityBackend` test double is available. -//! -//! All three test functions are gated on -//! `any(feature = "libsql", feature = "postgres")` (the same gate as the -//! `credential_refresh_worker` that powers arm b) so the file compiles and -//! produces zero tests when neither database feature is active. +//! All three tests gated on `any(feature = "libsql", feature = "postgres")`. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; #[cfg(any(feature = "libsql", feature = "postgres"))] @@ -63,13 +49,8 @@ fn test_scope() -> AuthProductScope { // ─── arm a: pure-store revoke ───────────────────────────────────────────────── -/// Create a `Configured` credential account, mark it `Revoked` via -/// `update_status`, and verify the durable read-back carries `Revoked`. -/// -/// `Revoked` is a terminal status: no further OAuth flow is expected. -/// The assertion on the read-back proves the status change committed to the -/// durable `FilesystemAuthProductServices` store and was not -/// merely returned in-memory by `update_status`. +/// `Revoked` is terminal; the read-back proves the status committed to the +/// durable store, not just `update_status`'s in-memory return value. #[cfg(any(feature = "libsql", feature = "postgres"))] #[tokio::test] async fn revoked_account_reads_back_revoked() { @@ -96,8 +77,7 @@ async fn revoked_account_reads_back_revoked() { "update_status return value must carry Revoked" ); - // Durable read-back proves the mutation committed to the store, not - // merely returned from update_status in-memory. + // Durable read-back, not just update_status's in-memory return. let read_back = bundle .services .credential_account_service() @@ -118,30 +98,20 @@ async fn revoked_account_reads_back_revoked() { // ─── arm b: invalid_grant sweep marks account Revoked ───────────────────────── // -// Proves the end-to-end path: -// sweep_once → select_idle_candidates → RebornProductAuthServices::refresh_credential_account -// → ProviderBackedCredentialAccountService::refresh_account -// → HostOAuthProviderClient::refresh_token (HTTP egress: 400 invalid_grant) -// → AuthProductError::InvalidGrant -// → report_terminal_refresh_status(.., Revoked) +// Path: sweep_once -> refresh_credential_account -> refresh_account -> +// refresh_token (HTTP egress: 400 invalid_grant) -> InvalidGrant -> +// report_terminal_refresh_status(.., Revoked). // -// The egress mock uses the `push_response` mechanism: the bundle's default -// egress returns 200 (so the initial token exchange succeeds and stores a real -// refresh-secret handle), then a 400 `invalid_grant` response is queued for -// the sweep's refresh call. +// Egress: default 200 (initial exchange), then a queued 400 invalid_grant for +// the sweep's refresh call (`push_response`). -/// An idle Google OAuth account that receives `{"error":"invalid_grant"}` from -/// the token endpoint during a credential-refresh sweep is persistently marked -/// `Revoked` by `refresh_account`. +/// An idle Google OAuth account receiving `invalid_grant` from the token +/// endpoint during a credential-refresh sweep is persistently marked +/// `Revoked`. /// -/// Egress call count asserts: -/// 1 = initial token exchange (connect flow, 200) -/// 2 = sweep refresh attempt (queued 400 invalid_grant) -/// -/// The account write-back is verified through the durable -/// `CredentialAccountService::get_account` path (not just the refresh report -/// return value), guarding against the "HTTP fired but account write dropped" -/// failure mode. +/// Egress count: 1 = initial exchange (200), 2 = sweep refresh (400 +/// invalid_grant). Verified via durable `get_account` read-back, guarding +/// against "HTTP fired but account write dropped". #[cfg(any(feature = "libsql", feature = "postgres"))] #[tokio::test] async fn invalid_grant_sweep_marks_account_revoked() { @@ -152,9 +122,8 @@ async fn invalid_grant_sweep_marks_account_revoked() { let account = connect_google_account(&bundle, &scope, 0x22).await; let account_id = account.id; - // The sweep makes exactly one refresh call per candidate, so a single - // queued invalid_grant response is enough; the constructor-default 200 - // body stays in place for any calls after that. + // Sweep makes exactly one refresh call per candidate; the default 200 + // body covers any calls after this queued invalid_grant response. bundle.egress.push_response( 400, serde_json::json!({"error": "invalid_grant"}) @@ -205,10 +174,8 @@ async fn invalid_grant_sweep_marks_account_revoked() { // ─── negative guard ──────────────────────────────────────────────────────────── /// A credential-refresh sweep with a normal `200` egress MUST NOT mark the -/// account `Revoked`. -/// -/// After a successful sweep the account is `Configured` (tokens rotated) and -/// the egress call count is 2 (initial exchange + refresh). +/// account `Revoked`; it stays `Configured` (tokens rotated), egress count 2 +/// (initial exchange + refresh). #[cfg(any(feature = "libsql", feature = "postgres"))] #[tokio::test] async fn normal_sweep_does_not_mark_account_revoked() { @@ -260,15 +227,10 @@ async fn normal_sweep_does_not_mark_account_revoked() { // ─── FIFO + default-fallback unit test ─────────────────────────────────────── -/// [`ScriptedOAuthTokenEgress`] queued responses are consumed in FIFO order; -/// once the queue is exhausted, subsequent calls fall back to the constructor -/// default. -/// -/// This test drives [`RuntimeHttpEgress::execute`] directly — no OAuth flow, -/// no real network — to verify the FIFO queue + default-fallback path and -/// the [`captured_count`] accessor in isolation, exercising the -/// `push_response` / `with_error_response` pairing described in the -/// `ScriptedOAuthTokenEgress` documentation. +/// [`ScriptedOAuthTokenEgress`] queued responses are consumed FIFO; once +/// exhausted, calls fall back to the constructor default. Drives +/// [`RuntimeHttpEgress::execute`] directly (no OAuth flow, no real network) to +/// verify the FIFO + fallback path and [`captured_count`] in isolation. #[tokio::test] async fn scripted_oauth_token_egress_consumes_queued_responses_fifo_then_default() { use ironclaw_host_api::{ @@ -284,9 +246,8 @@ async fn scripted_oauth_token_egress_consumes_queued_responses_fifo_then_default egress.push_response(200, body_a.clone()); egress.push_response(500, body_b.clone()); - // `ScriptedOAuthTokenEgress::execute` only reads `request.body.len()` and - // records the full request; the other fields are unused by the scripted - // impl, so this dummy request leaves them empty/default. + // Only `request.body.len()` is read by the scripted impl; other fields + // are left empty/default. let dummy_request = || RuntimeHttpEgressRequest { runtime: RuntimeKind::Wasm, scope: ResourceScope::local_default( diff --git a/tests/reborn_integration_auth_gate.rs b/tests/integration/auth_gate.rs similarity index 50% rename from tests/reborn_integration_auth_gate.rs rename to tests/integration/auth_gate.rs index 0aa44a6965f..5b3ebec5385 100644 --- a/tests/reborn_integration_auth_gate.rs +++ b/tests/integration/auth_gate.rs @@ -1,49 +1,29 @@ -//! E-AUTHGATE seam test: a capability whose credential account resolves to -//! `AuthRequired` raises a real `TurnStatus::BlockedAuth` gate, and denying -//! that gate resumes the run to completion without re-dispatching the parked -//! capability (no loop, no silent re-execution). +//! E-AUTHGATE: a capability whose credential account resolves `AuthRequired` +//! raises a real `TurnStatus::BlockedAuth` gate; denying it resumes to +//! completion without re-dispatching the parked capability. //! -//! Drives the production auth path end-to-end: scripted `github.*` tool call → -//! real credential-account injection (`FixedRuntimeCredentialAccountResolver` -//! returns `AuthRequired`) → `CapabilityObligationError::AuthRequired` → the -//! agent loop blocks the run at `BlockedAuth` with a `gate:auth-` ref → deny + -//! resume → the executor's deny short-circuit -//! (`crates/ironclaw_agent_loop/src/executor/capabilities.rs`, the -//! `state.pending_auth_resume` disposition check right after the -//! `visible_calls.is_empty()` guard) surfaces a model-visible gate-declined -//! failure for the parked capability instead of re-dispatching it → the run -//! completes. Nothing is faked except the model at the vendor-SDK seam. +//! Path: scripted `github.*` call -> `FixedRuntimeCredentialAccountResolver` +//! returns `AuthRequired` -> `BlockedAuth` (`gate:auth-` ref) -> deny+resume -> +//! the deny short-circuit in `executor/capabilities.rs` +//! (`state.pending_auth_resume` check) surfaces gate-declined instead of +//! re-dispatching. Only the model is faked. //! -//! DEFERRED here, COVERED elsewhere: the happy "submit credentials → resume -//! completes" arm. The `live_auth_gate` fixture wires a FIXED `AuthRequired` -//! credential-account resolver with no toggle to flip it to resolved mid-test -//! (and no `run_state` store, so the capability host's auth-resume path cannot -//! complete on this fixture at all). That arm is now covered by -//! `tests/reborn_group_journeys/` (C-JOURNEY) on the -//! `RebornIntegrationGroup::live_auth_and_approval()` group, whose auth gate -//! resolves through the REAL `ProductAuthRuntimeCredentialResolver` + -//! production manual-token flow — no settable-resolver fake needed. +//! DEFERRED here, COVERED by `tests/reborn_group_journeys/` (C-JOURNEY) via +//! `RebornIntegrationGroup::live_auth_and_approval()`: the "submit credentials +//! -> resume completes" arm, since this fixture's resolver is fixed +//! `AuthRequired` with no `run_state` store to complete a real resume. //! -//! `assert_tool_error` IS used below, despite the general guidance to prefer -//! `wait_for_status(Completed)` as the sole discriminator (as in -//! `tests/reborn_group_approvals/scenario_gate_then_deny.rs`): mutation-testing -//! this specific short-circuit (deleting the disposition check) showed that -//! `wait_for_status(Completed)` alone does NOT fail. This harness's mock -//! capability host has no support for a genuine auth-resume completion (see -//! the DEFERRED note above), so a neutralized short-circuit still reaches -//! `Completed` — just via a *different*, harness-specific path: the -//! re-dispatched call comes back `Failed(Backend, "... resume requires -//! run_state")` instead of being denied, and that Failed observation is ALSO -//! surfaced to the model as a non-blocking failure, which also finalizes to -//! `Completed`. `wait_for_status(Completed)` cannot tell these apart; the -//! persisted tool-error class/reason can, because a real re-dispatch is the -//! only way a `Failed{Backend}` result is ever recorded for this capability in -//! this test. +//! `assert_tool_error` (not just `wait_for_status(Completed)`) is required: +//! mutation-testing the deny short-circuit proved `Completed` alone doesn't +//! discriminate — a bypassed short-circuit still re-dispatches, fails +//! `Backend`, and that failure ALSO finalizes `Completed`. Only the persisted +//! tool-error class/reason distinguishes them. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use ironclaw_turns::{GateRef, TurnStatus}; @@ -77,30 +57,23 @@ async fn github_auth_gate_denied_resume_completes_without_loop() { .await .expect("run blocks on an auth gate"); // `submit_turn_until_auth_blocked` already validates the `gate:auth-` - // prefix and returns `Err` otherwise, so the `.expect` above is the real - // failure point — no redundant assert needed here. + // prefix; the `.expect` above is the real failure point. harness .deny_auth_gate(run_id, &gate_ref) .await .expect("deny + resume auth gate"); - // The scripted final reply text is intentionally NOT asserted: `TraceLlm` - // emits scripted replies by call order regardless of what the model - // actually observed, so asserting its text would not distinguish a correct - // deny-and-continue from a regression that loops or fails differently - // (same reasoning as - // tests/reborn_group_approvals/scenario_gate_then_deny.rs). Do not call - // `assert_reply_contains` here. + // Final reply text intentionally NOT asserted: `TraceLlm` emits scripted + // replies by call order regardless of model behavior, so it wouldn't + // discriminate a correct deny from a looping regression. harness .wait_for_status(run_id, TurnStatus::Completed) .await .expect("denied auth resume completes without re-blocking / looping"); - // The discriminating proof: no `Failed{Backend}` tool-error was persisted - // for this capability. Mutation-verified — see the module doc above — a - // `Failed{Backend, "resume requires run_state"}` result only exists when - // the deny short-circuit was bypassed and re-dispatched. + // Mutation-verified: a `Failed{Backend}` result only exists if the deny + // short-circuit was bypassed and re-dispatched (see module doc). harness .assert_no_tool_error(ToolErrorClass::Failed, "backend") .await @@ -109,12 +82,9 @@ async fn github_auth_gate_denied_resume_completes_without_loop() { re-dispatch)", ); - // Positive proof of the CORRECT outcome: `short_circuit_denied_resume` - // (capabilities.rs ~1149) persists its raw planner summary via - // `SanitizedStrategySummary::from_trusted_static("auth gate denied by - // user")`, deliberately bypassing the "capability denied with " prefix - // (no host-returned text to prefix for a gate denial) — so - // `assert_tool_error(Denied, ..)` cannot express this; use the raw-summary + // `short_circuit_denied_resume` (capabilities.rs ~1149) persists a raw + // planner summary bypassing the "capability denied with " prefix, so + // `assert_tool_error(Denied, ..)` can't express this; use the raw-summary // assertion instead. harness .assert_tool_error_summary_contains("auth gate denied by user") @@ -122,27 +92,21 @@ async fn github_auth_gate_denied_resume_completes_without_loop() { .expect("the deny short-circuit's planner summary was persisted"); } -/// W4-AUTHGATE-WIRE (flagship): a GitHub capability whose credential account -/// resolves OK (`.with_github_issue_tools()` — unlike the sibling test above, -/// whose `live_auth_gate` fixture is wired via `github_issue_tools_auth_required`'s -/// credential-*missing* resolver) but whose injected token draws a runtime -/// `401` from the (scripted) GitHub API must raise `TurnStatus::BlockedAuth` -/// with `credential_requirements` POPULATED (provider=github, ManualToken -/// setup) — the #5174/#5180 bug class: an empty `credential_requirements` left -/// `AuthPromptView.provider` null, so the WebUI's manual-token card threw -/// client-side and silently dropped the submit ("Could not save the token", no -/// network request ever sent). +/// W4-AUTHGATE-WIRE (flagship): a GitHub capability with a valid credential +/// account but a runtime `401` from the GitHub API must raise +/// `TurnStatus::BlockedAuth` with `credential_requirements` POPULATED +/// (provider=github, ManualToken) — the #5174/#5180 bug class: an empty +/// `credential_requirements` left `AuthPromptView.provider` null, silently +/// dropping the WebUI manual-token submit ("Could not save the token"). /// -/// This runs the FULL scripted-gateway integration harness (`submit_turn` -> -/// product workflow -> turn coordinator -> agent loop -> capability host -> the -/// real GitHub WASM module), a tier below neither of the two existing pins for -/// this fix covers: `ironclaw_capabilities/tests/capability_host_auth_required_enrichment_contract.rs` -/// drives `CapabilityHost::invoke_json` directly (no turn/loop), and -/// `ironclaw_host_runtime/tests/github_wasm_runtime_contract.rs` -/// (`host_runtime_services_maps_google_drive_wasm_401_to_auth_required`) drives -/// `HostRuntimeServices::invoke_capability` directly (no coordinator/loop, and a -/// different first-party extension). Neither exercises the real submit-turn -> -/// `BlockedAuth` wire the WebUI actually depends on. +/// Runs the full scripted-gateway harness (submit_turn -> workflow -> +/// coordinator -> agent loop -> capability host -> real GitHub WASM module) — +/// a tier neither existing pin covers: +/// `capability_host_auth_required_enrichment_contract.rs` drives +/// `CapabilityHost::invoke_json` directly (no turn/loop); +/// `github_wasm_runtime_contract.rs` drives `HostRuntimeServices::invoke_capability` +/// directly (no coordinator/loop). Neither exercises the real submit-turn -> +/// `BlockedAuth` wire the WebUI depends on. #[tokio::test] async fn runtime_401_after_injection_populates_provider_credential_requirement() { let harness = RebornIntegrationHarness::test_default() @@ -163,10 +127,8 @@ async fn runtime_401_after_injection_populates_provider_credential_requirement() .await .expect("run blocks on an auth gate"); - // `submit_turn_until_auth_blocked` only returns the gate ref; re-fetch the - // full state (already at `BlockedAuth`, so this returns immediately) to - // read `credential_requirements` — the exact field the #5180 fix enriches - // from the capability's declared `InjectCredentialAccountOnce` obligation. + // Re-fetch full state to read `credential_requirements` -- the field the + // #5180 fix enriches from the InjectCredentialAccountOnce obligation. let state = harness .wait_for_status(run_id, TurnStatus::BlockedAuth) .await @@ -204,13 +166,10 @@ async fn runtime_401_after_injection_populates_provider_credential_requirement() .expect("denied auth resume completes"); } -/// W4-AUTHGATE-WIRE: cancelling a run parked at `BlockedAuth` must land -/// directly on `TurnStatus::Cancelled` (unlike a mid-model park -- see -/// `reborn_integration_cancel.rs` -- a blocked-gate run has no active worker to -/// cooperate with cancellation, so `request_cancel_once` transitions it -/// straight through) and must leave no stale replay: once cancelled, the SAME -/// real gate ref can no longer resume the run (closes the #5067/#4957 class of -/// auth gates staying "live"/resumable after the user has moved on). +/// W4-AUTHGATE-WIRE: cancelling a run parked at `BlockedAuth` lands directly +/// on `Cancelled` (no active worker to cooperate with, unlike a mid-model +/// park) and leaves no stale replay -- the SAME gate ref can't resume after +/// cancel (closes the #5067/#4957 "gate stays live" class). #[tokio::test] async fn cancel_blocked_auth_gate_leaves_no_stale_replay() { let harness = RebornIntegrationHarness::test_default() @@ -248,10 +207,8 @@ async fn cancel_blocked_auth_gate_leaves_no_stale_replay() { .await .expect("run stays Cancelled"); - // Stale-replay guard: resuming the now-terminal run with the SAME real gate - // ref (not a bogus one -- this proves the run's terminal status is what - // blocks the resume, not a gate-ref mismatch) must fail, not silently - // re-dispatch the parked github capability. + // Stale-replay guard: the SAME gate ref (not bogus) must be rejected by + // the run's terminal status, not a gate-ref mismatch. let resume_err = harness .deny_auth_gate(run_id, &gate_ref) .await @@ -270,11 +227,9 @@ async fn cancel_blocked_auth_gate_leaves_no_stale_replay() { .expect("cancel + failed resume must not trigger a second github dispatch"); } -/// Regression guard for the flip side of the stale-replay test above: an -/// invalid/unknown `GateRef` string against a still-open (non-terminal) run -/// must also fail cleanly rather than resuming under a synthesized ref. -/// Cheap, discriminating companion assertion -- not a full scenario -- so it -/// stays inline here instead of a third full harness build. +/// Regression guard, flip side of the stale-replay test above: an +/// invalid/unknown `GateRef` against a still-open run must also fail cleanly, +/// not resume under a synthesized ref. #[tokio::test] async fn deny_auth_gate_rejects_a_non_auth_gate_ref_prefix() { let harness = RebornIntegrationHarness::test_default() diff --git a/tests/reborn_integration_backend_matrix.rs b/tests/integration/backend_matrix.rs similarity index 64% rename from tests/reborn_integration_backend_matrix.rs rename to tests/integration/backend_matrix.rs index 57a82e0dd3a..01220f1e92d 100644 --- a/tests/reborn_integration_backend_matrix.rs +++ b/tests/integration/backend_matrix.rs @@ -1,23 +1,19 @@ -//! Reborn integration-test framework — slice 3: storage-backend matrix. +//! Reborn integration-test framework — storage-backend matrix. //! -//! Two things this tier did not cover before: -//! 1. **Backend parity** — one golden scenario through BOTH -//! `StorageMode::InMemory` and `StorageMode::LibSql` (real SQLite on a -//! tmp `.db`, real SQL + migrations + CAS), asserting an identical -//! outcome. The canonical `rstest` matrix exemplar for this tier. -//! 2. **Persistence correctness (LibSql-only)** — a write-then-read-back -//! test that reopens the SQLite file through a *fresh* database handle and -//! asserts the assistant reply survived to disk, proving real -//! serialization + durability (design §3.8 guardrail). +//! Covers: backend parity (one golden scenario through `StorageMode::InMemory` +//! and `StorageMode::LibSql`, asserting an identical outcome — the canonical +//! `rstest` matrix exemplar for this tier) and LibSql persistence correctness +//! (write-then-reopen through a fresh database handle, design §3.8 guardrail). //! //! Runs under default features, no services, no keys, no Docker, no //! `integration` feature — libSQL is an embedded SQLite file in a `TempDir` //! dropped at test end. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::builder::{RebornIntegrationHarness, StorageMode}; @@ -48,11 +44,9 @@ async fn backend_parity_replies_to_greeting(#[case] storage: StorageMode) { .expect("reply finalized in thread history"); } -/// Persistence correctness, LibSql-only (design §3.8): the reply must survive -/// to the SQLite file and read back through a *fresh* database handle — real -/// serialization + durability, not an in-process cache. InMemory cannot make -/// this assertion (nothing reaches disk), so this test legitimately requires -/// `StorageMode::LibSql`. +/// Persistence correctness (design §3.8): the reply must survive to the +/// SQLite file and read back through a fresh database handle, not an +/// in-process cache. InMemory cannot make this assertion (nothing reaches disk). #[tokio::test] async fn libsql_persists_reply_across_reopen() { let harness = RebornIntegrationHarness::test_default() @@ -72,10 +66,8 @@ async fn libsql_persists_reply_across_reopen() { } /// Guard: `assert_reply_persists_after_reopen` must return `Err` when the -/// expected text is absent — proving the LibSql reopen read-back assertion is -/// not vacuously green (it really inspects the reopened on-disk history, and a -/// wrong expectation fails). Mirrors the negative-guard tests the other slices -/// carry (e.g. `assertions_fail_when_tool_did_not_run`). +/// expected text is absent, proving the reopen assertion isn't vacuously +/// green — it inspects real on-disk history. #[tokio::test] async fn persistence_assertion_fails_on_mismatch_after_reopen() { let harness = RebornIntegrationHarness::test_default() diff --git a/tests/reborn_integration_budget.rs b/tests/integration/budget.rs similarity index 58% rename from tests/reborn_integration_budget.rs rename to tests/integration/budget.rs index 3ab3b0e0181..8d03ef985bb 100644 --- a/tests/reborn_integration_budget.rs +++ b/tests/integration/budget.rs @@ -3,28 +3,25 @@ //! `build_default_budget_accountant` into `DefaultPlannedRuntimeParts::model_budget_accountant`, //! and that accountant fires on a real coordinator-path turn. //! -//! Budget SEMANTICS (ledger, warn/deny thresholds, approval-unblock, -//! `BudgetEvent` cascade) are ALREADY covered at crate tier via -//! `build_reborn_runtime` (`budget_e2e.rs` / `budget_approval_e2e.rs`) — this -//! binary does NOT re-author them. It proves only the group/flat harness (which -//! bypasses the `build_reborn_services` shell) now composes the accountant live: -//! on the turn's first model call the accountant's compiled-default seeding -//! policy installs the run owner's daily USD cap, observable through the -//! retained in-memory governor. +//! Budget SEMANTICS (ledger, thresholds, approval-unblock, `BudgetEvent` +//! cascade) are covered at crate tier via `build_reborn_runtime` +//! (`budget_e2e.rs` / `budget_approval_e2e.rs`) — not re-authored here. This +//! proves only that the accountant is live: on the turn's first model call it +//! seeds the run owner's daily USD cap, observable via the in-memory governor. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::builder::RebornIntegrationHarness; use reborn_support::reply::RebornScriptedReply; -/// The wired accountant seeds the run owner's compiled-default daily cap on the -/// turn's first model call — the liveness proof that -/// `build_default_budget_accountant` reaches the coordinator → loop → model-port -/// path through the harness's `DefaultPlannedRuntimeParts` wiring. +/// Liveness proof: the wired accountant seeds the run owner's daily cap on +/// the turn's first model call, reaching coordinator → loop → model-port via +/// `DefaultPlannedRuntimeParts` wiring. #[tokio::test] async fn budget_accountant_seeds_user_cap_on_turn_model_call() { let h = RebornIntegrationHarness::test_default() @@ -39,10 +36,9 @@ async fn budget_accountant_seeds_user_cap_on_turn_model_call() { .expect("wired budget accountant seeded the run owner's daily cap"); } -/// Guard: without `with_budget_accounting`, no accountant is wired, so the -/// liveness assertion must FAIL (there is no governor to read). Pins that the -/// assertion above is not vacuously passing and that the default path is -/// behavior-identical (no accountant). +/// Guard: without `with_budget_accounting` wired, no governor exists, so the +/// liveness assertion above must fail — not vacuously passing — and the +/// default path stays behavior-identical (no accountant). #[tokio::test] async fn budget_assertion_requires_wiring() { let h = RebornIntegrationHarness::test_default() diff --git a/tests/reborn_integration_cancel.rs b/tests/integration/cancel.rs similarity index 64% rename from tests/reborn_integration_cancel.rs rename to tests/integration/cancel.rs index 99f5dbcd6d8..beef332ca93 100644 --- a/tests/reborn_integration_cancel.rs +++ b/tests/integration/cancel.rs @@ -1,23 +1,21 @@ //! Reborn integration test — mid-turn cancellation + related failure paths //! (E-GATEWAY seam, C-ERRORS). //! -//! Proves the cancel path end-to-end at the int tier: the model call parks at -//! the vendor-SDK seam, the test cancels the in-flight run, releases the park, -//! and the run reaches `TurnStatus::Cancelled` (not `Completed`). Exercises the -//! parking provider (`park_model`) and `cancel_run`. Cancellation is observed -//! by the loop-driver host's own default `TurnStateRunCancellationFactory` -//! (`group.rs` leaves the optional `cancellation_factory` as `None`), not a +//! Proves the cancel path end-to-end: the model call parks at the vendor-SDK +//! seam, the test cancels the in-flight run, releases the park, and the run +//! reaches `TurnStatus::Cancelled` (not `Completed`). Cancellation is observed +//! by the loop-driver host's default `TurnStateRunCancellationFactory`, not a //! wired coordinator fan-out. //! -//! The tests below extend the same seam with C-ERRORS coverage: a -//! leaked-permit regression guard on the cancel path (precedent: PR #5206's -//! RAII `ReservationGuard` bugs), thread-busy rejection, and a non-retryable -//! provider-`Err` reaching a categorized `TurnStatus::Failed`. +//! Also covers C-ERRORS: a leaked-permit regression guard (precedent: PR +//! #5206's RAII `ReservationGuard` bugs), thread-busy rejection, and a +//! non-retryable provider-`Err` reaching a categorized `TurnStatus::Failed`. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use std::time::Duration; @@ -58,12 +56,10 @@ async fn cancels_a_parked_mid_turn_run() { .expect("parked run reaches Cancelled after cancel"); } -/// Regression guard: cancelling a parked run must release whatever -/// per-actor/tenant admission slot or run-concurrency permit it held. If -/// cancellation leaked one (precedent: PR #5206's leaked WASM -/// permit/reservation bugs), a second turn on the SAME thread right after -/// would either hang past `wait_for_status`'s internal deadline or come back -/// `RejectedBusy` instead of completing. +/// Regression guard: cancelling a parked run must release its per-actor/ +/// tenant admission permit. If leaked (precedent: PR #5206's WASM +/// permit/reservation bugs), a second turn on the SAME thread would hang or +/// come back `RejectedBusy` instead of completing. #[tokio::test] async fn cancelled_run_does_not_block_a_second_turn_on_the_same_thread() { let gate = ParkingModelGate::new(); @@ -91,9 +87,9 @@ async fn cancelled_run_does_not_block_a_second_turn_on_the_same_thread() { .await .expect("parked run reaches Cancelled after cancel"); - // The gate's channels are already consumed after the first park+release - // cycle, so this second call passes through the same `ParkingLlm` instantly - // (see `ParkingModelGate`'s "second call does not block" guarantee). + // The gate's channels are already consumed, so this second call passes + // through the same `ParkingLlm` instantly (`ParkingModelGate`'s "second + // call does not block" guarantee). harness .submit_turn("do another thing") .await @@ -104,11 +100,9 @@ async fn cancelled_run_does_not_block_a_second_turn_on_the_same_thread() { .expect("second turn's reply persisted"); } -/// A second submit on a thread whose first run is still active (parked, not -/// yet cancelled) must be rejected — `InboundTurnOutcome::RejectedBusy` -/// (`ironclaw_product_workflow::inbound_turn`) — not silently queued or -/// accepted. Releases the park afterward so the first run still completes and -/// the harness doesn't leak a parked task past the test. +/// A second submit on a thread whose first run is still active (parked) must +/// be rejected — `InboundTurnOutcome::RejectedBusy` — not silently queued or +/// accepted. Releases the park afterward so no parked task leaks past the test. #[tokio::test] async fn busy_reject_when_thread_already_has_an_active_run() { let gate = ParkingModelGate::new(); @@ -143,11 +137,10 @@ async fn busy_reject_when_thread_already_has_an_active_run() { .expect("first turn still completes after the rejected second submit"); } -/// A raw provider `Err` that the real `ironclaw_llm` decorator chain classifies -/// as non-retryable (`LlmError::ContextLengthExceeded`, excluded from -/// `ironclaw_llm::retry::is_retryable`) must reach the model a `TurnStatus::Failed` -/// run categorized `"model_error"` (`LoopFailureKind::ModelError`), not silently -/// retry forever or surface as a different failure category. +/// A raw provider `Err` classified non-retryable by `ironclaw_llm` +/// (`LlmError::ContextLengthExceeded`, excluded from `is_retryable`) must +/// reach `TurnStatus::Failed` categorized `"model_error"`, not retry forever +/// or surface as a different category. #[tokio::test] async fn mid_turn_provider_error_reaches_failed_with_model_error_category() { let harness = RebornIntegrationHarness::test_default() @@ -176,29 +169,21 @@ async fn mid_turn_provider_error_reaches_failed_with_model_error_category() { } /// Regression guard, `Failed`-path sibling of -/// `cancelled_run_does_not_block_a_second_turn_on_the_same_thread`: whatever -/// per-thread busy/admission lock a run holds must be released when the run -/// reaches `TurnStatus::Failed`, not just `Cancelled`. If that release leaked -/// (same "wedge class" as PR #5206's leaked WASM permit/reservation bugs), a -/// second submit on the SAME thread right after would come back -/// `RejectedBusy` instead of being admitted. +/// `cancelled_run_does_not_block_a_second_turn_on_the_same_thread`: the +/// per-thread busy/admission lock must release on `TurnStatus::Failed`, not +/// just `Cancelled` (same "wedge class" as PR #5206's leaked WASM +/// permit/reservation bugs) — a leak would make a second submit on the SAME +/// thread come back `RejectedBusy`. /// -/// This cannot mirror the cancel-path test's shape exactly (second turn -/// reaching `Completed`): `fail_model()` swaps in `ErrLlm` — see -/// `tests/support/reborn/scripted_provider.rs` — as this thread's ENTIRE raw -/// model provider, unconditionally, for every future call (no per-call -/// counting; `ErrLlm::complete`/`complete_with_tools` always return -/// `LlmError::ContextLengthExceeded`). There is also no builder seam to swap -/// in a fresh, successful script for a second turn on the same thread: a -/// second `RebornIntegrationGroup::thread(...)` build for the same -/// `conversation_id` would resolve the same `TurnScope` and panic on -/// `ScopeRegistryGateway::register`'s duplicate-registration guard (see -/// `tests/support/reborn/scope_gateway.rs::duplicate_register_for_same_scope_panics`). -/// So the second turn here also fails (same `"model_error"` category) — the -/// regression signal is that it is *admitted* (`Accepted`, not `RejectedBusy`) -/// and reaches its own terminal status promptly rather than hanging or being -/// silently dropped, proving the lock was genuinely released and not just -/// accepted-then-stuck. +/// The second turn also fails (same `"model_error"` category), not +/// completes: `fail_model()` swaps in `ErrLlm` as the thread's entire raw +/// model provider permanently (no per-call counting), and there is no +/// builder seam to swap in a fresh script for a second turn on the same +/// thread (a second `group.thread(...)` for the same `conversation_id` would +/// panic on `ScopeRegistryGateway::register`'s duplicate-registration guard). +/// The regression signal is that it is *admitted* (`Accepted`, not +/// `RejectedBusy`) and reaches its own terminal status promptly, proving the +/// lock was genuinely released. #[tokio::test] async fn failed_run_does_not_block_a_second_turn_on_the_same_thread() { let harness = RebornIntegrationHarness::test_default() diff --git a/tests/reborn_integration_comm_context.rs b/tests/integration/comm_context.rs similarity index 70% rename from tests/reborn_integration_comm_context.rs rename to tests/integration/comm_context.rs index aa00e3d5480..fba108f3961 100644 --- a/tests/reborn_integration_comm_context.rs +++ b/tests/integration/comm_context.rs @@ -2,18 +2,17 @@ //! pipeline — the delivery-preference / connected-channel slice it resolves //! renders into the model request on a real coordinator-path turn. //! -//! Distinct from the outbound delivery **sink** (E-OUTBOUND, a sibling lane): -//! this covers prompt **context** (delivery preferences/targets), not a delivery -//! recorder. The production `RuntimeCommunicationContextProvider`'s -//! facade→context mapping is densely unit-tested at crate tier -//! (`ironclaw_reborn_composition::communication_context`); this binary covers -//! only the int-tier wiring gap — that the `communication_context_provider` -//! field threads through the coordinator path into the model request. +//! Distinct from the outbound delivery sink (E-OUTBOUND): this covers prompt +//! context, not a delivery recorder. The facade→context mapping itself is +//! unit-tested at crate tier (`ironclaw_reborn_composition::communication_context`); +//! this binary covers only that the field threads through the coordinator path +//! into the model request. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::builder::RebornIntegrationHarness; @@ -57,19 +56,15 @@ async fn no_communication_section_without_provider() { .expect("harness builds"); h.submit_turn("hello").await.expect("turn completes"); - // Baseline: a model request WAS captured at all (the turn actually - // reached the scripted provider), so the negative assertion below is - // proving absence of the communication section, not a vacuous pass - // against zero captured requests. + // Baseline: a request WAS captured, so the negative assertion below + // proves absence of the section, not a vacuous pass on zero requests. h.assert_model_request_contains("hello") .await .expect("the turn's own text must reach the captured model request"); - // Specific error check (not generic `is_err()`): pin that the failure is - // the "not found" path over exactly the one captured request, so an - // infra-level failure (e.g. JSON serialization) can't masquerade as proof - // the communication section was absent, and a regression that silently - // captures zero requests can't slip through either. + // Specific error check (not `is_err()`): pins the failure to the + // "not found" path over the one captured request, ruling out an infra + // failure or a zero-capture regression masquerading as proof of absence. let err = h .assert_model_request_contains("Outbound delivery target:") .await diff --git a/tests/integration/coverage-exemptions.toml b/tests/integration/coverage-exemptions.toml new file mode 100644 index 00000000000..99861ea0582 --- /dev/null +++ b/tests/integration/coverage-exemptions.toml @@ -0,0 +1,31 @@ +# Reborn integration-tier coverage exemptions. +# +# Each [[exemption]] entry excludes ONE source file from the per-crate +# "uncovered" accounting in the Reborn integration-tier coverage report: +# - Computed by scripts/ci/reborn-coverage-summary.sh from the merged lcov +# tracefile (scripts/ci/reborn-coverage-merge-lcov.sh), scoped to +# crates/ironclaw_* (all workspace crates the int-tier suites exercise). +# - Rendered by the `coverage-report` job in +# .github/workflows/reborn-tests.yml into $GITHUB_STEP_SUMMARY and a +# sticky PR comment. +# +# An exempted file's lines are dropped from the per-crate table's +# covered/total line counts entirely — they neither help nor hurt a crate's +# percentage — and the file is listed verbatim, with its reason and tracking +# issue, in the report's own "Exemptions" section. The exclusion must stay +# visible, never silent. +# +# This coverage signal is INFORMATIONAL ONLY: no ratchet, no fail threshold, +# on this report yet. Exemptions exist for visibility/triage today, not to +# game a gate that doesn't exist — every entry still needs a reason and a +# tracking issue so a future ratchet effort (or a reviewer) can tell "known, +# tracked gap" from "nobody looked". +# +# Schema (repeat the [[exemption]] table once per exempted file): +# +# [[exemption]] +# module = "crates/ironclaw_x/src/y.rs" # required: repo-relative source path +# reason = "why this file cannot be exercised by int-tier tests" # required +# issue = "https://github.com/nearai/ironclaw/issues/NNNN" # required: tracking issue +# +# No entries yet — add one per exemption, each with `reason` and `issue`. diff --git a/tests/reborn_integration_durable.rs b/tests/integration/durable.rs similarity index 96% rename from tests/reborn_integration_durable.rs rename to tests/integration/durable.rs index 4921115b9b2..5db9f506621 100644 --- a/tests/reborn_integration_durable.rs +++ b/tests/integration/durable.rs @@ -7,9 +7,10 @@ //! `assert_reply_persists_after_reopen` for capability state. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::group::RebornIntegrationGroup; diff --git a/tests/reborn_integration_golden_payload.rs b/tests/integration/golden_payload.rs similarity index 55% rename from tests/reborn_integration_golden_payload.rs rename to tests/integration/golden_payload.rs index 3cd9d2e7450..15fdf34db95 100644 --- a/tests/reborn_integration_golden_payload.rs +++ b/tests/integration/golden_payload.rs @@ -1,28 +1,22 @@ //! Reborn integration — golden inference-payload coverage. //! //! Exact-matches the FULL model-visible inference payload (system prompt + -//! conversation turns + tool-call/tool-result messages + ordered tool surface) -//! per inference iteration against a committed `insta` snapshot, plus the exact -//! final user-visible reply. Where `assert_system_prompt_contains` proves a -//! substring reached the model, this pins end-to-end prompt construction byte -//! for byte — catching silent drift in prompt assembly, history accumulation, -//! and tool-result feed-back. See `tests/support/reborn/golden.rs` for the -//! canonicalization + single-filter normalization rationale. Regenerate drift -//! with `cargo insta review` (or `INSTA_UPDATE=always cargo test`). +//! turns + tool-call/tool-result messages + ordered tool surface) per +//! inference iteration against a committed `insta` snapshot, plus the exact +//! final reply — catching silent drift in prompt assembly, history +//! accumulation, and tool-result feed-back that a substring check can't see. +//! See `tests/integration/support/golden.rs` for canonicalization rationale. +//! Regenerate with `cargo insta review` (or `INSTA_UPDATE=always cargo test`). //! -//! This is the dedicated suite for watching prompt-construction drift: base -//! system-prompt assembly, context/capability surfacing (communication -//! context, capability surface), message appending across turns, and (once -//! reachable — see the "(e) Compaction" blocker note below) compaction — on a -//! deliberately small, curated set of scenarios (full payload matches are -//! expensive to review and maintain; substring checks elsewhere cover -//! everything else). Add a new scenario here only when an existing one can't +//! Deliberately small, curated scenario set (full payload matches are +//! expensive to review); add a scenario only when an existing one can't //! absorb it — see root `CLAUDE.md` Testing Discipline. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::builder::RebornIntegrationHarness; @@ -56,10 +50,9 @@ async fn golden_single_turn_greeting() { .expect("final reply matches exactly"); } -/// (b) Tool-call turn: BOTH inference iterations exact-matched — the initial -/// call and the post-tool-result call. Pins tool-result feed-back construction -/// (the assistant `tool_calls[].id` and the following `tool` message's -/// `tool_call_id` must match). +/// (b) Tool-call turn: both inference iterations exact-matched (initial call +/// + post-tool-result call), pinning that `tool_calls[].id` matches the +/// following `tool` message's `tool_call_id`. #[tokio::test] async fn golden_tool_call_feedback() { let h = RebornIntegrationHarness::test_default() @@ -112,12 +105,10 @@ async fn golden_multi_turn_history() { .expect("final reply matches exactly"); } -/// (d) Context surfacing: a wired `CommunicationContextProvider` PLUS the real +/// (d) Context surfacing: a wired `CommunicationContextProvider` plus the real /// builtin capability surface, on one plain-text turn. Pins byte-for-byte how -/// the communication/context section and the capability surface render into -/// the system prompt alongside each other (a single-value substring check, -/// e.g. `assert_model_request_contains`, cannot see whether the two sections -/// interleave, reorder, or duplicate content). +/// the two sections render together in the system prompt — ordering/duplication +/// a substring check like `assert_model_request_contains` cannot see. #[tokio::test] async fn golden_context_surfacing() { let provider = RecordingCommunicationContextProvider::with_target_and_channel( @@ -141,14 +132,10 @@ async fn golden_context_surfacing() { .expect("final reply matches exactly"); } -/// (f) Parallel tool_calls: ONE assistant response carrying TWO `tool_calls[]` -/// entries (`RebornScriptedReply::tool_calls`), both exact-matched — the -/// initial multi-call request and the post-both-results call. Distinct from -/// (b) `golden_tool_call_feedback`: that pins a SINGLE tool_call's id/result -/// feedback; this pins that MULTIPLE tool_calls in one assistant message each -/// get a distinct id and each following `tool` role message's `tool_call_id` -/// lines up with the right one, in order — a shape (b) cannot exercise with -/// only one call. +/// (f) Parallel tool_calls: one assistant response carrying TWO `tool_calls[]` +/// entries, both exact-matched. Distinct from (b): pins that multiple +/// tool_calls in one message each get a distinct id, and each following `tool` +/// message's `tool_call_id` lines up with the right one, in order. #[tokio::test] async fn golden_parallel_tool_calls() { let h = RebornIntegrationHarness::test_default() @@ -173,14 +160,11 @@ async fn golden_parallel_tool_calls() { } /// (g) Image/user-parts: an inline image attachment landed through the real -/// `submit_inbound_with_attachments` entry point (`RebornIntegrationGroup::attachment_tools()`) -/// and routed through a vision-pattern model id. Pins byte-for-byte how the -/// user turn's multimodal content parts (`ContentPart::ImageUrl` `data:` URL -/// alongside the text part) render into the request — a shape none of the -/// plain-text scenarios above can exercise. Complements -/// `tests/reborn_integration_attach.rs`'s substring assertion -/// (`assert_model_saw_image_attachment`) with the full-payload byte-for-byte -/// pin, catching drift in part ORDERING/shape a substring check cannot see. +/// `submit_inbound_with_attachments` entry point, routed through a +/// vision-pattern model id. Pins byte-for-byte how `ContentPart::ImageUrl` +/// renders alongside the text part — complements +/// `tests/reborn_integration_attach.rs`'s substring check by catching drift in +/// part ordering/shape a substring check can't see. #[tokio::test] async fn golden_image_attachment_turn() { let group = RebornIntegrationGroup::attachment_tools() @@ -207,15 +191,10 @@ async fn golden_image_attachment_turn() { .expect("final reply matches exactly"); } -/// (h) Gated turn (approve arm): a real `BlockedApproval` gate -/// (`RebornIntegrationGroup::live_approvals()`) raised by a scripted -/// `builtin.write_file` call, approved, and resumed. Snapshots BOTH captured -/// inference calls around the gate in one golden — the pre-gate tool_call -/// request AND the post-resume request that reacts to the granted write's -/// result — pinning that a gate resume does not silently drop, duplicate, or -/// reorder the accumulated turn history the way a mid-run resume easily -/// could. Distinct from (b): that tool call dispatches immediately -/// (auto-approved, `test_default()`'s Echo backend has no gate); this one +/// (h) Gated turn (approve arm): a real `BlockedApproval` gate raised by a +/// scripted `builtin.write_file` call, approved, and resumed. Snapshots both +/// captured inference calls around the gate, pinning that a resume doesn't +/// silently drop, duplicate, or reorder history. Distinct from (b): this one /// actually parks on `TurnStatus::BlockedApproval` between the two calls. #[tokio::test] async fn golden_gated_turn_approve() { @@ -250,39 +229,14 @@ async fn golden_gated_turn_approve() { .expect("final reply matches exactly"); } -// (e) Compaction: NOT implemented — blocked. Investigated the byte-cap-overflow -// force-compaction path (`ByteCapStrategy` / `PostCapabilityStage`, -// crates/ironclaw_agent_loop/src/executor/post_capability.rs:73-91) end-to-end -// through this harness with a scripted `builtin.http` result exceeding the -// 32,000-byte cap (crates/ironclaw_agent_loop/src/strategies/compaction.rs:190-213, -// `ByteCapStrategy::with_defaults`). Confirmed via instrumented local runs (not -// committed) that `pending_capability_bytes` correctly accumulates past the cap -// and `state.compaction_state.force_compact_on_next_iteration` / -// `skip_model_this_iteration` DO get set, and that `PromptStage` -// (crates/ironclaw_agent_loop/src/executor/prompt.rs:219-241) correctly detects -// the flag on the very next iteration and enters `PromptCompactionStep`. -// -// The wall: `PromptCompactionStep::run` decides via -// `self.ctx.planner.compaction().should_compact(..)` -// (crates/ironclaw_agent_loop/src/executor/prompt.rs:442-446), and the Reborn -// `DefaultPlanner` wires that seam to `ActiveTaskPreservingCompactionStrategy`, -// not the bare `DefaultCompactionStrategy` -// (crates/ironclaw_agent_loop/src/default_planner.rs:339). +// (e) Compaction: NOT implemented — blocked. The byte-cap-overflow force- +// compaction path (`ByteCapStrategy` / `PostCapabilityStage`) sets +// `state.compaction_state.force_compact_on_next_iteration`, but Reborn's +// `DefaultPlanner` wires `PromptCompactionStep` to // `ActiveTaskPreservingCompactionStrategy::should_compact` -// (crates/ironclaw_agent_loop/src/strategies/active_task_compaction.rs:41-61) -// never reads `state.compaction_state.force_compact_on_next_iteration` at all — -// it always falls through to `active_task_preserving_user_boundary`, which -// requires a genuine accumulated tail of at least `preserve_tail_tokens` -// (8,000 tokens, `DefaultCompactionStrategy::DEFAULT_PRESERVE_TAIL_TOKENS`) of -// real message content plus `minimum_tail_messages`/`minimum_compacted_messages` -// (3 each) before it will trigger at all. The byte-cap-overflow force path -// (`CompactionInitiator::CapabilityResultOverflow`) is consequently a dead -// letter under the strategy Reborn's `default_planner.rs` actually installs — -// it sets state flags this strategy's `should_compact` never consults. -// -// Exercising compaction here would need either a production fix (out of scope -// per this suite's test-only mandate) or inflating the scripted transcript to a -// genuine ~8,000+ token natural tail, which no longer tests the reported -// byte-cap mechanism, would make the golden snapshot unreviewable, and is -// fragile against token-estimation drift. Reporting the blocker instead of -// faking the scenario. +// (crates/ironclaw_agent_loop/src/strategies/active_task_compaction.rs:41-61), +// which never reads that flag — it only compacts on a genuine ~8,000+ token +// accumulated tail. The force path is a dead letter under Reborn's installed +// strategy; exercising it here needs a production fix (out of scope) or a +// fragile multi-thousand-token scripted transcript that breaks golden-snapshot +// reviewability. diff --git a/tests/reborn_integration_greeting.rs b/tests/integration/greeting.rs similarity index 60% rename from tests/reborn_integration_greeting.rs rename to tests/integration/greeting.rs index 318fc5f2591..161d0b66a55 100644 --- a/tests/reborn_integration_greeting.rs +++ b/tests/integration/greeting.rs @@ -1,24 +1,23 @@ -//! Reborn integration-test framework — slice 1 smoke test. +//! Reborn integration-test framework — smoke test. //! //! Proves the single LLM seam end-to-end: synthetic inbound → product workflow //! → scheduler → planned agent loop → real `LlmProviderModelGateway` → real //! `ironclaw_llm` decorator chain (hermetic passthrough) → scripted `TraceLlm` -//! → assistant reply finalized in thread history. InMemory storage, no services, -//! no keys, no Docker, no `integration` feature. +//! → assistant reply finalized in thread history. InMemory storage, no +//! services, no keys, no Docker, no `integration` feature. //! -//! Asserts BOTH facets of the one default turn: the finalized reply (output -//! seam) and the model-visible system prompt (input seam, T0-SYSPROMPT). The -//! system-prompt assertion rides this smoke test rather than a redundant file — -//! it exercises the same `build → submit_turn` path, so consolidating avoids a -//! second full support-tree compile for zero new path coverage. +//! Asserts both facets of the turn: the finalized reply (output seam) and the +//! model-visible system prompt (input seam, T0-SYSPROMPT) — consolidated here +//! rather than a redundant file, since both ride the same `build → submit_turn` +//! path. -// The support tree is large and shared; a single-test file exercises only a -// slice of it, so suppress dead-code warnings on the includes (matches -// `reborn_qa_recorded_behavior.rs`). +// The support tree is large and shared; a single-test file only exercises a +// slice, so suppress dead-code warnings on the includes. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::builder::RebornIntegrationHarness; @@ -45,9 +44,8 @@ async fn replies_to_greeting() { .assert_system_prompt_contains("Use only visible capabilities.") .await .expect("composed capability policy reached the model as a system prompt"); - // Negative guard: the user's own turn text appears in the captured request - // but only in a `User`-role message, so the `System`-only filter must not - // match it — proves the assertion discriminates on role, not mere presence. + // Negative guard: the user's text appears only in a `User`-role message, + // so the `System`-only filter must not match it — proves role discrimination. assert!( harness .assert_system_prompt_contains("hi there") diff --git a/tests/integration/group_approvals/main.rs b/tests/integration/group_approvals/main.rs new file mode 100644 index 00000000000..dc25e15de31 --- /dev/null +++ b/tests/integration/group_approvals/main.rs @@ -0,0 +1,141 @@ +//! Group integration tests for the Reborn approval flow — the real gate path. +//! +//! One sequential `#[tokio::test]` drives eight scenarios over a shared +//! [`RebornIntegrationGroup::live_approvals`] group (one approval-request store, +//! one capability-lease store, one `(tenant, user)` auto-approve toggle, all +//! shared across threads). See `tests/integration/CLAUDE.md` §"Group tests". +//! +//! Every scenario drives the REAL gate path: scripted `builtin.write_file` call +//! → real `TurnStatus::BlockedApproval` gate (auto-approve disabled for the +//! group at construction) → real `ApprovalResolver` (`approve_gate`/`deny_gate`) +//! → `coordinator.resume_turn`. Only the model is faked. Exception: +//! `failure_category_demasked` drives a genuinely-FAILED run (no gate) to prove +//! the loop-exit de-mask wiring. +//! +//! ## Ordering (state machine over the shared auto-approve store) +//! +//! Independent gate scenarios run first while auto-approve is OFF (the control +//! proving gates are real): `gate_then_approve`, `gate_then_deny`, +//! `concurrent_dual_gate_resume` (HEADLINE, Option P — two threads parked on +//! `BlockedApproval` simultaneously on the shared `TurnCoordinator`, resolved +//! independently by `run_id`), `failure_category_demasked`, +//! `gate_ref_edge_cases::{stale_gate_ref_resume, missing_gate_bare_resolve}` +//! (C-DENYEDGE rows 7 & 10), `approval_request_persists_after_reopen` +//! (C-DURABLE). Then `approve_always_persists_cross_thread` (HEADLINE) flips +//! the toggle ON and MUST run before `ask_each_time_resumes_once` +//! (W4-ASK-EACH-ONCE, #5306 class), which installs a persistent group-wide +//! `AskEachTime` override on `builtin.write_file` and so must run LAST. + +#[allow(dead_code)] +#[path = "../support/mod.rs"] +mod reborn_support; +#[allow(dead_code)] +#[path = "../../support/mod.rs"] +mod support; + +mod scenario_approval_request_persists_after_reopen; +mod scenario_approve_always_persists_cross_thread; +mod scenario_ask_each_time_resumes_once; +mod scenario_concurrent_dual_gate_resume; +mod scenario_failure_category_demasked; +mod scenario_gate_ref_edge_cases; +mod scenario_gate_then_approve; +mod scenario_gate_then_deny; + +use reborn_support::builder::StorageMode; +use reborn_support::group::{RebornIntegrationGroup, ScenarioReport}; + +#[tokio::test] +async fn approvals_group_e2e() { + let g = RebornIntegrationGroup::live_approvals() + .await + .expect("group builds"); + + let mut report = ScenarioReport::new(); + // Independent gate scenarios, run while auto-approve is still OFF (see + // module doc's ordering section). + report.record( + "gate_then_approve", + scenario_gate_then_approve::run(&g).await, + ); + report.record("gate_then_deny", scenario_gate_then_deny::run(&g).await); + report.record( + "concurrent_dual_gate_resume", + scenario_concurrent_dual_gate_resume::run(&g).await, + ); + report.record( + "failure_category_demasked", + scenario_failure_category_demasked::run(&g).await, + ); + report.record( + "stale_gate_ref_resume", + scenario_gate_ref_edge_cases::stale_gate_ref_resume(&g).await, + ); + report.record( + "missing_gate_bare_resolve", + scenario_gate_ref_edge_cases::missing_gate_bare_resolve(&g).await, + ); + // C-DURABLE: approval-request store is always on-disk regardless of the + // group's `StorageMode`, so no `StorageMode::LibSql` variant is needed. + report.record( + "approval_request_persists_after_reopen", + scenario_approval_request_persists_after_reopen::run(&g).await, + ); + // Dependent: must run last (flips the (tenant, user) auto-approve toggle ON). + scenario_approve_always_persists_cross_thread::run(&g) + .await + .expect("approve-always persists cross-thread"); + // W4-ASK-EACH-ONCE: must run after every other `builtin.write_file` + // scenario above -- installs a persistent group-wide `AskEachTime` + // override that would otherwise force-gate their plain-Ask-mode writes. + report.record( + "ask_each_time_resumes_once", + scenario_ask_each_time_resumes_once::run(&g).await, + ); + report.assert_all_passed(); +} + +#[tokio::test] +async fn approvals_group_libsql_e2e() { + let g = RebornIntegrationGroup::builder() + .storage(StorageMode::LibSql) + .live_approvals() + .await + .expect("group builds"); + + let mut report = ScenarioReport::new(); + // Independent gate scenarios, run while auto-approve is still OFF (see + // module doc's ordering section). + report.record( + "gate_then_approve", + scenario_gate_then_approve::run(&g).await, + ); + report.record("gate_then_deny", scenario_gate_then_deny::run(&g).await); + report.record( + "concurrent_dual_gate_resume", + scenario_concurrent_dual_gate_resume::run(&g).await, + ); + report.record( + "failure_category_demasked", + scenario_failure_category_demasked::run(&g).await, + ); + report.record( + "stale_gate_ref_resume", + scenario_gate_ref_edge_cases::stale_gate_ref_resume(&g).await, + ); + report.record( + "missing_gate_bare_resolve", + scenario_gate_ref_edge_cases::missing_gate_bare_resolve(&g).await, + ); + // Dependent: must run last (flips the (tenant, user) auto-approve toggle ON). + scenario_approve_always_persists_cross_thread::run(&g) + .await + .expect("approve-always persists cross-thread"); + // W4-ASK-EACH-ONCE: must run after every other `builtin.write_file` + // scenario above -- see the non-libsql variant's comment above. + report.record( + "ask_each_time_resumes_once", + scenario_ask_each_time_resumes_once::run(&g).await, + ); + report.assert_all_passed(); +} diff --git a/tests/reborn_group_approvals/scenario_approval_request_persists_after_reopen.rs b/tests/integration/group_approvals/scenario_approval_request_persists_after_reopen.rs similarity index 71% rename from tests/reborn_group_approvals/scenario_approval_request_persists_after_reopen.rs rename to tests/integration/group_approvals/scenario_approval_request_persists_after_reopen.rs index e61bc6bc18b..daef3c54232 100644 --- a/tests/reborn_group_approvals/scenario_approval_request_persists_after_reopen.rs +++ b/tests/integration/group_approvals/scenario_approval_request_persists_after_reopen.rs @@ -1,16 +1,12 @@ //! C-DURABLE: a pending approval request survives an independent reopen of //! the approval-request store at the SAME on-disk local-dev `storage_root` — -//! proving capability-produced approval state persists to disk, not just to -//! in-memory state. Parallels `assert_reply_persists_after_reopen` (thread -//! history) and `reborn_integration_durable.rs` (extension installs) for the -//! approval-request store. +//! proving approval state persists to disk, not just in memory. Parallels +//! `assert_reply_persists_after_reopen` (thread history) and +//! `reborn_integration_durable.rs` (extension installs). //! -//! Raises a real `BlockedApproval` gate (auto-approve is disabled for the -//! group), reopens a FRESH `ApprovalRequestStore` at the capability harness's -//! `storage_root_for_test()`, and asserts the `Pending` record is there — -//! independent of the live `Arc` the running group holds. Then resolves the -//! gate through the normal path so the scenario leaves no run permanently -//! blocked. +//! Raises a real `BlockedApproval` gate, reopens a FRESH `ApprovalRequestStore` +//! at the same root, and asserts the `Pending` record is there independent of +//! the live `Arc` the running group holds, then resolves the gate normally. use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; use super::reborn_support::reply::RebornScriptedReply; @@ -39,8 +35,8 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { .ok_or("live_approvals always uses HostRuntime")?; let (request_id, scope) = capability_harness.approval_request_scope_for_test(&gate_ref)?; - // Reopen a FRESH, independent store at the same on-disk root — not the live - // `Arc` the running group holds — and confirm the pending request is there. + // Reopen a FRESH store at the same on-disk root, independent of the live + // `Arc`, and confirm the pending request is there. let reopened = ironclaw_reborn_composition::test_support::open_local_dev_approval_request_store_for_test( &capability_harness.storage_root_for_test(), diff --git a/tests/reborn_group_approvals/scenario_approve_always_persists_cross_thread.rs b/tests/integration/group_approvals/scenario_approve_always_persists_cross_thread.rs similarity index 65% rename from tests/reborn_group_approvals/scenario_approve_always_persists_cross_thread.rs rename to tests/integration/group_approvals/scenario_approve_always_persists_cross_thread.rs index 8687ecf4aa9..1a3f4c00ad9 100644 --- a/tests/reborn_group_approvals/scenario_approve_always_persists_cross_thread.rs +++ b/tests/integration/group_approvals/scenario_approve_always_persists_cross_thread.rs @@ -2,18 +2,15 @@ //! DIFFERENT thread of the same group. //! //! Thread A flips the per-`(tenant, user)` auto-approve toggle ON via the real -//! CAS-persisted `AutoApproveSettingStore` (shared across the group). Thread B — -//! a distinct conversation/thread — then runs the SAME gated capability and -//! completes WITHOUT blocking, because it reads thread A's persisted setting. -//! This is the exact user requirement: set approve-always once, and subsequent -//! threads invoking the same tool are not prompted. +//! CAS-persisted `AutoApproveSettingStore`. Thread B, a distinct +//! conversation/thread, then runs the SAME gated capability and completes +//! WITHOUT blocking because it reads thread A's persisted setting. //! -//! Non-vacuity: the sibling `gate_then_approve`/`gate_then_deny` scenarios run -//! FIRST (auto-approve still OFF) and prove the gate genuinely fires — so thread -//! B's no-gate completion here is the setting flip, not a vacuous pass. The -//! `submit_turn` call waits for `Completed`; if the setting had NOT crossed the -//! thread boundary, the write would block and `submit_turn` would time out and -//! error. +//! Non-vacuity: sibling `gate_then_approve`/`gate_then_deny` run FIRST +//! (auto-approve still OFF) and prove the gate genuinely fires, so thread B's +//! no-gate completion here is the setting flip, not a vacuous pass — if the +//! setting had not crossed the thread boundary, `submit_turn` would time out +//! waiting for `Completed` instead of returning. use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; use super::reborn_support::reply::RebornScriptedReply; diff --git a/tests/reborn_group_approvals/scenario_ask_each_time_resumes_once.rs b/tests/integration/group_approvals/scenario_ask_each_time_resumes_once.rs similarity index 54% rename from tests/reborn_group_approvals/scenario_ask_each_time_resumes_once.rs rename to tests/integration/group_approvals/scenario_ask_each_time_resumes_once.rs index 9ce54047396..921d1b987b1 100644 --- a/tests/reborn_group_approvals/scenario_ask_each_time_resumes_once.rs +++ b/tests/integration/group_approvals/scenario_ask_each_time_resumes_once.rs @@ -3,20 +3,15 @@ //! `BlockedApproval` gate, and approving it resumes the run to `Completed` in //! ONE round trip -- it must NOT re-gate the just-approved resume. //! -//! Pre-#5306, `require_approval_for_profile_policy` checked the explicit -//! `ask_each_time` override (and the "hard floor" force-approval class) -//! BEFORE consulting the matching one-shot approval lease a resume carries. -//! So a resumed dispatch for an `AskEachTime`-overridden capability hit the -//! `ask_each_time` branch first and gated AGAIN, ignoring the lease that was -//! just issued for exactly this invocation -- the run could never reach -//! `Completed` (an unresumable BlockedApproval loop). The fix reordered the -//! one-shot-lease check to run FIRST. +//! Regression: pre-#5306, `require_approval_for_profile_policy` checked the +//! `ask_each_time` override BEFORE consulting the resume's one-shot approval +//! lease, so a resumed dispatch re-hit the `ask_each_time` branch and gated +//! AGAIN -- an unresumable `BlockedApproval` loop. Fixed by checking the +//! lease first. //! -//! `live_approvals()`'s plain `write_file`/`read_file` @ `PermissionMode::Ask` -//! gate (exercised by `scenario_gate_then_approve.rs`) does NOT reach the -//! `ask_each_time` branch at all (no override is installed there), so it -//! cannot exercise this ordering bug -- this scenario installs the override -//! explicitly via `set_ask_each_time_override_for_test` before submitting. +//! `live_approvals()`'s plain Ask-mode gate (`scenario_gate_then_approve.rs`) +//! never reaches the `ask_each_time` branch, so this scenario installs the +//! override explicitly via `set_ask_each_time_override_for_test`. use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; use super::reborn_support::reply::RebornScriptedReply; @@ -37,9 +32,8 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { .build() .await?; - // Install the explicit AskEachTime override for this run's real dispatch - // (tenant, user) -- the same scope key `require_approval_for_profile_policy` - // reads `tool_override` under. + // Install the explicit AskEachTime override under the same (tenant, user) + // scope key `require_approval_for_profile_policy` reads `tool_override` under. g.capability_harness() .ok_or("live_approvals always uses HostRuntime")? .set_ask_each_time_override_for_test( @@ -55,11 +49,8 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { .submit_turn_until_blocked("write the ask-each-time file") .await?; - // Approve through the real resolver + resume. This is the discriminating - // assertion: pre-#5306, the resumed dispatch re-hits the `ask_each_time` - // branch and re-blocks (the run never reaches `Completed`, so - // `wait_for_status(Completed)` times out against the still-`BlockedApproval` - // status instead of returning `Ok`). + // Discriminating assertion: pre-#5306 the resumed dispatch re-blocks, so + // `wait_for_status(Completed)` times out instead of returning `Ok`. h.approve_gate(run_id, &gate_ref).await?; h.wait_for_status(run_id, TurnStatus::Completed).await?; @@ -68,11 +59,9 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { h.assert_workspace_file_contains("ask-each-time.txt", "ask each time payload") .await?; - // "Resumes exactly once" companion proof: the SAME gate_ref is already - // resolved (Approved) from the single resume above, so a second approve - // on it must fail NotPending, not succeed against a fresh re-raised gate - // (which would indicate a second, distinct BlockedApproval was silently - // created and resolved somewhere in between). + // "Resumes exactly once" companion proof: re-approving the same + // already-resolved gate_ref must fail NotPending, not succeed against a + // silently re-raised gate. let err = h.approve_gate(run_id, &gate_ref).await.err().ok_or( "expected err: re-approving the already-resolved ask-each-time gate must fail", diff --git a/tests/integration/group_approvals/scenario_concurrent_dual_gate_resume.rs b/tests/integration/group_approvals/scenario_concurrent_dual_gate_resume.rs new file mode 100644 index 00000000000..be7d40937d2 --- /dev/null +++ b/tests/integration/group_approvals/scenario_concurrent_dual_gate_resume.rs @@ -0,0 +1,118 @@ +//! HEADLINE scenario for Option P: two threads simultaneously parked on the +//! group's ONE shared `TurnCoordinator`, resolved independently with opposite +//! dispositions, asserting resume is dispatched by `run_id` and neither run's +//! resume disturbs the other's pending gate or terminal state. +//! +//! Scope: `GroupSharedStorage` gives the whole group ONE workspace root, so +//! this does NOT prove per-thread workspace isolation — confirmed +//! empirically: an added `thread_b.assert_workspace_file_absent("concurrent_a.txt")` +//! check failed because A's write is visible through B's handle on the +//! shared workspace. Only resume-by-`run_id` is asserted here. +//! +//! New scenario needed (consolidate-don't-proliferate, CLAUDE.md "Testing +//! Discipline"): the sibling gate scenarios resolve one thread's gate before +//! the next submits, so at most one run is ever `Blocked` at a time — that +//! can't distinguish "resume keys on `run_id`" from "resume keys on +//! registration order". This scenario puts two runs in `Blocked` state on +//! the shared coordinator SIMULTANEOUSLY, then resolves them with opposite +//! dispositions. + +use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; +use super::reborn_support::reply::RebornScriptedReply; +use ironclaw_turns::TurnStatus; +use serde_json::json; + +pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { + // Two distinct threads, two distinct gated writes, different target files + // so each run's own disposition (A approved, B denied) is independently + // verifiable on disk (see module note on shared-workspace scope). + let thread_a = g + .thread("conv-concurrent-dual-gate-a") + .script([ + RebornScriptedReply::tool_call( + "builtin.write_file", + json!({"path": "/workspace/concurrent_a.txt", "content": "thread A approved"}), + ), + RebornScriptedReply::text("A: write approved"), + ]) + .build() + .await?; + let thread_b = g + .thread("conv-concurrent-dual-gate-b") + .script([ + RebornScriptedReply::tool_call( + "builtin.write_file", + json!({"path": "/workspace/concurrent_b.txt", "content": "thread B should not persist"}), + ), + RebornScriptedReply::text("B: write was not authorized"), + ]) + .build() + .await?; + + // Submit both turns and drive both to `BlockedApproval`, then resolve them + // one at a time. The essential coverage is COEXISTENCE: two distinct runs + // are simultaneously parked on the ONE shared coordinator, each then + // resolved by its own `run_id`. We deliberately do NOT `tokio::join!` the + // submit/resume calls -- truly parallel read-modify-write turns against + // the shared CAS turn-state store is a separate, prod-relevant concern + // orthogonal to what this scenario proves. + let (run_a, gate_a) = thread_a + .submit_turn_until_blocked("write the concurrent A file") + .await?; + let (run_b, gate_b) = thread_b + .submit_turn_until_blocked("write the concurrent B file") + .await?; + + // Non-vacuity: catches the harness collapsing both threads onto one run + // before the resolution step even starts. + if run_a == run_b { + return Err(format!("expected distinct run ids, both runs were {run_a}").into()); + } + if gate_a.as_str() == gate_b.as_str() { + return Err(format!("expected distinct gate refs, both were {gate_a:?}").into()); + } + + // Resolve independently with OPPOSITE dispositions. APPROVE A first while + // B is STILL blocked -- if resume keyed on anything other than `run_id`, + // approving A would disturb B's still-pending gate. + thread_a.approve_gate(run_a, &gate_a).await?; + let state_a = thread_a + .wait_for_status(run_a, TurnStatus::Completed) + .await?; + + // Now DENY B. Its gate must still be intact and independently resolvable. + thread_b.deny_gate(run_b, &gate_b).await?; + let state_b = thread_b + .wait_for_status(run_b, TurnStatus::Completed) + .await?; + + // No error-category leakage: cross-resume bugs here historically surface + // as `driver_protocol_violation` or `TraceLlm exhausted` (wrong thread's + // scripted-reply deque drained) -- assert neither leaked onto either run. + for (label, state) in [("A", &state_a), ("B", &state_b)] { + if let Some(failure) = &state.failure { + return Err(format!( + "thread {label} run reached Completed but recorded a failure \ + (no error category should leak on a clean concurrent resume): {failure:?}" + ) + .into()); + } + } + + // A's gate was APPROVED: the write landed on disk under A's own path. + thread_a + .assert_workspace_file_contains("concurrent_a.txt", "thread A approved") + .await?; + // B's gate was DENIED: its file must be absent -- would exist if A's + // approval cross-bled onto B's run. + thread_b + .assert_workspace_file_absent("concurrent_b.txt") + .await?; + // Re-checked via `thread_a`'s handle (shared workspace): rules out a + // resume path keyed on "whichever gate resolved last" instead of run_id. + thread_a + .assert_workspace_file_absent("concurrent_b.txt") + .await?; + + Ok(()) +} diff --git a/tests/integration/group_approvals/scenario_failure_category_demasked.rs b/tests/integration/group_approvals/scenario_failure_category_demasked.rs new file mode 100644 index 00000000000..062e3d8a5bc --- /dev/null +++ b/tests/integration/group_approvals/scenario_failure_category_demasked.rs @@ -0,0 +1,69 @@ +//! Scenario: a genuinely-FAILED group run reports its TRUE failure category, +//! not the masking `driver_protocol_violation`. +//! +//! `RebornIntegrationGroupBuilder::into_group` (`tests/integration/support/group.rs` +//! ~line 472) wires `.with_checkpoint_state_store(..)` onto the group-level +//! `ThreadCheckpointLoopExitEvidencePort` -- the de-mask fix. Without it, +//! `verify_failure_evidence` (`crates/ironclaw_reborn/src/loop_exit_applier.rs`) +//! short-circuits `Ok(false)` on a `None` store, so `validate_failed_exit` +//! (`crates/ironclaw_turns/src/loop_exit.rs`) rewrites every `Failed` exit to +//! the opaque `"driver_protocol_violation"` category. With the store wired, a +//! verified failed exit's TRUE category survives onto `TurnRunState::failure`. +//! +//! New scenario needed: every other scenario in this binary asserts clean +//! completion/gate resolution and never drives a run to `TurnStatus::Failed`, +//! so none exercises `verify_failure_evidence`; the `loop_exit_applier` unit +//! tests cover the fixture but not `group.rs`'s end-to-end wiring. +//! +//! Failure is produced deterministically: an EMPTY scripted-reply list +//! (`.script([])`) makes `TraceLlm::next_step` return `LlmError::RequestFailed` +//! on the first model call; once the bounded-retry `RecoveryStrategy` budget +//! is exhausted, the loop emits `LoopExit::Failed` with +//! `LoopFailureKind::ModelError` (category `"model_error"`). + +use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; +use ironclaw_turns::TurnStatus; + +pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { + // No scripted replies: the first model call exhausts the scripted + // provider, deterministically driving the run to `Failed` (see module doc). + let h = g + .thread("conv-failure-category-demasked") + .script([]) + .build() + .await?; + + let run_id = h.submit_turn_async("trigger a model failure").await?; + let state = h.wait_for_status(run_id, TurnStatus::Failed).await?; + + let failure = state + .failure + .as_ref() + .ok_or("run reached Failed but TurnRunState::failure was None")?; + + // The de-mask fix's entire point (see module doc): the TRUE category + // must survive, not the masking sentinel. + if failure.category() == "driver_protocol_violation" { + return Err(format!( + "failure category was the masking sentinel \"driver_protocol_violation\"; \ + the group-level checkpoint_state_store wiring (group.rs ~line 472) is not \ + de-masking the real failure category (got: {failure:?})" + ) + .into()); + } + + // Empirically confirmed: `TraceLlm` exhaustion surfaces as + // `LoopFailureKind::ModelError` -> category `"model_error"` + // (`crates/ironclaw_turns/src/loop_exit.rs`). Asserting the exact value + // proves the de-masked path produces the loop's REAL reason, not just + // any non-sentinel category. + if failure.category() != "model_error" { + return Err(format!( + "expected de-masked failure category \"model_error\" (TraceLlm exhaustion -> \ + LoopFailureKind::ModelError), got {failure:?}" + ) + .into()); + } + + Ok(()) +} diff --git a/tests/reborn_group_approvals/scenario_gate_ref_edge_cases.rs b/tests/integration/group_approvals/scenario_gate_ref_edge_cases.rs similarity index 83% rename from tests/reborn_group_approvals/scenario_gate_ref_edge_cases.rs rename to tests/integration/group_approvals/scenario_gate_ref_edge_cases.rs index 4a65cb28f92..d39b8eb7846 100644 --- a/tests/reborn_group_approvals/scenario_gate_ref_edge_cases.rs +++ b/tests/integration/group_approvals/scenario_gate_ref_edge_cases.rs @@ -57,10 +57,8 @@ pub async fn stale_gate_ref_resume(g: &RebornIntegrationGroup) -> HarnessResult< ); } - // Non-vacuity follow-up: the run is STILL blocked (the failed stale-ref - // resume never reached the point of clearing the gate) — resuming with - // the REAL gate ref (resume-only; the approval was already resolved - // above) completes the run normally. + // Non-vacuity: the run is STILL blocked (stale-ref resume never cleared + // the gate); resuming with the REAL gate ref completes it normally. h.resume_gate(run_id, &gate_ref).await?; h.wait_for_status(run_id, TurnStatus::Completed).await?; h.assert_workspace_file_contains("stale_ref.txt", "stale ref write") @@ -78,12 +76,9 @@ pub async fn missing_gate_bare_resolve(g: &RebornIntegrationGroup) -> HarnessRes // No tool call, so no gate is ever raised on this thread. let run_id = h.submit_turn("just say hello, no tools").await?; - // Syntactically well-formed (`"gate:approval-"` shape, matching what - // `approval_request_id_from_gate_ref` parses) but NEVER-ISSUED gate ref — - // no capability call on this thread ever recorded this request id in the - // harness's `pending_approval_scopes` bookkeeping. `approve_gate`'s - // local-dev resolve step fails on this lookup before `resume_run` is ever - // reached, so the (already-`Completed`) `run_id` is never touched. + // Syntactically well-formed but NEVER-ISSUED gate ref: no capability call + // on this thread recorded this request id, so `approve_gate`'s local-dev + // resolve fails on the lookup before `resume_run` is ever reached. let bogus_gate_ref = GateRef::new("gate:approval-11111111-1111-1111-1111-111111111111") .expect("valid bounded gate ref string"); diff --git a/tests/reborn_group_approvals/scenario_gate_then_approve.rs b/tests/integration/group_approvals/scenario_gate_then_approve.rs similarity index 71% rename from tests/reborn_group_approvals/scenario_gate_then_approve.rs rename to tests/integration/group_approvals/scenario_gate_then_approve.rs index 6f96583c048..d126188624b 100644 --- a/tests/reborn_group_approvals/scenario_gate_then_approve.rs +++ b/tests/integration/group_approvals/scenario_gate_then_approve.rs @@ -33,21 +33,14 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { // Approve through the real resolver + resume; the gated capability re-runs. h.approve_gate(run_id, &gate_ref).await?; h.wait_for_status(run_id, TurnStatus::Completed).await?; - // The approved write actually re-ran AND PERSISTED: the real file on disk - // holds the written content. This proves approve→resume re-dispatched the - // gated capability and the write took effect — not merely that the scripted - // reply was emitted (`builtin.write_file`'s result does not echo content). + // The approved write actually re-ran AND PERSISTED to disk -- not merely + // that the scripted reply was emitted. h.assert_workspace_file_contains("approved.txt", "approved write") .await?; - // Double-resolve regression guard (C-DENYEDGE row 6): approving the SAME - // already-`Completed` gate a second time must fail loudly, not silently - // no-op or hang. The approval record is already `Approved` from the first - // `approve_gate` call above, so this second call's `approve_local_dev_gate` - // hits `ApprovalResolver::approve_capability_action`'s - // `record.status != Pending` check and returns - // `ApprovalResolutionError::NotPending { status: Approved }` before the - // resume/coordinator layer is even reached. + // Double-resolve regression guard (C-DENYEDGE row 6): re-approving the + // already-`Approved` gate must fail loudly (`NotPending`), not silently + // no-op or hang. let err = h .approve_gate(run_id, &gate_ref) .await diff --git a/tests/reborn_group_approvals/scenario_gate_then_deny.rs b/tests/integration/group_approvals/scenario_gate_then_deny.rs similarity index 65% rename from tests/reborn_group_approvals/scenario_gate_then_deny.rs rename to tests/integration/group_approvals/scenario_gate_then_deny.rs index b0ee6a117a1..7e5a2397667 100644 --- a/tests/reborn_group_approvals/scenario_gate_then_deny.rs +++ b/tests/integration/group_approvals/scenario_gate_then_deny.rs @@ -27,18 +27,14 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { let (run_id, gate_ref) = h.submit_turn_until_blocked("write the denied file").await?; - // Deny + resume. Real guard (non-vacuous): the deny→resume pipeline drives the - // run to terminal `Completed` (the model sees a non-retryable authorization - // failure and finalizes a reply) — it would hang/Fail if deny or resume were - // broken. (The scripted final reply text is NOT asserted: the scripted model - // emits it unconditionally, so it would not discriminate.) + // Non-vacuous: the deny→resume pipeline must drive the run to terminal + // `Completed`, not hang/Fail. (Scripted final reply text is not asserted + // -- the model emits it unconditionally, so it wouldn't discriminate.) h.deny_gate(run_id, &gate_ref).await?; h.wait_for_status(run_id, TurnStatus::Completed).await?; - // The denied write must NOT have executed: unlike the approve path, the gated - // capability is never re-dispatched, so the target file is never created. We - // assert the real persisted state — the file is absent on disk — which proves - // deny blocked the side effect, not merely that the run terminated. + // The denied write must NOT have executed: the gated capability is never + // re-dispatched, so the file is absent on disk. h.assert_workspace_file_absent("denied.txt").await?; Ok(()) } diff --git a/tests/reborn_group_extensions/main.rs b/tests/integration/group_extensions/main.rs similarity index 52% rename from tests/reborn_group_extensions/main.rs rename to tests/integration/group_extensions/main.rs index c37da0d1579..b0a4d4260e4 100644 --- a/tests/reborn_group_extensions/main.rs +++ b/tests/integration/group_extensions/main.rs @@ -11,16 +11,14 @@ //! state. One orchestrating function gives deterministic ordering for free. #[allow(dead_code)] -#[path = "../support/reborn/mod.rs"] +#[path = "../support/mod.rs"] mod reborn_support; #[allow(dead_code)] -#[path = "../support/mod.rs"] +#[path = "../../support/mod.rs"] mod support; -// Scenario modules, declared alphabetically because rustfmt reorders `mod` -// declarations. The execution order — install → remove → activate — is set by -// the `report.record(...)` call sequence in `extensions_group_e2e` below, not -// by declaration order. +// Modules are alphabetical (rustfmt reorders `mod` decls); execution order is +// set by the `report.record(...)` sequence below, not declaration order. mod scenario_activate_then_active_cross_thread; mod scenario_install_then_visible_cross_thread; mod scenario_install_unknown_extension_id_fails_safely; @@ -35,39 +33,31 @@ async fn extensions_group_e2e() { .expect("group builds"); let mut report = ScenarioReport::new(); - // Scenario 1 (HEADLINE): install in thread A → search in thread B over the - // shared store. Installer must succeed before the viewer runs, so we use - // `report.record` which records the result without early-aborting. + // Scenario 1 (HEADLINE): install in thread A, search in thread B. Installer + // must succeed before the viewer runs; `report.record` avoids early-abort. report.record( "install_then_visible_cross_thread", scenario_install_then_visible_cross_thread::run(&g).await, ); - // Scenario 2: install + remove in thread A → search in thread B confirms - // the extension is no longer installed over the shared store. Independent - // of Scenario 1: Scenario 1 installs "github" and never removes it; - // Scenario 2 installs + removes "notion" so it is self-contained and does - // not depend on Scenario 1's shared-store state. + // Scenario 2: install+remove in thread A, search in thread B. Uses "notion" + // (not "github", which Scenario 1 never removes) to stay self-contained. report.record( "remove_then_absent_cross_thread", scenario_remove_then_absent_cross_thread::run(&g).await, ); - // Scenario 3: install in thread A → activate in thread B → search in thread C - // confirms the extension reports `installation_phase:active` over the shared - // store. Closes the `extension_activate` int-tier gap. Independent of - // Scenarios 1 & 2: it uses "web-access" (the only credential-free bundled - // extension), untouched by "github"/"notion", so it is self-contained. + // Scenario 3: install → activate → search across three threads; closes the + // extension_activate int-tier gap. Uses "web-access" (credential-free, + // untouched by Scenarios 1-2) to stay self-contained. report.record( "activate_then_active_cross_thread", scenario_activate_then_active_cross_thread::run(&g).await, ); - // Scenario 4 (W4-EXT-MANIFEST-ERR, narrowed): an extension_id absent from - // the bundled catalog fails `builtin.extension_install` safely with a - // model-visible `Failed{InputEncode}` tool error. Independent of Scenarios - // 1-3: it uses a nonexistent id, touching no shared-store state any other - // scenario reads. + // Scenario 4 (W4-EXT-MANIFEST-ERR): an unknown extension_id fails + // `builtin.extension_install` safely with `Failed{InputEncode}`. Uses a + // nonexistent id, touching no shared-store state other scenarios read. report.record( "install_unknown_extension_id_fails_safely", scenario_install_unknown_extension_id_fails_safely::run(&g).await, diff --git a/tests/integration/group_extensions/scenario_activate_then_active_cross_thread.rs b/tests/integration/group_extensions/scenario_activate_then_active_cross_thread.rs new file mode 100644 index 00000000000..8522ac114ef --- /dev/null +++ b/tests/integration/group_extensions/scenario_activate_then_active_cross_thread.rs @@ -0,0 +1,121 @@ +//! Scenario 3 (HEADLINE): install in thread A, ACTIVATE in thread B, confirm +//! thread C observes ACTIVE (not merely installed) over the shared store — +//! closes the `extension_activate` int-tier gap (install/search/remove already +//! had cross-thread coverage). +//! +//! Uses "web-access": the only bundled extension that activates without +//! credentials (others raise an auth gate — see +//! `local_dev_extension_activate_returns_auth_gate_for_missing_extension_credentials` +//! in `extension_lifecycle_capabilities.rs`), and it's untouched by Scenarios +//! 1-2 ("github"/"notion"), so this is a genuine fresh transition. +//! +//! Per `extension_lifecycle.rs::commit_activation` / `search_installation_phase`: +//! a successful activate yields `"activated":true` + `visible_capability_ids`; +//! `extension_search` renders an active extension as +//! `"installation_phase":"active"` vs `"installed"` for installed-but-inactive. + +use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; +use super::reborn_support::reply::RebornScriptedReply; +use serde_json::json; + +pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { + // ── Thread A: installer ───────────────────────────────────────────────── + let installer = g + .thread("ext-activate-phase-install") + .script([ + RebornScriptedReply::tool_call( + "builtin.extension_install", + json!({"extension_id": "web-access"}), + ), + RebornScriptedReply::text("installed"), + ]) + .build() + .await?; + installer.submit_turn("install web-access").await?; + installer + .assert_tool_invoked("builtin.extension_install") + .await?; + installer + .assert_tool_result_contains("\"installed\":true") + .await?; + + // ── Thread B: activator (DIFFERENT conversation, SAME shared store) ────── + let activator = g + .thread("ext-activate-phase-activate") + .script([ + RebornScriptedReply::tool_call( + "builtin.extension_activate", + json!({"extension_id": "web-access"}), + ), + RebornScriptedReply::text("activated"), + ]) + .build() + .await?; + activator.submit_turn("activate web-access").await?; + activator + .assert_tool_invoked("builtin.extension_activate") + .await?; + // Assert the VALUE, not just the key, so an `activated:false` / auth-gate + // outcome cannot satisfy this. + activator + .assert_tool_result_contains("\"activated\":true") + .await?; + // `web-access.search` coming online is the observable proof that activation + // published the tool surface (mere install does NOT publish capabilities). + activator + .assert_tool_result_contains(r#""web-access.search""#) + .await?; + + // ── Thread C: viewer (DIFFERENT conversation, SAME shared store) ───────── + let viewer = g + .thread("ext-activate-phase-viewer") + .script([ + RebornScriptedReply::tool_call( + "builtin.extension_search", + json!({"query": "web-access"}), + ), + RebornScriptedReply::text("searched"), + ]) + .build() + .await?; + viewer + .submit_turn("search web-access after activation") + .await?; + viewer + .assert_tool_invoked("builtin.extension_search") + .await?; + // Assert the VALUE so a still-`installed` phase cannot satisfy this. + viewer + .assert_tool_result_contains(r#""installation_phase":"active""#) + .await?; + + // Discriminating guard: a no-op activate would still surface "installed" + // here; its absence proves the phase genuinely advanced. + if viewer + .assert_tool_result_contains(r#""installation_phase":"installed""#) + .await + .is_ok() + { + return Err( + "web-access still shows installation_phase:installed after a cross-thread activate; \ + builtin.extension_activate did not advance the lifecycle through the shared store" + .into(), + ); + } + + // Non-vacuity guard: catalog entry must still appear, so the active-phase + // assertion above isn't vacuously true on an empty/errored result. + if viewer + .assert_tool_result_contains("\"web-access\"") + .await + .is_err() + { + return Err( + "non-vacuity guard failed: web-access catalog entry must appear in search results \ + after activation; the active-phase assertion would otherwise be vacuous" + .into(), + ); + } + + Ok(()) +} diff --git a/tests/reborn_group_extensions/scenario_install_then_visible_cross_thread.rs b/tests/integration/group_extensions/scenario_install_then_visible_cross_thread.rs similarity index 55% rename from tests/reborn_group_extensions/scenario_install_then_visible_cross_thread.rs rename to tests/integration/group_extensions/scenario_install_then_visible_cross_thread.rs index a85b6ba33e9..0eae387334b 100644 --- a/tests/reborn_group_extensions/scenario_install_then_visible_cross_thread.rs +++ b/tests/integration/group_extensions/scenario_install_then_visible_cross_thread.rs @@ -1,12 +1,9 @@ //! Scenario 1 (HEADLINE): install an extension in thread A; thread B (a //! DIFFERENT conversation) sees it installed over the shared store. //! -//! Thread A calls `builtin.extension_install` for "github". Thread B calls -//! `builtin.extension_search` for "github" and asserts the result carries -//! `installation_phase: "installed"` — a field that only appears in search -//! results for an already-installed extension. Because the two threads use -//! different conversation IDs but the same `Arc`, -//! this proves cross-thread extension persistence. +//! `extension_search` renders `installation_phase: "installed"` only for an +//! already-installed extension. Different conversation IDs but the same +//! `Arc` prove cross-thread persistence. use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; use super::reborn_support::reply::RebornScriptedReply; @@ -14,9 +11,6 @@ use serde_json::json; pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { // ── Thread A: installer ───────────────────────────────────────────────── - // Install the "github" extension. The installation is persisted to the - // shared HostRuntimeCapabilityHarness filesystem so subsequent threads - // see it immediately. let installer = g .thread("ext-installer") .script([ @@ -32,16 +26,11 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { installer .assert_tool_invoked("builtin.extension_install") .await?; - // Verify the install succeeded: output JSON contains `"installed":true`. installer .assert_tool_result_contains("\"installed\":true") .await?; // ── Thread B: viewer (DIFFERENT conversation, SAME shared store) ───────── - // A distinct conversation_id produces a distinct binding and thread scope, - // but the underlying `HostRuntimeCapabilityHarness` is Arc-cloned, so the - // viewer reads from the exact same extension-install store the installer - // just wrote to. let viewer = g .thread("ext-viewer") .script([ @@ -54,19 +43,14 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { viewer .assert_tool_invoked("builtin.extension_search") .await?; - // The search result carries `installation_phase: "installed"` for the github - // package only when it is already installed (before installation the field is - // absent entirely). Assert the VALUE, not just the key — a `pending`/`failed` - // phase must not satisfy this — so the check proves thread B observes thread - // A's *successful* installation over the shared store. + // Assert the VALUE, not just the key — a `pending`/`failed` phase must not + // satisfy this — so this proves thread B observes thread A's success. viewer .assert_tool_result_contains(r#""installation_phase":"installed""#) .await?; - // Committed negative guard (non-vacuity): a marker for a never-installed, - // non-existent extension must be ABSENT from the search result, so - // `assert_tool_result_contains` is proven to discriminate rather than pass - // unconditionally. + // Non-vacuity guard: a never-installed marker must be absent, proving + // `assert_tool_result_contains` discriminates rather than passing unconditionally. if viewer .assert_tool_result_contains("this-extension-does-not-exist-zzz") .await diff --git a/tests/integration/group_extensions/scenario_install_unknown_extension_id_fails_safely.rs b/tests/integration/group_extensions/scenario_install_unknown_extension_id_fails_safely.rs new file mode 100644 index 00000000000..9871fa88a57 --- /dev/null +++ b/tests/integration/group_extensions/scenario_install_unknown_extension_id_fails_safely.rs @@ -0,0 +1,47 @@ +//! Scenario 4 (W4-EXT-MANIFEST-ERR, narrowed): `builtin.extension_install` +//! with an `extension_id` not in the bundled catalog fails safely with a +//! model-visible tool error, instead of panicking or silently no-oping. +//! +//! `extension_id` resolves against a fixed, compile-time-embedded catalog +//! (`AvailableExtensionCatalog::resolve`) — every bundled manifest is +//! asset-embedded and always valid, so the originally-scoped schema/reserved-id/ +//! trust-level `ManifestV2Error` arms are unreachable through this capability in +//! production. The one reachable arm is an unknown `extension_id`: +//! `catalog.resolve` returns `InvalidBindingRequest`, mapped by +//! `extension_lifecycle_capabilities.rs::lifecycle_error` to +//! `RuntimeDispatchErrorKind::InputEncode` — the `"invalid_input"` reason token, +//! a `Failed` (not `Denied`) outcome distinct from Scenario 1's success path. + +use super::reborn_support::assertions::ToolErrorClass; +use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; +use super::reborn_support::reply::RebornScriptedReply; +use serde_json::json; + +pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { + let h = g + .thread("ext-install-unknown-id") + .script([ + RebornScriptedReply::tool_call( + "builtin.extension_install", + json!({"extension_id": "not-a-real-bundled-extension"}), + ), + RebornScriptedReply::text("could not install that extension"), + ]) + .build() + .await?; + h.submit_turn("install the not-a-real-bundled-extension extension") + .await?; + h.assert_tool_invoked("builtin.extension_install").await?; + + // Proved as a *class* (not a needle-prefix convention): the same reason + // string could in principle render under either class. + h.assert_tool_error(ToolErrorClass::Failed, "invalid_input") + .await?; + + // Discriminating negative arm: the same reason token under the OTHER class + // must be absent, proving the class argument is load-bearing here. + h.assert_no_tool_error(ToolErrorClass::Denied, "invalid_input") + .await?; + + Ok(()) +} diff --git a/tests/reborn_group_extensions/scenario_remove_then_absent_cross_thread.rs b/tests/integration/group_extensions/scenario_remove_then_absent_cross_thread.rs similarity index 52% rename from tests/reborn_group_extensions/scenario_remove_then_absent_cross_thread.rs rename to tests/integration/group_extensions/scenario_remove_then_absent_cross_thread.rs index 730b1b92b2d..096f63bb497 100644 --- a/tests/reborn_group_extensions/scenario_remove_then_absent_cross_thread.rs +++ b/tests/integration/group_extensions/scenario_remove_then_absent_cross_thread.rs @@ -1,23 +1,13 @@ //! Scenario 2 (HEADLINE): install then remove an extension in thread A; thread B //! (a DIFFERENT conversation) does NOT see it as installed over the shared store. //! -//! Uses "notion" (NOT "github") so this scenario is self-contained: Scenario 1 -//! installs "github" into the same shared store and never removes it, so a fresh -//! install→remove cycle here must use a different bundled extension to observe a -//! real install result rather than an already-installed no-op. +//! Uses "notion" (not "github", which Scenario 1 installs into this same store +//! and never removes) so the install→remove cycle is self-contained. //! -//! Thread A-install calls `builtin.extension_install` for "notion". Thread -//! A-remove (a distinct conversation) calls `builtin.extension_remove` for the -//! same package (arg shape `{"extension_id": ""}`, identical to install/ -//! activate; success output carries `"removed":true` — both confirmed from -//! `extension_lifecycle_capabilities.rs`'s unit tests). Thread B calls -//! `builtin.extension_search` for "notion" and asserts the result does NOT -//! carry `installation_phase: "installed"` — the field is entirely absent for -//! extensions that are not installed (confirmed from the same unit tests), -//! and reappears/disappears as the lifecycle state changes. Because all three -//! conversations use different conversation IDs but the same -//! `Arc`, this proves cross-thread extension -//! removal persistence: a remove in thread A is durably visible to thread B. +//! `installation_phase` is entirely absent from `extension_search` for a +//! not-installed extension (confirmed in `extension_lifecycle_capabilities.rs`'s +//! unit tests). Three threads, different conversation IDs, same +//! `Arc`, prove cross-thread removal persistence. use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; use super::reborn_support::reply::RebornScriptedReply; @@ -25,11 +15,6 @@ use serde_json::json; pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { // ── Phase 1: install "notion" ──────────────────────────────────────────── - // Install a fresh extension so there is something to remove. "notion" is - // chosen because Scenario 1 installs "github" into this same shared store - // and never removes it; re-installing an already-installed extension would - // be a no-op and produce no `"installed":true` result. "notion" is untouched - // by Scenario 1, so this is a genuine fresh install. let installer = g .thread("ext-remove-phase-install") .script([ @@ -45,15 +30,11 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { installer .assert_tool_invoked("builtin.extension_install") .await?; - // Confirm the install succeeded: output carries `"installed":true`. installer .assert_tool_result_contains("\"installed\":true") .await?; // ── Phase 2: remove "notion" (DIFFERENT conversation, SAME shared store) ─ - // A distinct conversation_id → distinct binding/thread scope, but the - // same `HostRuntimeCapabilityHarness`, so the remover can see and delete - // the installation that Phase 1 just wrote. let remover = g .thread("ext-remove-phase-remove") .script([ @@ -69,15 +50,11 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { remover .assert_tool_invoked("builtin.extension_remove") .await?; - // Confirm the real remove succeeded: capability output carries `"removed":true`. remover .assert_tool_result_contains("\"removed\":true") .await?; // ── Phase 3: cross-thread search — "notion" must NOT be installed ─────── - // A DIFFERENT conversation_id produces a distinct binding and thread scope - // but Arc-clones the same `HostRuntimeCapabilityHarness`, so the viewer - // reads from the exact same extension-install store the remover just wrote to. let viewer = g .thread("ext-remove-phase-viewer") .script([ @@ -91,9 +68,8 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { .assert_tool_invoked("builtin.extension_search") .await?; - // Absence assertion: after removal `installation_phase` reverts to absent. - // `assert_tool_result_contains` returns `Ok` when the needle IS present and - // `Err` when it is absent — invert to assert absence. + // `assert_tool_result_contains` returns `Ok` when present, `Err` when + // absent — invert to assert absence. if viewer .assert_tool_result_contains(r#""installation_phase":"installed""#) .await @@ -106,11 +82,8 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { ); } - // Non-vacuity guard: "notion" must still appear in the catalog search result - // (as an available-but-not-installed bundled extension), proving the search - // actually ran and returned catalog entries. The absence of - // `installation_phase` is therefore meaningful — not a symptom of an empty - // or errored result. + // Non-vacuity guard: catalog entry must still appear, so the absence + // assertion above isn't vacuously true on an empty/errored result. if viewer .assert_tool_result_contains("\"notion\"") .await diff --git a/tests/integration/group_journeys/main.rs b/tests/integration/group_journeys/main.rs new file mode 100644 index 00000000000..b8a59b1c3dc --- /dev/null +++ b/tests/integration/group_journeys/main.rs @@ -0,0 +1,112 @@ +//! C-JOURNEY — multi-turn Reborn journeys: deterministic twins of the live +//! canary use cases that chain gate → resume → next turn on ONE +//! conversation/harness. Distinct from `reborn_group_approvals` / +//! `reborn_integration_auth_gate` (single-gate mechanics): the value here is +//! the chaining across turns. +//! +//! | scenario | inbound | gate(s) | outcome | +//! |---|---|---|---| +//! | interactive_approval_journey | interactive | approval → approval | approve, deny, follow-up | +//! | auth_then_approval_journey | interactive | (approval→auth) → approval | approve+resolve, approve, follow-up | +//! | auth_deny_then_retry_journey | interactive | (approval→auth) → (approval→auth) | approve+deny, approve+resolve | +//! | multi_actor_gate_isolation | interactive×2 | approval (A) / approval (B) | per-actor gate + resume isolation | +//! +//! Gate-arm discipline: a gated tool-call turn consumes exactly TWO script +//! entries regardless of approve/deny; a plain follow-up turn consumes ONE. +//! +//! `auth_then_approval_journey` and `auth_deny_then_retry_journey` run on a +//! SECOND group, `RebornIntegrationGroup::live_auth_and_approval()` (built +//! from `HostRuntimeCapabilityHarness::file_and_github_auth_tools`), NOT +//! `live_approvals` above — it converges the auth gate onto the SAME +//! `build_reborn_services` runtime (unlike `live_auth_gate`'s separate, +//! lower-level `HostRuntimeServices` build with no `run_state`/ +//! `approval_requests`/`capability_leases` stores). No GitHub credential is +//! seeded at construction, so `github.get_repo` raises `BlockedApproval` then, +//! post-approve, a real `BlockedAuth`; `resolve_auth_gate` seeds a credential +//! through the real `ProductAuthRuntimeCredentialResolver` and resumes the +//! same parked capability. See +//! `HostRuntimeCapabilityHarness::file_and_github_auth_tools`'s doc comment +//! for the composition-seam mechanism this required. +//! +//! ## Deferred / blocked permutations +//! +//! - **Triggered-origin chained journey**: needs the scripted-gateway seam +//! (`RebornIntegrationHarness::submit_triggered_turn_scripted`) to reconcile +//! the trigger's minted owner scope with the journey approval helpers' +//! binding scope. Single-turn triggered-origin coverage already exists (see +//! `reborn_integration_triggered_submit`). +//! - **Multi-actor GATED journey** (`multi_actor_gate_isolation`): runs on +//! `RebornIntegrationGroup::multiuser_approvals()`, whose C-MULTIUSER +//! `scope_capability_by_run_owner` seam scopes each actor's gated write to +//! its own run owner. Plain (non-gated) distinct-actor isolation is covered +//! by `reborn_group_multiuser::two_actors_own_threads`. + +#[allow(dead_code)] +#[path = "../support/mod.rs"] +mod reborn_support; +#[allow(dead_code)] +#[path = "../../support/mod.rs"] +mod support; + +mod scenario_auth_deny_then_retry_journey; +mod scenario_auth_then_approval_journey; +mod scenario_interactive_approval_journey; +mod scenario_multi_actor_gate_isolation; + +use reborn_support::group::{RebornIntegrationGroup, ScenarioReport}; + +#[tokio::test] +async fn journeys_group_e2e() { + let g = RebornIntegrationGroup::live_approvals() + .await + .expect("group builds"); + + let mut report = ScenarioReport::new(); + report.record( + "interactive_approval_journey", + scenario_interactive_approval_journey::run(&g).await, + ); + report.assert_all_passed(); +} + +/// C-JOURNEY: auth→approval convergence journeys, on SEPARATE +/// `live_auth_and_approval` groups (not the `live_approvals` group above). +/// +/// ONE GROUP PER SCENARIO: `resolve_auth_gate` seeds a `UserReusable` GitHub +/// credential under the group's canonical scope, so a shared group would let +/// scenario 2's `github.get_repo` resolve immediately instead of raising the +/// fresh `BlockedAuth` gate it pins (verified: shared-group variant fails +/// with "expected BlockedAuth but run reached terminal status Completed"). +#[tokio::test] +async fn journeys_group_auth_convergence_e2e() { + let mut report = ScenarioReport::new(); + + let g = RebornIntegrationGroup::live_auth_and_approval() + .await + .expect("auth+approval group builds"); + report.record( + "auth_then_approval_journey", + scenario_auth_then_approval_journey::run(&g).await, + ); + + let g_deny = RebornIntegrationGroup::live_auth_and_approval() + .await + .expect("auth+approval deny group builds"); + report.record( + "auth_deny_then_retry_journey", + scenario_auth_deny_then_retry_journey::run(&g_deny).await, + ); + report.assert_all_passed(); +} + +/// The multi-actor GATED journey — see the module doc's "Multi-actor GATED +/// journey" note for the `scope_capability_by_run_owner` seam this requires. +#[tokio::test] +async fn multi_actor_gate_isolation() { + let g = RebornIntegrationGroup::multiuser_approvals() + .await + .expect("group builds"); + scenario_multi_actor_gate_isolation::run(&g) + .await + .expect("multi-actor gated journey"); +} diff --git a/tests/integration/group_journeys/scenario_auth_deny_then_retry_journey.rs b/tests/integration/group_journeys/scenario_auth_deny_then_retry_journey.rs new file mode 100644 index 00000000000..1ecb9e40aae --- /dev/null +++ b/tests/integration/group_journeys/scenario_auth_deny_then_retry_journey.rs @@ -0,0 +1,83 @@ +//! C-JOURNEY convergence scenario (companion to +//! `scenario_auth_then_approval_journey`): a single conversation whose FIRST +//! auth gate is DENIED, then a SECOND auth gate on the SAME thread is +//! RESOLVED — proves a denied auth gate does not poison a later successful +//! resolve on the same run/thread. Each turn also chains an approval gate +//! before the auth gate (see `scenario_auth_then_approval_journey`'s module +//! doc for why `github.get_repo` raises both in sequence on this harness). +//! +//! turn 1: approve -> BlockedAuth -> DENY -> Completed (no re-dispatch) +//! turn 2: fresh approve -> fresh BlockedAuth -> RESOLVE -> re-dispatches +//! for real -> Completed + +use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; +use super::reborn_support::reply::RebornScriptedReply; +use ironclaw_turns::TurnStatus; +use serde_json::json; + +pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { + let h = g + .thread("conv-journey-auth-deny-then-retry") + .script([ + // turn 1 (2 entries: approval+auth-gated call + post-deny reply) + RebornScriptedReply::tool_call( + "github.get_repo", + json!({"owner": "octocat", "repo": "hello-world"}), + ), + RebornScriptedReply::text("could not look up the repo without authorization"), + // turn 2 (2 entries: approval+auth-gated call + post-resolve reply) + RebornScriptedReply::tool_call( + "github.get_repo", + json!({"owner": "octocat", "repo": "hello-world"}), + ), + RebornScriptedReply::text( + "AUTHDENYRETRY_TURN2 repo info retrieved after connecting github", + ), + ]) + .build() + .await?; + + // --- turn 1: approve the action, then DENY the auth gate --- + let (run1, approval_gate1) = h + .submit_turn_until_blocked("AUTHDENYRETRY_TURN1 look up the repo the first time") + .await?; + h.approve_gate(run1, &approval_gate1).await?; + let auth_state1 = h.wait_for_status(run1, TurnStatus::BlockedAuth).await?; + let auth_gate1 = auth_state1 + .gate_ref + .ok_or("blocked auth run missing gate ref")?; + h.deny_auth_gate(run1, &auth_gate1).await?; + h.wait_for_status(run1, TurnStatus::Completed).await?; + // Pin WHAT the deny path produced: turn 1's own scripted reply, and zero + // network egress despite the model being told about the declined + // capability — anchors the "turn 2 result can't be turn 1 residue" + // reasoning below. + h.assert_reply_contains("could not look up the repo without authorization") + .await?; + h.assert_egress_count(0).await?; + + // --- turn 2: SAME conversation; turn 1's deny left no credential or + // stale approval, so the dispatch raises both gates fresh again --- + let (run2, approval_gate2) = h + .submit_turn_until_blocked("AUTHDENYRETRY_TURN2 look up the repo the second time") + .await?; + if run2 == run1 { + return Err("turn 2 reused turn 1's run id -- turns did not chain".into()); + } + h.approve_gate(run2, &approval_gate2).await?; + let auth_state2 = h.wait_for_status(run2, TurnStatus::BlockedAuth).await?; + let auth_gate2 = auth_state2 + .gate_ref + .ok_or("blocked auth run missing gate ref")?; + h.resolve_auth_gate(run2, &auth_gate2).await?; + h.wait_for_status(run2, TurnStatus::Completed).await?; + h.assert_reply_contains("AUTHDENYRETRY_TURN2").await?; + // Proves the post-resolve dispatch actually executed: turn 1's denied + // dispatch never produces a result, and `assert_tool_result_contains` has + // no `*_since`-scoped variant (unlike `assert_tool_error_since`), so its + // discriminating power here comes from pairing with the turn-1 negative + // arm above (zero egress + turn 1's own distinct reply) — this can only + // be satisfied by turn 2's own credential-backed dispatch. + h.assert_tool_result_contains("octocat/hello-world").await?; + Ok(()) +} diff --git a/tests/reborn_group_journeys/scenario_auth_then_approval_journey.rs b/tests/integration/group_journeys/scenario_auth_then_approval_journey.rs similarity index 56% rename from tests/reborn_group_journeys/scenario_auth_then_approval_journey.rs rename to tests/integration/group_journeys/scenario_auth_then_approval_journey.rs index 271adcc50d7..0b2a5ee90d4 100644 --- a/tests/reborn_group_journeys/scenario_auth_then_approval_journey.rs +++ b/tests/integration/group_journeys/scenario_auth_then_approval_journey.rs @@ -1,30 +1,14 @@ -//! C-JOURNEY convergence scenario: a single conversation that raises an -//! APPROVAL gate then an AUTH gate on the SAME capability call (turn 1, -//! `github.get_repo` — a real WASM capability chains BOTH gate classes: -//! the user must first consent to the action, then the missing GitHub -//! credential blocks it), followed by a plain APPROVAL gate on a different -//! capability (turn 2, `write_file`), chained by a plain follow-up (turn 3). -//! All on the SAME `HostRuntimeCapabilityHarness` runtime. Distinct from -//! `scenario_interactive_approval_journey` (approval → approval only): the -//! value here is that gate classes chain WITHIN one capability call AND -//! ACROSS turns, all resolving happily on ONE `build_reborn_services` runtime, -//! with history carried across the gate-class boundary. +//! C-JOURNEY convergence scenario: one conversation chains an APPROVAL gate +//! then an AUTH gate on the SAME capability call (turn 1, `github.get_repo`), +//! then a plain APPROVAL gate on a different capability (turn 2, `write_file`), +//! then a follow-up (turn 3) — all on ONE `HostRuntimeCapabilityHarness` +//! runtime. Distinct from `scenario_interactive_approval_journey` +//! (approval → approval only): here gate classes chain WITHIN one capability +//! call AND across turns, with history carried across the gate-class boundary. //! -//! turn 1: `github.get_repo` -> `BlockedApproval` -> APPROVE -> resumes, -//! the still-uncredentialed capability re-dispatches and blocks -//! again at `BlockedAuth` -> RESOLVE (seed a real GitHub credential -//! account through product-auth, then resume) -> the SAME parked -//! capability re-dispatches for real -> `Completed`; -//! turn 2: gated `write_file` -> `BlockedApproval` -> APPROVE -> resumes, -//! the write re-dispatches and persists -> `Completed`; -//! turn 3: a plain follow-up user message -> the model's request carries -//! BOTH turn 1's and turn 3's history -> replies -> `Completed`. -//! -//! Requires `RebornIntegrationGroup::live_auth_and_approval()` — the -//! converged group whose capability harness -//! (`HostRuntimeCapabilityHarness::file_and_github_auth_tools`) surfaces -//! both an unseeded `github.get_repo` capability and real file-tool approval -//! stores on the SAME runtime. +//! Requires `RebornIntegrationGroup::live_auth_and_approval()` — the converged +//! group whose capability harness surfaces both an unseeded `github.get_repo` +//! capability and real file-tool approval stores on the SAME runtime. use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; use super::reborn_support::reply::RebornScriptedReply; @@ -71,13 +55,9 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { } h.resolve_auth_gate(run1, &auth_gate1).await?; h.wait_for_status(run1, TurnStatus::Completed).await?; - // The parked github capability actually EXECUTED post-resolve: the scripted - // network fixture body (`octocat/hello-world`) surfaced back as a recorded - // Completed-path capability result. `assert_tool_invoked` alone is NOT - // discriminating here (mutation-verified): invocations are recorded at - // dispatch ENTRY, and a resume without credentials completes the run anyway - // by surfacing the second `AuthRequired` as a model-visible failure — only - // the result CONTENT proves the credential-backed re-dispatch really ran. + // Mutation-verified: `assert_tool_invoked` alone doesn't discriminate here + // (recorded at dispatch entry); only the scripted body surfacing back + // proves the credential-backed re-dispatch actually ran. h.assert_tool_result_contains("octocat/hello-world").await?; // --- turn 2: write_file approval gate (same conversation, next turn) --- @@ -101,10 +81,9 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { return Err("turn 3 reused a prior run id -- turns did not chain".into()); } h.assert_reply_contains("AUTH_JOURNEY_FINAL_REPLY").await?; - // The model request for turn 3 carries the earlier turns' history: ONE - // captured request contains BOTH turn 1's and turn 3's user text. Only - // turn 3's request can (turn 1's request predates turn 3), so this is a - // real context-carryover proof spanning the approval->auth->approval chain. + // Turn 3's request must contain BOTH turn 1's and turn 3's text — real + // context-carryover across the approval->auth->approval chain, not a + // tautology (only turn 3's request can, since turn 1 predates it). h.assert_model_request_contains_all(&["AUTHJOURNEY_TURN1", "AUTHJOURNEY_TURN3"]) .await?; Ok(()) diff --git a/tests/reborn_group_journeys/scenario_interactive_approval_journey.rs b/tests/integration/group_journeys/scenario_interactive_approval_journey.rs similarity index 67% rename from tests/reborn_group_journeys/scenario_interactive_approval_journey.rs rename to tests/integration/group_journeys/scenario_interactive_approval_journey.rs index eb38d1fd470..426a07afc5b 100644 --- a/tests/reborn_group_journeys/scenario_interactive_approval_journey.rs +++ b/tests/integration/group_journeys/scenario_interactive_approval_journey.rs @@ -1,21 +1,13 @@ -//! Canonical multi-turn JOURNEY over ONE conversation/harness — the deterministic -//! twin of the live approval-gate canary flow. Chains, on a single thread: -//! -//! turn 1: gated `write_file` → `BlockedApproval` → APPROVE → resumes, the -//! write re-dispatches and persists → `Completed`; -//! turn 2: gated `write_file` → `BlockedApproval` → DENY → resumes, the model -//! sees a non-retryable authorization failure, side effect suppressed -//! → `Completed`; -//! turn 3: a plain follow-up user message → the model's request carries the -//! PRIOR turns' history → replies → `Completed`. +//! Canonical multi-turn JOURNEY over ONE conversation/harness: chains, on a +//! single thread, turn 1 (gated `write_file` → APPROVE → persists), turn 2 +//! (gated `write_file` → DENY → suppressed, non-retryable auth failure), turn 3 +//! (plain follow-up carrying prior turns' history). //! //! Journey value (not re-tested here) = the CHAINING across turns on one -//! conversation: gate-resolve → next turn → gate-resolve → follow-up, all over -//! the group's ONE shared coordinator/turn-store with the SAME binding. The -//! single-gate mechanics (approve/deny/resume correctness) are already pinned by -//! `reborn_group_approvals`; this asserts they COMPOSE across a live session and -//! that per-turn state does not bleed (turn 2's deny does not un-persist turn 1's -//! approved write; turn 3 still sees both). +//! conversation over the group's ONE shared coordinator/turn-store. Single-gate +//! mechanics (approve/deny/resume) are already pinned by `reborn_group_approvals`; +//! this asserts they COMPOSE across a live session with no per-turn state bleed +//! (turn 2's deny doesn't un-persist turn 1's write; turn 3 still sees both). use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; use super::reborn_support::reply::RebornScriptedReply; @@ -75,10 +67,9 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { return Err("turn 3 reused a prior run id — turns did not chain".into()); } h.assert_reply_contains("JOURNEY_FINAL_REPLY").await?; - // The model request for turn 3 carries the earlier turns' history: ONE - // captured request contains BOTH turn 1's and turn 3's user text. Only turn - // 3's request can (turn 1's request predates turn 3), so this is a real - // context-carryover proof, not a tautology. + // Turn 3's request must carry BOTH turn 1's and turn 3's text — real + // context-carryover, not a tautology (only turn 3's request can, since + // turn 1 predates it). h.assert_model_request_contains_all(&["JOURNEY_TURN1", "JOURNEY_TURN3"]) .await?; Ok(()) diff --git a/tests/reborn_group_journeys/scenario_multi_actor_gate_isolation.rs b/tests/integration/group_journeys/scenario_multi_actor_gate_isolation.rs similarity index 79% rename from tests/reborn_group_journeys/scenario_multi_actor_gate_isolation.rs rename to tests/integration/group_journeys/scenario_multi_actor_gate_isolation.rs index 51744087b72..29660a6efc4 100644 --- a/tests/reborn_group_journeys/scenario_multi_actor_gate_isolation.rs +++ b/tests/integration/group_journeys/scenario_multi_actor_gate_isolation.rs @@ -2,16 +2,12 @@ //! actor B (via `with_actor_id`), each hitting its OWN approval gate over the //! group's ONE shared coordinator. Pins that gate resolution + resume state stay //! bound to the RAISING actor's turn: approving A's gate does NOT resolve B's, -//! and B's turn does not inherit A's approval — B still blocks on its own gate -//! and must be resolved independently under B's actor. +//! and B still blocks on its own gate, resolved independently under B's actor. //! -//! Runs on `RebornIntegrationGroup::multiuser_approvals()`, whose per-actor -//! capability dispatch (the C-MULTIUSER `scope_capability_by_run_owner` -//! harness seam) scopes each actor's gated write to ITS OWN run owner, so -//! actor B's dispatch no longer dies with `driver_protocol_violation` under -//! actor A's user. Production already isolates capability dispatch by run -//! owner correctly; this seam makes that isolation observable at the harness -//! level. +//! Runs on `RebornIntegrationGroup::multiuser_approvals()` (C-MULTIUSER +//! `scope_capability_by_run_owner` harness seam), which scopes each actor's +//! gated write to ITS OWN run owner so actor B's dispatch doesn't die with +//! `driver_protocol_violation` under actor A's user. //! //! Complementary to (not a duplicate of): `reborn_group_approvals`'s //! `concurrent_dual_gate_resume` (SAME actor, two threads parked simultaneously) @@ -100,10 +96,8 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { a.assert_workspace_file_contains("journey-actor-a.txt", "ACTOR_PAYLOAD") .await?; - // Actor B: its turn MUST still block on ITS OWN gate. If A's approval had - // leaked to B (a resolution-scope bug), B's write would auto-complete and - // `submit_turn_until_blocked` would fail fast on a terminal `Completed` - // instead of returning a gate ref — this is the load-bearing isolation pin. + // Load-bearing isolation pin: if A's approval leaked to B, B's write would + // auto-complete and this call would fail fast on `Completed` instead of a gate ref. let (run_b, gate_b) = b .submit_turn_until_blocked("ACTOR_B write the file") .await @@ -125,12 +119,9 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { // Owner isolation: neither actor's owner scope may read the other's thread // history (each owner's records live under a separate - // `/tenants//users//threads` subtree). - // - // Positive control FIRST: each actor must still be able to read its OWN - // thread's history under its OWN scope, so the negative checks below - // aren't vacuously true because `history()` never resolves anything on - // this harness. + // `/tenants//users//threads` subtree). Positive control + // FIRST: each actor must read its OWN history, so the negative checks + // below aren't vacuously true. let a_own_history = a .thread_harness .history(a.binding.thread_id.clone()) @@ -168,12 +159,10 @@ async fn assert_history_isolated( {reader_name}'s owner scope" ) .into()), - // Pin the SPECIFIC failure reason (cross-owner lookup resolves to - // `SessionThreadError::UnknownThread` — scope-relative path - // resolution, not a separate ACL check) rather than accepting any - // `Err(_)` — an unrelated failure (e.g. a driver/backend error) - // would otherwise vacuously satisfy a bare `.is_ok()` check without - // proving isolation. + // Pin the SPECIFIC failure (cross-owner lookup -> UnknownThread via + // scope-relative path resolution, not a separate ACL check) rather + // than accepting any `Err(_)`, which an unrelated backend error + // could satisfy without proving isolation. Err(RebornThreadHarnessError::Thread(SessionThreadError::UnknownThread { .. })) => Ok(()), Err(other_err) => Err(format!( "isolation check for actor {other_name} under actor {reader_name}'s scope failed \ diff --git a/tests/reborn_group_memory/main.rs b/tests/integration/group_memory/main.rs similarity index 79% rename from tests/reborn_group_memory/main.rs rename to tests/integration/group_memory/main.rs index dad4b6f94c3..8bfe7e283ab 100644 --- a/tests/reborn_group_memory/main.rs +++ b/tests/integration/group_memory/main.rs @@ -4,19 +4,16 @@ //! (one filesystem, one memory backend). State written by thread A is visible //! to thread B because both share the same underlying store — the whole point. //! -//! ## Why one sequential `#[tokio::test]` -//! -//! Each scenario's writer must complete before its reader/searcher/lister runs; -//! a shared group instance cannot be split across Cargo test cases without -//! fragile global state. One orchestrating function gives deterministic ordering -//! for free. Each scenario seeds its own data, so they are independent and the -//! ordering between scenarios does not matter. +//! One sequential `#[tokio::test]`: a shared group instance can't split across +//! Cargo test cases without fragile global state, and each scenario's +//! writer must complete before its reader/searcher/lister runs. Scenarios seed +//! their own data, so ordering between them doesn't matter. #[allow(dead_code)] -#[path = "../support/reborn/mod.rs"] +#[path = "../support/mod.rs"] mod reborn_support; #[allow(dead_code)] -#[path = "../support/mod.rs"] +#[path = "../../support/mod.rs"] mod support; mod scenario_memory_search_finds_seeded; diff --git a/tests/reborn_group_memory/scenario_memory_search_finds_seeded.rs b/tests/integration/group_memory/scenario_memory_search_finds_seeded.rs similarity index 85% rename from tests/reborn_group_memory/scenario_memory_search_finds_seeded.rs rename to tests/integration/group_memory/scenario_memory_search_finds_seeded.rs index 24aeac6af91..241bbc0cdd5 100644 --- a/tests/reborn_group_memory/scenario_memory_search_finds_seeded.rs +++ b/tests/integration/group_memory/scenario_memory_search_finds_seeded.rs @@ -35,9 +35,8 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { writer.assert_tool_invoked("builtin.memory_write").await?; // ── Thread B: searcher (DIFFERENT conversation, SAME shared store) ────── - // A natural-language query whose terms overlap the seeded sentence. The - // matched chunk's `content` snippet surfaces the marker token, proving the - // search actually located the seeded document rather than returning empty. + // Query overlaps the seeded sentence; the matched chunk's snippet must + // surface the marker token, proving search located the doc (not empty). let searcher = g .thread("conv-memory-searcher") .script([ @@ -60,10 +59,8 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { .assert_tool_result_contains("osprey-meridian-7") .await?; - // Committed negative guard (non-vacuity): a marker that was never written - // must be ABSENT from the search result, so `assert_tool_result_contains` - // is proven to discriminate rather than pass unconditionally (e.g. if the - // search silently returned every document or an empty-but-stringified set). + // Non-vacuity: an unwritten marker must be ABSENT, so the assertion above + // is proven to discriminate rather than pass unconditionally. if searcher .assert_tool_result_contains("tungsten-mirage-88") .await diff --git a/tests/reborn_group_memory/scenario_memory_tree_reflects_structure.rs b/tests/integration/group_memory/scenario_memory_tree_reflects_structure.rs similarity index 93% rename from tests/reborn_group_memory/scenario_memory_tree_reflects_structure.rs rename to tests/integration/group_memory/scenario_memory_tree_reflects_structure.rs index 4163a43f851..acc53916807 100644 --- a/tests/reborn_group_memory/scenario_memory_tree_reflects_structure.rs +++ b/tests/integration/group_memory/scenario_memory_tree_reflects_structure.rs @@ -51,9 +51,8 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { lister.assert_tool_result_contains("atlas/").await?; lister.assert_tool_result_contains("runbook.md").await?; - // Committed negative guard (non-vacuity): a directory that was never created - // must be ABSENT, so the positive assertions discriminate rather than pass - // unconditionally. + // Non-vacuity: an uncreated directory must be ABSENT, so the positive + // assertions discriminate rather than pass unconditionally. if lister.assert_tool_result_contains("phantom/").await.is_ok() { return Err("negative guard failed: tree must not contain an uncreated directory".into()); } diff --git a/tests/reborn_group_memory/scenario_write_then_read_cross_thread.rs b/tests/integration/group_memory/scenario_write_then_read_cross_thread.rs similarity index 94% rename from tests/reborn_group_memory/scenario_write_then_read_cross_thread.rs rename to tests/integration/group_memory/scenario_write_then_read_cross_thread.rs index 47d9d44719b..e54de10d59b 100644 --- a/tests/reborn_group_memory/scenario_write_then_read_cross_thread.rs +++ b/tests/integration/group_memory/scenario_write_then_read_cross_thread.rs @@ -48,8 +48,7 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { // asserting on the marker proves thread B reads thread A's write. reader.assert_tool_result_contains("plum-42").await?; - // Committed negative guard (non-vacuity): a marker that was never written - // must be ABSENT from the same read result, so `assert_tool_result_contains` + // Non-vacuity: an unwritten marker must be ABSENT, so the assertion above // is proven to discriminate rather than pass unconditionally. if reader .assert_tool_result_contains("banana-99") diff --git a/tests/reborn_group_multiuser/main.rs b/tests/integration/group_multiuser/main.rs similarity index 98% rename from tests/reborn_group_multiuser/main.rs rename to tests/integration/group_multiuser/main.rs index 9d4941d57e0..a542a63cfcf 100644 --- a/tests/reborn_group_multiuser/main.rs +++ b/tests/integration/group_multiuser/main.rs @@ -9,10 +9,10 @@ //! exercises a second owner over the shared runtime. #[allow(dead_code)] -#[path = "../support/reborn/mod.rs"] +#[path = "../support/mod.rs"] mod reborn_support; #[allow(dead_code)] -#[path = "../support/mod.rs"] +#[path = "../../support/mod.rs"] mod support; mod scenario_auto_approve_isolation_across_actors; diff --git a/tests/reborn_group_multiuser/scenario_auto_approve_isolation_across_actors.rs b/tests/integration/group_multiuser/scenario_auto_approve_isolation_across_actors.rs similarity index 68% rename from tests/reborn_group_multiuser/scenario_auto_approve_isolation_across_actors.rs rename to tests/integration/group_multiuser/scenario_auto_approve_isolation_across_actors.rs index 8e1600aec13..9ed34184850 100644 --- a/tests/reborn_group_multiuser/scenario_auto_approve_isolation_across_actors.rs +++ b/tests/integration/group_multiuser/scenario_auto_approve_isolation_across_actors.rs @@ -1,24 +1,19 @@ -//! C-MULTIUSER scenario: per-actor AUTO-APPROVE (always-allow) isolation. -//! -//! Covers both the "approval-settings non-leak" and "auto-approve non-leak" -//! requirements — they are the SAME mechanic: the always-allow toggle is keyed -//! on the run owner's `(tenant, user)` scope (`AutoApproveSettingKey`, keyed on -//! `{tenant_id, user_id}`), read at capability dispatch. A grant made for actor -//! A's owner must NOT let actor B's identical, otherwise-gated call through. +//! C-MULTIUSER scenario: per-actor AUTO-APPROVE (always-allow) isolation. The +//! always-allow toggle is keyed on the run owner's `(tenant, user)` scope +//! (`AutoApproveSettingKey`), read at capability dispatch — a grant made for +//! actor A's owner must NOT let actor B's identical, otherwise-gated call +//! through. Covers both "approval-settings non-leak" and "auto-approve +//! non-leak", which are the SAME mechanic. //! //! Actor A grants always-allow for its own owner, then A's gated `write_file` -//! completes with NO approval gate. Actor B — a DISTINCT actor over the group's -//! ONE shared auto-approve store — issues the IDENTICAL `write_file` call and -//! still raises a real `BlockedApproval` gate, because A's grant is scoped to -//! A's owner alone. +//! completes with NO gate. Actor B — a DISTINCT actor over the SAME shared +//! auto-approve store — issues the IDENTICAL call and still raises a real +//! `BlockedApproval` gate, because A's grant is scoped to A's owner alone. //! //! Seam: `RebornIntegrationGroup::multiuser_approvals` builds the file-approval -//! backend with `with_run_owner_scoped_capability_dispatch`, so each actor's -//! dispatch (and thus the auto-approve lookup, the persisted approval request, -//! and the gate-evidence lookup) is keyed on that actor's OWN owner — matching -//! production, where the run owner IS the capability user. Without per-actor -//! scoping, all actors collapse onto one fixed capability user and this -//! isolation is unobservable. +//! backend with `with_run_owner_scoped_capability_dispatch`, keying dispatch +//! (auto-approve lookup, approval request, gate evidence) on each actor's OWN +//! owner — matching production, where the run owner IS the capability user. use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; use super::reborn_support::reply::RebornScriptedReply; @@ -43,11 +38,8 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { .subject_user_id .clone() .ok_or("actor A binding missing subject user id")?; - // Prove the grant is load-bearing, not just present: `AUTO_APPROVE_DEFAULT_ENABLED` - // is `true`, so a genuinely no-op `enable_auto_approve_for_owner` would still let - // A's write through by default and mask a broken grant. Force A OFF first so "A - // completes without a gate" is only reachable if the enable call below genuinely - // flips A's owner scope back ON. + // Force A OFF first: `AUTO_APPROVE_DEFAULT_ENABLED` is `true`, so a no-op + // `enable_auto_approve_for_owner` could otherwise mask a broken grant. g.disable_auto_approve_for_owner(&a_owner) .await .map_err(|e| format!("[A pre-disable] {e}"))?; @@ -86,9 +78,8 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { if a.binding.subject_user_id == b.binding.subject_user_id { return Err("with_actor_id seam no-op: both actors resolved the same owner".into()); } - // Give B its OWN explicit always-allow=OFF (auto-approve defaults ON per - // `AUTO_APPROVE_DEFAULT_ENABLED`), so B is a genuine gating actor. The - // isolation claim is then unambiguous: A's ON and B's OFF coexist per user. + // Give B its OWN explicit always-allow=OFF so B is a genuine gating actor; + // A's ON and B's OFF then coexist unambiguously per owner. let b_owner = b .binding .subject_user_id diff --git a/tests/reborn_group_multiuser/scenario_memory_isolation_across_actors.rs b/tests/integration/group_multiuser/scenario_memory_isolation_across_actors.rs similarity index 62% rename from tests/reborn_group_multiuser/scenario_memory_isolation_across_actors.rs rename to tests/integration/group_multiuser/scenario_memory_isolation_across_actors.rs index e09dd9cc71c..1bdcd18cf89 100644 --- a/tests/reborn_group_multiuser/scenario_memory_isolation_across_actors.rs +++ b/tests/integration/group_multiuser/scenario_memory_isolation_across_actors.rs @@ -1,28 +1,19 @@ //! C-MULTIUSER scenario: per-actor MEMORY isolation over the group's ONE shared -//! capability backend. +//! capability backend. Actor A writes a private memory; a DISTINCT actor B +//! cannot read or search it, while A still can — the tool-tier proof that +//! Reborn memory is scoped by the run's owner (`MemoryDocumentScope` keys the +//! path on the caller's user id, `crates/ironclaw_memory_native/src/path.rs`). //! -//! Actor A writes a private memory; a DISTINCT actor B on the same group cannot -//! read or search it, while A still can. This is the tool/capability-tier proof -//! that Reborn memory is scoped by the run's owner — production keys the memory -//! document path on the caller's user id (`MemoryDocumentScope` → -//! `/memory/tenants//users//agents//projects//…`, -//! `crates/ironclaw_memory_native/src/path.rs`), so two users in one workspace -//! never share a memory subtree. +//! Related QA: issue #5460 ("memories visible to every user in the workspace") +//! — the reporter noted the TOOL path is already isolated; this scenario pins +//! that tool-path isolation in-process, independent of the WebUI read surface +//! #5460 targets. //! -//! Related QA: issue #5460 ("Memories in the WebUI workspace are visible to -//! every user in the workspace"). The reporter noted the memory TOOL path is -//! already isolated ("the memories are isolated if you ask to expose other's -//! through tools") — this scenario pins exactly that tool-path isolation so a -//! regression at the capability tier (e.g. dropping the per-user path scope) -//! is caught in-process, independent of the WebUI read surface #5460 targets. -//! -//! The seam that makes this observable at int tier is -//! `RebornIntegrationGroup::multiuser_memory_tools` — it builds the capability -//! backend with `with_run_owner_scoped_capability_dispatch`, so each actor's -//! `memory_*` call dispatches under its OWN `(tenant, user)` scope instead of -//! the harness's single fixed capability user (which -//! `RebornIntegrationGroup::builtin_tools` uses, collapsing all actors onto one -//! shared memory subtree — see `tests/reborn_group_memory/`). +//! Seam: `RebornIntegrationGroup::multiuser_memory_tools` builds the backend +//! with `with_run_owner_scoped_capability_dispatch`, so each actor's `memory_*` +//! call dispatches under its OWN `(tenant, user)` scope instead of the single +//! fixed capability user `builtin_tools` uses (which collapses all actors onto +//! one shared memory subtree — see `tests/reborn_group_memory/`). use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; use super::reborn_support::reply::RebornScriptedReply; @@ -68,9 +59,8 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { ]) .build() .await?; - // Non-vacuity: the seam must resolve a genuinely DISTINCT owner for B. If - // `with_actor_id` regressed to a no-op, both actors would share one owner - // and this scenario would degrade to the already-covered same-owner case. + // Non-vacuity: if `with_actor_id` regressed to a no-op, both actors would + // share one owner and this scenario would degrade to the same-owner case. if a.binding.subject_user_id == b.binding.subject_user_id { return Err("with_actor_id seam no-op: both actors resolved the same owner".into()); } @@ -89,9 +79,7 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { } // ── Actor A again (SAME owner, new conversation): still finds its own memory - // Proves the write persisted and B's miss is genuine per-owner isolation, - // not an empty/broken store (the banana-99 discrimination pattern applied to - // the positive side). + // Proves B's miss is genuine isolation, not an empty/broken store. let a_reader = g .thread("conv-mem-iso-a-read") .script([ diff --git a/tests/reborn_group_multiuser/scenario_two_actors_own_threads.rs b/tests/integration/group_multiuser/scenario_two_actors_own_threads.rs similarity index 87% rename from tests/reborn_group_multiuser/scenario_two_actors_own_threads.rs rename to tests/integration/group_multiuser/scenario_two_actors_own_threads.rs index d2556c8fc02..1bfe870d76b 100644 --- a/tests/reborn_group_multiuser/scenario_two_actors_own_threads.rs +++ b/tests/integration/group_multiuser/scenario_two_actors_own_threads.rs @@ -58,13 +58,11 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { Ok(()) } -/// One direction of the owner-isolation negative guard (prove genuine owner -/// separation, not just "didn't crash" — the banana-99 pattern): -/// `reader`'s own thread must not surface `other`'s reply, and `other`'s -/// thread must not be readable through `reader`'s owner scope at all — each -/// owner's records live under a separate `/tenants//users//threads` -/// subtree. One helper -/// called once per direction, so the symmetric check cannot silently de-sync. +/// One direction of the owner-isolation negative guard: `reader`'s thread must +/// not surface `other`'s reply, and `other`'s thread must not be readable +/// through `reader`'s owner scope at all (each owner's records live under a +/// separate `/tenants//users//threads` subtree). Called once per +/// direction so the symmetric check can't silently de-sync. async fn assert_cannot_read_other_actor( reader: &RebornIntegrationHarness, reader_name: &str, diff --git a/tests/reborn_group_skills/main.rs b/tests/integration/group_skills/main.rs similarity index 96% rename from tests/reborn_group_skills/main.rs rename to tests/integration/group_skills/main.rs index 7314ba35048..317955ee108 100644 --- a/tests/reborn_group_skills/main.rs +++ b/tests/integration/group_skills/main.rs @@ -10,10 +10,10 @@ //! tiers never drift on capability ids / mounts / policy. #[allow(dead_code)] -#[path = "../support/reborn/mod.rs"] +#[path = "../support/mod.rs"] mod reborn_support; #[allow(dead_code)] -#[path = "../support/mod.rs"] +#[path = "../../support/mod.rs"] mod support; mod scenario_install_list_remove; diff --git a/tests/reborn_group_skills/scenario_install_list_remove.rs b/tests/integration/group_skills/scenario_install_list_remove.rs similarity index 100% rename from tests/reborn_group_skills/scenario_install_list_remove.rs rename to tests/integration/group_skills/scenario_install_list_remove.rs diff --git a/tests/reborn_group_triggers/main.rs b/tests/integration/group_triggers/main.rs similarity index 55% rename from tests/reborn_group_triggers/main.rs rename to tests/integration/group_triggers/main.rs index ea5b34d9479..a4bdadd8b24 100644 --- a/tests/reborn_group_triggers/main.rs +++ b/tests/integration/group_triggers/main.rs @@ -17,10 +17,10 @@ //! orchestrating function gives deterministic ordering for free. #[allow(dead_code)] -#[path = "../support/reborn/mod.rs"] +#[path = "../support/mod.rs"] mod reborn_support; #[allow(dead_code)] -#[path = "../support/mod.rs"] +#[path = "../../support/mod.rs"] mod support; mod scenario_trigger_persists_after_reopen; @@ -49,34 +49,18 @@ async fn triggers_group_e2e() { scenario_trigger_persists_after_reopen::run(&g).await, ); - // Triggered-turn coverage map (via `RebornIntegrationHarness::submit_triggered_turn`, - // E-TRIGGERED-SUBMIT) — do NOT duplicate any of this here: - // - `TurnOriginKind::ScheduledTrigger` propagation (with a discriminating - // interactive-origin `Inbound` contrast arm) — - // `tests/reborn_integration_triggered_submit.rs`. Flat single-thread - // test, so it doesn't belong in this multi-thread group binary. - // - triggered fire → real `BlockedApproval` gate → approve/deny → resume — - // `triggered_gate_group` below (`scenario_triggered_gate::{run_approve,run_deny}`), - // driven through `submit_triggered_turn_scripted`. - // - one-shot `Once` fire → `Completed` derivation — - // `crates/ironclaw_reborn_composition/tests/trigger_poller_e2e.rs` + - // `crates/ironclaw_triggers/tests/repository_contract.rs`. - // - triggered run completes + final reply persists in the trigger's own - // thread (the state the production push leg reads) — - // `triggered_run_completes_and_persists_reply_in_trigger_thread` in - // `tests/reborn_integration_triggered_submit.rs`. - // - the trigger → Slack outbound-delivery leg — - // `crates/ironclaw_reborn_composition/src/slack_host_beta.rs`. + // Triggered-turn coverage map (E-TRIGGERED-SUBMIT via `submit_triggered_turn`) + // — do NOT duplicate any of this here: + // - origin propagation: `tests/reborn_integration_triggered_submit.rs` + // - gate raise/approve/deny/resume: `triggered_gate_group` below + // - one-shot fire -> Completed: `trigger_poller_e2e.rs` + `repository_contract.rs` + // - reply persists in trigger's own thread: `reborn_integration_triggered_submit.rs` + // - push leg (trigger -> Slack outbound delivery): `slack_host_beta.rs` // - // Still BLOCKED at int tier: the PUSH half (triggered run → outbound - // delivery sink). `deliver_triggered_run` is a PRIVATE fn in the Slack - // services-shell (`slack_delivery.rs`), reachable only via a detached - // `tokio::spawn` entry (`PostSubmitDeliveryHook`) not wired into any - // harness turn lifecycle by construction — covered instead by - // `slack_delivery.rs`'s own `#[cfg(test)]` module + - // `product_workflow/tests/outbound_delivery_contract.rs`. Requires a - // services-shell disposition, not an authorable harness seam; do not - // reconstruct it here. + // Still BLOCKED at int tier: the PUSH half. `deliver_triggered_run` is a + // private fn reachable only via a detached `tokio::spawn` hook, not wired + // into any harness turn lifecycle — covered instead by `slack_delivery.rs`'s + // own `#[cfg(test)]` module + `outbound_delivery_contract.rs`. // C-DENYEDGE row 4: a scheduled-trigger fire must not be able to create // its own follow-up trigger. Uses THIS group's `triggers()` capability @@ -91,16 +75,12 @@ async fn triggers_group_e2e() { report.assert_all_passed(); } -/// Triggered-origin runs raise, park on, and resume from REAL approval gates -/// (mid-fire gate → approve/deny → resume), exactly like interactive runs. -/// -/// Lives in this binary (not `reborn_group_approvals`) because the scenario's -/// subject is the TRIGGERED submit wire, not the approval machinery — the -/// approval arms mirror `reborn_group_approvals/scenario_gate_then_{approve,deny}` -/// over the trusted-trigger origin. Each arm gets its OWN `live_approvals` -/// group (see `scenario_triggered_gate` docs), so this is a separate -/// `#[tokio::test]` rather than more scenarios on the verbs group above (the -/// `triggers()` group has auto-approve ENABLED — a gate can never raise there). +/// Triggered-origin runs raise, park on, and resume from REAL approval gates, +/// exactly like interactive runs. Lives in this binary (not +/// `reborn_group_approvals`) because the subject is the TRIGGERED submit wire. +/// Separate `#[tokio::test]` (own `live_approvals` group per arm) because the +/// `triggers()` group above has auto-approve ENABLED — a gate can never raise +/// there. #[tokio::test] async fn triggered_gate_group() { let mut report = ScenarioReport::new(); @@ -122,10 +102,8 @@ async fn triggered_gate_group() { ); // C-DENYEDGE row 1: a resume for the right run_id but a mutated - // (wrong-tenant) TurnScope must be rejected with ScopeNotFound, and the - // gate must remain live/resolvable afterward. Own group: mutating the - // approval store mid-scenario should not be attributed to the approve/ - // deny arms above. + // (wrong-tenant) TurnScope must be rejected with ScopeNotFound, gate still + // live/resolvable after. Own group: isolates the mutation from the arms above. let g_wrong_scope = RebornIntegrationGroup::live_approvals() .await .expect("wrong-scope-arm group builds"); @@ -134,12 +112,10 @@ async fn triggered_gate_group() { scenario_triggered_gate::run_wrong_scope_resume_rejected(&g_wrong_scope).await, ); - // C-JOURNEY (wave-4 carry-over): a triggered fire whose run raises a - // gate, gets resolved, then CHAINS into a SECOND gate/action in the SAME - // run — pins ScheduledTrigger origin propagation across BOTH resume hops - // (not just the first) plus reply persistence. Own group: two gates over - // the group's shared approval store should not be attributed to the - // single-gate arms above. + // C-JOURNEY: a triggered fire raises a gate, resolves, then CHAINS into a + // SECOND gate in the SAME run — pins ScheduledTrigger origin across BOTH + // resume hops plus reply persistence. Own group: isolates the two-gate case + // from the single-gate arms above. let g_chained = RebornIntegrationGroup::live_approvals() .await .expect("chained-gate-arm group builds"); diff --git a/tests/reborn_group_triggers/scenario_trigger_persists_after_reopen.rs b/tests/integration/group_triggers/scenario_trigger_persists_after_reopen.rs similarity index 90% rename from tests/reborn_group_triggers/scenario_trigger_persists_after_reopen.rs rename to tests/integration/group_triggers/scenario_trigger_persists_after_reopen.rs index fd928d0dd36..01f29c580b9 100644 --- a/tests/reborn_group_triggers/scenario_trigger_persists_after_reopen.rs +++ b/tests/integration/group_triggers/scenario_trigger_persists_after_reopen.rs @@ -45,10 +45,8 @@ pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { let capability_harness = g .capability_harness() .ok_or("triggers group always uses HostRuntime")?; - // Same run-scope tenant every group's `build_base` resolves (the fixed - // itest scope) — read off the group's own product-workflow scope rather - // than a separately hardcoded literal, so this can never drift from the - // tenant `trigger_create` actually stored under. + // Read off the group's own scope rather than a hardcoded literal, so this + // can never drift from the tenant `trigger_create` actually stored under. let tenant_id = g.shared.product_harness.scope.tenant_id.clone(); // Reopen a FRESH, independent repository at the same on-disk root — not the diff --git a/tests/integration/group_triggers/scenario_trigger_self_create_denied.rs b/tests/integration/group_triggers/scenario_trigger_self_create_denied.rs new file mode 100644 index 00000000000..4569db738c0 --- /dev/null +++ b/tests/integration/group_triggers/scenario_trigger_self_create_denied.rs @@ -0,0 +1,112 @@ +//! C-DENYEDGE (row 4): a scheduled-trigger fire must not be able to create (or +//! remove/pause/resume) triggers of its own — int-tier twin of the +//! `ironclaw_reborn::runtime` unit coverage for issue #5505 +//! (`SCHEDULED_TRIGGER_DENIED_CAPABILITY_IDS`, PR #5515). +//! +//! Drives a triggered-origin run (`submit_triggered_turn_scripted`) that scripts +//! `builtin.trigger_create`. The host's `PerSurfaceCapabilityDenyDecorator`, +//! keyed on the `scheduled_trigger` run profile's capability-surface id, strips +//! trigger_create/remove/pause/resume from the model-visible surface +//! (`trigger_list` stays visible). +//! +//! Traced, not assumed: denial happens at the model-gateway seam +//! (`ironclaw_reborn::model_gateway`'s `validate_provider_tool_call`, via +//! `CapabilitySurfaceDenyFilter`), BEFORE a `CapabilityCallCandidate` is ever +//! constructed — so `CapabilityStage` never runs and nothing is appended via +//! `append_tool_result_reference` (confirmed empirically: persisted history is +//! exactly `[User, Assistant]`, no `ToolResultReference`). The executor +//! transparently re-issues the model call rather than gating or failing. +//! +//! Because nothing is persisted for this seam, `assert_tool_error`/ +//! `assert_tool_error_summary_contains` (which read persisted envelopes) can't +//! observe it. This scenario instead asserts the security property directly: +//! (1) `builtin.trigger_create` was never dispatched, (2) the run completes +//! cleanly, (3) no trigger with the attempted name exists afterward (verified +//! via a real, non-triggered `builtin.trigger_list` call). +//! +//! Distinct from `scenario_verbs_lifecycle`'s `trigger_create` coverage: that +//! submits through the plain `submit_turn` wire (interactive run profile), +//! where the trigger-mutator surface is fully visible and never exercises +//! this deny map. + +use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; +use super::reborn_support::reply::RebornScriptedReply; +use ironclaw_turns::TurnStatus; +use serde_json::json; + +/// Distinctive enough that a false-positive match against another scenario's +/// trigger name (e.g. `scenario_verbs_lifecycle`'s `"t0-triggers-once"`) is +/// not a concern. +const SELF_CREATE_ATTEMPT_TRIGGER_NAME: &str = "self-created-follow-up-should-not-exist"; + +pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { + let h = g.thread("conv-trigger-self-create-denied").build().await?; + + let submission = h + .submit_triggered_turn_scripted( + "create a follow-up reminder", + [ + RebornScriptedReply::tool_call( + "builtin.trigger_create", + json!({ + "name": SELF_CREATE_ATTEMPT_TRIGGER_NAME, + "prompt": "remind me again", + "schedule": {"kind": "once", "at": "2999-01-01T00:00:00", "timezone": "UTC"}, + }), + ), + RebornScriptedReply::text("understood, I can't schedule that myself"), + ], + ) + .await?; + + // Must NOT hang or fail: the denial is a model-recoverable outcome, not + // a gate or a terminal failure. + h.wait_for_status_in_scope( + &submission.turn_scope, + submission.run_id, + TurnStatus::Completed, + ) + .await?; + + // The capability must never have reached dispatch — reads the SAME + // invocation recorder `assert_tool_invoked` uses for the positive case in + // `scenario_verbs_lifecycle`, not a bare `.is_err()` on something unrelated. + if h.assert_tool_invoked("builtin.trigger_create") + .await + .is_ok() + { + return Err( + "expected builtin.trigger_create to be denied for a scheduled-trigger fire, \ + but the capability recorder shows it was invoked" + .into(), + ); + } + + // Strongest proof: no trigger with the attempted name exists, verified via + // a genuine INTERACTIVE `trigger_list` call (read-only, visible on every + // profile) rather than anything scoped to the denied run. + let verifier = g + .thread("conv-trigger-self-create-denied-verify") + .script([ + RebornScriptedReply::tool_call("builtin.trigger_list", json!({})), + RebornScriptedReply::text("listed"), + ]) + .build() + .await?; + verifier.submit_turn("list my triggers").await?; + let listed = verifier.tool_result_output("builtin.trigger_list").await?; + let triggers = listed["triggers"] + .as_array() + .ok_or("trigger_list output missing triggers array")?; + if triggers + .iter() + .any(|t| t["name"] == json!(SELF_CREATE_ATTEMPT_TRIGGER_NAME)) + { + return Err(format!( + "expected no trigger named {SELF_CREATE_ATTEMPT_TRIGGER_NAME:?} to exist, \ + but trigger_list returned {listed}" + ) + .into()); + } + Ok(()) +} diff --git a/tests/reborn_group_triggers/scenario_triggered_chained_gate.rs b/tests/integration/group_triggers/scenario_triggered_chained_gate.rs similarity index 76% rename from tests/reborn_group_triggers/scenario_triggered_chained_gate.rs rename to tests/integration/group_triggers/scenario_triggered_chained_gate.rs index ec701337350..b8a2f212318 100644 --- a/tests/reborn_group_triggers/scenario_triggered_chained_gate.rs +++ b/tests/integration/group_triggers/scenario_triggered_chained_gate.rs @@ -1,18 +1,12 @@ //! C-JOURNEY (triggered-origin): a triggered fire whose run raises a real -//! `BlockedApproval` gate, gets resolved, and then CHAINS into a SECOND -//! `BlockedApproval` gate in the SAME run (the post-resume model call issues -//! another gated tool call instead of finalizing) — pinning that -//! `TurnOriginKind::ScheduledTrigger` origin survives BOTH resume hops, not -//! just the first. +//! `BlockedApproval` gate, gets resolved, then CHAINS into a SECOND +//! `BlockedApproval` gate in the SAME run — pins that `TurnOriginKind:: +//! ScheduledTrigger` survives BOTH resume hops, not just the first. //! -//! Distinct from `scenario_triggered_gate::run_approve`: that scenario is -//! ONE gate (tool_call, text) — a single resume hop. This scenario scripts -//! THREE model calls (tool_call, tool_call, text) so the run parks TWICE, -//! and reads `state.product_context.origin` at both parked states AND at -//! final `Completed`, closing the gap that a regression which only -//! preserved origin across the FIRST resume (e.g. a resume path that -//! rebuilds `product_context` from a fresh, non-trigger-aware default on the -//! second hop) would otherwise slip through undetected. +//! Distinct from `scenario_triggered_gate::run_approve` (ONE gate, single +//! resume hop): this scripts THREE model calls so the run parks TWICE, and +//! reads `state.product_context.origin` at both parks AND at final +//! `Completed` — closing the gap where origin regresses only on the second hop. use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; use super::reborn_support::reply::RebornScriptedReply; @@ -117,13 +111,10 @@ pub async fn run_chained_approve(g: &RebornIntegrationGroup) -> HarnessResult<() Ok(()) } -/// Read the run state fresh at the coordinator boundary (not the state handed -/// in from an earlier `wait_for_status_in_scope`/`approve_gate_in_scope` -/// call) and assert its `product_context.origin` is -/// `TurnOriginKind::ScheduledTrigger`. Re-reading independently at each of -/// the three checkpoints (not just trusting the ONE `TriggeredSubmission` for -/// the whole scenario) is what actually proves origin persists ACROSS both -/// resume hops, rather than merely being set once at submit time. +/// Read the run state fresh at each checkpoint (not the state handed in from +/// an earlier call, and not just the ONE `TriggeredSubmission`) and assert +/// `product_context.origin` is `ScheduledTrigger` — proves origin persists +/// ACROSS both resume hops, not merely set once at submit time. fn assert_scheduled_trigger_origin( state: &ironclaw_turns::TurnRunState, checkpoint: &str, diff --git a/tests/reborn_group_triggers/scenario_triggered_gate.rs b/tests/integration/group_triggers/scenario_triggered_gate.rs similarity index 80% rename from tests/reborn_group_triggers/scenario_triggered_gate.rs rename to tests/integration/group_triggers/scenario_triggered_gate.rs index 8aa8bf01a3b..bb1e06bf262 100644 --- a/tests/reborn_group_triggers/scenario_triggered_gate.rs +++ b/tests/integration/group_triggers/scenario_triggered_gate.rs @@ -118,34 +118,23 @@ pub async fn run_deny(g: &RebornIntegrationGroup) -> HarnessResult<()> { Ok(()) } -/// C-DENYEDGE (row 1): a resume request carrying the RIGHT `run_id` but a -/// MUTATED `TurnScope` (different tenant) must be rejected, not silently -/// resolved against the wrong tenant's run. +/// C-DENYEDGE (row 1): a resume for the RIGHT `run_id` but a MUTATED +/// `TurnScope` (different tenant) must be rejected, not silently resolved +/// against the wrong tenant's run. /// -/// Drives a triggered fire to a real `BlockedApproval` gate exactly like -/// [`run_approve`], then attempts to resume the SAME `run_id` under a -/// scope with a different `tenant_id`. `InMemoryTurnStateStore::resume_turn_once` -/// (`crates/ironclaw_turns/src/memory/mod.rs`) looks the run up by `run_id` -/// alone, then checks `record.scope != request.scope` BEFORE consulting the -/// gate/precondition/actor at all — so a mis-scoped resume against an -/// otherwise-valid pending gate deterministically hits -/// `TurnError::ScopeNotFound` (`#[error("turn run not found")]`), and the -/// taken-out record is unconditionally reinserted afterward regardless of -/// outcome, so the run's `BlockedApproval` state is untouched by the -/// rejected attempt. +/// Drives a triggered fire to a real `BlockedApproval` gate like +/// [`run_approve`], then resumes the SAME `run_id` under a mismatched tenant +/// scope. `InMemoryTurnStateStore::resume_turn_once` checks `record.scope != +/// request.scope` BEFORE consulting gate/precondition/actor, so this +/// deterministically hits `TurnError::ScopeNotFound` and reinserts the +/// taken-out record unconditionally — the gate's `BlockedApproval` state is +/// untouched by the rejected attempt. /// -/// This drives `resume_run_in_scope` (the resume-only tail -/// `approve_gate_in_scope`/`deny_gate_in_scope` share) rather than those two -/// helpers themselves: both first resolve the approval request in the -/// (scope-independent) local-dev approval store via -/// `capability_recorder.approve_local_dev_gate`/`deny_local_dev_gate`, which -/// is keyed on `gate_ref` alone and would irreversibly flip the approval -/// record to `Approved`/`Denied` even though the resume itself is rejected — -/// corrupting the approval-store state this scenario's non-vacuity check -/// depends on being still-`Pending`. The resume-only call isolates the -/// assertion to exactly the mechanism under test (the `TurnScope` equality -/// check inside `resume_turn_once`) and leaves the gate genuinely live for -/// the real resume that follows. +/// Drives `resume_run_in_scope` directly (not `approve_gate_in_scope`/ +/// `deny_gate_in_scope`): those first flip the approval record to +/// Approved/Denied in the scope-independent local-dev store even when the +/// resume itself is rejected, which would corrupt the still-`Pending` state +/// this scenario's non-vacuity check depends on. pub async fn run_wrong_scope_resume_rejected(g: &RebornIntegrationGroup) -> HarnessResult<()> { let h = g.thread("conv-triggered-gate-wrong-scope").build().await?; diff --git a/tests/reborn_group_triggers/scenario_verbs_lifecycle.rs b/tests/integration/group_triggers/scenario_verbs_lifecycle.rs similarity index 82% rename from tests/reborn_group_triggers/scenario_verbs_lifecycle.rs rename to tests/integration/group_triggers/scenario_verbs_lifecycle.rs index 65599fc58c0..ec18c99350e 100644 --- a/tests/reborn_group_triggers/scenario_verbs_lifecycle.rs +++ b/tests/integration/group_triggers/scenario_verbs_lifecycle.rs @@ -23,25 +23,16 @@ use serde_json::json; const ONCE_AT: &str = "2999-01-01T00:00:00"; const TRIGGER_NAME: &str = "t0-triggers-once"; -// TODO(T0-TRIGGERS, no enabler needed): distinct verb branches this same group -// can grow in a follow-up without any harness seam — add as new `scenario_*` -// files, not by bloating this happy-path lifecycle: -// - cron-schedule create (`{kind:"cron", expression, timezone}`) → list renders -// `is_recurring`/next_run_at; contrast with the Once path here. -// - `trigger_list` `limit`/`run_limit` params (bounded output). -// - deny/error branches through the capability path (model-recoverable, NOT -// terminal per `.claude/rules/agent-loop-capabilities.md`): remove/pause a -// non-existent `trigger_id` → `{"removed":false}` / `{"updated":false}`; -// malformed `trigger_id` → surfaced input error the model can retry. +// TODO(T0-TRIGGERS, no enabler needed): follow-ups this group can grow as new +// `scenario_*` files (not by bloating this happy-path lifecycle): cron-schedule +// create/list contrast, `trigger_list` limit params, deny/error branches +// (remove/pause a non-existent id, malformed id). // -// Two gotchas for follow-up scenarios in THIS group binary: -// - `tool_result_output(cap)` returns the MOST RECENT result for `cap` in the -// thread's slice. If a scenario dispatches the same verb twice in one thread, -// read the intermediate result before the second call — `.rev()` will -// otherwise silently return the later one. -// - the group's trigger repository is shared across scenarios with NO cleanup -// between them; keep list assertions id-scoped (`.any(|t| t["trigger_id"]…)`) -// and never assert an exact `triggers.len()`, which would flake on leftovers. +// Gotchas for follow-ups in THIS binary: `tool_result_output(cap)` returns the +// MOST RECENT result for `cap` in the thread's slice — read intermediate +// results before a second call to the same verb. The trigger repository has NO +// cleanup between scenarios — keep list assertions id-scoped, never assert +// exact `triggers.len()`. pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { // ── Thread A: create a one-shot Once trigger, then list it ─────────────── let creator = g diff --git a/tests/reborn_integration_hooks.rs b/tests/integration/hooks.rs similarity index 64% rename from tests/reborn_integration_hooks.rs rename to tests/integration/hooks.rs index 30ed293fd53..01ac8d84bf3 100644 --- a/tests/reborn_integration_hooks.rs +++ b/tests/integration/hooks.rs @@ -4,43 +4,34 @@ //! //! # BLOCKED by a production bug (both scenarios are `#[ignore]`d RED regressions) //! -//! Wiring ANY hook dispatcher into a full coordinator-path turn (via the same -//! `build_default_planned_runtime` → `RebornLoopDriverHostFactory::with_hook_dispatcher_builder_factory` -//! path production uses when the hooks flag is enabled) fails EVERY turn with -//! `driver_unavailable`. Root cause, confirmed via trace capture: +//! Wiring ANY hook dispatcher into a full coordinator-path turn fails EVERY +//! turn with `driver_unavailable`: +//! `HostUnavailableWithDiagnostics { stage: Checkpoint, kind: Unavailable, +//! safe_summary: "stage_checkpoint_payload not implemented" }`. //! -//! ```text -//! HostUnavailableWithDiagnostics { stage: Checkpoint, kind: Unavailable, -//! safe_summary: "stage_checkpoint_payload not implemented" } -//! ``` +//! Root cause: `ironclaw_hooks::middleware::checkpoint_port:: +//! HookedLoopCheckpointPort` overrides only `LoopCheckpointPort::checkpoint`; +//! it does NOT forward `stage_checkpoint_payload`/`load_checkpoint_payload` to +//! its inner port, so those fall through to the trait's fail-closed defaults. +//! A planned run stages a checkpoint payload before the first model call, so +//! with a hook dispatcher active the turn dies there before any hook fires. +//! Hooks are off by default in production, so this latent bug has never been +//! exercised — this is the first full-turn-with-hooks path. //! -//! `ironclaw_hooks::middleware::checkpoint_port::HookedLoopCheckpointPort` -//! (crates/ironclaw_hooks/src/middleware/checkpoint_port.rs) overrides ONLY -//! `LoopCheckpointPort::checkpoint`; it does NOT forward `stage_checkpoint_payload` -//! (nor `load_checkpoint_payload`) to its inner port, so those fall through to the -//! trait's fail-closed defaults (`ironclaw_turns::run_profile::host::LoopCheckpointPort`, -//! host.rs:2214/2227 — "…not implemented"). A planned run stages a checkpoint -//! payload on `checkpoint_before_model` (before the first model call), so with a -//! hook dispatcher active the turn dies at that checkpoint before any hook -//! meaningfully fires. Hooks are off by default in production -//! (`HOOKS_ENABLED_ENV`), so this latent wrapper bug has never been exercised — -//! this is the first full-turn-with-hooks path (all existing hook tests drive -//! mock ports via `host.invoke_capability`, never a coordinator submit). +//! TODO(reborn-hooks-checkpoint-forward): forward both methods through +//! `HookedLoopCheckpointPort` to `self.inner` (mirroring `checkpoint()`), then +//! remove the `#[ignore]`s below. Cross-crate fix in `ironclaw_hooks`, out of +//! scope for this tests-only lane. //! -//! TODO(reborn-hooks-checkpoint-forward): forward `stage_checkpoint_payload` + -//! `load_checkpoint_payload` through `HookedLoopCheckpointPort` to `self.inner` -//! (mirroring `checkpoint()`), then remove the `#[ignore]`s below. Fix is a -//! cross-crate change in `ironclaw_hooks`, out of scope for this tests-only lane. -//! -//! The E-HOOK-INFRA enabler (recording hook doubles + `HookDispatcherBuilderFactory` -//! builders in `tests/support/reborn/hooks.rs`, the `hook_dispatcher_builder_factory` -//! group-builder seam, and the `with_hook_factory` harness seam) DOES land and is -//! correct — these scenarios go green the moment the wrapper bug is fixed. +//! The E-HOOK-INFRA enabler (recording hook doubles, the +//! `hook_dispatcher_builder_factory` group-builder seam, `with_hook_factory`) +//! DOES land and is correct — these scenarios go green once the wrapper bug is fixed. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::assertions::ToolErrorClass; @@ -82,12 +73,9 @@ async fn hooks_fire_at_lifecycle_points_on_coordinator_turn() { h.assert_tool_invoked("builtin.http") .await .expect("http tool ran through the real capability path"); - // Both lifecycle points fired, in the expected order: the AfterModel - // observer dispatches once per `finalize_assistant_message` call (see - // `HookedLoopTranscriptPort::finalize_assistant_message`), and the script - // has two assistant turns (the tool-call reply, then the final text - // reply) with the BeforeCapability gate hook firing between them, right - // before the dispatched `builtin.http` capability. + // Both lifecycle points fired in order: AfterModel dispatches once per + // `finalize_assistant_message` call, and the script's two assistant turns + // sandwich the BeforeCapability gate hook right before dispatch. assert_eq!( log.fires(), vec![ diff --git a/tests/reborn_integration_http_matcher.rs b/tests/integration/http_matcher.rs similarity index 96% rename from tests/reborn_integration_http_matcher.rs rename to tests/integration/http_matcher.rs index 1c89a2bda12..a78e2939b8e 100644 --- a/tests/reborn_integration_http_matcher.rs +++ b/tests/integration/http_matcher.rs @@ -11,9 +11,10 @@ //! network, no services, no keys, no Docker, no `integration` feature. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use ironclaw_threads::MessageKind; @@ -218,11 +219,9 @@ async fn http_oversize_response_surfaces_recoverable_failed() { .expect("run recovered and finalized (not terminal driver_unavailable)"); } -/// Guards `assert_tool_error` against a vacuous pass, mirroring the sibling -/// negative guards (`shell_assertions_fail_when_no_shell_call_ran`, -/// `assert_mcp_tool_called_fails_when_no_mcp_call_ran`). Three ways it must -/// return `Err`: (a) a completed turn that persisted NO tool-error reference at -/// all, (b) a real `Denied` turn probed with the wrong reason, and (c) that same +/// Guards `assert_tool_error` against a vacuous pass. Three ways it must +/// return `Err`: (a) a completed turn with NO persisted tool-error reference, +/// (b) a real `Denied` turn probed with the wrong reason, (c) that same /// `Denied` turn probed with the WRONG CLASS but the right reason token — /// proving the class discriminates structurally, not just the reason. #[tokio::test] @@ -406,9 +405,8 @@ async fn multi_turn_baseline_sliced_history_assertions() { .is_err(), "containment assert must reject text absent from the whole transcript" ); - // Fail-check: an out-of-range baseline (stale/foreign-thread value) is a - // loud error, not an empty slice — otherwise `assert_no_tool_error_since` - // would pass vacuously on a caller bug. + // Fail-check: an out-of-range baseline must be a loud error, not an empty + // slice — otherwise `assert_no_tool_error_since` passes vacuously. let past_end = h.history_len().await.expect("history len readable") + 1; assert!( h.assert_no_tool_error_since(past_end, ToolErrorClass::Denied, "policy_denied") diff --git a/tests/reborn_integration_mcp.rs b/tests/integration/mcp.rs similarity index 86% rename from tests/reborn_integration_mcp.rs rename to tests/integration/mcp.rs index f8725504101..64e6ce04e18 100644 --- a/tests/reborn_integration_mcp.rs +++ b/tests/integration/mcp.rs @@ -7,9 +7,10 @@ //! no real network, no services, no API keys, no Docker, no `integration` feature. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::assertions::ToolErrorClass; @@ -234,22 +235,15 @@ async fn mcp_server_5xx_surfaces_recoverable_failed() { .expect("run recovered and finalized (not terminal driver_unavailable)"); } -/// Error path — MCP `tools/call` returns a valid, successful JSON-RPC response -/// whose parsed `result` payload exceeds the host's MCP output-size limit -/// (`McpRuntimeConfig::default().max_output_bytes` = 1 MiB, wired by -/// `build_loopback_mcp_runtime` in `tests/support/reborn/harness_mcp.rs`). The -/// client surfaces this as `Failed{OutputTooLarge}` (recoverable, model-visible -/// — `reason` token `"output_too_large"`), and the run completes. Distinct wire -/// path from both cases above: this trips neither the HTTP status gate nor the -/// JSON-RPC error-field guard — the HTTP call and JSON-RPC parse both succeed — -/// it trips the POST-parse `output_bytes > max_output_bytes` check in -/// `McpRuntime::execute_extension_json` (`crates/ironclaw_mcp/src/lib.rs`), -/// which is a wholly different mechanism from an HTTP-egress -/// `response_body_limit` rejection: `LoopbackMcpRuntimeHttpEgress` (this -/// harness's test-only egress) never enforces `response_body_limit` against the -/// real bytes it reads back, so only this post-parse size check is reachable at -/// this test tier — a genuinely oversized HTTP response body would just be read -/// in full and only then rejected once parsed. +/// Error path — a valid, successful JSON-RPC response whose parsed `result` +/// exceeds the host's MCP output-size limit (`McpRuntimeConfig::default() +/// .max_output_bytes` = 1 MiB). Surfaces as `Failed{OutputTooLarge}` +/// (recoverable, model-visible), run completes. Distinct wire path from both +/// cases above: HTTP and JSON-RPC parse both succeed; this trips the +/// POST-parse `output_bytes > max_output_bytes` check in `McpRuntime:: +/// execute_extension_json` — `LoopbackMcpRuntimeHttpEgress` never enforces +/// `response_body_limit` against real bytes, so this is the only size check +/// reachable at this test tier. #[tokio::test] async fn mcp_tool_call_output_too_large_surfaces_failed() { // `content` serializes (inside the mock server's `{"content":[{"type": @@ -287,9 +281,7 @@ async fn mcp_tool_call_output_too_large_surfaces_failed() { } // MCP "auth mismatch" (HTTP 401/403) is deliberately NOT covered here: unlike -// the two cases above, a 401/403 maps to `McpClientError::AuthRequired` — an -// auth *gate* (park-and-resume), not a model-visible `Failed`/`Denied` tool -// error. Exercising a credentialed-backend 401 through to a raised auth gate -// requires a capability-backend credential stub that this tier does not yet -// have; it is the same "live-401 re-auth arm" already deferred in -// `reborn_integration_auth_failure.rs`. Track with that follow-up. +// the cases above, a 401/403 maps to `McpClientError::AuthRequired` — an auth +// *gate*, not a model-visible tool error — and requires a capability-backend +// credential stub this tier lacks. Same deferred "live-401 re-auth arm" as +// `reborn_integration_auth_failure.rs`. diff --git a/tests/reborn_integration_oauth_connect.rs b/tests/integration/oauth_connect.rs similarity index 100% rename from tests/reborn_integration_oauth_connect.rs rename to tests/integration/oauth_connect.rs diff --git a/tests/reborn_integration_oauth_refresh.rs b/tests/integration/oauth_refresh.rs similarity index 83% rename from tests/reborn_integration_oauth_refresh.rs rename to tests/integration/oauth_refresh.rs index 6c1174701a6..cac36e63fee 100644 --- a/tests/reborn_integration_oauth_refresh.rs +++ b/tests/integration/oauth_refresh.rs @@ -12,9 +12,10 @@ // The support tree is large and shared; a single-test file exercises only a // slice of it, so suppress dead-code warnings on the includes. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use chrono::{Duration, Utc}; @@ -36,21 +37,13 @@ fn test_scope() -> AuthProductScope { /// idle threshold) triggers a token-refresh HTTP call for the idle account AND /// commits the rotated credential to the durable store. /// -/// The egress count alone only proves the refresh HTTP call *fired*; it would -/// still pass if the refresh made the call but silently dropped the account -/// write. To close that gap we also re-read the account through the durable -/// `CredentialAccountService` and assert the persisted access-token handle was -/// rewritten to the refresh-path handle. On a successful refresh, -/// `ProviderBackedCredentialAccountService::refresh_account` persists -/// `HostOAuthProviderClient::store_refreshed_tokens`'s output via -/// `create_or_update_account`; that handle (`…-oauth-refresh-access-`) -/// is produced *only* by the refresh write-back path and differs from the -/// connect-exchange handle (`…-oauth-access--`). A -/// dropped account write would leave the original connect handle in place, so -/// the handle assertions fail in that case. Because `store_refreshed_tokens` -/// returns the handle only after `put`-ing the new token material, the rotated -/// handle on the account also transitively proves the refreshed material was -/// persisted to the secret store. +/// Egress count alone only proves the HTTP call *fired*; it would still pass +/// if the write was silently dropped. To close that gap, re-read the account +/// and assert the persisted access-token handle was rewritten to the +/// refresh-path handle (`…-oauth-refresh-access-`, produced only +/// by `store_refreshed_tokens`'s write-back) rather than the original +/// connect-exchange handle — which also transitively proves the refreshed +/// material was persisted to the secret store. #[tokio::test] async fn credential_refresh_sweep_refreshes_idle_google_account() { let bundle = build_google_oauth_product_auth_for_test(); diff --git a/tests/reborn_integration_outbound_target.rs b/tests/integration/outbound_target.rs similarity index 83% rename from tests/reborn_integration_outbound_target.rs rename to tests/integration/outbound_target.rs index a96fad61dff..b0823175a33 100644 --- a/tests/reborn_integration_outbound_target.rs +++ b/tests/integration/outbound_target.rs @@ -1,34 +1,23 @@ //! C-SYNTH outbound seam: the `outbound_target_tools` group surfaces the two -//! local-dev synthetic `outbound_delivery_*` capabilities, and scripted tool -//! calls dispatch through the REAL production synthetic-capability wrap -//! (`wrap_local_dev_synthetic_capabilities` + `outbound_delivery_capabilities`) -//! over an injected `FakeOutboundPreferencesFacade` at the production-wired -//! facade trait seam. +//! local-dev synthetic `outbound_delivery_*` capabilities, dispatched through +//! the REAL production synthetic-capability wrap over an injected +//! `FakeOutboundPreferencesFacade` at the production-wired facade trait seam. //! -//! Covers the reachable model-visible (kind-A) routes for these capabilities: -//! - `targets_list` happy path (its only reachable route — every facade error is -//! kind-B `driver_unavailable`, so only the happy path is pinned). -//! - `target_set` happy path (settings decision `Allow` via default-ON -//! auto-approve → facade succeeds). -//! - `target_set` settings-`Deny` → `Failed{policy_denied}` (a `Disabled` tool -//! override, `OutboundDeliveryTargetSetHandler`'s -//! `OutboundDeliveryApprovalSettingsDecision::Deny` → `PolicyDenied` arm in -//! `runtime/local_dev/outbound_delivery.rs`). -//! - `target_set` facade `NotFound` → `Failed{invalid_input}` (unknown target, -//! `OutboundDeliveryTargetSetHandler`'s `NotFound` → -//! `CapabilityFailureKind::InvalidInput` arm in the same file). -//! - `target_set` approval gate: `Ask` (auto-approve disabled) → real -//! `BlockedApproval` gate → approve → resume applies the preference; deny → -//! resume leaves the preference unchanged. +//! Covers the reachable model-visible routes: `targets_list` happy path (its +//! only reachable route — every facade error is `driver_unavailable`); +//! `target_set` happy path; settings-`Deny` → `Failed{policy_denied}`; facade +//! `NotFound` → `Failed{invalid_input}`; approval gate `Ask` → approve applies +//! the preference / deny leaves it unchanged. //! -//! Read-back through the SAME facade double (`recorded_set_target_ids`) proves a -//! `Completed`/applied outcome actually reached the facade seam — a no-op set +//! Read-back through the SAME facade double (`recorded_set_target_ids`) proves +//! a `Completed`/applied outcome actually reached the facade seam — a no-op set //! that still fabricated a success payload would leave it empty. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::assertions::ToolErrorClass; @@ -337,19 +326,12 @@ async fn target_set_approval_gate_deny_leaves_preference_unchanged() { .await .expect("run resumes to Completed after denial"); - // A bare `Completed` status also matches a silent no-op/vanish bug (the - // resumed capability call could simply disappear rather than surface a - // model-visible denial). Pin the persisted gate-declined failure summary: - // `short_circuit_denied_resume` (capabilities.rs) surfaces this scenario as - // `CapabilityOutcome::Failed(GateDeclined)` with a fixed host-authored - // planner summary — `SanitizedStrategySummary::from_trusted_static( - // "approval gate denied by user")` — NOT the `capability_denied_summary`/ - // `capability_failed_summary` prefix wrapper (those apply only when a - // capability itself returns Denied/Failed, not this executor-level - // gate-declined short-circuit). `assert_tool_error_summary_contains` reads - // that raw safe_summary without the class-prefix requirement, mirroring - // the analogous auth-gate-deny assertion in - // `reborn_integration_auth_gate.rs`. + // A bare `Completed` also matches a silent no-op/vanish bug. Pin the + // gate-declined failure summary directly: `short_circuit_denied_resume` + // surfaces this as a fixed host-authored planner summary, NOT the + // `capability_denied_summary`/`capability_failed_summary` prefix wrapper + // (those apply only when a capability itself returns Denied/Failed). + // Mirrors the analogous assertion in `reborn_integration_auth_gate.rs`. harness .assert_tool_error_summary_contains("approval gate denied by user") .await diff --git a/tests/reborn_integration_process_port.rs b/tests/integration/process_port.rs similarity index 57% rename from tests/reborn_integration_process_port.rs rename to tests/integration/process_port.rs index 5e6ca83d562..db745bdf324 100644 --- a/tests/reborn_integration_process_port.rs +++ b/tests/integration/process_port.rs @@ -6,9 +6,10 @@ //! Docker, or `integration` feature. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::assertions::ToolErrorClass; @@ -109,8 +110,72 @@ async fn shell_timeout_surfaces_recoverable_failed() { .expect("run recovered and finalized (not terminal driver_unavailable)"); } -// `.with_live_shell()` test omitted: a live `echo` is hermetic but offers no -// assertion the recording-port test above doesn't already cover — the scripted -// model reply is fixed regardless of actual shell output, so there is nothing -// to assert on the real execution result beyond "the tool was invoked", which -// the recording test already proves end-to-end through the real dispatch path. +/// Live-shell path (`.with_live_shell()`) — proves `builtin.shell` dispatches +/// through the real `LocalHostProcessPort`, not the inert `RecordingProcessPort` +/// the other tests in this file cover. Asserts both directions: (a) the real +/// command's echoed output is visible in the model-facing tool result — only +/// possible if an actual OS process ran — and (b) `assert_shell_ran_through_inert_port` +/// (which passes only when the inert port recorded a command) reports `Err`, +/// proving the recording port's command buffer stayed empty. Guards against a +/// regression that routes live-shell requests back through +/// `core_builtin_tools_default()` (the inert path) while every other test in +/// this file — which never exercises `.with_live_shell()` — would stay green. +/// +/// Runs on a larger-stack thread (mirrors +/// `tests/reborn_qa_smoke_scenarios_e2e.rs::run_async_test_with_stack`): the +/// real `LocalHostProcessPort` subprocess path (spawn + piped-stream capture) +/// adds enough async-state-machine depth on top of the full +/// `product_workflow → composition → webui_v2 → runtime` chain to overflow +/// the default `#[tokio::test]` thread stack in a debug build. +#[test] +fn live_shell_uses_local_process_port() { + run_with_larger_stack(async { + let h = RebornIntegrationHarness::test_default() + .with_live_shell() + .script([ + RebornScriptedReply::tool_call( + "builtin.shell", + json!({"command": "echo live-shell-probe"}), + ), + RebornScriptedReply::text("done"), + ]) + .build() + .await + .expect("harness builds"); + h.submit_turn("run shell").await.expect("turn completes"); + h.assert_tool_result_contains("live-shell-probe") + .await + .expect("real process output surfaced in the model-visible tool result"); + assert!( + h.assert_shell_ran_through_inert_port().await.is_err(), + "live shell must not route through the inert RecordingProcessPort" + ); + h.assert_reply_contains("done") + .await + .expect("final reply finalized"); + }); +} + +/// Spawns `test` on a dedicated 16MB-stack thread with a current-thread tokio +/// runtime. See the doc comment on `live_shell_uses_local_process_port` for +/// why this one test needs it (matches the existing fix in +/// `tests/reborn_qa_smoke_scenarios_e2e.rs::run_async_test_with_stack`). +fn run_with_larger_stack(test: F) +where + F: std::future::Future + Send + 'static, +{ + let handle = std::thread::Builder::new() + .name("live_shell_uses_local_process_port".to_string()) + .stack_size(16 * 1024 * 1024) + .spawn(move || { + tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("tokio test runtime") + .block_on(test); + }) + .expect("spawn stack-sized test thread"); + if let Err(panic) = handle.join() { + std::panic::resume_unwind(panic); + } +} diff --git a/tests/reborn_integration_profile.rs b/tests/integration/profile.rs similarity index 64% rename from tests/reborn_integration_profile.rs rename to tests/integration/profile.rs index 6d5e0674203..b6cdc279163 100644 --- a/tests/reborn_integration_profile.rs +++ b/tests/integration/profile.rs @@ -1,25 +1,21 @@ -//! E-PROFILE seam smoke test: `profile_tools()` builds a `RebornIntegrationGroup` -//! whose ONE planned runtime is wired with a real `HostUserProfileSource` -//! (`build_user_profile_source_for_test`, backed by the local-dev memory -//! filesystem `builtin.profile_set` writes through) instead of the default -//! `EmptyUserProfileSource`. +//! E-PROFILE seam smoke test: `profile_tools()` wires ONE planned runtime with +//! a real `HostUserProfileSource` (backed by the local-dev memory filesystem +//! `builtin.profile_set` writes through) instead of `EmptyUserProfileSource`. //! -//! A scripted `builtin.profile_set` tool call dispatches through the REAL -//! production capability (`crates/ironclaw_host_runtime/src/first_party_tools/profile_set.rs`), -//! and the test then reads the profile back through the SAME -//! `Arc` instance the group's planned runtime is -//! built with (`RebornIntegrationGroup::user_profile_source_for_test`) — not a -//! re-derived equivalent — so a regression in the `into_group` wiring itself -//! (not just `build_user_profile_source_for_test`) fails this test. +//! A scripted `builtin.profile_set` call dispatches through the real +//! production capability; the test reads back through the SAME +//! `Arc` instance the group's runtime uses +//! (`user_profile_source_for_test`), so a regression in `into_group` wiring +//! itself — not just `build_user_profile_source_for_test` — fails this test. //! -//! Mutation-catching: if the profile source ignores its filesystem input (or -//! always returns `EmptyUserProfileSource`/`None`), `resolve_user_profile` -//! returns `None` here and the `expect` below fails. +//! Mutation-catching: an ignored filesystem input or an always-`None` source +//! makes `resolve_user_profile` return `None`, failing the `expect` below. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use ironclaw_turns::run_profile::{ @@ -30,13 +26,10 @@ use reborn_support::group::RebornIntegrationGroup; use reborn_support::reply::RebornScriptedReply; /// Build the `LoopRunContext` `resolve_user_profile` reads from, scoped to the -/// same `(tenant, user)` the `profile_set` write dispatched under: the -/// thread's binding tenant and the capability harness's dispatch user, which -/// `RebornIntegrationGroup::profile_tools()` aligns to the run's -/// canonical binding subject user via `with_user_id` (mirroring -/// `live_approvals`) — see `HostRuntimeCapabilityHarness::with_user_id` doc -/// comment — dispatch's `ResourceScope` is keyed on the harness user, not the -/// binding owner, unless overridden. +/// same `(tenant, user)` the `profile_set` write dispatched under — the +/// capability harness's dispatch user, which `profile_tools()` aligns to the +/// binding subject user via `with_user_id` (mirroring `live_approvals`; see +/// `HostRuntimeCapabilityHarness::with_user_id`). async fn read_back_run_context(tenant_id: &str, user_id: &str) -> LoopRunContext { let resolved_run_profile = InMemoryRunProfileResolver::default() .resolve_run_profile(RunProfileResolutionRequest::interactive_default()) @@ -85,10 +78,8 @@ async fn profile_set_write_is_readable_through_the_wired_profile_source() { .await .expect("profile_set dispatched through the real capability"); - // Read back through the SAME `HostUserProfileSource` the group's planned - // runtime was built with (E-PROFILE seam), keyed by the dispatch tenant - // and the capability harness's dispatch user (see `read_back_run_context` - // above for how that user is derived). + // Read back through the SAME HostUserProfileSource (E-PROFILE seam), keyed + // by the dispatch tenant/user (see `read_back_run_context` above). let dispatch_user = group .capability_harness() .expect("profile_tools always uses HostRuntime") @@ -119,10 +110,8 @@ async fn profile_set_write_is_readable_through_the_wired_profile_source() { "location must survive the round trip" ); - // The profile is resolved once per loop spawn (before turn 1's profile_set write), so a - // SECOND loop is needed to observe it in the model-visible prompt. A second thread in the - // same group shares the profile source under the SAME aligned (tenant, user) as turn 1, so - // its fresh loop renders the newly written profile into the runtime-context system message. + // Profile resolves once per loop spawn, before turn 1's write — a second + // thread's fresh loop is needed to observe it in the model-visible prompt. let prompt_thread = group .thread("conv-profile-prompt") .script([RebornScriptedReply::text("ok")]) diff --git a/tests/reborn_integration_project_create.rs b/tests/integration/project_create.rs similarity index 57% rename from tests/reborn_integration_project_create.rs rename to tests/integration/project_create.rs index a70d6cea8b5..439521441b4 100644 --- a/tests/reborn_integration_project_create.rs +++ b/tests/integration/project_create.rs @@ -1,23 +1,21 @@ //! E-PROJ seam smoke test: the `project_lifecycle` group surfaces the local-dev -//! synthetic `project_create` capability, and a scripted `builtin.project_create` -//! tool call dispatches through the REAL production synthetic-capability wrap +//! synthetic `project_create` capability; a scripted `builtin.project_create` +//! call dispatches through the REAL synthetic-capability wrap //! (`wrap_local_dev_synthetic_capabilities` + `project_create_capability`) and //! persists a project via the real `ProjectService`. //! -//! The result-contains assertion proves dispatch + a recorded output payload, -//! but not actual persistence: a regression that made `create_project` a -//! silent no-op while still fabricating a `{project_id, name}` success payload -//! would pass it. The read-back below closes that gap by re-querying the REAL -//! `ProjectService` (through `RebornIntegrationGroup::capability_harness` -> -//! `project_service_for_test`, the SAME instance -//! `apply_synthetic_capability_wrappers` dispatched the write through) and -//! asserting the created project is actually present — mirrors the E-PROFILE -//! `reborn_integration_profile` write -> read-back pattern. +//! A result-contains assertion alone would pass a silent-no-op regression that +//! still fabricates a success payload, so the read-back below re-queries the +//! SAME `ProjectService` instance the write dispatched through +//! (`capability_harness` -> `project_service_for_test`) and asserts the +//! project is actually present — mirrors the E-PROFILE write -> read-back +//! pattern. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use ironclaw_product_workflow::{ProjectCaller, RebornListProjectsRequest}; @@ -62,13 +60,9 @@ async fn project_create_capability_dispatches_and_persists_project() { // Persistence read-back (E-PROJ): re-fetch through the SAME `ProjectService` // instance the capability wrote through, scoped to the same `(tenant, user)` - // `project_create_capability::effective_user_id` derived the caller from. - // `effective_user_id` prefers the run scope's explicit thread owner, then - // the run actor, and only falls back to the capability harness's fixed - // constructor user when neither is set — this thread's binding has neither - // an explicit owner nor an override (`project_tools()` never calls - // `with_user_id`), so the actual dispatch caller is the thread's binding - // actor, not `capability_harness.user_id()`. + // `project_create_capability::effective_user_id` derived the caller from — + // here, the thread's binding actor (`project_tools()` never calls + // `with_user_id`, so it's not `capability_harness.user_id()`). let capability_harness = group .capability_harness() .expect("project_lifecycle always uses HostRuntime"); @@ -92,16 +86,12 @@ async fn project_create_capability_dispatches_and_persists_project() { ); } -/// An oversized `name` (201 ASCII bytes, over `MAX_PROJECT_NAME_BYTES=200`) -/// passes `parse_project_create_input`'s non-empty check but fails -/// `ProjectRecord::validate()` inside the real `ProjectService`, which returns -/// `ProjectServiceError::InvalidInput`. `project_service_outcome` maps that to -/// `CapabilityOutcome::Failed(CapabilityFailureKind::InvalidInput)`, persisted -/// as a `ToolResultReference` with `safe_summary` `"capability failed with -/// invalid_input: ..."` — a recoverable, model-visible tool error, not a -/// terminal `driver_unavailable` crash. Distinct from the happy-path test -/// above: this proves the reject path routes through the same -/// Completed-turn/Failed-outcome plumbing instead of aborting the run. +/// An oversized `name` (201 bytes, over `MAX_PROJECT_NAME_BYTES=200`) passes +/// input parsing but fails `ProjectRecord::validate()` inside the real +/// `ProjectService`; `project_service_outcome` maps the resulting +/// `InvalidInput` error to a recoverable, model-visible `Failed` tool error — +/// proving the reject path routes through Completed-turn/Failed-outcome +/// plumbing rather than aborting the run. #[tokio::test] async fn project_create_invalid_input_routes_to_recoverable_tool_error() { let group = RebornIntegrationGroup::project_lifecycle() @@ -136,35 +126,20 @@ async fn project_create_invalid_input_routes_to_recoverable_tool_error() { .expect("oversized name surfaces as a Failed(InvalidInput) capability outcome"); } -/// C-SYNTH fault-injection arm — `project_create` against a -/// `ProjectService::Denied` failure. Distinct from the two tests above: those -/// exercise the real `ProjectService` end-to-end (happy path) or a real, -/// model-controlled input-validation reject (`ProjectServiceError::InvalidInput`, -/// a caller mistake the real store itself detects). This test drives a -/// genuine host-side reject the real local-dev store cannot be coerced into -/// on demand — `create_project` calling through a `FaultInjectingProjectService` -/// decorator wrapping the real service (`project_lifecycle_fault_injected()`), -/// which forces `ProjectServiceError::Denied` only for the sentinel -/// `FAULT_INJECT_DENIED_PROJECT_NAME` and delegates every other call -/// (including a same-turn happy-path create) to the real store. -/// `project_service_outcome`'s `Denied` arm maps this to a model-visible, -/// recoverable `Failed(PolicyDenied)` tool error — NOT the terminal -/// `Internal` arm, which would instead kill the whole run — proving the -/// distinction end-to-end through the real capability dispatch rather than -/// only at `project_create.rs`'s unit-test level. +/// C-SYNTH fault-injection arm — `project_create` against a genuine host-side +/// `ProjectService::Denied` reject the real local-dev store can't be coerced +/// into on demand: `create_project` calls through a +/// `FaultInjectingProjectService` decorator (`project_lifecycle_fault_injected()`) +/// that forces `Denied` only for `FAULT_INJECT_DENIED_PROJECT_NAME` and +/// delegates everything else to the real store. `project_service_outcome`'s +/// `Denied` arm maps this to a recoverable `Failed(PolicyDenied)` tool error on +/// the FIRST attempt — not the terminal `Internal` arm. /// -/// Deliberately NOT `ProjectServiceError::Unavailable`/`Internal`: both route -/// through `DefaultRecoveryStrategy`'s capability-retry branch, whose retry -/// re-dispatch hits an independently confirmed production bug for -/// provider-tool-call-originated invocations under local-dev composition -/// (`LocalDevCapabilityIo::resolve_capability_input` rejects the reused -/// `input_ref` on the retry attempt with `InvalidInvocation`/"capability -/// input ref was not staged for this loop run", collapsing the documented -/// "retry twice, then a model-visible `Failed`" contract into an immediate -/// terminal `driver_unavailable`) — see issue #5608. `Denied` -/// surfaces to the model on the FIRST attempt -/// (`capability_error_is_model_visible_tool_failure`), so it exercises a -/// genuine `project_service_outcome` arm without tripping that unrelated bug. +/// Deliberately NOT `Unavailable`/`Internal`: both route through the +/// capability-retry branch, whose retry re-dispatch hits an unrelated, +/// independently confirmed bug for provider-tool-call-originated invocations +/// under local-dev composition (issue #5608) that collapses the "retry twice, +/// then Failed" contract into an immediate `driver_unavailable`. #[tokio::test] async fn project_create_denied_fault_routes_to_recoverable_tool_error() { let group = RebornIntegrationGroup::project_lifecycle_fault_injected() diff --git a/tests/reborn_integration_safety.rs b/tests/integration/safety.rs similarity index 97% rename from tests/reborn_integration_safety.rs rename to tests/integration/safety.rs index 753dea03a27..b68e54b929c 100644 --- a/tests/reborn_integration_safety.rs +++ b/tests/integration/safety.rs @@ -9,9 +9,10 @@ //! content rather than passing vacuously. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use ironclaw_turns::run_profile::InstructionSafetyContext; diff --git a/tests/reborn_integration_secret_injection.rs b/tests/integration/secret_injection.rs similarity index 79% rename from tests/reborn_integration_secret_injection.rs rename to tests/integration/secret_injection.rs index 04e032857b8..47bf4ae43ed 100644 --- a/tests/reborn_integration_secret_injection.rs +++ b/tests/integration/secret_injection.rs @@ -1,26 +1,22 @@ //! Reborn integration-test tier — T0-SECRET-INJECT. //! //! Proves credential/secret injection reaches the wire: a scripted `github.*` -//! tool call executes the real first-party GitHub WASM capability behind a +//! call executes the real first-party GitHub WASM capability behind a //! `GithubHarnessAuthorizer` that attaches an `InjectCredentialAccountOnce` //! obligation. The host egress pipeline resolves the synthetic access token -//! (`ghp_fake_fixture_token`, from the harness `StaticSecretStore`) and injects -//! it as `Authorization: Bearer ` onto the outbound request before the -//! recording network egress captures it. The assertion reads that captured -//! request and confirms the injected credential is present on the header. -//! -//! Note on the egress lane: this harness's runtime egress recorder -//! (`runtime_http_requests()`) is inert — `try_with_host_http_egress` overwrites -//! the runtime port with the host pipeline over the recording *network* egress — -//! so injection is observable on the network lane. See -//! `assert_network_egress_header_contains` for the full mechanism. +//! (from the harness `StaticSecretStore`) and injects it as `Authorization: +//! Bearer ` onto the outbound request before the recording *network* +//! egress captures it (the runtime egress lane is inert here — +//! `try_with_host_http_egress` overwrites it — see +//! `assert_network_egress_header_contains`). //! //! Security: the token is a synthetic test fixture, never a real credential. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::builder::RebornIntegrationHarness; diff --git a/tests/reborn_integration_secrets.rs b/tests/integration/secrets.rs similarity index 80% rename from tests/reborn_integration_secrets.rs rename to tests/integration/secrets.rs index b4bc251cdde..604b3e127f8 100644 --- a/tests/reborn_integration_secrets.rs +++ b/tests/integration/secrets.rs @@ -2,20 +2,19 @@ //! //! Store-level durability proof: writes a secret to a `FilesystemSecretStore` //! backed by a libSQL composite, then reopens a genuinely fresh store over the -//! same on-disk database file and reads the secret back. Proves real on-disk -//! secret durability without exercising the turn/model layer. +//! same on-disk database file and reads the secret back — real on-disk +//! durability, without the turn/model layer. //! -//! Gated on the `libsql` feature because the test directly instantiates -//! `LibSqlRootFilesystem`, a type that compiles only under -//! `feature = "libsql"`. Running without the feature produces 0 tests -//! (compile-safe); running with `--features libsql` produces 2 tests. +//! Gated on `feature = "libsql"` (directly instantiates `LibSqlRootFilesystem`, +//! which only compiles under that feature); 0 tests without it, 2 with it. #![cfg(feature = "libsql")] #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use std::sync::Arc; @@ -73,12 +72,8 @@ async fn secret_persists_across_libsql_reopen() { drop(composite); // --- Reopen: fresh libsql database, fresh composite, fresh store --- - // - // Mirrors the `assert_reply_persists_after_reopen` pattern in - // `tests/support/reborn/builder.rs`: `libsql::Builder::new_local` opens - // the existing file, `run_migrations` is idempotent (schema already - // exists), and the SAME `root` path yields the SAME cached master-key - // file so decryption succeeds on the fresh store. + // Mirrors `assert_reply_persists_after_reopen` (builder.rs): same `root` + // path yields the same cached master-key file, so decryption succeeds. let db_path = dir.path().join(LOCAL_DEV_DB_FILENAME); let db = Arc::new( libsql::Builder::new_local(&db_path) @@ -156,22 +151,15 @@ async fn secret_read_back_fails_for_unknown_handle() { } /// Prove secrets don't leak across tenants: leasing a secret under a scope -/// that differs ONLY in `tenant_id` (same host user, agent, and project) -/// must fail, and must fail with the same discriminating error as the -/// unknown-handle case above — `FilesystemSecretStore::read_secret` -/// (`crates/ironclaw_secrets/src/filesystem_store.rs`) enforces this via -/// `same_scope_owner`, which compares `tenant_id`/`user_id`/`agent_id`/ -/// `project_id` on the stored secret against the caller's scope and treats -/// any mismatch as "not found" — the same code path a genuinely unknown -/// handle hits, not a distinct "forbidden" branch. That is still a real, -/// discriminating assertion: it proves the tenant mismatch is what's -/// gating the read (not some unrelated store failure), and the -/// non-vacuity check below proves the store isn't just broadly broken. +/// that differs ONLY in `tenant_id` must fail with the same +/// `UnknownSecret` error as the unknown-handle case — `same_scope_owner` +/// (`FilesystemSecretStore::read_secret`) treats any scope-field mismatch as +/// "not found" rather than a distinct "forbidden" branch. The non-vacuity +/// check below proves the store isn't just broadly broken. /// -/// (A same-tenant-different-user cross-read is a distinct mechanic — same -/// `same_scope_owner` comparison, different axis — and is intentionally -/// left for a separate test; this one isolates the tenant dimension only, -/// per the C-DENYEDGE row-2 scope.) +/// (Same-tenant-different-user cross-read is a distinct axis of the same +/// `same_scope_owner` comparison, left for a separate test — C-DENYEDGE row 2 +/// isolates tenant only.) #[tokio::test] async fn secret_read_back_fails_for_wrong_tenant_scope() { let dir = tempfile::tempdir().expect("temp dir"); diff --git a/tests/reborn_integration_skill_activate.rs b/tests/integration/skill_activate.rs similarity index 77% rename from tests/reborn_integration_skill_activate.rs rename to tests/integration/skill_activate.rs index 7638b695009..0cd02430ae4 100644 --- a/tests/reborn_integration_skill_activate.rs +++ b/tests/integration/skill_activate.rs @@ -3,31 +3,24 @@ //! //! A `greet` system skill is seeded by `skill_activation_tools()`. The model //! explicitly activates it through the REAL local-dev synthetic capability -//! (`builtin.skill_activate`, dispatched via -//! `wrap_skill_activation_capability_for_test`), and the test then proves BOTH -//! halves of the seam: +//! (`builtin.skill_activate`), and the test proves BOTH halves of the seam: +//! the capability dispatched and reported the skill activated (`count: 1`), +//! and its instructions reached a subsequent model request through the wired +//! `skill_context_source` (`assert_model_request_contains`). //! -//! - the capability dispatched and reported the skill activated (`count: 1`), -//! and -//! - the activated skill's instructions reached a subsequent model request -//! through the runtime's wired `skill_context_source` -//! (`assert_model_request_contains`). -//! -//! The user message deliberately omits the skill's `greet` activation keyword, -//! so the injected `GREET_SKILL_PROMPT_SENTINEL` can only originate from the -//! explicit `skill_activate` call — not from keyword auto-activation. If either -//! the capability wrap or the `into_group` `skill_context_source` wiring -//! regresses, the sentinel never reaches a captured request and the assert fails. +//! The user message omits the skill's `greet` keyword, so the injected +//! `GREET_SKILL_PROMPT_SENTINEL` can only originate from the explicit +//! `skill_activate` call, not keyword auto-activation. //! //! The second test below pins the OTHER half of E-SKILL: criteria-based //! (keyword) auto-activation stays OFF on the coordinator path (product -//! decision, closed issue #5530) — see the mechanism note on that test for -//! why. +//! decision, closed issue #5530). #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::assertions::ToolErrorClass; @@ -75,14 +68,12 @@ async fn skill_activate_dispatches_and_injects_skill_context() { } // INTENTIONAL: `SkillActivationMode::ActivationCriteria` (keyword/regex -// auto-activation) does not fire on the modern `TurnCoordinator`/agent-loop -// path — disabled on purpose (product decision, closed issue #5530). The -// explicit `builtin.skill_activate` capability path (proven above) is the -// supported mechanism. Mechanically: criteria selection only runs when -// `take_message_for_run` returns `Some`, populated by `record_user_message`, -// whose sole production caller is the legacy `RebornRuntime::submit_user_turn` -// — the coordinator stack never records the message, so criteria selection -// stays inert here by design. +// auto-activation) does not fire on the `TurnCoordinator`/agent-loop path — +// disabled on purpose (product decision, closed issue #5530). It only runs +// when `record_user_message` populates `take_message_for_run`, whose sole +// caller is the legacy `RebornRuntime::submit_user_turn`; the coordinator +// stack never records the message, so criteria selection stays inert here. +// The explicit `builtin.skill_activate` path (proven above) is supported. #[tokio::test] async fn skill_criteria_auto_activation_stays_off_on_coordinator_path() { let group = RebornIntegrationGroup::skill_activation_tools() @@ -133,13 +124,9 @@ async fn skill_criteria_auto_activation_stays_off_on_coordinator_path() { /// C-SYNTH failure route — `skill_activate` `ContextBudgetExceeded` is a /// MODEL-VISIBLE `Failed` tool error (recoverable), not a terminal driver -/// error. -/// -/// An oversized system skill (prompt ≈ 10k tokens at the ~4-bytes-per-token -/// estimate, over `DEFAULT_MAX_SKILL_CONTEXT_TOKENS = 4000`) is seeded via -/// `seed_system_skill_for_test`. Activating it drives the real selection path -/// (`reserve_skill_budget` → `SkillActivationSelectionError::ContextBudgetExceeded` -/// → `CapabilityOutcome::Failed`, skill_activation.rs `selection_outcome`). +/// error. An oversized system skill (~10k tokens, over +/// `DEFAULT_MAX_SKILL_CONTEXT_TOKENS = 4000`) drives the real selection path +/// (`reserve_skill_budget` → `ContextBudgetExceeded` → `Failed`). #[tokio::test] async fn skill_activate_over_budget_surfaces_recoverable_failed() { let group = RebornIntegrationGroup::skill_activation_tools() @@ -192,17 +179,12 @@ async fn skill_activate_over_budget_surfaces_recoverable_failed() { ); } -/// C-SYNTH `AmbiguousSkill` seeding arm — a skill name that resolves to TWO -/// Trusted candidates under different `SkillSourceKind`s (a system-scoped -/// skill AND a user-scoped skill sharing the same name) drives the real -/// `validate_explicit_mentions_are_unambiguous` reject path -/// (`SkillActivationSelectionError::AmbiguousSkill` -> -/// `skill_activation_selection_outcome`'s `AmbiguousSkill` arm), NOT the -/// `ContextBudgetExceeded` arm the sibling test above already covers. Both -/// arms map to a model-visible, recoverable `Failed(InvalidInput)` — this -/// test proves the OTHER arm reaches that same outcome through the real -/// capability dispatch, and that neither candidate's instructions leak into a -/// later model request despite the name matching both. +/// C-SYNTH `AmbiguousSkill` seeding arm — a skill name resolving to TWO +/// Trusted candidates (system-scoped + user-scoped, same name) drives the +/// real `validate_explicit_mentions_are_unambiguous` reject path +/// (`AmbiguousSkill` → `Failed(InvalidInput)`), distinct from the +/// `ContextBudgetExceeded` arm above. Proves neither candidate's instructions +/// leak into a later model request despite the name matching both. #[tokio::test] async fn skill_activate_ambiguous_name_surfaces_recoverable_failed() { let group = RebornIntegrationGroup::skill_activation_tools() diff --git a/tests/support/reborn/assertions.rs b/tests/integration/support/assertions.rs similarity index 82% rename from tests/support/reborn/assertions.rs rename to tests/integration/support/assertions.rs index f1968b61d40..d28073ac7c7 100644 --- a/tests/support/reborn/assertions.rs +++ b/tests/integration/support/assertions.rs @@ -1,21 +1,15 @@ //! Egress + tool-result + model-prompt assertions for [`RebornIntegrationHarness`]. //! -//! These read the captured Tier-2 `RuntimeHttpEgressRequest`s and recorded -//! capability results through the `pub(super)` accessors on the harness -//! (`captured_egress_requests` / `captured_capability_results`) rather than -//! re-reaching internals. +//! Read the captured Tier-2 `RuntimeHttpEgressRequest`s and recorded +//! capability results through the harness's `pub(super)` accessors rather than +//! reaching internals directly. //! -//! The egress-assertion group (`assert_egress_count` / `assert_egress_url_order` -//! / `assert_egress_method_order` / `assert_egress_body_contains`) all assert -//! over the SAME captured `RecordingRuntimeHttpEgress` request log — there is -//! one runtime-lane egress-assertion API, not a parallel one. The one -//! exception is `assert_network_egress_header_contains`, which reads the -//! recording *network* egress lane — required for the T0-SECRET-INJECT -//! credential-injection proof, whose harness routes through the host egress -//! pipeline over the network recorder (see that method's docs for why). -//! `assert_system_prompt_contains` reads a different capture source — the -//! scripted `TraceLlm`'s captured requests, via the harness's -//! `captured_system_prompts` accessor. +//! The egress-assertion group (`assert_egress_count`/`assert_egress_url_order`/ +//! `assert_egress_method_order`/`assert_egress_body_contains`) all read the SAME +//! `RecordingRuntimeHttpEgress` log — one runtime-lane API, not a parallel one. +//! `assert_network_egress_header_contains` is the exception: it reads the +//! *network* egress lane (required for T0-SECRET-INJECT). `assert_system_prompt_contains` +//! reads yet another source — the scripted `TraceLlm`'s captured requests. // Shared integration-test support: not every binary that mounts the // `reborn_support` tree consumes this module (e.g. `support_unit_tests.rs`), so @@ -161,13 +155,11 @@ impl RebornIntegrationHarness { /// Assert that ANY captured egress request whose URL contains `url_substr` /// carried a body containing `body_substr` — checks every matching request, /// not just the first. Needed for a multi-request handshake where every leg - /// shares the same URL (e.g. web-access's Exa MCP `initialize` / - /// `notifications/initialized` / `tools/call` sequence, C-WEBACCESS) and - /// only one leg's body carries the substring under test. Prefer - /// [`assert_egress_body_contains`] when `url_substr` is expected to match - /// exactly one request — its first-match semantics catch a false pass that - /// this looser check would miss if a later, unrelated same-URL request also - /// happened to satisfy `body_substr`. + /// shares the same URL (e.g. web-access's Exa MCP handshake, C-WEBACCESS) + /// and only one leg's body carries the substring. Prefer + /// [`assert_egress_body_contains`] when `url_substr` matches exactly one + /// request — its first-match semantics catch a false pass this looser + /// check would miss. pub async fn assert_egress_body_contains_any( &self, url_substr: &str, @@ -246,16 +238,13 @@ impl RebornIntegrationHarness { .into()) } - /// Assert that some SINGLE model request this thread sent to the scripted - /// provider contains EVERY needle in `needles` (all in one request, not - /// spread across several). This is the multi-turn "sees prior context" - /// proof: pass a needle unique to an earlier turn plus one unique to the - /// current turn — only the current turn's request carries BOTH, because the - /// earlier turn's own request predates the later text. Scanning with the - /// single-needle [`assert_model_request_contains`] cannot express this (each - /// needle would trivially match its own originating request), so a genuine - /// context-carryover regression (the loop rebuilding the request without - /// prior history) would slip through it but not through this. + /// Assert some SINGLE model request contains EVERY needle in `needles` + /// (all in one request, not spread across several) — the multi-turn "sees + /// prior context" proof: an earlier-turn needle plus a current-turn needle + /// only co-occur if history carried over. The single-needle + /// [`assert_model_request_contains`] can't express this (each needle + /// trivially matches its own originating request), so it would miss a + /// context-carryover regression this catches. pub async fn assert_model_request_contains_all(&self, needles: &[&str]) -> HarnessResult<()> { let requests = self.scripted_llm.captured_requests(); for messages in &requests { @@ -273,15 +262,12 @@ impl RebornIntegrationHarness { } /// Collects the persisted `safe_summary` field of every `ToolResultReference` - /// message on this thread's FULL history (not baseline-sliced — same caveat - /// as `assert_tool_error`/`assert_tool_error_summary_contains`: safe only for - /// single-turn harnesses today). Shared collector for [`assert_tool_error`], - /// [`assert_no_tool_error`], and [`assert_tool_error_summary_contains`]. - /// - /// A `ToolResultReference` message with `content: None`, or with `content` - /// that fails to decode as a `ToolResultReferenceEnvelope`, is an `Err` — - /// never silently skipped. Both would otherwise vanish from `summaries` - /// and degrade into a misleading "not found; saw [...]" for the caller. + /// message on this thread's FULL history (not baseline-sliced — safe only + /// for single-turn harnesses today). Shared collector for + /// [`assert_tool_error`], [`assert_no_tool_error`], and + /// [`assert_tool_error_summary_contains`]. Fail loud (never silently skip): + /// missing or undecodable `content` is an `Err`, not an omission that would + /// degrade into a misleading "not found" for the caller. async fn persisted_tool_error_summaries(&self) -> HarnessResult> { let history = self .thread_harness @@ -367,35 +353,23 @@ impl RebornIntegrationHarness { } /// Assert a model-visible tool error of `class` carrying `reason` was - /// persisted for this thread. Unlike [`assert_tool_result_contains`] (which - /// reads the in-process recorder, populated only on the *Completed* write - /// path), a `Failed`/`Denied` capability outcome is persisted through a - /// different pipeline — `append_capability_result_ref` → - /// `append_tool_result_reference` — as a `MessageKind::ToolResultReference` - /// message whose `content` is the JSON-serialized `ToolResultReferenceEnvelope`. - /// Reaching this state at all (rather than a terminal `driver_unavailable`) - /// also proves the failure was a recoverable, model-visible tool error. + /// persisted for this thread. Unlike [`assert_tool_result_contains`] + /// (in-process recorder, *Completed* path only), a `Failed`/`Denied` + /// outcome persists as a `MessageKind::ToolResultReference` message — + /// reaching this state at all proves it was recoverable, not a terminal + /// `driver_unavailable`. /// - /// This parses the envelope and checks its **`safe_summary` field** — NOT a - /// raw-JSON substring — so `reason` cannot match incidentally inside - /// `model_observation`/`result_ref`, and JSON escaping can't skew the match. - /// The summary reads `"capability with : …"`; the - /// assertion requires the summary to start with [`class`](ToolErrorClass)'s - /// prefix AND contain `reason`. `class` therefore discriminates - /// Failed-vs-Denied structurally: a regression that flips one into the other - /// fails even when both classes render the same `reason` token (e.g. - /// `policy_denied`). + /// Parses the envelope's **`safe_summary` field** (not a raw-JSON + /// substring, so `reason` can't match incidentally elsewhere) and requires + /// it to start with [`class`](ToolErrorClass)'s prefix AND contain + /// `reason` — `class` discriminates Failed-vs-Denied structurally even + /// when both render the same `reason` token (e.g. `policy_denied`). /// - /// **Scans the full thread history, not baseline-sliced** (unlike the - /// sibling `assert_egress_*`/`assert_tool_result_contains`, which slice - /// `[baseline..]` off the shared in-process recorder). Every current caller - /// is a single-turn, single-tool-call harness, so there is at most one - /// `ToolResultReference` message and no earlier-thread bleed-through is - /// reachable. A future multi-turn or group-thread reuse of this assertion - /// MUST add baseline scoping first (thread a `baseline_history_len` through - /// harness construction, mirroring `baseline_egress_count` etc.) — do not - /// assume this helper is safe to reuse as-is once a thread has more than one - /// turn. + /// **Scans the full thread history, not baseline-sliced.** Safe today + /// because every caller is single-turn/single-tool-call; a future + /// multi-turn or group-thread reuse MUST add baseline scoping first + /// (mirroring `baseline_egress_count`) before assuming this is reusable + /// as-is. pub async fn assert_tool_error( &self, class: ToolErrorClass, @@ -443,14 +417,11 @@ impl RebornIntegrationHarness { /// Assert some persisted `ToolResultReference`'s raw `safe_summary` text /// contains `text` — NO class-prefix requirement. Complements - /// [`assert_tool_error`] for `CapabilityErrorSummary`s the executor builds - /// via `SanitizedStrategySummary::from_trusted_static` in - /// `crates/ironclaw_agent_loop/src/executor/capabilities.rs` (filtered-surface - /// denial, stale-surface retry, auth/approval gate-declined short-circuit) — - /// those are fixed host-authored literals with no host-returned text to - /// prefix, so `assert_tool_error`'s `capability_{failed,denied}_summary` - /// prefix match can never succeed for them. Use only for known - /// executor-synthesized literals. + /// [`assert_tool_error`] for executor-synthesized `CapabilityErrorSummary`s + /// (filtered-surface denial, stale-surface retry, gate-declined + /// short-circuit — `executor/capabilities.rs`) that are fixed host-authored + /// literals with no host-returned text, so `assert_tool_error`'s prefix + /// match can never succeed for them. pub async fn assert_tool_error_summary_contains(&self, text: &str) -> HarnessResult<()> { let summaries = self.persisted_tool_error_summaries().await?; if summaries.iter().any(|summary| summary.contains(text)) { @@ -464,24 +435,16 @@ impl RebornIntegrationHarness { /// Assert that any captured **network** egress request whose URL /// contains `url_substr` carried a header named `header_name` - /// (case-insensitive) whose value contains `value_substr`. This is the - /// credential-injection-on-the-wire proof for T0-SECRET-INJECT: a - /// host-injected `Authorization: Bearer ` lands on the outbound - /// request only after the egress pipeline's `apply_credential_injections` - /// step, which the recording network egress captures. + /// (case-insensitive) whose value contains `value_substr` — the + /// credential-injection-on-the-wire proof for T0-SECRET-INJECT. /// - /// **Why the network lane, not the runtime lane:** the GitHub WASM harness - /// (`with_github_issue_tools`) wires its recording `RuntimeHttpEgress` and - /// then calls `try_with_host_http_egress`, which overwrites the runtime port - /// with the host egress pipeline over the recording *network* egress. So the - /// injected request flows through the network recorder, and the runtime-lane - /// `assert_egress_*` family (which reads `runtime_http_requests()`) is inert - /// for this wiring. Assert here instead. + /// **Why the network lane, not the runtime lane:** `with_github_issue_tools`'s + /// `try_with_host_http_egress` overwrites the runtime port with the host + /// egress pipeline over the recording *network* egress, so the runtime-lane + /// `assert_egress_*` family is inert for this wiring — assert here instead. /// - /// Checks only the `[baseline_network_count..]` delta so a group thread never - /// spuriously matches a prior thread's request (R2), mirroring the runtime-lane - /// `assert_egress_*` family's baseline discipline even though no group - /// constructor wires `GithubIssueTools` today. + /// Checks only `[baseline_network_count..]` so a group thread never + /// spuriously matches a prior thread's request (R2). pub async fn assert_network_egress_header_contains( &self, url_substr: &str, @@ -528,18 +491,13 @@ impl RebornIntegrationHarness { /// C-BUDGET liveness assertion: the group's wired `model_budget_accountant` /// seeded the run owner's daily USD cap on the turn's first model call. + /// Reads the in-memory `ResourceGovernor` behind `build_default_budget_accountant` + /// (wired via `with_budget_accounting()`); asserting the cap equals the + /// compiled default (`$5.00`) proves it came from `BudgetDefaults` through + /// the real coordinator → loop → model-port path, not an incidental path. /// - /// Reads the in-memory `ResourceGovernor` retained behind the production - /// `build_default_budget_accountant` accountant (wired via - /// `with_budget_accounting()` / `budget_accounting()`). Before any turn the - /// run-owner account does not exist; after a completed turn the accountant's - /// `pre_model_call` has fired through the real coordinator → loop → model-port - /// path and its compiled-default seeding policy has installed the daily cap. - /// Asserting the cap equals the compiled default (`$5.00`) proves the value - /// came from the production helper's `BudgetDefaults`, not an incidental path. - /// - /// This is wiring-liveness only — budget SEMANTICS (thresholds, gates, - /// `BudgetEvent` cascade) are covered at crate tier (`budget_e2e.rs`). + /// Wiring-liveness only — budget SEMANTICS are covered at crate tier + /// (`budget_e2e.rs`). pub async fn assert_budget_user_cap_seeded(&self) -> HarnessResult<()> { let governor = self._shared.budget_governor.as_ref().ok_or( "harness was not built with budget accounting wired (call with_budget_accounting)", diff --git a/tests/support/reborn/builder.rs b/tests/integration/support/builder.rs similarity index 85% rename from tests/support/reborn/builder.rs rename to tests/integration/support/builder.rs index 0a781a5fcd3..262900ecb67 100644 --- a/tests/support/reborn/builder.rs +++ b/tests/integration/support/builder.rs @@ -1,19 +1,16 @@ //! `RebornIntegrationHarness` — the integration test tier that runs the full //! internal Reborn stack and intercepts the model at the vendor-SDK seam. //! -//! Unlike `RebornBinaryE2EHarness` (which swaps the whole `HostManagedModelGateway` -//! with `RebornTraceReplayModelGateway`), this tier wires the REAL -//! `LlmProviderModelGateway` over the REAL `ironclaw_llm` decorator chain -//! (`apply_decorator_chain`, hermetic passthrough) and only scripts the raw -//! provider underneath via `TraceLlm`. A turn therefore exercises model-profile -//! resolution, `CompletionRequest`/tool-definition assembly, and the -//! retry/routing/circuit/cache decorators for real. +//! Unlike `RebornBinaryE2EHarness` (swaps the whole `HostManagedModelGateway`), +//! this tier wires the REAL `LlmProviderModelGateway` over the REAL +//! `ironclaw_llm` decorator chain and only scripts the raw provider underneath +//! via `TraceLlm` — a turn exercises model-profile resolution, request/tool-def +//! assembly, and the retry/routing/circuit/cache decorators for real. //! -//! `StorageMode { InMemory, LibSql }` — the builder defaults to `InMemory`; +//! `StorageMode { InMemory, LibSql }` — defaults to `InMemory`; //! `.storage(StorageMode::LibSql)` selects a real SQLite file in a -//! per-`build()` `TempDir`. Both modes ride **one** `CompositeRootFilesystem` -//! at `/tenants/...` so thread history and turn state share the same backend -//! and the same production path layout. +//! per-`build()` `TempDir`. Both ride **one** `CompositeRootFilesystem` at +//! `/tenants/...` so thread history and turn state share the same backend. // Shared integration-test support: not every binary that mounts the // `reborn_support` tree consumes this module — `support_unit_tests.rs` mounts @@ -71,17 +68,14 @@ pub(crate) const HARNESS_ACTOR_ID: &str = "host-user"; pub(crate) const INTERACTIVE_MODEL_PROFILE: &str = "interactive_model"; /// Selects the durable storage backend mounted into the integration harness's -/// `CompositeRootFilesystem` (design spec §3.2, §3.8). +/// `CompositeRootFilesystem`. Both modes ride **one** composite at the +/// production path layout `/tenants//users//...` — the only +/// difference is which `RootFilesystem` is mounted under `/tenants`, +/// `/memory`, and `/events`. /// -/// Both modes ride **one** composite at the production path layout -/// `/tenants//users//...` — the only difference is which -/// `RootFilesystem` is mounted under `/tenants`, `/memory`, and `/events`. -/// -/// `InMemory` is the default: it's fast, needs no filesystem, and covers -/// all assertion cases that don't require on-disk durability. -/// `LibSql` creates a real SQLite file in a per-`build()` `TempDir`, runs -/// the full libSQL migration suite, and lets `assert_reply_persists_after_reopen` -/// verify that data survived serialization to disk (design §3.8 guardrail). +/// `InMemory` (default): fast, no filesystem, covers all cases that don't +/// need on-disk durability. `LibSql`: real SQLite in a per-`build()` +/// `TempDir`, full migrations, enables `assert_reply_persists_after_reopen`. #[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] pub enum StorageMode { /// In-memory backend: fast, no filesystem I/O, default. @@ -153,7 +147,7 @@ impl RebornIntegrationHarnessBuilder { /// into the harness's underlying group. Rendered verbatim as a `system`-role /// prompt message ahead of any per-turn instructions; read back via /// `assert_system_prompt_contains`. Defaults to `None` (no banner, matching - /// today's behavior) — see `tests/support/reborn/group.rs`'s + /// today's behavior) — see `tests/integration/support/group.rs`'s /// `RebornIntegrationGroupBuilder::safety_context` for the underlying wiring. pub fn with_safety_context(mut self, ctx: InstructionSafetyContext) -> Self { self.safety_context = Some(ctx); @@ -292,16 +286,13 @@ impl RebornIntegrationHarnessBuilder { } /// W4-AUTHGATE-WIRE: script the GitHub WASM capability's real HTTP call to - /// come back with `status` instead of the default `200` fixture (FIFO, - /// one call consumed per queued status). Unlike `with_keyed_http_responses` - /// (the runtime-egress lane), the `GithubIssueTools` backend's real call - /// flows through the **network** egress lane — see - /// `reborn_integration_secret_injection.rs`'s module doc - /// (`try_with_host_http_egress` overwrites the runtime port with the host - /// pipeline over the recording network egress) — so a runtime-401 (the - /// credential-injected-but-401 auth-gate path, distinct from - /// `github_issue_tools_auth_required`'s credential-*missing* path) must be - /// scripted here instead. Implies [`with_github_issue_tools`](Self::with_github_issue_tools). + /// come back with `status` instead of the default `200` (FIFO, one call + /// consumed per queued status). The `GithubIssueTools` backend's real call + /// flows through the **network** egress lane, not the runtime-egress lane + /// `with_keyed_http_responses` scripts (`try_with_host_http_egress` + /// overwrites the runtime port), so a runtime-401 (credential-injected-but-401, + /// distinct from `github_issue_tools_auth_required`'s credential-missing + /// path) must be scripted here. Implies [`with_github_issue_tools`](Self::with_github_issue_tools). pub fn with_github_network_status(mut self, status: u16) -> Self { self.capability = RebornCapabilityBackend::GithubIssueTools; self.github_network_statuses.push(status); @@ -347,11 +338,8 @@ impl RebornIntegrationHarnessBuilder { /// the scripted provider, and start the planned runtime. /// /// Routes through an internal, degenerate one-thread `RebornIntegrationGroup` - /// (matching this builder's capability/storage/shell_mode/keyed_http_responses - /// selections) so there is exactly ONE assembly path for both groups and - /// single-shot harnesses (design §3: no de-facto fork). Behavior is - /// byte-identical to the old inline build — existing tests are unaffected - /// (R1/R5). + /// so there is exactly ONE assembly path for both groups and single-shot + /// harnesses — no de-facto fork. pub async fn build(self) -> HarnessResult { apply_hermetic_env(); @@ -410,24 +398,16 @@ pub struct RebornIntegrationHarness { pub(crate) conversation_id: String, /// External (raw, pre-resolution) actor id every submit for this thread is /// made under. Defaults to `HARNESS_ACTOR_ID`; a group thread built with - /// `with_actor_id` (the E-MULTIUSER seam) carries its distinct actor here - /// so submit-time envelopes resolve the SAME binding (and owner scope) as - /// the build-time probe. + /// `with_actor_id` (E-MULTIUSER seam) carries its distinct actor here so + /// submit-time envelopes resolve the SAME binding as the build-time probe. /// - /// NOT redundant with `binding.actor_user_id`: the binding's field is a - /// one-way SHA-256-derived opaque `UserId` (`user_id_for_binding` / - /// `scoped_id`, product_workflow.rs), while `verified_text_envelope_with_trigger` - /// (called by both the build-time probe and every submit) needs the raw, - /// pre-hash external actor-id string to compute the SAME `binding_path` - /// hash the probe persisted under. Substituting `binding.actor_user_id` - /// here would make submit resolve a *different*, unrelated binding (new - /// `actor_user_id`/`subject_user_id`, same `thread_id` since that hash is - /// actor-independent) instead of reusing this harness's own — silently - /// breaking every `submit_turn`/`submit_turn_async` call, not just the - /// multi-actor scenario. `binding.actor_user_id` IS the right source of - /// truth for `resume_run`'s `TurnActor` (an already-resolved identity, no - /// envelope round-trip involved) — that is a materially different use - /// case from this field's. + /// NOT redundant with `binding.actor_user_id` (a one-way hashed opaque + /// `UserId`): `verified_text_envelope_with_trigger` needs the raw, + /// pre-hash string to compute the SAME `binding_path` hash the probe + /// persisted under — substituting the hashed field here would silently + /// resolve a different binding on every submit. `binding.actor_user_id` + /// remains the right source for `resume_run`'s `TurnActor` (no envelope + /// round-trip there). pub(crate) actor_id: String, pub(crate) binding: ResolvedBinding, pub(crate) turn_scope: TurnScope, @@ -440,15 +420,11 @@ pub struct RebornIntegrationHarness { pub(crate) event_seq: AtomicU64, pub(crate) capability_recorder: HarnessCapabilityRecorder, /// The concrete scripted `TraceLlm` retained before it was upcast to - /// `dyn LlmProvider` in the per-thread gateway build. Its - /// `captured_requests()` lets assertions inspect the exact model-visible - /// requests: the system prompt (safety banners, profile lines — - /// `captured_system_prompts()`/`assert_system_prompt_contains`) and any - /// host-injected context such as activated-skill instructions - /// (`assert_model_request_contains`, E-SKILL half B). Retained even when - /// the thread parks the model (`park_model`, E-GATEWAY): parking mode is - /// only a wrapper (`ParkingLlm`) around this SAME `TraceLlm`, so captured - /// requests are still inspectable for a parked thread. + /// `dyn LlmProvider`. Its `captured_requests()` lets assertions inspect the + /// exact model-visible requests (system prompt, host-injected context — + /// `assert_system_prompt_contains`/`assert_model_request_contains`). + /// Retained even when parked (`park_model`, E-GATEWAY): `ParkingLlm` only + /// wraps this SAME `TraceLlm`. pub(crate) scripted_llm: Arc, /// Shared storage bundle keeping the composite, TempDir, product harness, and /// capability alive for this harness's lifetime. For a single-shot harness the @@ -470,13 +446,10 @@ pub struct RebornIntegrationHarness { pub(crate) baseline_network_count: usize, /// Turn-lifecycle-event count on the group-shared `InMemoryTurnEventSink` at /// harness construction, if `.with_turn_event_sink()` opted in. The sink has - /// no per-thread channel — every thread in a group publishes to the same - /// `Arc` — so without this baseline a group thread's - /// `assert_turn_event_recorded` could pass on an EARLIER thread's event - /// (e.g. a prior thread's `Completed`) even if this thread's own turn never - /// published anything, silently defeating the assertion. `recorded_turn_events` - /// slices `[baseline_turn_event_count..]` the same way the other - /// `baseline_*_count` fields scope their recorders to this thread only (R2). + /// no per-thread channel, so without this baseline a group thread's + /// `assert_turn_event_recorded` could pass on an earlier thread's event. + /// `recorded_turn_events` slices `[baseline_turn_event_count..]` like the + /// other `baseline_*_count` fields (R2). pub(crate) baseline_turn_event_count: usize, } @@ -561,13 +534,10 @@ impl RebornIntegrationHarness { /// Submit a user turn carrying N inline attachments of any mime type, and /// wait for it to complete (W4-ATTACH-VARIANTS). Generalizes /// `submit_turn_with_image_attachment` to multiple attachments and - /// non-image kinds (e.g. `text/plain`, which `AttachmentKind::from_mime_type` - /// classifies as `Document` — its bytes are extracted to text by - /// `land_inbound_attachments` and rendered into the `` block - /// `augment_model_content` appends to the message, rather than read back as - /// a multimodal part). Same production entry point - /// (`DefaultProductWorkflow::submit_inbound_with_attachments`) and lander - /// requirement as `submit_turn_with_image_attachment`. + /// non-image kinds (e.g. `text/plain`, classified `Document` — extracted + /// to text and rendered into the `` block rather than read + /// back as a multimodal part). Same entry point and lander requirement as + /// `submit_turn_with_image_attachment`. pub async fn submit_turn_with_attachments( &self, text: &str, @@ -701,16 +671,10 @@ impl RebornIntegrationHarness { /// service (design §3.8 durability guardrail). /// /// For `StorageMode::LibSql`: opens a **genuinely fresh** `libsql::Database` - /// connection to the on-disk `.db` file — the live `CompositeRootFilesystem` - /// Arc is deliberately NOT reused. Only data that was actually serialized and - /// committed to disk is visible through the new handle, so this assertion - /// proves real on-disk durability, not an in-process cache. - /// - /// For `StorageMode::InMemory`: re-instantiates the - /// `FilesystemSessionThreadService` over the same in-process handle (no disk - /// involved). This asserts service re-instantiation but cannot prove durability - /// — there is nothing on disk to read back. Use `StorageMode::LibSql` for the - /// durability guarantee. + /// connection to the on-disk file (the live composite `Arc` is deliberately + /// NOT reused), so this proves real on-disk durability. For `InMemory`: + /// re-instantiates the service over the same in-process handle — asserts + /// re-instantiation only, not durability (nothing on disk to read back). pub async fn assert_reply_persists_after_reopen(&self, text: &str) -> HarnessResult<()> { if let Some(db_path) = &self._shared.libsql_db_path { // Open a fresh libsql connection — independent of the live composite. @@ -1081,14 +1045,9 @@ impl RebornIntegrationHarness { /// Test-support variant of [`approve_gate`](Self::approve_gate) for the /// stale-gate-ref-resume regression guard (C-DENYEDGE row 7): resolves the /// LOCAL-DEV approval using `real_gate_ref` (so the approval-store lookup - /// succeeds, unlike passing a bogus ref straight through `approve_gate`, - /// which fails earlier inside `approve_local_dev_gate`'s own request-id - /// lookup) but issues the COORDINATOR resume with a DIFFERENT - /// `stale_gate_ref`. This reaches `resume_turn_once`'s - /// `record.gate_ref.as_ref() != Some(&request.gate_resolution_ref)` check — - /// `TurnError::InvalidRequest { reason: "gate resolution reference mismatch" }` — - /// a path `approve_gate` itself can never reach because it always resolves - /// and resumes with the SAME gate_ref. + /// succeeds) but issues the COORDINATOR resume with a DIFFERENT + /// `stale_gate_ref`, reaching `resume_turn_once`'s gate-ref-mismatch check + /// (`InvalidRequest`) — a path `approve_gate` itself can never reach. pub async fn approve_gate_with_stale_resume_ref( &self, run_id: TurnRunId, @@ -1109,13 +1068,11 @@ impl RebornIntegrationHarness { /// Resume-only companion to /// [`approve_gate_with_stale_resume_ref`](Self::approve_gate_with_stale_resume_ref): - /// issues the coordinator resume for `gate_ref` WITHOUT re-running the - /// local-dev approval resolve step. Needed for a non-vacuity follow-up - /// after a failed stale-ref resume attempt — the approval record is - /// already `Approved` at that point (the resolve step succeeded before - /// the stale-ref resume failed), so calling `approve_gate` again would - /// hit the double-resolve `NotPending` error instead of completing the - /// still-blocked run. + /// issues the coordinator resume WITHOUT re-running the local-dev approval + /// resolve step. Needed for a non-vacuity follow-up after a failed + /// stale-ref resume: the record is already `Approved`, so re-calling + /// `approve_gate` would hit a double-resolve `NotPending` error instead of + /// completing the still-blocked run. pub async fn resume_gate(&self, run_id: TurnRunId, gate_ref: &GateRef) -> HarnessResult<()> { self.resume_run( run_id, @@ -1147,22 +1104,16 @@ impl RebornIntegrationHarness { } /// Deny a blocked AUTH gate and resume the run (user-declines path). Unlike - /// [`deny_gate`](Self::deny_gate) (approval gates, which resolve a persisted request in the - /// local-dev approval store), auth gates have no such store entry — there is nothing to - /// resolve — so this resumes directly with `GateResumeDisposition::Denied`. The executor's - /// `short_circuit_denied_resume` then surfaces a model-visible gate-declined failure for the - /// parked capability instead of re-dispatching it (which would re-block on the still-missing - /// credential → infinite loop). + /// [`deny_gate`](Self::deny_gate) (approval gates resolve a persisted + /// request), auth gates have no such store entry, so this resumes directly + /// with `GateResumeDisposition::Denied` — `short_circuit_denied_resume` + /// then surfaces a model-visible gate-declined failure instead of + /// re-dispatching (which would re-block on the missing credential forever). /// - /// Like `deny_gate` (which resumes with its own gate-class-specific - /// `ResumeTurnPrecondition::BlockedApprovalGate`), this resumes with a - /// gate-class-specific precondition — `ResumeTurnPrecondition::BlockedAuthGate` — the same - /// precondition `AuthInteractionService` uses for its production auth-resume path. That precondition is - /// enforced server-side (`resume_turn_once` requires `record.status == BlockedAuth`), so a - /// stale or wrong (non-auth) gate ref fails the resume with `TurnError::InvalidTransition` - /// instead of silently resuming whatever gate class happens to be blocked. A client-side - /// `gate:auth-` prefix check (mirroring `submit_turn_until_auth_blocked`) adds cheap - /// defense-in-depth on top of that server-side check. + /// Resumes with the gate-class-specific `ResumeTurnPrecondition::BlockedAuthGate` + /// (server-enforced: `resume_turn_once` requires `status == BlockedAuth`), + /// same shape as `deny_gate`'s `BlockedApprovalGate`. A client-side + /// `gate:auth-` prefix check adds cheap defense-in-depth on top. pub async fn deny_auth_gate(&self, run_id: TurnRunId, gate_ref: &GateRef) -> HarnessResult<()> { if !gate_ref.as_str().starts_with("gate:auth-") { return Err(format!("expected an auth gate ref, got {gate_ref:?}").into()); @@ -1178,19 +1129,16 @@ impl RebornIntegrationHarness { /// Resolve a blocked AUTH gate the "user submitted credentials" way /// (C-JOURNEY convergence seam): seed a real GitHub credential account - /// through product-auth (`HostRuntimeCapabilityHarness::seed_github_credential_account`) - /// so the parked capability's next `ProductAuthRuntimeCredentialResolver` - /// lookup resolves, then resume with `ResumeTurnPrecondition::BlockedAuthGate` - /// and NO deny disposition so the parked `github.*` capability re-dispatches - /// and the run completes. + /// (`seed_github_credential_account`) so the parked capability's next + /// credential-resolver lookup resolves, then resume with NO deny + /// disposition so the parked `github.*` capability re-dispatches and the + /// run completes. /// - /// Only valid on a capability harness built via - /// `HostRuntimeCapabilityHarness::file_and_github_auth_tools` (reached - /// through `_shared.capability`) — the credential-seeding path needs the - /// `build_reborn_services` product-auth wiring that `deny_auth_gate`'s - /// sibling fixture (`RebornIntegrationGroup::live_auth_gate`, a lower-level - /// `HostRuntimeServices` build with a hardcoded credential resolver and no - /// run_state store) does not have. + /// Only valid on a harness built via + /// `HostRuntimeCapabilityHarness::file_and_github_auth_tools` — the + /// credential-seeding path needs `build_reborn_services` product-auth + /// wiring that `deny_auth_gate`'s sibling fixture (`live_auth_gate`, a + /// lower-level build with a hardcoded resolver) does not have. pub async fn resolve_auth_gate( &self, run_id: TurnRunId, @@ -1208,13 +1156,9 @@ impl RebornIntegrationHarness { } }; // Seed under THIS run's actual (tenant, user, agent, project) — the - // credential resolver's `account_visible_from_runtime_scope` check - // matches on all four, so a scope built from a differently-scoped - // group/harness (e.g. a fixed "*-e2e" literal) would silently seed an - // account the real dispatch-time lookup never finds, leaving the run - // stuck at `BlockedAuth` forever. `self.turn_scope` + `self.binding` - // are this harness's own resolved run scope (same fields - // `resume_run` uses for `TurnScope`/`TurnActor`). + // resolver's `account_visible_from_runtime_scope` check matches all + // four, so a differently-scoped seed would leave the run stuck at + // `BlockedAuth` forever. let scope = ResourceScope { tenant_id: self.turn_scope.tenant_id.clone(), user_id: self.binding.actor_user_id.clone(), @@ -1357,15 +1301,12 @@ impl RebornIntegrationHarness { // --------------------------------------------------------------------------- /// Build the one `CompositeRootFilesystem` for a harness, selecting the durable -/// backend by `mode`. The `dir` argument is used only for `LibSql` (the SQLite -/// file is created there by the production `build_default_local_dev_database_roots` -/// sequence); `InMemory` ignores it. +/// backend by `mode`. `dir` is used only for `LibSql` (the SQLite file is +/// created there); `InMemory` ignores it. /// -/// Returns the composite alongside the path to the on-disk SQLite file for -/// `LibSql` (`None` for `InMemory`). The path is stored on -/// `RebornIntegrationHarness` so `assert_reply_persists_after_reopen` can open -/// a genuinely fresh database connection — independent of the live -/// `CompositeRootFilesystem` Arc — and confirm real on-disk durability. +/// Returns the composite alongside the on-disk SQLite path for `LibSql` +/// (`None` for `InMemory`) — stored on the harness so +/// `assert_reply_persists_after_reopen` can open a genuinely fresh connection. pub(crate) async fn build_storage_composite( mode: StorageMode, dir: &Path, diff --git a/tests/support/reborn/capability_backend.rs b/tests/integration/support/capability_backend.rs similarity index 91% rename from tests/support/reborn/capability_backend.rs rename to tests/integration/support/capability_backend.rs index 1c6c2da483f..3357c500bc9 100644 --- a/tests/support/reborn/capability_backend.rs +++ b/tests/integration/support/capability_backend.rs @@ -5,7 +5,7 @@ use std::sync::Arc; use super::group::GroupCapability; -use super::harness::HostRuntimeCapabilityHarness; +use super::harness::profiles::core_builtin::{self, CoreBuiltinOptions}; use super::http_matcher::ScriptedHttpResponse; use super::process::ScriptedProcessResult; @@ -80,10 +80,13 @@ impl RebornCapabilityBackend { // latter with a canned result installed below). let host_runtime = match shell_mode { ShellMode::Live => { - HostRuntimeCapabilityHarness::core_builtin_tools_with_live_shell().await? + core_builtin::core_builtin_tools( + CoreBuiltinOptions::default().with_live_shell(), + ) + .await? } ShellMode::Inert | ShellMode::Scripted(_) => { - HostRuntimeCapabilityHarness::core_builtin_tools().await? + core_builtin::core_builtin_tools_default().await? } }; host_runtime.install_http_responses(keyed_http_responses)?; @@ -93,7 +96,7 @@ impl RebornCapabilityBackend { GroupCapability::HostRuntime(Arc::new(host_runtime)) } RebornCapabilityBackend::MockMcp { mcp_url } => { - let host_runtime = HostRuntimeCapabilityHarness::mock_mcp_tools( + let host_runtime = super::harness::profiles::mock_mcp::mock_mcp_tools( &mcp_url, MOCK_MCP_PROVIDER_ID, &format!("{MOCK_MCP_PROVIDER_ID}.search"), @@ -105,7 +108,7 @@ impl RebornCapabilityBackend { // T0-SECRET-INJECT (see the `GithubIssueTools` variant docs above): // no approval gate / user alignment — the authorizer allows every // dispatch outright. - let host_runtime = HostRuntimeCapabilityHarness::github_issue_tools().await?; + let host_runtime = super::harness::profiles::github::github_issue_tools().await?; // W4-AUTHGATE-WIRE: wire keyed HTTP responses onto this backend too // (previously only `BuiltinHttpTools` installed them). A no-op for // existing callers that never populate `keyed_http_responses` for @@ -124,7 +127,7 @@ impl RebornCapabilityBackend { } RebornCapabilityBackend::WebAccessTools => { // C-WEBACCESS — see the `WebAccessTools` variant docs above. - let host_runtime = HostRuntimeCapabilityHarness::web_access_tools().await?; + let host_runtime = super::harness::profiles::web_access::web_access_tools().await?; host_runtime.install_web_access_responses(web_access_response_bodies)?; GroupCapability::HostRuntime(Arc::new(host_runtime)) } diff --git a/tests/support/reborn/comm_context.rs b/tests/integration/support/comm_context.rs similarity index 75% rename from tests/support/reborn/comm_context.rs rename to tests/integration/support/comm_context.rs index 3bb2a200cad..b6f02e1194f 100644 --- a/tests/support/reborn/comm_context.rs +++ b/tests/integration/support/comm_context.rs @@ -1,20 +1,15 @@ //! C-COMMCTX: a recording [`CommunicationContextProvider`] test double. //! -//! Wired into a harness/group via `with_communication_context_provider` / -//! `RebornIntegrationGroupBuilder::communication_context_provider`, this double -//! returns a fixed delivery-preference / connected-channel slice so a test can -//! prove the wired `communication_context_provider` reaches the turn pipeline — -//! the slice renders into the model request (assert via +//! Wired via `with_communication_context_provider`, this double returns a +//! fixed delivery-preference/connected-channel slice so a test can prove it +//! reaches the turn pipeline and renders into the model request (assert via //! `assert_model_request_contains`). //! -//! This is DISTINCT from the outbound delivery **sink** (E-OUTBOUND, a sibling -//! lane): this is prompt **context** (delivery preferences/targets), not a -//! delivery recorder. The production `RuntimeCommunicationContextProvider`'s -//! facade→context mapping is already densely unit-tested in -//! `crates/ironclaw_reborn_composition/src/communication_context.rs`; this double -//! deliberately covers only the int-tier gap — that the `communication_context_provider` -//! field wires through the coordinator path into the model request — without -//! re-authoring that crate-tier mapping coverage. +//! DISTINCT from the outbound delivery **sink** (E-OUTBOUND): this is prompt +//! **context**, not a delivery recorder. The production facade→context +//! mapping is already unit-tested in +//! `crates/ironclaw_reborn_composition/src/communication_context.rs`; this +//! double covers only the int-tier wiring gap. // Shared integration-test support: not every binary that mounts the // `reborn_support` tree consumes this module, so its symbols read as dead there diff --git a/tests/support/reborn/config.rs b/tests/integration/support/config.rs similarity index 100% rename from tests/support/reborn/config.rs rename to tests/integration/support/config.rs diff --git a/tests/integration/support/doubles/empty_identity_context_source.rs b/tests/integration/support/doubles/empty_identity_context_source.rs new file mode 100644 index 00000000000..cc0448aeabf --- /dev/null +++ b/tests/integration/support/doubles/empty_identity_context_source.rs @@ -0,0 +1,18 @@ +use async_trait::async_trait; +use ironclaw_loop_support::{ + HostIdentityContextBuildError, HostIdentityContextCandidate, HostIdentityContextSource, +}; +use ironclaw_turns::run_profile::{LoopRunContext, PromptMode}; + +pub(crate) struct EmptyIdentityContextSource; + +#[async_trait] +impl HostIdentityContextSource for EmptyIdentityContextSource { + async fn load_identity_candidates( + &self, + _run_context: &LoopRunContext, + _mode: PromptMode, + ) -> Result, HostIdentityContextBuildError> { + Ok(Vec::new()) + } +} diff --git a/tests/integration/support/doubles/fixed_runtime_credential_account_resolver.rs b/tests/integration/support/doubles/fixed_runtime_credential_account_resolver.rs new file mode 100644 index 00000000000..16dfc08e358 --- /dev/null +++ b/tests/integration/support/doubles/fixed_runtime_credential_account_resolver.rs @@ -0,0 +1,28 @@ +use async_trait::async_trait; +use ironclaw_host_api::{CredentialStageError, SecretHandle}; +use ironclaw_host_runtime::{ + RuntimeCredentialAccessSecret, RuntimeCredentialAccountRequest, + RuntimeCredentialAccountResolver, +}; + +#[derive(Debug)] +pub(crate) struct FixedRuntimeCredentialAccountResolver { + pub(crate) result: Result, +} + +#[async_trait] +impl RuntimeCredentialAccountResolver for FixedRuntimeCredentialAccountResolver { + async fn resolve_access_secret( + &self, + request: RuntimeCredentialAccountRequest<'_>, + ) -> Result { + assert_eq!(request.provider.as_str(), "github"); + assert_eq!(request.requester_extension.as_str(), "github"); + self.result + .clone() + .map(|handle| RuntimeCredentialAccessSecret { + scope: request.scope.clone(), + handle, + }) + } +} diff --git a/tests/integration/support/doubles/github_harness_authorizer.rs b/tests/integration/support/doubles/github_harness_authorizer.rs new file mode 100644 index 00000000000..13e6242fdcd --- /dev/null +++ b/tests/integration/support/doubles/github_harness_authorizer.rs @@ -0,0 +1,60 @@ +use super::super::github as github_support; +use async_trait::async_trait; +use ironclaw_authorization::TrustAwareCapabilityDispatchAuthorizer; +use ironclaw_host_api::{ + CapabilityDescriptor, Decision, ExecutionContext, ExtensionId, Obligation, Obligations, + ResourceEstimate, RuntimeCredentialAccountProviderId, SecretHandle, +}; +use ironclaw_trust::TrustDecision; + +use super::super::harness::HarnessResult; + +pub(crate) struct GithubHarnessAuthorizer { + obligations: Obligations, +} + +impl GithubHarnessAuthorizer { + pub(crate) fn new() -> HarnessResult { + Ok(Self { + obligations: Obligations::new(vec![ + Obligation::ApplyNetworkPolicy { + policy: github_support::api_policy(), + }, + Obligation::InjectCredentialAccountOnce { + handle: SecretHandle::new("github_runtime_token")?, + provider: RuntimeCredentialAccountProviderId::new("github")?, + setup: ironclaw_host_api::RuntimeCredentialAccountSetup::ManualToken, + provider_scopes: Vec::new(), + requester_extension: ExtensionId::new("github")?, + }, + ])?, + }) + } +} + +#[async_trait] +impl TrustAwareCapabilityDispatchAuthorizer for GithubHarnessAuthorizer { + async fn authorize_dispatch_with_trust( + &self, + _context: &ExecutionContext, + _descriptor: &CapabilityDescriptor, + _estimate: &ResourceEstimate, + _trust_decision: &TrustDecision, + ) -> Decision { + Decision::Allow { + obligations: self.obligations.clone(), + } + } + + async fn authorize_spawn_with_trust( + &self, + _context: &ExecutionContext, + _descriptor: &CapabilityDescriptor, + _estimate: &ResourceEstimate, + _trust_decision: &TrustDecision, + ) -> Decision { + Decision::Allow { + obligations: self.obligations.clone(), + } + } +} diff --git a/tests/integration/support/doubles/harness_capability_port_factory.rs b/tests/integration/support/doubles/harness_capability_port_factory.rs new file mode 100644 index 00000000000..870af3a210e --- /dev/null +++ b/tests/integration/support/doubles/harness_capability_port_factory.rs @@ -0,0 +1,24 @@ +/// Test double substituting the production `LoopCapabilityPortFactory` wiring +/// (`LocalDevLoopCapabilityPortFactory` / `HostRuntimeLoopCapabilityPortFactory`) +/// for the Echo (`RecordingTestCapabilityPort`) backend. +use std::sync::Arc; + +use async_trait::async_trait; +use ironclaw_loop_support::LoopCapabilityPortFactory; +use ironclaw_turns::run_profile::{AgentLoopHostError, LoopCapabilityPort, LoopRunContext}; + +use super::recording_test_capability_port::RecordingTestCapabilityPort; + +pub(crate) struct HarnessCapabilityPortFactory { + pub(crate) port: Arc, +} + +#[async_trait] +impl LoopCapabilityPortFactory for HarnessCapabilityPortFactory { + async fn create_capability_port( + &self, + _run_context: &LoopRunContext, + ) -> Result, AgentLoopHostError> { + Ok(self.port.clone()) + } +} diff --git a/tests/integration/support/doubles/host_runtime_harness_capability_port_factory.rs b/tests/integration/support/doubles/host_runtime_harness_capability_port_factory.rs new file mode 100644 index 00000000000..c70acd0e17e --- /dev/null +++ b/tests/integration/support/doubles/host_runtime_harness_capability_port_factory.rs @@ -0,0 +1,27 @@ +/// Test double substituting the production `LoopCapabilityPortFactory` wiring: +/// `LocalDevLoopCapabilityPortFactory` (`crates/ironclaw_reborn_composition/src/runtime/local_dev.rs`) +/// and `HostRuntimeLoopCapabilityPortFactory` (`crates/ironclaw_loop_support/src/capability_port.rs`). +use std::sync::Arc; + +use async_trait::async_trait; +use ironclaw_loop_support::LoopCapabilityPortFactory; +use ironclaw_turns::run_profile::{AgentLoopHostError, LoopCapabilityPort, LoopRunContext}; + +use super::super::harness::HostRuntimeCapabilityHarness; + +pub(crate) struct HostRuntimeHarnessCapabilityPortFactory { + pub(crate) harness: Arc, + pub(crate) milestone_sink: Arc, +} + +#[async_trait] +impl LoopCapabilityPortFactory for HostRuntimeHarnessCapabilityPortFactory { + async fn create_capability_port( + &self, + run_context: &LoopRunContext, + ) -> Result, AgentLoopHostError> { + self.harness + .create_recording_capability_port(run_context, &self.milestone_sink) + .await + } +} diff --git a/tests/integration/support/doubles/mod.rs b/tests/integration/support/doubles/mod.rs new file mode 100644 index 00000000000..1e9cacebca1 --- /dev/null +++ b/tests/integration/support/doubles/mod.rs @@ -0,0 +1,37 @@ +//! Test doubles substituting production ports for the Reborn binary-E2E +//! and host-runtime capability harnesses. One file per substituted port. + +mod empty_identity_context_source; +mod fixed_runtime_credential_account_resolver; +mod github_harness_authorizer; +mod harness_capability_port_factory; +mod host_runtime_harness_capability_port_factory; +mod recording_approval_request_store; +mod recording_capability_result_writer; +mod recording_delegating_capability_port; +mod recording_host_runtime; +mod recording_network_http_egress; +mod recording_runtime_http_egress; +mod recording_test_capability_port; +mod static_capability_surface_profile_resolver; +mod static_secret_store; + +pub(crate) use empty_identity_context_source::EmptyIdentityContextSource; +pub(crate) use fixed_runtime_credential_account_resolver::FixedRuntimeCredentialAccountResolver; +pub(crate) use github_harness_authorizer::GithubHarnessAuthorizer; +pub(crate) use harness_capability_port_factory::HarnessCapabilityPortFactory; +pub(crate) use host_runtime_harness_capability_port_factory::HostRuntimeHarnessCapabilityPortFactory; +pub(crate) use recording_approval_request_store::RecordingApprovalRequestStore; +pub(crate) use recording_capability_result_writer::RecordingCapabilityResultWriter; +pub(crate) use recording_delegating_capability_port::RecordingDelegatingCapabilityPort; +pub(crate) use recording_host_runtime::RecordingHostRuntime; +pub(crate) use recording_network_http_egress::RecordingNetworkHttpEgress; +pub(crate) use recording_runtime_http_egress::RecordingRuntimeHttpEgress; +// Consts consumed only by the binary-E2E harness in the parity/QA support tree +// (unused in bins that don't mount it). +#[allow(unused_imports)] +pub(crate) use recording_test_capability_port::{ + RecordingTestCapabilityPort, TEST_CAPABILITY_ID, TEST_CAPABILITY_SURFACE_VERSION, +}; +pub(crate) use static_capability_surface_profile_resolver::StaticCapabilitySurfaceProfileResolver; +pub(crate) use static_secret_store::StaticSecretStore; diff --git a/tests/integration/support/doubles/recording_approval_request_store.rs b/tests/integration/support/doubles/recording_approval_request_store.rs new file mode 100644 index 00000000000..0b9fb33d4b0 --- /dev/null +++ b/tests/integration/support/doubles/recording_approval_request_store.rs @@ -0,0 +1,77 @@ +/// Test double substituting the production `ApprovalRequestStore` impls +/// (`InMemoryApprovalRequestStore` / `FilesystemApprovalRequestStore`, +/// `crates/ironclaw_run_state/src/lib.rs`). +use std::{ + collections::HashMap, + sync::{Arc, Mutex}, +}; + +use async_trait::async_trait; +use ironclaw_host_api::{ApprovalRequestId, ResourceScope}; + +/// Records `(ApprovalRequestId, ResourceScope)` on `save_pending`, then delegates +/// every method to the inner store. Synthetic local-dev capabilities (e.g. +/// `outbound_delivery_target_set`) persist approval requests directly to the +/// store rather than through the host runtime, so [`RecordingHostRuntime`] +/// never captures their scope — wrapping the store they write through +/// restores the `pending_approval_scopes` bookkeeping `approve_local_dev_gate` +/// / `deny_local_dev_gate` depend on. Delegation is total, so the inner store +/// stays the single source of truth. +pub(crate) struct RecordingApprovalRequestStore { + pub(crate) inner: Arc, + pub(crate) pending_approval_scopes: Arc>>, +} + +#[async_trait] +impl ironclaw_run_state::ApprovalRequestStore for RecordingApprovalRequestStore { + async fn save_pending( + &self, + scope: ResourceScope, + request: ironclaw_host_api::approval::ApprovalRequest, + ) -> Result { + self.pending_approval_scopes + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .insert(request.id, scope.clone()); + self.inner.save_pending(scope, request).await + } + + async fn get( + &self, + scope: &ResourceScope, + request_id: ApprovalRequestId, + ) -> Result, ironclaw_run_state::RunStateError> { + self.inner.get(scope, request_id).await + } + + async fn approve( + &self, + scope: &ResourceScope, + request_id: ApprovalRequestId, + ) -> Result { + self.inner.approve(scope, request_id).await + } + + async fn deny( + &self, + scope: &ResourceScope, + request_id: ApprovalRequestId, + ) -> Result { + self.inner.deny(scope, request_id).await + } + + async fn discard_pending( + &self, + scope: &ResourceScope, + request_id: ApprovalRequestId, + ) -> Result { + self.inner.discard_pending(scope, request_id).await + } + + async fn records_for_scope( + &self, + scope: &ResourceScope, + ) -> Result, ironclaw_run_state::RunStateError> { + self.inner.records_for_scope(scope).await + } +} diff --git a/tests/integration/support/doubles/recording_capability_result_writer.rs b/tests/integration/support/doubles/recording_capability_result_writer.rs new file mode 100644 index 00000000000..c45a4700586 --- /dev/null +++ b/tests/integration/support/doubles/recording_capability_result_writer.rs @@ -0,0 +1,60 @@ +/// Test double substituting the production `LoopCapabilityResultWriter` impl +/// (`LocalDevCapabilityIo`, `crates/ironclaw_reborn_composition/src/runtime/local_dev.rs`). +use std::sync::{Arc, Mutex}; + +use async_trait::async_trait; +use ironclaw_host_api::CapabilityId; +use ironclaw_loop_support::{ + CapabilityResultWrite, CapabilityWriteResult, LoopCapabilityResultWriter, +}; +use ironclaw_reborn_composition::ProductLiveCapabilityIo; +use ironclaw_turns::{ + LoopResultRef, + run_profile::{AgentLoopHostError, AgentLoopHostErrorKind, LoopRunContext}, +}; + +use super::super::harness::RecordedCapabilityResult; + +pub(crate) struct RecordingCapabilityResultWriter { + pub(crate) inner: Arc, + pub(crate) results: Arc>>, +} + +#[async_trait] +impl LoopCapabilityResultWriter for RecordingCapabilityResultWriter { + async fn write_capability_result( + &self, + write: CapabilityResultWrite<'_>, + ) -> Result { + let capability_id = write.capability_id.clone(); + let output = write.output.clone(); + let write_result = self.inner.write_capability_result(write).await?; + self.results.lock().unwrap().push(RecordedCapabilityResult { + capability_id, + output, + }); + Ok(write_result) + } + + async fn update_capability_result( + &self, + run_context: &LoopRunContext, + result_ref: &LoopResultRef, + output: serde_json::Value, + ) -> Result { + let byte_len = self + .inner + .update_capability_result(run_context, result_ref, output.clone()) + .await?; + self.results.lock().unwrap().push(RecordedCapabilityResult { + capability_id: CapabilityId::new( + ironclaw_loop_support::DEFAULT_SPAWN_SUBAGENT_CAPABILITY_ID, + ) + .map_err(|error| { + AgentLoopHostError::new(AgentLoopHostErrorKind::Internal, error.to_string()) + })?, + output, + }); + Ok(byte_len) + } +} diff --git a/tests/integration/support/doubles/recording_delegating_capability_port.rs b/tests/integration/support/doubles/recording_delegating_capability_port.rs new file mode 100644 index 00000000000..595131cff10 --- /dev/null +++ b/tests/integration/support/doubles/recording_delegating_capability_port.rs @@ -0,0 +1,62 @@ +/// Test double substituting the production `LoopCapabilityPort` produced by +/// `HostRuntimeLoopCapabilityPortFactory` (`crates/ironclaw_loop_support/src/capability_port.rs`). +use std::sync::{Arc, Mutex}; + +use async_trait::async_trait; +use ironclaw_turns::run_profile::{ + AgentLoopHostError, CapabilityBatchInvocation, CapabilityBatchOutcome, CapabilityCallCandidate, + CapabilityInvocation, CapabilityOutcome, LoopCapabilityPort, ProviderToolCall, + ProviderToolDefinition, VisibleCapabilityRequest, VisibleCapabilitySurface, +}; + +pub(crate) struct RecordingDelegatingCapabilityPort { + pub(crate) inner: Arc, + pub(crate) invocations: Arc>>, +} + +#[async_trait] +impl LoopCapabilityPort for RecordingDelegatingCapabilityPort { + fn tool_definitions(&self) -> Result, AgentLoopHostError> { + self.inner.tool_definitions() + } + + fn validate_provider_tool_call( + &self, + tool_call: &ProviderToolCall, + ) -> Result<(), AgentLoopHostError> { + self.inner.validate_provider_tool_call(tool_call) + } + + async fn register_provider_tool_call( + &self, + request: ironclaw_turns::run_profile::RegisterProviderToolCallRequest, + ) -> Result { + self.inner.register_provider_tool_call(request).await + } + + async fn visible_capabilities( + &self, + request: VisibleCapabilityRequest, + ) -> Result { + self.inner.visible_capabilities(request).await + } + + async fn invoke_capability( + &self, + request: CapabilityInvocation, + ) -> Result { + self.invocations.lock().unwrap().push(request.clone()); + self.inner.invoke_capability(request).await + } + + async fn invoke_capability_batch( + &self, + request: CapabilityBatchInvocation, + ) -> Result { + self.invocations + .lock() + .unwrap() + .extend(request.invocations.iter().cloned()); + self.inner.invoke_capability_batch(request).await + } +} diff --git a/tests/integration/support/doubles/recording_host_runtime.rs b/tests/integration/support/doubles/recording_host_runtime.rs new file mode 100644 index 00000000000..34fac8ab8b2 --- /dev/null +++ b/tests/integration/support/doubles/recording_host_runtime.rs @@ -0,0 +1,126 @@ +/// Test double substituting the production `HostRuntime` impl +/// (`DefaultHostRuntime`, `crates/ironclaw_host_runtime/src/production.rs`). +use std::{ + collections::HashMap, + sync::{Arc, Mutex}, +}; + +use async_trait::async_trait; +use ironclaw_host_api::{ApprovalRequestId, ResourceScope}; +use ironclaw_host_runtime::{ + CancelRuntimeWorkOutcome, CancelRuntimeWorkRequest, HostRuntime, HostRuntimeError, + HostRuntimeHealth, HostRuntimeStatus, RuntimeCapabilityOutcome, RuntimeCapabilityRequest, + RuntimeCapabilityResumeRequest, RuntimeStatusRequest, + VisibleCapabilityRequest as RuntimeVisibleCapabilityRequest, + VisibleCapabilitySurface as RuntimeVisibleCapabilitySurface, +}; + +pub(crate) struct RecordingHostRuntime { + inner: Arc, + pending_approval_scopes: Arc>>, +} + +impl RecordingHostRuntime { + pub(crate) fn new( + inner: Arc, + pending_approval_scopes: Arc>>, + ) -> Self { + Self { + inner, + pending_approval_scopes, + } + } +} + +#[async_trait] +impl HostRuntime for RecordingHostRuntime { + async fn invoke_capability( + &self, + request: RuntimeCapabilityRequest, + ) -> Result { + let scope = request.context.resource_scope.clone(); + let outcome = self.inner.invoke_capability(request).await?; + if let RuntimeCapabilityOutcome::ApprovalRequired(gate) = &outcome { + self.pending_approval_scopes + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .insert(gate.approval_request_id, scope); + } + Ok(outcome) + } + + async fn spawn_capability( + &self, + request: RuntimeCapabilityRequest, + ) -> Result { + let scope = request.context.resource_scope.clone(); + let outcome = self.inner.spawn_capability(request).await?; + if let RuntimeCapabilityOutcome::ApprovalRequired(gate) = &outcome { + self.pending_approval_scopes + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .insert(gate.approval_request_id, scope); + } + Ok(outcome) + } + + async fn resume_capability( + &self, + request: RuntimeCapabilityResumeRequest, + ) -> Result { + self.inner.resume_capability(request).await + } + + /// C-JOURNEY: forward auth-resume to the real runtime. `auth_resume_capability` + /// is a DEFAULTED trait method whose default fails loudly, so a wrapper + /// that forgets to forward it fails visibly — this wrapper's missing + /// forward was latent until the first auth→resolve→re-dispatch journey + /// drove it. Mirrors `invoke_capability`'s ApprovalRequired scope + /// recording because `auth_resume_json` can itself raise an approval gate. + async fn auth_resume_capability( + &self, + request: ironclaw_host_runtime::RuntimeCapabilityAuthResumeRequest, + ) -> Result { + let scope = request.context.resource_scope.clone(); + let outcome = self.inner.auth_resume_capability(request).await?; + if let RuntimeCapabilityOutcome::ApprovalRequired(gate) = &outcome { + self.pending_approval_scopes + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .insert(gate.approval_request_id, scope); + } + Ok(outcome) + } + + async fn resume_spawn_capability( + &self, + request: RuntimeCapabilityResumeRequest, + ) -> Result { + self.inner.resume_spawn_capability(request).await + } + + async fn visible_capabilities( + &self, + request: RuntimeVisibleCapabilityRequest, + ) -> Result { + self.inner.visible_capabilities(request).await + } + + async fn cancel_work( + &self, + request: CancelRuntimeWorkRequest, + ) -> Result { + self.inner.cancel_work(request).await + } + + async fn runtime_status( + &self, + request: RuntimeStatusRequest, + ) -> Result { + self.inner.runtime_status(request).await + } + + async fn health(&self) -> Result { + self.inner.health().await + } +} diff --git a/tests/integration/support/doubles/recording_network_http_egress.rs b/tests/integration/support/doubles/recording_network_http_egress.rs new file mode 100644 index 00000000000..7b80d7fb13e --- /dev/null +++ b/tests/integration/support/doubles/recording_network_http_egress.rs @@ -0,0 +1,74 @@ +/// Test double substituting the production `NetworkHttpEgress` impl: +/// `PolicyNetworkHttpEgress` (`crates/ironclaw_network/src/egress.rs`) over +/// `ReqwestNetworkTransport` (`crates/ironclaw_network/src/transport.rs`). +use std::{ + collections::VecDeque, + sync::{Arc, Mutex}, +}; + +use ironclaw_network::{ + NetworkHttpEgress, NetworkHttpError, NetworkHttpRequest, NetworkHttpResponse, NetworkUsage, +}; + +#[derive(Debug, Clone)] +pub(crate) struct RecordingNetworkHttpEgress { + default_body: Vec, + response_bodies: Arc>>>, + /// W4-AUTHGATE-WIRE: FIFO of scripted non-default statuses, consumed ahead + /// of the hardcoded `200` default. Lets a test drive the runtime-401 path + /// for capabilities whose real HTTP call flows through this **network** + /// lane (`GithubIssueTools`, via `try_with_host_http_egress`) rather than + /// the runtime egress matcher. Empty by default — pre-existing callers + /// keep the old hardcoded-200 behavior byte-identical. + status_queue: Arc>>, + requests: Arc>>, +} + +impl RecordingNetworkHttpEgress { + pub(crate) fn with_body(body: Vec) -> Self { + Self { + default_body: body, + response_bodies: Arc::new(Mutex::new(VecDeque::new())), + status_queue: Arc::new(Mutex::new(VecDeque::new())), + requests: Arc::new(Mutex::new(Vec::new())), + } + } + + pub(crate) fn requests(&self) -> Vec { + self.requests.lock().unwrap().clone() + } + + /// Enqueue one FIFO scripted status, consumed by the next `execute` call + /// ahead of the hardcoded `200` default. + pub(crate) fn push_status(&self, status: u16) { + self.status_queue.lock().unwrap().push_back(status); + } +} + +#[async_trait::async_trait] +impl NetworkHttpEgress for RecordingNetworkHttpEgress { + async fn execute( + &self, + request: NetworkHttpRequest, + ) -> Result { + let request_bytes = request.body.len() as u64; + self.requests.lock().unwrap().push(request); + let body = self + .response_bodies + .lock() + .unwrap() + .pop_front() + .unwrap_or_else(|| self.default_body.clone()); + let status = self.status_queue.lock().unwrap().pop_front().unwrap_or(200); + Ok(NetworkHttpResponse { + status, + headers: vec![("content-type".to_string(), "application/json".to_string())], + body: body.clone(), + usage: NetworkUsage { + request_bytes, + response_bytes: body.len() as u64, + resolved_ip: None, + }, + }) + } +} diff --git a/tests/integration/support/doubles/recording_runtime_http_egress.rs b/tests/integration/support/doubles/recording_runtime_http_egress.rs new file mode 100644 index 00000000000..6ea04090ddc --- /dev/null +++ b/tests/integration/support/doubles/recording_runtime_http_egress.rs @@ -0,0 +1,115 @@ +/// Test double substituting the production `RuntimeHttpEgress` impl +/// (`HostHttpEgressService`, `crates/ironclaw_host_runtime/src/egress/mod.rs`). +use std::{ + collections::VecDeque, + sync::{Arc, Mutex}, +}; + +use async_trait::async_trait; +use ironclaw_host_api::{ + RuntimeHttpEgress, RuntimeHttpEgressError, RuntimeHttpEgressRequest, RuntimeHttpEgressResponse, +}; + +#[derive(Debug, Clone)] +pub(crate) struct RecordingRuntimeHttpEgress { + default_body: Vec, + /// URL/method/capability-keyed scripted responses (§3.6 P1 ergonomics). + /// Consulted before the FIFO queue; first match wins. + scripted: Arc>>, + response_bodies: Arc>>>, + requests: Arc>>, +} + +impl RecordingRuntimeHttpEgress { + pub(crate) fn with_body(body: Vec) -> Self { + Self { + default_body: body, + scripted: Arc::new(Mutex::new(Vec::new())), + response_bodies: Arc::new(Mutex::new(VecDeque::new())), + requests: Arc::new(Mutex::new(Vec::new())), + } + } + + pub(crate) fn requests(&self) -> Vec { + self.requests.lock().unwrap().clone() + } + + /// Append keyed scripted responses (the canonical keyed-matcher install). + pub(crate) fn install_scripted( + &self, + responses: impl IntoIterator, + ) { + self.scripted.lock().unwrap().extend(responses); + } + + /// Enqueue one FIFO response body (C-WEBACCESS), consumed in call order + /// ahead of `default_body`. Mirrors `install_scripted`'s shape but for the + /// plain FIFO queue rather than keyed matchers — used to script the + /// three-leg Exa MCP handshake (`initialize` → `notifications/initialized` + /// → `tools/call`), which all target the same URL/method/capability and so + /// cannot be told apart by the keyed matcher. + pub(crate) fn push_response_body(&self, body: Vec) { + self.response_bodies.lock().unwrap().push_back(body); + } +} + +#[async_trait::async_trait] +impl RuntimeHttpEgress for RecordingRuntimeHttpEgress { + async fn execute( + &self, + request: RuntimeHttpEgressRequest, + ) -> Result { + let request_bytes = request.body.len() as u64; + // Resolve the keyed outcome BEFORE recording the request: `push(request)` + // moves `request` by value into the log, so any code reading its fields + // (the `.matches()` lookup) must run first. (`RuntimeHttpEgressRequest` + // does implement `Drop`/`ZeroizeOnDrop` to scrub its URL/headers, but that + // fires later when the logged entry is actually dropped, not on push.) + let keyed_outcome = { + let scripted = self.scripted.lock().unwrap(); + scripted + .iter() + .find(|response| response.matches(&request)) + .map(|response| response.outcome()) + }; + self.requests.lock().unwrap().push(request); + // A scripted egress error short-circuits with `Err`, driving the tool's + // error mapping. A body outcome (or the FIFO/default fallback) returns + // `Ok` with the scripted status/body. + let (status, body) = match keyed_outcome { + Some(super::super::http_matcher::ScriptedHttpOutcome::Error(error)) => { + return Err(error); + } + Some(super::super::http_matcher::ScriptedHttpOutcome::Body { status, bytes }) => { + (status, bytes) + } + None => ( + 200, + self.response_bodies + .lock() + .unwrap() + .pop_front() + .unwrap_or_else(|| self.default_body.clone()), + ), + }; + Ok(RuntimeHttpEgressResponse { + status, + headers: vec![("content-type".to_string(), "application/json".to_string())], + body: body.clone(), + saved_body: None, + request_bytes, + response_bytes: body.len() as u64, + redaction_applied: false, + }) + } +} + +#[async_trait] +impl ironclaw_host_runtime::ToolCallHttpEgress for RecordingRuntimeHttpEgress { + async fn execute_for_model_visible_output( + &self, + request: RuntimeHttpEgressRequest, + ) -> Result { + RuntimeHttpEgress::execute(self, request).await + } +} diff --git a/tests/integration/support/doubles/recording_test_capability_port.rs b/tests/integration/support/doubles/recording_test_capability_port.rs new file mode 100644 index 00000000000..d754fe4a65d --- /dev/null +++ b/tests/integration/support/doubles/recording_test_capability_port.rs @@ -0,0 +1,259 @@ +#![allow(dead_code)] // Carried from harness.rs's blanket allow: shared across bins with differing usage. + +/// Test double substituting the whole production capability-port dispatch +/// pipeline (`HostRuntimeLoopCapabilityPortFactory` + +/// `LocalDevLoopCapabilityPortFactory`) with a lightweight in-memory Echo backend. +use std::sync::{ + Arc, Mutex, + atomic::{AtomicUsize, Ordering}, +}; + +use async_trait::async_trait; +use ironclaw_host_api::{CapabilityId, ExtensionId, ProviderToolName, RuntimeKind}; +use ironclaw_host_runtime::READ_FILE_CAPABILITY_ID; +use ironclaw_loop_support::DEFAULT_SPAWN_SUBAGENT_CAPABILITY_ID; +use ironclaw_turns::{ + LoopGateRef, + run_profile::{ + AgentLoopHostError, CapabilityBatchInvocation, CapabilityBatchOutcome, + CapabilityCallCandidate, CapabilityDescriptorView, CapabilityInputRef, + CapabilityInvocation, CapabilityOutcome, CapabilityResultMessage, CapabilitySurfaceVersion, + ConcurrencyHint, LoopCapabilityPort, ProviderToolCallReplay, ProviderToolDefinition, + VisibleCapabilityRequest, VisibleCapabilitySurface, + }, +}; +use serde_json::json; + +pub(crate) const TEST_CAPABILITY_ID: &str = "test.echo"; +pub(crate) const TEST_CAPABILITY_SURFACE_VERSION: &str = "trace_replay_v1"; +const SUBAGENT_ALLOWED_TEST_TOOL_NAME: &str = "test_read_file"; + +#[derive(Clone)] +pub struct RecordingTestCapabilityPort { + mode: CapabilityMode, + expose_spawn_subagent: bool, + use_subagent_allowed_tool: bool, + invocations: Arc>>, + next_result: Arc, + approval_calls: Arc, +} + +#[derive(Debug, Clone, Copy)] +enum CapabilityMode { + Echo, + ApprovalThenEcho, + SpawnAuthThenApprovalThenEcho, +} + +impl RecordingTestCapabilityPort { + pub fn echo() -> Self { + Self::new(CapabilityMode::Echo, false, false) + } + + pub fn echo_with_spawn_subagent() -> Self { + Self::new(CapabilityMode::Echo, true, false) + } + + pub fn approval_then_echo() -> Self { + Self::new(CapabilityMode::ApprovalThenEcho, false, false) + } + + pub fn approval_then_echo_with_spawn_subagent() -> Self { + Self::new(CapabilityMode::ApprovalThenEcho, true, false) + } + + pub fn approval_then_allowed_tool_with_spawn_subagent() -> Self { + Self::new(CapabilityMode::ApprovalThenEcho, true, true) + } + + pub fn spawn_auth_then_approval_then_echo_with_spawn_subagent() -> Self { + Self::new(CapabilityMode::SpawnAuthThenApprovalThenEcho, true, false) + } + + pub fn spawn_auth_then_approval_then_allowed_tool_with_spawn_subagent() -> Self { + Self::new(CapabilityMode::SpawnAuthThenApprovalThenEcho, true, true) + } + + fn new( + mode: CapabilityMode, + expose_spawn_subagent: bool, + use_subagent_allowed_tool: bool, + ) -> Self { + Self { + mode, + expose_spawn_subagent, + use_subagent_allowed_tool, + invocations: Arc::new(Mutex::new(Vec::new())), + next_result: Arc::new(AtomicUsize::new(1)), + approval_calls: Arc::new(AtomicUsize::new(0)), + } + } + + fn primary_capability_id(&self) -> CapabilityId { + let id = if self.use_subagent_allowed_tool { + READ_FILE_CAPABILITY_ID + } else { + TEST_CAPABILITY_ID + }; + CapabilityId::new(id).expect("valid capability id") + } + + fn primary_tool_name(&self) -> &'static str { + if self.use_subagent_allowed_tool { + SUBAGENT_ALLOWED_TEST_TOOL_NAME + } else { + "test_echo" + } + } + + pub(crate) fn invocations(&self) -> Vec { + self.invocations.lock().unwrap().clone() + } + + pub fn invocation_count(&self) -> usize { + self.invocations.lock().unwrap().len() + } + + pub(crate) fn capability_allowlist(&self) -> Vec { + let mut allowlist = vec![self.primary_capability_id()]; + if self.expose_spawn_subagent { + allowlist.push( + CapabilityId::new(DEFAULT_SPAWN_SUBAGENT_CAPABILITY_ID) + .expect("valid capability id"), + ); + } + allowlist + } + + fn completed_result(&self) -> CapabilityOutcome { + let ordinal = self.next_result.fetch_add(1, Ordering::SeqCst); + CapabilityOutcome::Completed(CapabilityResultMessage { + result_ref: ironclaw_turns::LoopResultRef::new(format!("result:test-echo-{ordinal}")) + .expect("valid result ref"), + safe_summary: "echo: hi".to_string(), + progress: ironclaw_turns::run_profile::CapabilityProgress::MadeProgress, + terminate_hint: false, + byte_len: 0, + output_digest: None, + }) + } +} + +#[async_trait] +impl LoopCapabilityPort for RecordingTestCapabilityPort { + fn tool_definitions(&self) -> Result, AgentLoopHostError> { + let definitions = vec![ProviderToolDefinition { + capability_id: self.primary_capability_id(), + name: ProviderToolName::new(self.primary_tool_name()).expect("provider tool name"), + description: "Echo a test payload".to_string(), + parameters: json!({ + "type": "object", + "properties": { + "message": {"type": "string"} + } + }), + }]; + Ok(definitions) + } + + async fn register_provider_tool_call( + &self, + request: ironclaw_turns::run_profile::RegisterProviderToolCallRequest, + ) -> Result { + let call = request.tool_call; + let capability_id = self.primary_capability_id(); + Ok(CapabilityCallCandidate { + activity_id: ironclaw_turns::CapabilityActivityId::new(), + surface_version: CapabilitySurfaceVersion::new(TEST_CAPABILITY_SURFACE_VERSION) + .expect("valid surface version"), + capability_id: capability_id.clone(), + effective_capability_ids: vec![capability_id], + input_ref: CapabilityInputRef::new(format!("input:{}", call.id)) + .expect("valid input ref"), + provider_replay: Some(ProviderToolCallReplay { + provider_id: call.provider_id, + provider_model_id: call.provider_model_id, + provider_turn_id: call.turn_id.unwrap_or_else(|| "trace-turn".to_string()), + provider_call_id: call.id, + provider_tool_name: call.name, + arguments: call.arguments, + response_reasoning: call.response_reasoning, + reasoning: call.reasoning, + signature: call.signature, + }), + }) + } + + async fn visible_capabilities( + &self, + _request: VisibleCapabilityRequest, + ) -> Result { + let descriptors = vec![CapabilityDescriptorView { + capability_id: self.primary_capability_id(), + provider: Some(ExtensionId::new("test").expect("valid provider")), + runtime: RuntimeKind::FirstParty, + safe_name: self.primary_tool_name().to_string(), + safe_description: "Echo a test payload".to_string(), + concurrency_hint: ConcurrencyHint::SafeForParallel, + parameters_schema: json!({"type": "object"}), + }]; + Ok(VisibleCapabilitySurface { + version: CapabilitySurfaceVersion::new(TEST_CAPABILITY_SURFACE_VERSION) + .expect("valid surface version"), + descriptors, + callable_capability_ids: None, + }) + } + + async fn invoke_capability( + &self, + request: CapabilityInvocation, + ) -> Result { + self.invocations.lock().unwrap().push(request); + if matches!(self.mode, CapabilityMode::ApprovalThenEcho) + && self.approval_calls.fetch_add(1, Ordering::SeqCst) == 0 + { + return Ok(CapabilityOutcome::ApprovalRequired { + gate_ref: LoopGateRef::new("gate:test-approval").expect("valid gate ref"), + safe_summary: "test approval required".to_string(), + approval_resume: None, + }); + } + if matches!(self.mode, CapabilityMode::SpawnAuthThenApprovalThenEcho) { + match self.approval_calls.fetch_add(1, Ordering::SeqCst) { + 0 => return Ok(self.completed_result()), + 1 => { + return Ok(CapabilityOutcome::ApprovalRequired { + gate_ref: LoopGateRef::new("gate:test-approval").expect("valid gate ref"), + safe_summary: "test approval required".to_string(), + approval_resume: None, + }); + } + _ => {} + } + } + Ok(self.completed_result()) + } + + async fn invoke_capability_batch( + &self, + request: CapabilityBatchInvocation, + ) -> Result { + let stop_on_first_suspension = request.stop_on_first_suspension; + let mut outcomes = Vec::new(); + let mut stopped_on_suspension = false; + for invocation in request.invocations { + let outcome = self.invoke_capability(invocation).await?; + let is_suspension = outcome.is_suspension(); + outcomes.push(outcome); + if is_suspension && stop_on_first_suspension { + stopped_on_suspension = true; + break; + } + } + Ok(CapabilityBatchOutcome { + outcomes, + stopped_on_suspension, + }) + } +} diff --git a/tests/integration/support/doubles/static_capability_surface_profile_resolver.rs b/tests/integration/support/doubles/static_capability_surface_profile_resolver.rs new file mode 100644 index 00000000000..2d1fe6b6f71 --- /dev/null +++ b/tests/integration/support/doubles/static_capability_surface_profile_resolver.rs @@ -0,0 +1,19 @@ +use async_trait::async_trait; +use ironclaw_loop_support::{ + CapabilityAllowSet, CapabilityResolveError, CapabilitySurfaceProfileResolver, +}; +use ironclaw_turns::run_profile::LoopRunContext; + +pub(crate) struct StaticCapabilitySurfaceProfileResolver { + pub(crate) allow_set: CapabilityAllowSet, +} + +#[async_trait] +impl CapabilitySurfaceProfileResolver for StaticCapabilitySurfaceProfileResolver { + async fn resolve( + &self, + _run_context: &LoopRunContext, + ) -> Result { + Ok(self.allow_set.clone()) + } +} diff --git a/tests/integration/support/doubles/static_secret_store.rs b/tests/integration/support/doubles/static_secret_store.rs new file mode 100644 index 00000000000..dd0eb8866e0 --- /dev/null +++ b/tests/integration/support/doubles/static_secret_store.rs @@ -0,0 +1,112 @@ +use async_trait::async_trait; +use ironclaw_host_api::{ResourceScope, SecretHandle}; +use ironclaw_secrets::{ + SecretLease, SecretLeaseId, SecretLeaseStatus, SecretMaterial, SecretMetadata, SecretStore, + SecretStoreError, +}; + +pub(crate) struct StaticSecretStore { + handle: SecretHandle, + material: SecretMaterial, +} + +impl StaticSecretStore { + pub(crate) fn new(handle: SecretHandle, material: SecretMaterial) -> Self { + Self { handle, material } + } +} + +#[async_trait] +impl SecretStore for StaticSecretStore { + async fn put( + &self, + scope: ResourceScope, + handle: SecretHandle, + _material: SecretMaterial, + _expires_at: Option, + ) -> Result { + Ok(SecretMetadata { + scope, + handle, + expires_at: None, + }) + } + + async fn metadata( + &self, + scope: &ResourceScope, + handle: &SecretHandle, + ) -> Result, SecretStoreError> { + Ok((handle == &self.handle).then(|| SecretMetadata { + scope: scope.clone(), + handle: handle.clone(), + expires_at: None, + })) + } + + async fn metadata_for_scope( + &self, + scope: &ResourceScope, + ) -> Result, SecretStoreError> { + Ok(vec![SecretMetadata { + scope: scope.clone(), + handle: self.handle.clone(), + expires_at: None, + }]) + } + + async fn delete( + &self, + _scope: &ResourceScope, + _handle: &SecretHandle, + ) -> Result { + Ok(false) + } + + async fn lease_once( + &self, + scope: &ResourceScope, + handle: &SecretHandle, + ) -> Result { + if handle != &self.handle { + return Err(SecretStoreError::UnknownSecret { + scope: Box::new(scope.clone()), + handle: handle.clone(), + }); + } + Ok(SecretLease { + id: SecretLeaseId::new(), + scope: scope.clone(), + handle: handle.clone(), + status: SecretLeaseStatus::Active, + }) + } + + async fn consume( + &self, + _scope: &ResourceScope, + _lease_id: SecretLeaseId, + ) -> Result { + Ok(self.material.clone()) + } + + async fn revoke( + &self, + scope: &ResourceScope, + lease_id: SecretLeaseId, + ) -> Result { + Ok(SecretLease { + id: lease_id, + scope: scope.clone(), + handle: self.handle.clone(), + status: SecretLeaseStatus::Revoked, + }) + } + + async fn leases_for_scope( + &self, + _scope: &ResourceScope, + ) -> Result, SecretStoreError> { + Ok(Vec::new()) + } +} diff --git a/tests/support/reborn/extension_surface.rs b/tests/integration/support/extension_surface.rs similarity index 100% rename from tests/support/reborn/extension_surface.rs rename to tests/integration/support/extension_surface.rs diff --git a/tests/support/reborn/filesystem.rs b/tests/integration/support/filesystem.rs similarity index 87% rename from tests/support/reborn/filesystem.rs rename to tests/integration/support/filesystem.rs index c67990ae510..a551955b821 100644 --- a/tests/support/reborn/filesystem.rs +++ b/tests/integration/support/filesystem.rs @@ -12,17 +12,10 @@ use ironclaw_filesystem::{ }; use ironclaw_host_api::{HostPath, VirtualPath}; -/// Build the turn-state scope path for `binding`, with `root_prefix` -/// prepended before `/tenants/...`. -/// -/// The 4-arm match selects a path that isolates turn state by the combination -/// of tenant, optional agent, optional project, and owner user: -/// - Binary-E2E tier: `root_prefix = "/engine"` → `/engine/tenants/{t}/.../{u}/turns` -/// - Integration tier: `root_prefix = ""` → `/tenants/{t}/.../{u}/turns` -/// -/// Extracted here so `scoped_turns_fs` (harness.rs) and `scoped_turns_fs_composite` -/// (builder.rs) share one source of turn-path truth; adding a new dimension -/// (e.g. `mission_id`) only requires updating this function. +/// Turn-state scope path for `binding` (isolated by tenant/agent/project/ +/// owner user), with `root_prefix` prepended before `/tenants/...`. Shared by +/// `scoped_turns_fs` (harness.rs) and `scoped_turns_fs_composite` (builder.rs) +/// so both tiers derive turn paths from one source of truth. pub fn turns_scope_path(root_prefix: &str, binding: &ResolvedBinding) -> String { let owner_user_id = binding .subject_user_id diff --git a/tests/support/reborn/github.rs b/tests/integration/support/github.rs similarity index 90% rename from tests/support/reborn/github.rs rename to tests/integration/support/github.rs index 3808ef6e1da..071f378ecb8 100644 --- a/tests/support/reborn/github.rs +++ b/tests/integration/support/github.rs @@ -51,10 +51,9 @@ pub fn extension_registry() -> GithubSupportResult { Ok(registry) } -/// The parsed github `ExtensionPackage` alone (no registry wrapper). C-JOURNEY: -/// fed into `RebornServices::publish_bundled_extension_for_test` to make -/// `github.*` capabilities dispatchable on the `build_reborn_services` -/// local-dev runtime without a scripted install/activate handshake. +/// The parsed github `ExtensionPackage` alone (no registry wrapper); C-JOURNEY +/// feeds it into `publish_bundled_extension_for_test` for github.* dispatch +/// without a scripted install/activate handshake. pub fn extension_package() -> GithubSupportResult { let manifest = ExtensionManifest::parse_with_host_api_contracts( &std::fs::read_to_string(asset_root().join("manifest.toml"))?, diff --git a/tests/integration/support/golden.rs b/tests/integration/support/golden.rs new file mode 100644 index 00000000000..4cce2bb0ea1 --- /dev/null +++ b/tests/integration/support/golden.rs @@ -0,0 +1,123 @@ +//! Golden-payload assertions for [`RebornIntegrationHarness`]: exact-match of +//! the FULL model-visible inference payload (system prompt, turns, tool calls/ +//! results) per inference iteration, via `insta` snapshots (`cargo insta +//! review` / `INSTA_UPDATE=always` to accept drift). Complements the substring +//! checks in `assertions.rs` (`assert_system_prompt_contains` etc.) by catching +//! prompt/turn-assembly drift a substring check can't see. Payload is rendered +//! to canonical JSON (`serde_json::Value`, BTree-sorted keys) for deterministic +//! ordering across runs. +//! +//! `messages` render in FULL; `tool_surface` renders as ORDERED TOOL NAMES ONLY +//! (not full JSON schemas) — the full builtin schema (~1.2k lines) would couple +//! this golden to every unrelated tool schema edit, and schemas are already +//! pinned per-tool and via the system prompt's `surface sha256:` hash. The name +//! list still catches a tool appearing/disappearing/reordering and the +//! `.`→`__` provider-seam encoding. +//! +//! Two values are normalized (anchored on an exact literal prefix so nothing +//! else is touched), because production has no clock/date test seam (NO-WIRE +//! rule): the runtime context's model-visible wall clock (`Current date/time +//! at loop start: ...` → ``) and, for attachment-landing scenarios, +//! today's real UTC date embedded in the landed project path +//! (`/attachments//` → `/attachments//`). Everything else — tool- +//! call ids (canonicalized per-trace by `scripted_trace_llm`) and the `surface +//! sha256:` hash — stays exact. + +#![allow(dead_code)] + +use ironclaw_llm::{ChatMessage, ToolDefinition}; + +use super::builder::RebornIntegrationHarness; + +type HarnessResult = Result>; + +/// Render every captured inference request as one canonical, human-readable +/// block per call: `===== inference call {i} =====` followed by the pretty +/// JSON of `{ "messages": [...], "tool_surface": ["name", ...] }`. `messages` +/// is rendered in full; `tool_surface` is the ordered provider-seam tool names +/// only (see the module docs for why schemas are excluded). Key order is +/// deterministic (BTree-sorted through `serde_json::Value`); no volatile +/// normalization happens here — that is the caller's `insta` filter. +fn render_inference_payloads( + requests: &[Vec], + tool_definitions: &[Vec], +) -> String { + let empty = Vec::new(); + let mut out = String::new(); + for (index, messages) in requests.iter().enumerate() { + let tool_surface: Vec<&str> = tool_definitions + .get(index) + .unwrap_or(&empty) + .iter() + .map(|tool| tool.name.as_str()) + .collect(); + let payload = serde_json::json!({ "messages": messages, "tool_surface": tool_surface }); + let pretty = serde_json::to_string_pretty(&payload) + .expect("captured inference payload serializes to JSON"); + out.push_str(&format!("===== inference call {index} =====\n{pretty}\n")); + } + out +} + +/// Normalize the two nondeterministic values (clock, attachment date) — see +/// module docs. +fn normalize_volatile(rendered: &str) -> String { + let clock = + regex::Regex::new(r"Current date/time at loop start: \d{4}-\d{2}-\d{2}T\d{2}:\d{2}Z") + .expect("valid loop-start-clock regex"); + let rendered = clock.replace_all(rendered, "Current date/time at loop start: "); + let attachment_date = regex::Regex::new(r"/attachments/\d{4}-\d{2}-\d{2}/") + .expect("valid attachment-landing-date regex"); + attachment_date + .replace_all(&rendered, "/attachments//") + .into_owned() +} + +impl RebornIntegrationHarness { + /// Assert the FULL model-visible inference payload for this thread matches + /// the committed golden snapshot `golden_payload__{name}`. Panics (like the + /// sibling `assert_replay_snapshot!`) on mismatch; run `cargo insta review` + /// to inspect and accept drift. + pub fn assert_golden_payload(&self, name: &str) { + let rendered = normalize_volatile(&render_inference_payloads( + &self.scripted_llm.captured_requests(), + &self.scripted_llm.captured_tool_definitions(), + )); + let mut settings = insta::Settings::clone_current(); + settings.set_snapshot_path(concat!(env!("CARGO_MANIFEST_DIR"), "/tests/snapshots")); + settings.set_prepend_module_to_snapshot(false); + settings.set_omit_expression(true); + settings.bind(|| { + insta::assert_snapshot!(format!("golden_payload__{name}"), rendered); + }); + } + + /// Assert the finalized assistant reply on this thread is EXACTLY `expected` + /// (not a substring — the output-seam counterpart to `assert_golden_payload`, + /// pinning that the model's final text reaches the user verbatim). + pub async fn assert_reply_eq(&self, expected: &str) -> HarnessResult<()> { + let actual = self.final_reply_text().await?; + if actual == expected { + return Ok(()); + } + Err(format!("finalized reply {actual:?} does not exactly equal {expected:?}").into()) + } + + /// The exact finalized assistant reply text on this thread (last finalized + /// `Assistant` message). Errors if none is present. + async fn final_reply_text(&self) -> HarnessResult { + let history = self + .thread_harness + .history(self.binding.thread_id.clone()) + .await?; + history + .iter() + .rev() + .find(|message| { + message.kind == ironclaw_threads::MessageKind::Assistant + && message.status == ironclaw_threads::MessageStatus::Finalized + }) + .and_then(|message| message.content.clone()) + .ok_or_else(|| "no finalized assistant reply on thread".into()) + } +} diff --git a/tests/support/reborn/group.rs b/tests/integration/support/group.rs similarity index 77% rename from tests/support/reborn/group.rs rename to tests/integration/support/group.rs index 472661a090c..0559478e51d 100644 --- a/tests/support/reborn/group.rs +++ b/tests/integration/support/group.rs @@ -6,11 +6,9 @@ //! a per-thread workflow (binding + inbound service + scripted-gateway //! registration) over that one shared runtime. Within one group, state written //! by thread A is visible to thread B — the key e2e persistence contract. -//! -//! Separate groups are separate test binaries and run in parallel, fully -//! isolated. A single-shot [`RebornIntegrationHarness::test_default()`] is a -//! degenerate one-thread group (its own storage, baseline = 0), so all -//! existing tests are byte-identical after this refactor. +//! Separate groups are separate test binaries, fully isolated. A single-shot +//! [`RebornIntegrationHarness::test_default()`] is a degenerate one-thread +//! group (its own storage, baseline = 0). //! //! ## Group test binary layout //! @@ -21,43 +19,23 @@ //! scenario_approve_always_persists.rs //! ``` //! -//! ### Why one sequential `#[tokio::test]`, not N separate `#[test]` fns -//! -//! Cargo does not guarantee order or share an instance between multiple -//! `#[test]` fns in one binary, and `serial_test` + global statics are -//! fragile. One orchestrating fn is the only design that gives deterministic -//! ordering over a shared group instance without fragile machinery. -//! -//! ### Scenario shape -//! -//! ```rust,no_run -//! // scenario_approve_always_persists.rs -//! use crate::reborn_support::group::HarnessResult; -//! pub async fn run(g: &super::reborn_support::group::RebornIntegrationGroup) -//! -> HarnessResult<()> -//! { -//! // ... build thread, submit turn, assert ... -//! Ok(()) -//! } -//! ``` -//! -//! Use `?` for *dependent* scenarios (failure stops the driver) and +//! One sequential `#[tokio::test]` drives all scenarios (Cargo doesn't +//! guarantee order or share state across `#[test]` fns in one binary). Use `?` +//! for *dependent* scenarios (failure stops the driver) and //! `report.record(name, scenario::run(&g).await)` for *independent* ones //! (failure recorded, others continue). //! -//! ### Subdir module paths +//! ### Subdir module paths (required) //! //! Each group `main.rs` MUST declare BOTH `#[path]` overrides, each with -//! `#[allow(dead_code)]`: +//! `#[allow(dead_code)]` — bare `mod support;` resolves relative to the +//! group's own subdir and fails to compile: //! //! ```rust,no_run -//! #[allow(dead_code)] #[path = "../support/reborn/mod.rs"] mod reborn_support; -//! #[allow(dead_code)] #[path = "../support/mod.rs"] mod support; +//! #[allow(dead_code)] #[path = "../support/mod.rs"] mod reborn_support; +//! #[allow(dead_code)] #[path = "../../support/mod.rs"] mod support; //! ``` //! -//! Bare `mod support;` resolves to `tests/reborn_group_*/support.rs` (which -//! does not exist) and fails to compile. -//! //! ### Two composites — use the right one //! //! - [`RebornIntegrationGroup::turn_composite`]: thread/turn history read-back. @@ -148,6 +126,16 @@ use crate::support::trace_llm::TraceLlm; #[path = "group_constructors.rs"] mod group_constructors; +/// Optional-runtime-wiring setters (`storage`, `safety_context`, +/// `with_turn_event_sink`, `budget_accounting`, +/// `communication_context_provider`, `hook_dispatcher_builder_factory`) on +/// [`RebornIntegrationGroupBuilder`]. A private child module (not `pub mod` +/// from `mod.rs`), same precedent as `group_constructors` above — it reaches +/// the builder's private fields at plain module-private visibility instead +/// of widening them to `pub(crate)` for the whole test-support crate. +#[path = "group_options.rs"] +mod group_options; + /// Convenience alias matching `builder.rs` and `harness.rs`. pub type HarnessResult = Result>; @@ -204,33 +192,23 @@ pub(crate) struct GroupSharedStorage { /// only see that thread's own deltas (R2). pub(crate) capability_recorder: HarnessCapabilityRecorder, /// The exact `HostUserProfileSource` wired into the group's ONE planned - /// runtime (E-PROFILE seam; built once in `into_group` from - /// `capability_recorder.profile_filesystem()`). Kept so a profile-round-trip - /// test can call `resolve_user_profile` on the SAME instance the running - /// loop reads from, rather than re-deriving an equivalent one — a mutation - /// that breaks the `into_group` wiring (not just `build_user_profile_source_for_test` - /// itself) is caught. + /// runtime (E-PROFILE seam). Kept so a profile-round-trip test reads from + /// the SAME instance the running loop uses, not a re-derived equivalent — + /// catches wiring mutations, not just the builder itself. pub(crate) user_profile_source: Arc, - /// In-memory turn-lifecycle event sink wired into the group's ONE planned - /// runtime when `.with_turn_event_sink()` opted in (C-TRACECAP seam). - /// `None` for every group that did not opt in (`turn_event_sink` stays - /// `None` in `DefaultPlannedRuntimeParts`, matching prior behavior). - /// Kept as the concrete `InMemoryTurnEventSink` (not `Arc`) - /// so a test can read `.events()` back directly. + /// In-memory turn-lifecycle event sink wired in when `.with_turn_event_sink()` + /// opted in (C-TRACECAP seam); `None` otherwise. Concrete type (not `Arc`) so a test can read `.events()` back directly. pub(crate) turn_event_sink: Option>, - /// C-BUDGET: the in-memory `ResourceGovernor` wired behind the group's - /// `model_budget_accountant` (via the production `build_default_budget_accountant` - /// helper). Retained so a test can read back the account the accountant seeds - /// on the first model call of a turn — the liveness proof that the accountant - /// is wired through `DefaultPlannedRuntimeParts` and fires on the coordinator - /// path. `None` unless the group/harness was built with budget accounting - /// wired (`budget_accounting()` / `with_budget_accounting()`). + /// C-BUDGET: the in-memory `ResourceGovernor` behind the group's + /// `model_budget_accountant`. Retained so a test can read back the account + /// the accountant seeds on a turn's first model call — proof the + /// accountant is wired and fires. `None` unless budget accounting is wired. pub(crate) budget_governor: Option>, - /// C-BUDGET: the `(tenant, run-owner-user)` account the group's turns reserve - /// against — computed once from the canonical binding so a budget test reads - /// the SAME account the loop's accountant seeds (keyed on the run actor, i.e. - /// the binding subject user, per `GovernorBackedAccountant::resource_scope`). - /// `None` unless budget accounting is wired. + /// C-BUDGET: the `(tenant, run-owner-user)` account the group's turns + /// reserve against — computed once from the canonical binding so a test + /// reads the SAME account the loop's accountant seeds. `None` unless + /// budget accounting is wired. pub(crate) budget_account: Option, } @@ -324,16 +302,12 @@ impl GroupCapability { /// [`triggers`](Self::triggers), or via /// [`builder`](Self::builder) for custom storage mode. /// -/// The per-capability preset constructors themselves (`live_approvals`, -/// `builtin_tools`, `extension_lifecycle`, `live_auth_gate`, -/// `project_lifecycle`, `profile_tools`, `triggers`, `skill_activation_tools`, -/// `skill_management_tools`, `attachment_tools`, and their `RebornIntegrationGroupBuilder` -/// counterparts) live in the child `group_constructors` module (file: -/// `group_constructors.rs`, kept private to `group` — see the `mod -/// group_constructors` declaration below) — a thin catalog of "which -/// capability" selections layered over the one-shared-runtime assembly -/// mechanics (`build_base`/`into_group`) this file owns. Mirrors the -/// `harness_mcp.rs` split. +/// The per-capability preset constructors (`live_approvals`, `builtin_tools`, +/// `extension_lifecycle`, etc., and their `RebornIntegrationGroupBuilder` +/// counterparts) live in the private child module `group_constructors` — a +/// thin catalog of "which capability" selections layered over the +/// one-shared-runtime assembly mechanics (`build_base`/`into_group`) this +/// file owns. pub struct RebornIntegrationGroup { pub(crate) shared: Arc, } @@ -448,13 +422,14 @@ impl RebornIntegrationGroup { /// Option, Arc)` so each constructor can name fields rather than /// position-destructure a tuple. /// -/// Private (not `pub(crate)`): `group_constructors.rs` is a private child -/// module of `group` (see the `mod group_constructors` declaration above), so -/// module-private visibility already reaches it without widening these -/// internals to the whole test-support crate. The per-capability preset -/// constructors there take/return this type as the opaque handoff between -/// `build_base` and `into_group` — they never read its fields except -/// `canonical_binding`. +/// Plain module-private visibility: `group_constructors.rs` reaches this at +/// plain module-private visibility as a descendant of `group` (see the `mod +/// group_constructors` declaration above), so the fields stay private and the +/// per-capability preset constructors there — including their own +/// `build_group_capability_with_base` helper, which calls +/// `canonical_subject_user()` — take/return this type as the opaque handoff +/// between `build_base` and `into_group`; `build_base`/`into_group` themselves +/// stay module-private too. struct GroupBaseData { product_harness: RebornProductWorkflowHarness, composite: Arc, @@ -462,17 +437,11 @@ struct GroupBaseData { turn_root: Arc, /// A throwaway probe binding resolved once at group construction, used /// ONLY to derive the group-level shared turn store path and the - /// group-level `ThreadScope` for the single `ThreadCheckpointLoopExitEvidencePort`. - /// Every thread in a group shares `(tenant, agent, project)` — only - /// `thread_id` varies per conversation, and `ThreadScope` (unlike - /// `TurnScope`) has no `thread_id` field — so this canonical binding is a - /// valid stand-in for the whole group (see `ensure_thread_scope_matches_turn_scope`, - /// which checks only tenant/agent/project, never thread_id). - /// - /// `group_constructors.rs`'s `live_approvals`/`skill_activation_tools` - /// read the resolved tenant/subject user off this field directly (see - /// field docs above); reachable at module-private visibility since that - /// file is a child module of `group`. + /// group-level `ThreadScope`. Every thread in a group shares `(tenant, + /// agent, project)` — only `thread_id` varies, and `ThreadScope` has no + /// `thread_id` field — so this binding is a valid stand-in for the whole + /// group. `group_constructors.rs` reads tenant/subject user off this + /// field directly (module-private; it's a child module of `group`). canonical_binding: ResolvedBinding, } @@ -519,79 +488,6 @@ pub struct RebornIntegrationGroupBuilder { } impl RebornIntegrationGroupBuilder { - /// Select the durable storage backend (default: `StorageMode::InMemory`). - /// Use `StorageMode::LibSql` to exercise on-disk durability across - /// `assert_reply_persists_after_reopen`. - pub fn storage(mut self, mode: StorageMode) -> Self { - self.storage = mode; - self - } - - /// Wire a model-visible instruction-safety banner into the group's ONE - /// shared planned runtime (`DefaultPlannedRuntimeParts::safety_context`). - /// Rendered verbatim as a `system`-role prompt message ahead of any - /// per-turn instructions (`push_safety_context`); the only model-visible - /// artifact of instruction-safety scanning on this tier (T0-SYSPROMPT / - /// C-SAFETY). Defaults to `None` (no banner, matching today's behavior). - pub fn safety_context(mut self, ctx: InstructionSafetyContext) -> Self { - self.safety_context = Some(ctx); - self - } - - /// Install an in-memory `TurnEventSink` (`ironclaw_turns::InMemoryTurnEventSink`, - /// a real, already-shipped production type with zero callers today — this is the - /// seam production wires via `subscribe_best_effort` in `build_default_planned_runtime_inner`, - /// `crates/ironclaw_reborn/src/runtime.rs:613-619`) into the group's ONE planned - /// runtime (C-TRACECAP). Read the recorded events back with - /// [`RebornIntegrationHarness::recorded_turn_events`] — the ONLY read path; - /// it slices `[baseline_turn_event_count..]` so a group thread never sees a - /// sibling thread's events. Deliberately no raw group-level sink accessor: - /// one would bypass that slicing and reintroduce cross-thread bleed. - pub fn with_turn_event_sink(mut self) -> Self { - self.turn_event_sink = Some(Arc::new(InMemoryTurnEventSink::default())); - self - } - - /// Wire the production `build_default_budget_accountant` (over in-memory - /// governor, gate store, zero-cost table, and compiled-default seeding) into - /// the group's ONE shared planned runtime (`DefaultPlannedRuntimeParts::model_budget_accountant`), - /// and retain the governor for read-back. This is the C-BUDGET liveness seam: - /// on the first model call of any turn the accountant seeds the run owner's - /// daily USD cap into the governor, which - /// `RebornIntegrationHarness::assert_budget_user_cap_seeded` reads back. - /// Budget SEMANTICS (thresholds, gates, `BudgetEvent` cascade) are covered at - /// crate tier (`budget_e2e.rs`); this only proves the harness bypass path - /// (`build_default_planned_runtime`) wires the accountant live. Defaults off. - pub fn budget_accounting(mut self) -> Self { - self.budget = true; - self - } - - /// Wire a `CommunicationContextProvider` into the group's ONE shared planned - /// runtime (`DefaultPlannedRuntimeParts::communication_context_provider`), so - /// the delivery-preference / connected-channel slice it resolves renders into - /// the model request. This is the C-COMMCTX seam — distinct from the outbound - /// delivery **sink** (E-OUTBOUND): this is prompt **context**. Defaults `None`. - pub fn communication_context_provider( - mut self, - provider: Arc, - ) -> Self { - self.communication_context_provider = Some(provider); - self - } - - /// Wire a per-run `HookDispatcherBuilderFactory` into the group's ONE shared - /// planned runtime (`DefaultPlannedRuntimeParts::hook_dispatcher_builder_factory`), - /// so hooks fire at their lifecycle points on a coordinator-path turn. This is - /// the E-HOOK-INFRA / C-HOOKS seam. Defaults `None` (hook framework dormant). - pub fn hook_dispatcher_builder_factory( - mut self, - factory: HookDispatcherBuilderFactory, - ) -> Self { - self.hook_dispatcher_builder_factory = Some(factory); - self - } - /// Shared setup for every group constructor: hermetic env, the product /// workflow harness over the fixed itest scope, the per-group `TempDir`, and /// the thread/turn composite. Returns [`GroupBaseData`] so each constructor @@ -772,19 +668,12 @@ impl RebornIntegrationGroupBuilder { ..DefaultPlannedRuntimeConfig::default() }, model_route_resolver: None, - // E-GATEWAY: the parking scripted gateway (`park_model`) plus - // `cancel_run` are the covered seam. This optional `cancellation_factory` - // is intentionally left `None`: it does NOT gate whether a run reaches + // E-GATEWAY: left `None` — it does not gate whether a run reaches // `Cancelled`. `RebornLoopDriverHostFactory` always builds its own - // default `TurnStateRunCancellationFactory` internally - // (`ironclaw_reborn::loop_driver_host`), whose cancel poll loop - // (`DEFAULT_CANCEL_POLL_INTERVAL`, 25ms) observes the durable - // `CancelRequested` and drives the parked run to `Cancelled` on resume — - // which is what `reborn_integration_cancel` asserts (verified 12/12). - // Supplying a factory here would only add the coordinator's - // `CompositeTurnRunWakeNotifier` fan-out for product-live retained-run-handle - // observation (`ironclaw_reborn::runtime` `wake_notifier`), a path this - // test does not exercise — so wiring one would be dead, untested code. + // default `TurnStateRunCancellationFactory`, whose cancel poll loop + // drives a parked run to `Cancelled` on resume regardless (verified + // by `reborn_integration_cancel`). Supplying one here would only add + // the product-live wake-notifier fan-out, unexercised by this test. cancellation_factory: None, // E-SKILL: wire the local-dev skill context source so an activated // skill's instructions inject into the model request. `Some` only for @@ -794,16 +683,11 @@ impl RebornIntegrationGroupBuilder { skill_context_source: capability_recorder.skill_context_source(), input_queue: None, identity_context_source: Arc::new(EmptyIdentityContextSource), - // E-PROFILE: in HostRuntime mode, back the profile source with the - // local-dev memory filesystem so `profile_set` writes can be read back; - // non-HostRuntime backends (and HostRuntime harnesses without a profile - // filesystem) fall back to `EmptyUserProfileSource`. `resolve_user_profile` - // returns `None` when no `context/profile.json` exists, so existing - // HostRuntime group tests are behavior-identical. Built once here (not - // per-thread) because the group's ONE planned runtime is assembled once. - // Kept in a local (rather than built inline) so the SAME `Arc` can - // also be stashed on `GroupSharedStorage` for a profile-round-trip - // test to read from directly (see `user_profile_source` field docs). + // E-PROFILE: HostRuntime mode backs this with the local-dev memory + // filesystem so `profile_set` writes read back; other backends fall + // back to `EmptyUserProfileSource`. Built as a local (not inline) so + // the SAME `Arc` is also stashed on `GroupSharedStorage` for a + // profile-round-trip test to read directly. user_profile_source: Arc::clone(&user_profile_source), model_policy_guard: None, // C-BUDGET: production `build_default_budget_accountant` (Some only @@ -925,18 +809,10 @@ impl<'g> RebornThreadBuilder<'g> { } /// Resolve this thread's binding under a DISTINCT actor instead of the - /// group's default `HARNESS_ACTOR_ID` (E-MULTIUSER seam). The resulting - /// binding's `subject_user_id`/`actor_user_id` differ from every other - /// thread's, so the run's `TurnActor` and per-turn owner-scope resolution - /// (`ThreadScopeResolver::resolve_for_turn`, the same mechanism production - /// uses for multi-user WebChat) isolate this thread's reads/writes under - /// their own subtree. The owner axis of that subtree is the resolved - /// canonical `UserId` (`binding.subject_user_id` / `ThreadScope.owner_user_id`) - /// — NOT the external `actor_id` string passed here — the binding probe - /// maps this `actor_id` to that canonical user id once at build time, and - /// every subsequent op resolves its mount from the binding, not the raw - /// string. Unset (the default) keeps the existing `HARNESS_ACTOR_ID` - /// behavior byte-identical. + /// group's default `HARNESS_ACTOR_ID` (E-MULTIUSER seam), so per-turn + /// owner-scope resolution isolates this thread's reads/writes under their + /// own subtree (keyed on the resolved canonical `UserId`, not the raw + /// `actor_id` string). Unset keeps the default `HARNESS_ACTOR_ID` behavior. pub fn with_actor_id(mut self, actor_id: impl Into) -> Self { self.actor_id = Some(actor_id.into()); self @@ -1015,21 +891,14 @@ impl<'g> RebornThreadBuilder<'g> { // --- per-thread scripted gateway, registered before any submit --------- // Session path is per-conversation so group threads do not clobber each - // other's LLM session cache under the same `turn_root`. - // Retain the concrete `TraceLlm` before the `dyn LlmProvider` upcast so - // tests can inspect the model-visible requests via `captured_requests()`: - // the system prompt (`assert_system_prompt_contains`, T0-SYSPROMPT) and - // host-injected context such as activated-skill instructions - // (`assert_model_request_contains`, E-SKILL half B). + // other's LLM session cache under the same `turn_root`. Retain the + // concrete `TraceLlm` before the `dyn LlmProvider` upcast so tests can + // inspect captured requests via `captured_requests()`. // - // E-GATEWAY: when a park gate is set, swap the scripted provider for a - // parking one at the SAME vendor-SDK seam (the decorator chain still runs - // on top). Parking mode is only a wrapper around the same scripted - // provider, so the `TraceLlm` is built unconditionally first and the - // parking wrapper holds/clones that same `Arc` — the trace is retained - // either way, so tests can inspect captured requests - // (`assert_system_prompt_contains`, `assert_model_request_contains`) - // regardless of whether this thread is parked. + // E-GATEWAY: the `TraceLlm` is built unconditionally first; a park gate + // wraps it in a parking provider at the SAME vendor-SDK seam (decorator + // chain still runs on top), so captured requests stay inspectable either + // way. let scripted_llm: Arc = Arc::new(scripted_trace_llm(self.replies)); // C-ERRORS: `Failing` swaps in `ErrLlm` at the same vendor-SDK seam; // `Parked` swaps in the parking wrapper. `ThreadModelMode` makes the @@ -1102,16 +971,9 @@ impl<'g> RebornThreadBuilder<'g> { let workflow = DefaultProductWorkflow::new(inbound, ledger, binding_service); // Register the gateway only now that every fallible (`?`) step above has - // succeeded — `register` is infallible and interior-mutable, so deferring - // it here costs nothing, but registering any earlier risks leaving the - // scope registered for a harness that never finished building: a later - // `?` bailing out of `build()` would leave `turn_scope` registered, and - // retrying the same conversation would then hit the duplicate- - // registration panic. The loop-driver host only resolves - // `resolve_for_scope` at host construction (off the model hot path), and - // that happens strictly after this fn returns, so registering - // immediately before the harness is constructed still satisfies - // "registered before any submit for this scope". + // succeeded — registering earlier risks leaving the scope registered + // for a harness that never finished building (a later `?` bailing out + // would make a retry hit the duplicate-registration panic). shared .scope_gateway .register(turn_scope.clone(), thread_gateway); diff --git a/tests/support/reborn/group_constructors.rs b/tests/integration/support/group_constructors.rs similarity index 68% rename from tests/support/reborn/group_constructors.rs rename to tests/integration/support/group_constructors.rs index c6534db7672..d2647e62b6d 100644 --- a/tests/support/reborn/group_constructors.rs +++ b/tests/integration/support/group_constructors.rs @@ -1,18 +1,9 @@ //! Per-capability preset constructors for [`RebornIntegrationGroup`] / -//! [`RebornIntegrationGroupBuilder`]. -//! -//! `group.rs` owns the one-shared-runtime assembly mechanics -//! (`RebornIntegrationGroupBuilder::build_base` / `into_group`); this file is -//! a private child module of `group` (declared `#[path = "group_constructors.rs"] -//! mod group_constructors;` in `group.rs`, NOT `pub mod` from `mod.rs`) that -//! catalogs "which capability" selections layered on top of that mechanics — -//! one method per `HostRuntimeCapabilityHarness` preset. Keeping it a child -//! module (rather than a top-level sibling) lets it reach `build_base`/ -//! `into_group`/`GroupBaseData` at plain module-private visibility instead of -//! widening them to `pub(crate)` for the whole test-support crate. Split out -//! (design precedent: `harness_mcp.rs`) once `group.rs` crossed the 1000-line -//! ceiling with PR-E2's E-SKILL/E-DURABLE/E-GATEWAY additions; new capability -//! presets belong HERE, not back in `group.rs`. +//! [`RebornIntegrationGroupBuilder`] — one method per `HostRuntimeCapabilityHarness` +//! preset. Private child module of `group.rs` (owns the shared assembly +//! mechanics: `build_base`/`into_group`), so it reaches those + `GroupBaseData` +//! at module-private visibility instead of widening them to `pub(crate)` for +//! the whole test-support crate. New capability presets belong HERE. // Shared by all group test binaries; symbols read as dead when a binary does // not exercise every preset (mirrors the same attribute on `group.rs`/`builder.rs`). @@ -21,10 +12,30 @@ use std::sync::Arc; use super::super::harness::HostRuntimeCapabilityHarness; +use super::super::harness::options::ToolsProfile; use super::{ - GroupCapability, HarnessResult, RebornIntegrationGroup, RebornIntegrationGroupBuilder, + GroupBaseData, GroupCapability, HarnessResult, RebornIntegrationGroup, + RebornIntegrationGroupBuilder, }; +/// Shared "align user to the group's canonical binding subject, then build" +/// step for the preset constructors below whose capability executes under +/// the group's resolved binding user rather than a fixed constructor test +/// user (`live_approvals`, `live_auth_and_approval`, `profile_tools`, +/// `outbound_target_tools`). Does NOT cover `skill_activation_tools` +/// (alignment is a constructor-time tenant param plus a post-build skill +/// seed) or `multiuser_approvals` (alignment is +/// `.with_run_owner_scoped_capability_dispatch()`, not a fixed `user_id` +/// override) — those remain call-site-specific. +async fn build_group_capability_with_base( + profile: ToolsProfile, + base: &GroupBaseData, +) -> HarnessResult { + let subject_user = base.canonical_subject_user()?; + let harness = profile.build().await?; + Ok(harness.with_user_id(subject_user)) +} + impl RebornIntegrationGroup { /// Group with real file-tool approval stores (write_file/read_file at /// `PermissionMode::Ask`). Auto-approve is disabled for the group scope at @@ -56,18 +67,14 @@ impl RebornIntegrationGroup { Self::builder().live_auth_gate().await } - /// C-JOURNEY convergence seam: group surfacing BOTH an unseeded GitHub - /// capability (raises `TurnStatus::BlockedAuth`, resolvable with - /// `resolve_auth_gate`/`deny_auth_gate`) AND real file-tool approval - /// stores (`write_file`/`read_file` at `PermissionMode::Ask`, raises - /// `TurnStatus::BlockedApproval`, resolvable with `approve_gate`/`deny_gate`) - /// on the SAME `build_reborn_services` runtime — unlike `live_auth_gate` - /// (a separate, lower-level `HostRuntimeServices` build with a hardcoded - /// credential resolver and no run_state store), this group's auth gate - /// resolves through the REAL `ProductAuthRuntimeCredentialResolver`, so - /// `resolve_auth_gate`'s happy-path resume actually completes. Auto-approve - /// is disabled for the group scope at construction so gated file-tool calls - /// raise real `BlockedApproval` gates. + /// C-JOURNEY: surfaces BOTH an unseeded GitHub capability (`BlockedAuth`, + /// resolve via `resolve_auth_gate`/`deny_auth_gate`) AND real file-tool + /// approvals (`BlockedApproval`, via `approve_gate`/`deny_gate`) on ONE + /// `build_reborn_services` runtime. Unlike `live_auth_gate` (a hardcoded + /// credential resolver, no run_state store), the auth gate here resolves + /// through the REAL `ProductAuthRuntimeCredentialResolver`, so + /// `resolve_auth_gate` actually completes. Auto-approve disabled at + /// construction. pub async fn live_auth_and_approval() -> HarnessResult { Self::builder().live_auth_and_approval().await } @@ -191,27 +198,20 @@ impl RebornIntegrationGroupBuilder { /// Build a live-approvals group. See [`RebornIntegrationGroup::live_approvals`]. pub async fn live_approvals(self) -> HarnessResult { let base = self.build_base().await?; - // Execute first-party tools under the run's CANONICAL binding subject - // user (the hashed `UserId` the actor `host-user` resolves to), not the - // constructor's fixed test user, so capability dispatch, approval - // persistence, auto-approve keying, and gate-evidence lookup all share the - // run's `(tenant, user)` — matching production. Reuse the SAME canonical - // binding `build_base` already resolved for the shared turn-store / - // evidence scope, so the approval user and the turn-store scope are - // derived from one probe and cannot drift. - let subject_user = base.canonical_subject_user()?; - let host_runtime = HostRuntimeCapabilityHarness::file_tools_requiring_approval() - .await? - .with_user_id(subject_user); + // Align capability execution to the run's CANONICAL binding subject user + // (not the constructor's fixed test user) so dispatch, approval, auto- + // approve keying, and gate-evidence lookup share one `(tenant, user)` — + // matching production. `build_group_capability_with_base` (above) is the + // shared "build then align user" core. + let host_runtime = build_group_capability_with_base( + super::super::harness::profiles::file::file_tools_requiring_approval_profile()?, + &base, + ) + .await?; let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); let group = self.into_group(base, capability).await?; - // Disable auto-approve once at build time so every thread in this group - // faces real approval gates. The dispatch-time check is keyed on the - // capability harness's executor user (NOT the binding owner), so target - // `auto_approve_scope()` — `(run tenant, capability user)`. - // `live_approvals` always constructs `GroupCapability::HostRuntime`, so - // both `auto_approve_scope()` and `capability_harness()` are guaranteed - // `Some` — use `expect` rather than a redundant `if let`. + // Disable auto-approve once so every thread faces real approval gates; + // always `HostRuntime` here, so `Some` is guaranteed. let scope = group .shared .auto_approve_scope() @@ -225,14 +225,16 @@ impl RebornIntegrationGroupBuilder { /// Build a core built-in tools group. See [`RebornIntegrationGroup::builtin_tools`]. pub async fn builtin_tools(self) -> HarnessResult { - let host_runtime = HostRuntimeCapabilityHarness::core_builtin_tools().await?; + let host_runtime = + super::super::harness::profiles::core_builtin::core_builtin_tools_default().await?; let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); self.build_with_capability(capability).await } /// Build an extension-lifecycle group. See [`RebornIntegrationGroup::extension_lifecycle`]. pub async fn extension_lifecycle(self) -> HarnessResult { - let host_runtime = HostRuntimeCapabilityHarness::extension_lifecycle_tools().await?; + let host_runtime = + super::super::harness::profiles::extension::extension_lifecycle_tools().await?; let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); self.build_with_capability(capability).await } @@ -243,7 +245,8 @@ impl RebornIntegrationGroupBuilder { /// self-evidencing via the BeforeBlock checkpoint (loop_exit_applier.rs). Do /// NOT add approval-gate evidence here — that store is only for approval gates. pub async fn live_auth_gate(self) -> HarnessResult { - let host_runtime = HostRuntimeCapabilityHarness::github_issue_tools_auth_required().await?; + let host_runtime = + super::super::harness::profiles::github::github_issue_tools_auth_required().await?; let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); self.build_with_capability(capability).await } @@ -258,10 +261,13 @@ impl RebornIntegrationGroupBuilder { /// constructor's fixed test user. pub async fn live_auth_and_approval(self) -> HarnessResult { let base = self.build_base().await?; - let subject_user = base.canonical_subject_user()?; - let host_runtime = HostRuntimeCapabilityHarness::file_and_github_auth_tools() - .await? - .with_user_id(subject_user); + // `build_group_capability_with_base` (above) is the shared "build then + // align user" core — see `live_approvals` above. + let host_runtime = build_group_capability_with_base( + super::super::harness::profiles::github::file_and_github_auth_tools_profile()?, + &base, + ) + .await?; let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); let group = self.into_group(base, capability).await?; // `file_and_github_auth_tools` already disabled auto-approve under its @@ -281,7 +287,7 @@ impl RebornIntegrationGroupBuilder { /// Build a project-lifecycle group. See [`RebornIntegrationGroup::project_lifecycle`]. pub async fn project_lifecycle(self) -> HarnessResult { - let host_runtime = HostRuntimeCapabilityHarness::project_tools().await?; + let host_runtime = super::super::harness::profiles::project::project_tools().await?; let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); self.build_with_capability(capability).await } @@ -291,7 +297,7 @@ impl RebornIntegrationGroupBuilder { /// [`RebornIntegrationGroup::project_lifecycle_fault_injected`]. pub async fn project_lifecycle_fault_injected(self) -> HarnessResult { let host_runtime = - HostRuntimeCapabilityHarness::project_tools_with_fault_injection().await?; + super::super::harness::profiles::project::project_tools_with_fault_injection().await?; let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); self.build_with_capability(capability).await } @@ -299,24 +305,23 @@ impl RebornIntegrationGroupBuilder { /// Build a profile-tools group. See [`RebornIntegrationGroup::profile_tools`]. pub async fn profile_tools(self) -> HarnessResult { let base = self.build_base().await?; - // Execute `builtin.profile_set` under the run's canonical binding - // subject user, mirroring `live_approvals`'s alignment above. - // Without this, a second thread's loop resolves the profile under - // the canonical subject user while the write dispatched under the - // fixed constructor user, so the read-back never sees it. Also why - // this cannot go through `build_with_capability`: the capability - // depends on `base`, so `base` must be resolved first. - let subject_user = base.canonical_subject_user()?; - let host_runtime = HostRuntimeCapabilityHarness::profile_tools() - .await? - .with_user_id(subject_user); + // Align `builtin.profile_set`'s executor to the canonical subject user + // (mirrors `live_approvals`) — otherwise a write and its read-back + // resolve under different users. Needs `base` first, so can't go + // through `build_with_capability`. + let host_runtime = build_group_capability_with_base( + super::super::harness::profiles::profile::profile_tools_profile()?, + &base, + ) + .await?; let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); self.into_group(base, capability).await } /// Build a trigger-management group. See [`RebornIntegrationGroup::triggers`]. pub async fn triggers(self) -> HarnessResult { - let host_runtime = HostRuntimeCapabilityHarness::trigger_management_tools().await?; + let host_runtime = + super::super::harness::profiles::trigger::trigger_management_tools().await?; let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); self.build_with_capability(capability).await } @@ -333,9 +338,10 @@ impl RebornIntegrationGroupBuilder { // above) rather than a separately hardcoded literal, so the E-SKILL // skill context source is built for the same tenant the turn runs // under — see `HostRuntimeCapabilityHarness::skill_activation_tools`. - let host_runtime = - HostRuntimeCapabilityHarness::skill_activation_tools(&base.canonical_binding.tenant_id) - .await?; + let host_runtime = super::super::harness::profiles::skill::skill_activation_tools( + &base.canonical_binding.tenant_id, + ) + .await?; host_runtime.seed_system_skill_for_test( "greet", "greets the user warmly", @@ -345,56 +351,51 @@ impl RebornIntegrationGroupBuilder { self.into_group(base, capability).await } - /// Build a per-actor-scoped memory group. - /// See [`RebornIntegrationGroup::multiuser_memory_tools`]. Same capability + /// Per-actor-scoped memory group. See + /// [`RebornIntegrationGroup::multiuser_memory_tools`]. Same capability /// surface as [`builtin_tools`] but with per-actor capability dispatch, so - /// each actor's memory lands under its own owner subtree. Self-contained - /// (no shared helper) so it relocates trivially if the group constructors - /// are later split out. + /// each actor's memory lands under its own owner subtree. pub async fn multiuser_memory_tools(self) -> HarnessResult { - let host_runtime = HostRuntimeCapabilityHarness::core_builtin_tools() - .await? - .with_run_owner_scoped_capability_dispatch(); + let host_runtime = + super::super::harness::profiles::core_builtin::core_builtin_tools_default() + .await? + .with_run_owner_scoped_capability_dispatch(); let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); self.build_with_capability(capability).await } - /// Build a per-actor-scoped file-approval group. - /// See [`RebornIntegrationGroup::multiuser_approvals`]. Real approval stores - /// (write_file/read_file @ `Ask`) plus per-actor capability dispatch. Auto- - /// approve defaults ON per owner (`AUTO_APPROVE_DEFAULT_ENABLED = true`), so - /// a scenario that needs an owner to GATE sets that owner OFF explicitly via - /// `disable_auto_approve_for_owner` (and grants another owner via - /// `enable_auto_approve_for_owner`) — the per-user setting is what the test - /// asserts isolates. Because the seam makes the dispatch user equal the turn - /// owner, a raised approval request persists under — and its gate-evidence - /// lookup resolves through — the SAME owner, so the gate is verified (not - /// masked to `Failed`). Self-contained for trivial relocation. + /// Per-actor-scoped file-approval group. See + /// [`RebornIntegrationGroup::multiuser_approvals`]. Real approval stores + /// (write_file/read_file @ `Ask`) plus per-actor capability dispatch; + /// auto-approve defaults ON per owner, so a test that needs an owner to + /// GATE sets that owner OFF via `disable_auto_approve_for_owner` — the + /// per-user setting is what isolation asserts. Dispatch user == turn owner, + /// so the raised approval's gate-evidence lookup resolves under that same + /// owner (verified, not masked to `Failed`). pub async fn multiuser_approvals(self) -> HarnessResult { let base = self.build_base().await?; - let host_runtime = HostRuntimeCapabilityHarness::file_tools_requiring_approval() + let host_runtime = super::super::harness::profiles::file::file_tools_requiring_approval() .await? .with_run_owner_scoped_capability_dispatch(); let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); self.into_group(base, capability).await } - /// Build an outbound-target-tools group. See - /// [`RebornIntegrationGroup::outbound_target_tools`]. Mirrors `live_approvals` - /// / `profile_tools`: the synthetic `outbound_delivery_*` capabilities run - /// under the run's CANONICAL binding subject user (via `with_user_id`), so - /// the dispatch-time settings/auto-approve scope - /// (`_shared.auto_approve_scope()` = `(tenant, capability user)`) aligns with - /// the run's effective dispatch user — the approval-gate arm's - /// `disable_auto_approve` and the deny arm's `disable_outbound_target_set_tool` - /// both target that exact `(tenant, user)`. Auto-approve is left at its - /// default-ON state (no disable here); the gate arm disables it per-test. + /// Outbound-target-tools group. See + /// [`RebornIntegrationGroup::outbound_target_tools`]. Mirrors + /// `live_approvals`/`profile_tools`: the synthetic `outbound_delivery_*` + /// capabilities run under the run's canonical binding subject user, so the + /// dispatch-time auto-approve scope aligns with tests' per-test disables. + /// Auto-approve stays default-ON here; the gate arm disables it per-test. pub async fn outbound_target_tools(self) -> HarnessResult { let base = self.build_base().await?; - let subject_user = base.canonical_subject_user()?; - let host_runtime = HostRuntimeCapabilityHarness::outbound_target_tools() - .await? - .with_user_id(subject_user); + // `build_group_capability_with_base` (above) is the shared "build then + // align user" core — see `live_approvals` above. + let host_runtime = build_group_capability_with_base( + super::super::harness::profiles::outbound::outbound_target_tools_profile()?, + &base, + ) + .await?; let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); self.into_group(base, capability).await } @@ -402,14 +403,14 @@ impl RebornIntegrationGroupBuilder { /// Build a skill-management group. See /// [`RebornIntegrationGroup::skill_management_tools`]. pub async fn skill_management_tools(self) -> HarnessResult { - let host_runtime = HostRuntimeCapabilityHarness::skill_management_tools().await?; + let host_runtime = super::super::harness::profiles::skill::skill_management_tools().await?; let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); self.build_with_capability(capability).await } /// Build an attachment-tools group. See [`RebornIntegrationGroup::attachment_tools`]. pub async fn attachment_tools(self) -> HarnessResult { - let host_runtime = HostRuntimeCapabilityHarness::attachment_tools().await?; + let host_runtime = super::super::harness::profiles::attachment::attachment_tools().await?; let capability = GroupCapability::HostRuntime(Arc::new(host_runtime)); self.build_with_capability(capability).await } diff --git a/tests/integration/support/group_options.rs b/tests/integration/support/group_options.rs new file mode 100644 index 00000000000..0aefba0ec23 --- /dev/null +++ b/tests/integration/support/group_options.rs @@ -0,0 +1,89 @@ +//! Runtime-wiring setters for [`RebornIntegrationGroupBuilder`] — `storage`, +//! `safety_context`, `with_turn_event_sink`, `budget_accounting`, +//! `communication_context_provider`, `hook_dispatcher_builder_factory`. +//! Private child module of `group.rs` (owns the struct + `build_base`/ +//! `into_group`), so it reaches the builder's private fields at module- +//! private visibility instead of widening them to `pub(crate)`. New builder +//! setters belong HERE. + +// Shared by all group test binaries; symbols read as dead when a binary does +// not exercise every setter (mirrors the same attribute on `group.rs`/ +// `group_constructors.rs`). +#![allow(dead_code)] + +use std::sync::Arc; + +use ironclaw_reborn::loop_driver_host::HookDispatcherBuilderFactory; +use ironclaw_turns::InMemoryTurnEventSink; +use ironclaw_turns::run_profile::{CommunicationContextProvider, InstructionSafetyContext}; + +use super::super::builder::StorageMode; +use super::RebornIntegrationGroupBuilder; + +impl RebornIntegrationGroupBuilder { + /// Select the durable storage backend (default: `StorageMode::InMemory`). + /// Use `StorageMode::LibSql` to exercise on-disk durability across + /// `assert_reply_persists_after_reopen`. + pub fn storage(mut self, mode: StorageMode) -> Self { + self.storage = mode; + self + } + + /// Wire a model-visible instruction-safety banner into the group's ONE + /// shared planned runtime (`DefaultPlannedRuntimeParts::safety_context`). + /// Rendered verbatim as a `system`-role prompt message ahead of any + /// per-turn instructions (`push_safety_context`); the only model-visible + /// artifact of instruction-safety scanning on this tier (T0-SYSPROMPT / + /// C-SAFETY). Defaults to `None` (no banner, matching today's behavior). + pub fn safety_context(mut self, ctx: InstructionSafetyContext) -> Self { + self.safety_context = Some(ctx); + self + } + + /// Install an in-memory `InMemoryTurnEventSink` into the group's ONE + /// planned runtime via production's `subscribe_best_effort` seam + /// (C-TRACECAP). Read back via + /// [`RebornIntegrationHarness::recorded_turn_events`] — the ONLY read + /// path; it slices `[baseline_turn_event_count..]` so threads don't see + /// siblings' events. No raw sink accessor, deliberately. + pub fn with_turn_event_sink(mut self) -> Self { + self.turn_event_sink = Some(Arc::new(InMemoryTurnEventSink::default())); + self + } + + /// Wire the production `build_default_budget_accountant` into the group's + /// ONE planned runtime and retain the governor for read-back (C-BUDGET + /// liveness seam: the accountant seeds the run owner's daily cap on the + /// first model call; read back via `assert_budget_user_cap_seeded`). + /// Budget semantics are covered at crate tier (`budget_e2e.rs`); this only + /// proves the harness wires the accountant live. Defaults off. + pub fn budget_accounting(mut self) -> Self { + self.budget = true; + self + } + + /// Wire a `CommunicationContextProvider` into the group's ONE shared planned + /// runtime (`DefaultPlannedRuntimeParts::communication_context_provider`), so + /// the delivery-preference / connected-channel slice it resolves renders into + /// the model request. This is the C-COMMCTX seam — distinct from the outbound + /// delivery **sink** (E-OUTBOUND): this is prompt **context**. Defaults `None`. + pub fn communication_context_provider( + mut self, + provider: Arc, + ) -> Self { + self.communication_context_provider = Some(provider); + self + } + + /// Wire a per-run `HookDispatcherBuilderFactory` into the group's ONE shared + /// planned runtime (`DefaultPlannedRuntimeParts::hook_dispatcher_builder_factory`), + /// so hooks fire at their lifecycle points on a coordinator-path turn. This is + /// the E-HOOK-INFRA / C-HOOKS seam. Defaults `None` (hook framework dormant). + pub fn hook_dispatcher_builder_factory( + mut self, + factory: HookDispatcherBuilderFactory, + ) -> Self { + self.hook_dispatcher_builder_factory = Some(factory); + self + } +} diff --git a/tests/integration/support/harness/assembly.rs b/tests/integration/support/harness/assembly.rs new file mode 100644 index 00000000000..b983f12a5c9 --- /dev/null +++ b/tests/integration/support/harness/assembly.rs @@ -0,0 +1,462 @@ +use std::{ + path::{Path, PathBuf}, + sync::Arc, + time::Duration, +}; + +use super::super::{ + extension_surface::BUNDLED_EXTENSION_IDS, github as github_support, harness_web_access, +}; +use ironclaw_authorization::GrantAuthorizer; +use ironclaw_extensions::ExtensionRegistry; +use ironclaw_filesystem::{ + BackendCapabilities, BackendId, BackendKind, CompositeRootFilesystem, ContentKind, + InMemoryBackend, IndexPolicy, LocalFilesystem, MountDescriptor, RootFilesystem, StorageClass, +}; +use ironclaw_host_api::{ + CapabilityId, CredentialStageError, EffectKind, ExtensionId, HostPath, MountAlias, MountGrant, + MountPermissions, MountView, NetworkPolicy, NetworkScheme, NetworkTargetPattern, PackageId, + SecretHandle, VirtualPath, +}; +use ironclaw_host_runtime::{ + BUILTIN_FIRST_PARTY_PROVIDER, CapabilitySurfaceVersion as HostRuntimeCapabilitySurfaceVersion, + HostRuntime, HostRuntimeServices, RuntimeProcessPort, builtin_first_party_handlers, + builtin_first_party_package, +}; +use ironclaw_network::{PolicyNetworkHttpEgress, ReqwestNetworkTransport}; +use ironclaw_resources::InMemoryResourceGovernor; +use ironclaw_secrets::{InMemorySecretStore, SecretMaterial}; +use ironclaw_trust::{AdminConfig, AdminEntry, HostTrustAssignment, HostTrustPolicy}; +use ironclaw_wasm::{WitToolHost, WitToolRuntimeConfig}; + +use super::super::doubles::{ + FixedRuntimeCredentialAccountResolver, GithubHarnessAuthorizer, RecordingNetworkHttpEgress, + RecordingRuntimeHttpEgress, StaticSecretStore, +}; +use super::HarnessResult; + +pub(crate) fn local_dev_host_runtime_with_http_egress( + storage_root: PathBuf, + egress: Arc, + process_port: Option>, +) -> HarnessResult> { + let mut registry = ExtensionRegistry::new(); + registry.insert(builtin_first_party_package()?)?; + local_dev_host_runtime_with_registry_and_runtime_http_egress( + storage_root, + registry, + egress, + process_port, + ) +} + +pub(crate) fn host_runtime_storage_roots() +-> HarnessResult<(Arc, PathBuf, PathBuf)> { + let root = Arc::new(tempfile::tempdir()?); + let storage_root = root.path().join("local-dev"); + let workspace_root = storage_root.join("workspace"); + std::fs::create_dir_all(&workspace_root)?; + Ok((root, storage_root, workspace_root)) +} + +pub(crate) fn local_dev_host_runtime_with_registry_and_runtime_http_egress( + storage_root: PathBuf, + registry: ExtensionRegistry, + egress: Arc, + process_port: Option>, +) -> HarnessResult> { + let mut services = HostRuntimeServices::new( + Arc::new(registry), + local_dev_root_filesystem(storage_root, LocalDevRootMounts::core_builtins())?, + Arc::new(InMemoryResourceGovernor::new()), + Arc::new(GrantAuthorizer::new()), + ironclaw_processes::ProcessServices::in_memory(), + HostRuntimeCapabilitySurfaceVersion::new("reborn-app-v1")?, + ) + .with_secret_store(Arc::new(StaticSecretStore::new( + SecretHandle::new("github_manual_access")?, + SecretMaterial::from("ghp_fake_fixture_token"), + ))) + .with_runtime_credential_account_resolver(Arc::new(FixedRuntimeCredentialAccountResolver { + result: Ok(SecretHandle::new("github_manual_access")?), + })) + .with_first_party_capabilities(Arc::new(builtin_first_party_handlers(Arc::new( + ironclaw_triggers::InMemoryTriggerRepository::default(), + ))?)) + .with_first_party_http_egress(egress) + .with_trust_policy(Arc::new(first_party_trust_policy()?)); + // Inject the recording process port when provided; `None` defaults to + // `LocalHostProcessPort` (real execution). + if let Some(port) = process_port { + services = services.with_runtime_process_port_dyn(port); + } + + Ok(Arc::new(services.host_runtime_for_local_testing())) +} + +pub(crate) fn local_dev_host_runtime_with_registry_and_egress( + storage_root: PathBuf, + registry: ExtensionRegistry, + runtime_http_egress: Arc, + network_egress: Arc, + // E-AUTHGATE: `Ok(handle)` resolves the credential account (capability + // dispatches); `Err(AuthRequired)` raises a `BlockedAuth` gate at dispatch. + credential_account_result: Result, +) -> HarnessResult> { + let services = HostRuntimeServices::new( + Arc::new(registry), + local_dev_root_filesystem(storage_root, LocalDevRootMounts::github_assets())?, + Arc::new(InMemoryResourceGovernor::new()), + Arc::new(GithubHarnessAuthorizer::new()?), + ironclaw_processes::ProcessServices::in_memory(), + HostRuntimeCapabilitySurfaceVersion::new("reborn-app-v1")?, + ) + .with_secret_store(Arc::new(StaticSecretStore::new( + SecretHandle::new("github_manual_access")?, + SecretMaterial::from("ghp_fake_fixture_token"), + ))) + .with_runtime_credential_account_resolver(Arc::new(FixedRuntimeCredentialAccountResolver { + result: credential_account_result, + })) + .with_first_party_capabilities(Arc::new(builtin_first_party_handlers(Arc::new( + ironclaw_triggers::InMemoryTriggerRepository::default(), + ))?)) + .with_runtime_http_egress(runtime_http_egress) + .with_trust_policy(Arc::new(github_first_party_trust_policy()?)) + .try_with_host_http_egress((*network_egress).clone()) + .map_err(|report| std::io::Error::other(format!("host HTTP egress failed: {report:?}")))? + .try_with_wasm_runtime(WitToolRuntimeConfig::default(), WitToolHost::deny_all()) + .map_err(|report| std::io::Error::other(format!("WASM runtime failed: {report:?}")))?; + + Ok(Arc::new(services.host_runtime_for_local_testing())) +} + +pub(crate) fn local_dev_host_runtime_with_live_http_egress( + storage_root: PathBuf, +) -> HarnessResult> { + let mut registry = ExtensionRegistry::new(); + registry.insert(builtin_first_party_package()?)?; + + let services = HostRuntimeServices::new( + Arc::new(registry), + local_dev_root_filesystem(storage_root, LocalDevRootMounts::core_builtins())?, + Arc::new(InMemoryResourceGovernor::new()), + Arc::new(GrantAuthorizer::new()), + ironclaw_processes::ProcessServices::in_memory(), + HostRuntimeCapabilitySurfaceVersion::new("reborn-app-v1")?, + ) + .with_secret_store(Arc::new(InMemorySecretStore::new())) + .with_first_party_capabilities(Arc::new(builtin_first_party_handlers(Arc::new( + ironclaw_triggers::InMemoryTriggerRepository::default(), + ))?)) + .try_with_host_http_egress(PolicyNetworkHttpEgress::new(ReqwestNetworkTransport::new( + Duration::from_secs(2), + ))) + .map_err(|report| { + std::io::Error::other(format!( + "live HTTP egress production wiring failed: {report:?}" + )) + })? + .with_trust_policy(Arc::new(first_party_trust_policy()?)); + + Ok(Arc::new(services.host_runtime_for_local_testing())) +} + +pub(crate) fn local_dev_root_filesystem( + storage_root: PathBuf, + mounts: LocalDevRootMounts, +) -> HarnessResult> { + let mut local = LocalFilesystem::new(); + local.mount_local( + VirtualPath::new("/projects")?, + HostPath::from_path_buf(storage_root), + )?; + if mounts.github_assets { + local.mount_local( + VirtualPath::new("/system/extensions/github")?, + HostPath::from_path_buf(github_support::asset_root()), + )?; + } + if mounts.web_access_assets { + local.mount_local( + VirtualPath::new("/system/extensions/web-access")?, + HostPath::from_path_buf(harness_web_access::asset_root()), + )?; + } + + let local = Arc::new(local); + let mut root = CompositeRootFilesystem::new(); + root.mount( + local_dev_mount_descriptor( + "/projects", + "local-dev-projects", + BackendKind::LocalFilesystem, + StorageClass::FileContent, + ContentKind::ProjectFile, + IndexPolicy::NotIndexed, + BackendCapabilities::bytes_only(), + )?, + Arc::clone(&local), + )?; + if mounts.github_assets { + root.mount( + local_dev_mount_descriptor( + "/system/extensions/github", + "local-dev-github-assets", + BackendKind::LocalFilesystem, + StorageClass::FileContent, + ContentKind::ExtensionPackage, + IndexPolicy::NotIndexed, + BackendCapabilities::bytes_only(), + )?, + Arc::clone(&local), + )?; + } + if mounts.web_access_assets { + root.mount( + local_dev_mount_descriptor( + "/system/extensions/web-access", + "local-dev-web-access-assets", + BackendKind::LocalFilesystem, + StorageClass::FileContent, + ContentKind::ExtensionPackage, + IndexPolicy::NotIndexed, + BackendCapabilities::bytes_only(), + )?, + Arc::clone(&local), + )?; + } + if mounts.memory { + let memory = Arc::new(InMemoryBackend::new()); + root.mount( + local_dev_mount_descriptor( + "/memory", + "local-dev-memory", + BackendKind::MemoryDocuments, + StorageClass::StructuredRecords, + ContentKind::MemoryDocument, + IndexPolicy::FullTextAndVector, + memory.capabilities(), + )?, + memory, + )?; + } + Ok(Arc::new(root)) +} + +#[derive(Clone, Copy)] +pub(crate) struct LocalDevRootMounts { + github_assets: bool, + web_access_assets: bool, + memory: bool, +} + +impl LocalDevRootMounts { + pub(crate) fn core_builtins() -> Self { + Self { + github_assets: false, + web_access_assets: false, + memory: true, + } + } + + fn github_assets() -> Self { + Self { + github_assets: true, + web_access_assets: false, + memory: false, + } + } + + pub(crate) fn web_access_assets() -> Self { + Self { + github_assets: false, + web_access_assets: true, + memory: false, + } + } +} + +pub(crate) fn local_dev_mount_descriptor( + virtual_root: &str, + backend_id: &str, + backend_kind: BackendKind, + storage_class: StorageClass, + content_kind: ContentKind, + index_policy: IndexPolicy, + capabilities: BackendCapabilities, +) -> HarnessResult { + Ok(MountDescriptor { + virtual_root: VirtualPath::new(virtual_root)?, + backend_id: BackendId::new(backend_id)?, + backend_kind, + storage_class, + content_kind, + index_policy, + capabilities, + }) +} + +pub(crate) fn first_party_trust_policy() -> HarnessResult { + Ok(HostTrustPolicy::new(vec![Box::new( + AdminConfig::with_entries(vec![AdminEntry::for_local_manifest( + PackageId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, + "/system/extensions/builtin/manifest.toml".to_string(), + None, + HostTrustAssignment::first_party(), + vec![ + EffectKind::DispatchCapability, + EffectKind::ReadFilesystem, + EffectKind::WriteFilesystem, + EffectKind::DeleteFilesystem, + EffectKind::Network, + EffectKind::SpawnProcess, + EffectKind::ExecuteCode, + EffectKind::ExternalWrite, + ], + None, + )]), + )])?) +} + +pub(crate) fn github_first_party_trust_policy() -> HarnessResult { + Ok(HostTrustPolicy::new(vec![Box::new( + AdminConfig::with_entries(vec![AdminEntry::for_local_manifest( + PackageId::new("github")?, + "/system/extensions/github/manifest.toml".to_string(), + None, + HostTrustAssignment::first_party(), + vec![ + EffectKind::DispatchCapability, + EffectKind::Network, + EffectKind::UseSecret, + EffectKind::ExternalWrite, + ], + None, + )]), + )])?) +} + +pub(crate) fn http_test_policy() -> NetworkPolicy { + NetworkPolicy { + allowed_targets: vec![NetworkTargetPattern { + scheme: Some(NetworkScheme::Https), + host_pattern: "api.example.test".to_string(), + port: None, + }], + deny_private_ip_ranges: true, + max_egress_bytes: Some(10_000), + } +} + +pub(crate) fn wildcard_test_policy() -> NetworkPolicy { + NetworkPolicy { + allowed_targets: vec![NetworkTargetPattern { + scheme: None, + host_pattern: "*".to_string(), + port: None, + }], + deny_private_ip_ranges: true, + max_egress_bytes: Some(1_000_000), + } +} + +/// C-JOURNEY: recursively copy `src` into `dst` (creating `dst` and any +/// intermediate directories). Used to populate a harness's per-test +/// `/system/extensions/` mount with the real bundled-extension asset +/// directory (manifest + wasm module + schemas) so a WASM capability +/// published via `publish_bundled_extension_for_test` is genuinely loadable, +/// not just registered as metadata. +pub(crate) fn copy_dir_recursive(src: &Path, dst: &Path) -> HarnessResult<()> { + std::fs::create_dir_all(dst)?; + for entry in std::fs::read_dir(src)? { + let entry = entry?; + let file_type = entry.file_type()?; + let dst_path = dst.join(entry.file_name()); + if file_type.is_dir() { + copy_dir_recursive(&entry.path(), &dst_path)?; + } else { + std::fs::copy(entry.path(), &dst_path)?; + } + } + Ok(()) +} + +pub(crate) fn capability_ids_from_strs(ids: &[&str]) -> HarnessResult> { + ids.iter() + .map(|id| CapabilityId::new(*id).map_err(Into::into)) + .collect() +} + +pub(crate) fn bundled_extension_provider_trust() +-> HarnessResult)>> { + BUNDLED_EXTENSION_IDS + .iter() + .map(|id| Ok((ExtensionId::new(*id)?, local_dev_all_effects()))) + .collect() +} + +pub(crate) fn local_dev_all_effects() -> Vec { + vec![ + EffectKind::DispatchCapability, + EffectKind::ReadFilesystem, + EffectKind::WriteFilesystem, + EffectKind::DeleteFilesystem, + EffectKind::Network, + EffectKind::UseSecret, + EffectKind::SpawnProcess, + EffectKind::ExecuteCode, + EffectKind::ExternalWrite, + ] +} + +pub(crate) fn workspace_mounts(permissions: MountPermissions) -> HarnessResult { + Ok(MountView::new(vec![MountGrant::new( + MountAlias::new("/workspace")?, + VirtualPath::new("/projects/workspace")?, + permissions, + )])?) +} + +pub(crate) fn memory_mounts(permissions: MountPermissions) -> HarnessResult { + Ok(MountView::new(vec![MountGrant::new( + MountAlias::new("/memory")?, + VirtualPath::new("/memory")?, + permissions, + )])?) +} + +pub(crate) fn skill_mounts() -> HarnessResult { + Ok(MountView::new(vec![ + MountGrant::new( + MountAlias::new("/skills")?, + VirtualPath::new("/projects/skills")?, + MountPermissions::read_write_list_delete(), + ), + MountGrant::new( + MountAlias::new("/system/skills")?, + VirtualPath::new("/projects/system/skills")?, + MountPermissions::read_only(), + ), + ])?) +} + +pub(crate) fn qa_smoke_mounts() -> HarnessResult { + Ok(MountView::new(vec![ + MountGrant::new( + MountAlias::new("/workspace")?, + VirtualPath::new("/projects/workspace")?, + MountPermissions::read_write_list_delete(), + ), + MountGrant::new( + MountAlias::new("/skills")?, + VirtualPath::new("/projects/skills")?, + MountPermissions::read_write_list_delete(), + ), + MountGrant::new( + MountAlias::new("/system/skills")?, + VirtualPath::new("/projects/system/skills")?, + MountPermissions::read_only(), + ), + ])?) +} diff --git a/tests/integration/support/harness/mod.rs b/tests/integration/support/harness/mod.rs new file mode 100644 index 00000000000..328939f2d79 --- /dev/null +++ b/tests/integration/support/harness/mod.rs @@ -0,0 +1,1339 @@ +//! `HostRuntimeCapabilityHarness` — the integration-tier harness that +//! assembles real host-runtime capability wiring (a genuine `HostRuntime`, +//! real mounts, real capability dispatch) over recorded test doubles +//! substituted at the port boundaries (HTTP/network egress, process, +//! approval). See `tests/integration/CLAUDE.md` for the `harness/` file +//! split and the single-fake-at-the-vendor-SDK-seam contract it implements. + +#![allow(dead_code)] // Shared by staged Reborn binary-E2E validation ports. + +pub(crate) mod assembly; +pub(crate) mod options; +pub(crate) mod profiles; +pub(crate) mod recorder; + +pub(crate) use options::HostRuntimeHarnessOptions; +pub(crate) use recorder::{HarnessCapabilityRecorder, RecordedCapabilityResult}; + +use std::{ + collections::HashMap, + path::PathBuf, + sync::{Arc, Mutex}, +}; + +use super::{filesystem::BlockingTurnStatePutFilesystem, product_workflow::resource_scope}; +use ironclaw_approvals::{ApprovalResolver, AutoApproveSettingInput, DenyApproval, LeaseApproval}; +use ironclaw_auth::{AuthProductScope, AuthProviderId, AuthSurface, CredentialAccountLabel}; +use ironclaw_filesystem::{ + BackendKind, CompositeRootFilesystem, ContentKind, InMemoryBackend, IndexPolicy, + RootFilesystem, ScopedFilesystem, StorageClass, +}; +use ironclaw_host_api::{ + Action, AgentId, ApprovalRequestId, CapabilityGrant, CapabilityGrantId, CapabilityId, + CapabilitySet, EffectKind, ExtensionId, GrantConstraints, InvocationId, MountAlias, MountGrant, + MountPermissions, MountView, NetworkPolicy, Principal, ProjectId, ResourceScope, + RuntimeHttpEgressRequest, RuntimeKind, SecretHandle, TenantId, TrustClass, UserId, VirtualPath, +}; +use ironclaw_host_runtime::{CapabilitySurfacePolicy, HostRuntime, SurfaceKind}; +use ironclaw_loop_support::{ + CapabilityAllowSet, CapabilitySurfaceProfileResolver, HostRuntimeLoopCapabilityPortFactory, + LoopCapabilityPortFactory, LoopCapabilityResultWriter, +}; +use ironclaw_network::NetworkHttpRequest; +use ironclaw_product_workflow::{ProjectService, ResolvedBinding}; +use ironclaw_reborn_composition::test_support::SkillActivationTestSource; +use ironclaw_reborn_composition::{ + ProductLiveCapabilityIo, ProductLiveVisibleCapabilityRequestConfig, RebornBuildInput, + RebornLocalDevApprovalTestParts, RebornProductAuthServices, build_reborn_services, + visible_capability_request_for_run, +}; +use ironclaw_trust::EffectiveTrustClass; +use ironclaw_turns::{ + GateRef, + run_profile::{ + AgentLoopHostError, AgentLoopHostErrorKind, CapabilityInvocation, LoopCapabilityPort, + LoopHostMilestoneSink, LoopRunContext, + }, +}; + +pub(crate) use super::doubles::{ + EmptyIdentityContextSource, HarnessCapabilityPortFactory, + HostRuntimeHarnessCapabilityPortFactory, RecordingApprovalRequestStore, + RecordingCapabilityResultWriter, RecordingDelegatingCapabilityPort, RecordingHostRuntime, + RecordingNetworkHttpEgress, RecordingRuntimeHttpEgress, RecordingTestCapabilityPort, + StaticCapabilitySurfaceProfileResolver, +}; +pub(crate) use assembly::{ + LocalDevRootMounts, bundled_extension_provider_trust, capability_ids_from_strs, + copy_dir_recursive, host_runtime_storage_roots, http_test_policy, local_dev_all_effects, + local_dev_host_runtime_with_http_egress, local_dev_host_runtime_with_live_http_egress, + local_dev_host_runtime_with_registry_and_egress, local_dev_mount_descriptor, + local_dev_root_filesystem, memory_mounts, qa_smoke_mounts, skill_mounts, wildcard_test_policy, + workspace_mounts, +}; + +pub(crate) type HarnessResult = Result>; +pub(crate) type HarnessCapabilityParts = ( + Arc, + Arc, + Arc, + Arc, + HarnessCapabilityRecorder, +); +pub(crate) type HarnessTurnStorageBackend = BlockingTurnStatePutFilesystem; +pub(crate) type HarnessTurnBackend = CompositeRootFilesystem; + +pub(crate) enum HarnessCapabilityMode { + Recording(RecordingTestCapabilityPort), + HostRuntime(Arc), +} + +impl HarnessCapabilityMode { + pub(crate) fn into_parts( + self, + milestone_sink: Arc, + ) -> HarnessResult { + match self { + Self::Recording(port) => { + let port = Arc::new(port); + let capability_io = Arc::new(ProductLiveCapabilityIo::default()); + Ok(( + Arc::new(HarnessCapabilityPortFactory { + port: Arc::clone(&port), + }), + Arc::new(StaticCapabilitySurfaceProfileResolver { + allow_set: CapabilityAllowSet::allowlist(port.capability_allowlist()), + }), + capability_io.clone(), + capability_io, + HarnessCapabilityRecorder::Recording(port), + )) + } + Self::HostRuntime(harness) => Ok(( + harness.capability_factory(milestone_sink), + Arc::new(StaticCapabilitySurfaceProfileResolver { + allow_set: CapabilityAllowSet::allowlist(harness.capability_ids.clone()), + }), + harness.io.clone(), + harness.capability_result_writer(), + HarnessCapabilityRecorder::HostRuntime(harness), + )), + } + } +} + +/// Backing handles for the two synthetic `outbound_delivery_*` capabilities +/// (C-SYNTH outbound seam). `Some` only for `outbound_target_tools()`. Bundles +/// the injected facade double + the settings stores the production +/// `outbound_delivery_capabilities` wiring consumes, so the harness struct +/// widens by ONE field instead of four. The auto-approve store and +/// approval-request/lease stores are already held as sibling harness fields +/// (`auto_approve_settings` / `approval_parts`) and re-used, not duplicated here. +struct OutboundTargetToolsParts { + /// Concrete double (not the trait object) so tests can read `set` calls back; + /// upcast to `Arc` at wrap time. + facade: Arc, + requires_approval: bool, + tool_permission_overrides: Arc, + persistent_approval_policies: Arc, +} + +pub(crate) struct HostRuntimeCapabilityHarness { + runtime: Arc, + approval_parts: Option, + auto_approve_settings: Option>, + pending_approval_scopes: Arc>>, + io: Arc, + root: Arc, + workspace_root: PathBuf, + mounts: MountView, + capability_mount_overrides: Vec<(CapabilityId, MountView)>, + capability_ids: Vec, + runtime_kind: RuntimeKind, + effect_kinds: Vec, + network_policy: NetworkPolicy, + secrets: Vec, + provider_id: ExtensionId, + additional_provider_trust: Vec<(ExtensionId, Vec)>, + user_id: UserId, + invocations: Arc>>, + results: Arc>>, + http_egress: Option>, + network_egress: Option>, + /// Inert recording process port. `Some` when the harness injected a + /// `RecordingProcessPort`; `None` when the live `LocalHostProcessPort` was + /// used (`.with_live_shell()` path). + process_port: Option>, + /// Raw local-dev memory filesystem backing the user-profile source + /// (E-PROFILE seam). `Some` only for `new_with_options`-built harnesses (which + /// flow through `RebornServices`); `None` for the lower-level constructors and + /// the Echo backend. Read back via `profile_filesystem_for_test`. + profile_filesystem: Option>, + /// Project service for the local-dev synthetic `project_create` capability + /// (E-PROJ seam). `Some` for every `new_with_options`-built local-dev harness; + /// `None` for the lower-level constructors and the Echo backend. + /// `create_capability_port` wraps the port with the synthetic project-create + /// capability ONLY when `PROJECT_CREATE_CAPABILITY_ID` is also in + /// `capability_ids` (i.e. only `project_tools()` surfaces it), so other + /// local-dev groups are unaffected. Tests read projects back via `project_service`. + project_service: Option>, + /// Local-dev skill context source for the synthetic `skill_activate` + /// capability and runtime prompt injection (E-SKILL seam). `Some` only for + /// `skill_activation_tools()`; `None` otherwise. `create_capability_port` + /// wraps the port with the synthetic `skill_activate` capability ONLY when + /// `SKILL_ACTIVATE_CAPABILITY_ID` is also in `capability_ids`, and its + /// `context_source()` is wired as the runtime's `skill_context_source` in + /// `into_group`. Held as the opaque test-support handle so this crate never + /// names the crate-private source type. + skill_activation_source: Option, + /// Attachment read port + inbound lander backing the C-ATTACH seam. `Some` + /// only for `new_with_options`-built harnesses (which flow through + /// `RebornServices`, and thus have a local-dev workspace filesystem to build + /// both over); `None` for the lower-level constructors and the Echo backend. + /// Read back via `attachment_test_support_for_test`. + attachment_test_support: Option, + /// Backing handles for the synthetic `outbound_delivery_*` capabilities + /// (C-SYNTH outbound seam). `Some` only for `outbound_target_tools()`; + /// `create_capability_port` wraps the port with the two capabilities via + /// `apply_synthetic_capability_wrappers` when this is `Some`. + outbound_target_tools: Option, + /// C-MULTIUSER seam: when `true`, [`create_capability_port`] resolves the + /// capability-execution user from the RUN's owner/actor (mirroring + /// production `local_dev_visible_capability_request`, + /// `crates/ironclaw_reborn_composition/src/runtime/local_dev.rs`) instead of + /// this harness's single fixed `user_id`. That is what lets two distinct + /// actors dispatching over the group's ONE shared capability backend run + /// under DISTINCT `(tenant, user)` scopes, so memory, auto-approve, and + /// approval-settings isolate per actor — the real production behavior. + /// Defaults `false` so every existing fixed-user harness is byte-identical; + /// only the multiuser group constructors flip it on via + /// [`with_run_owner_scoped_capability_dispatch`]. + scope_capability_by_run_owner: bool, + /// Local-dev product-auth services (C-JOURNEY convergence seam). `Some` + /// only for `new_with_options`-built harnesses (which flow through + /// `RebornServices`); `None` for the lower-level constructors and the Echo + /// backend. `seed_github_credential_account` reads this to create a real + /// credential account through `credential_account_service()`, letting a + /// parked `github.*` auth gate's `ProductAuthRuntimeCredentialResolver` + /// lookup resolve on re-dispatch. + product_auth: Option>, + /// W4-ASK-EACH-ONCE: local-dev per-tool permission override store (mirrors + /// `auto_approve_settings`). `Some` only for `new_with_options`-built + /// harnesses (which flow through `RebornServices`); `None` for the + /// lower-level constructors and the Echo backend. Lets a test install a + /// dynamic `ToolPermissionOverride::AskEachTime` override on any capability + /// via `set_ask_each_time_override_for_test`, independent of the + /// `outbound_target_tools()`-only `OutboundTargetToolsParts` copy. + tool_permission_overrides: Option>, +} + +impl HostRuntimeCapabilityHarness { + /// C-JOURNEY: seed a real GitHub credential account — WITH real secret + /// material — through the PRODUCTION manual-token flow + /// (`request_manual_token_setup` → `submit_manual_token`), so a parked + /// `github.*` auth gate resolves on re-dispatch AND the WASM capability's + /// credential obligation can actually stage the token (a bare + /// `create_account(..)` with a dangling handle clears the gate but then + /// fails at `stage_credential_material`). + /// + /// `scope` MUST be the run's actual dispatch-time `(tenant, user, agent, + /// project)` — a mismatch silently seeds an account the dispatch-time + /// lookup never finds, leaving the run stuck at `BlockedAuth` (see + /// `RebornIntegrationHarness::resolve_auth_gate` for how callers derive it). + /// + /// Continuation is `AuthContinuationRef::SetupOnly`: the harness's + /// `resolve_auth_gate` performs the run resume itself, so this must not + /// ALSO dispatch a `TurnGateResume` continuation. + pub(crate) async fn seed_github_credential_account( + &self, + scope: &ResourceScope, + ) -> HarnessResult<()> { + let product_auth = self + .product_auth + .as_ref() + .ok_or("harness missing local-dev product auth (not built via new_with_options)")?; + let scope = AuthProductScope::credential_owner(scope, AuthSurface::Api); + let challenge = product_auth + .request_manual_token_setup( + ironclaw_reborn_composition::RebornManualTokenSetupRequest::new( + scope.clone(), + AuthProviderId::new("github")?, + CredentialAccountLabel::new("journey github")?, + ironclaw_auth::AuthContinuationRef::SetupOnly, + chrono::Utc::now() + chrono::Duration::minutes(10), + ), + ) + .await + .map_err(|error| format!("manual token setup failed: {error:?}"))?; + product_auth + .submit_manual_token( + ironclaw_reborn_composition::RebornManualTokenSubmitRequest::new( + scope.clone(), + challenge.interaction_id, + secrecy::SecretString::from("journey-github-token"), + ), + ) + .await + .map_err(|error| format!("manual token submit failed: {error:?}"))?; + Ok(()) + } + + async fn enable_global_auto_approve_for_product_and_harness_users(&self) -> HarnessResult<()> { + let product_scope = product_scope(); + self.enable_global_auto_approve(product_scope.clone()) + .await?; + let mut harness_user_scope = product_scope; + harness_user_scope.user_id = self.user_id.clone(); + self.enable_global_auto_approve(harness_user_scope).await?; + Ok(()) + } + + pub(crate) async fn enable_global_auto_approve( + &self, + scope: ResourceScope, + ) -> HarnessResult<()> { + let store = self + .auto_approve_settings + .as_ref() + .ok_or("host runtime harness missing local-dev auto-approve settings")?; + store + .set(AutoApproveSettingInput { + updated_by: Principal::User(scope.user_id.clone()), + scope, + enabled: true, + }) + .await?; + Ok(()) + } + + /// Global auto-approve now defaults ON. A test that needs to exercise the + /// per-tool approval gate must flip it OFF for the product and harness-user + /// scopes the run authorizes against, as an explicit precondition. + pub async fn disable_global_auto_approve_for_product_and_harness_users( + &self, + ) -> HarnessResult<()> { + let product_scope = product_scope(); + self.disable_global_auto_approve(product_scope.clone()) + .await?; + let mut harness_user_scope = product_scope; + harness_user_scope.user_id = self.user_id.clone(); + self.disable_global_auto_approve(harness_user_scope).await?; + Ok(()) + } + + pub(crate) async fn disable_global_auto_approve( + &self, + scope: ResourceScope, + ) -> HarnessResult<()> { + let store = self + .auto_approve_settings + .as_ref() + .ok_or("host runtime harness missing local-dev auto-approve settings")?; + store + .set(AutoApproveSettingInput { + updated_by: Principal::User(scope.user_id.clone()), + scope, + enabled: false, + }) + .await?; + Ok(()) + } + + async fn new( + service_label: &'static str, + capability_ids: Vec, + effect_kinds: Vec, + secrets: Vec, + provider_id: ExtensionId, + user_id: UserId, + runtime_policy: Option, + ) -> HarnessResult { + Self::new_with_options( + service_label, + capability_ids, + effect_kinds, + secrets, + provider_id, + user_id, + HostRuntimeHarnessOptions::new( + workspace_mounts(MountPermissions::read_write_list_delete())?, + runtime_policy, + ), + ) + .await + } + + async fn new_with_options( + service_label: &'static str, + capability_ids: Vec, + effect_kinds: Vec, + secrets: Vec, + provider_id: ExtensionId, + user_id: UserId, + options: HostRuntimeHarnessOptions, + ) -> HarnessResult { + let HostRuntimeHarnessOptions { + mounts, + runtime_policy, + seed_extension_credentials, + skill_activation_tenant, + outbound_target_facade, + network_http_egress_for_test, + activate_bundled_extensions_for_test, + project_service_fault_injection, + } = options; + let root = Arc::new(tempfile::tempdir()?); + let storage_root = root.path().join("local-dev"); + let workspace_root = storage_root.join("workspace"); + std::fs::create_dir_all(&workspace_root)?; + let mut input = if runtime_policy.as_ref().is_some_and(|policy| { + policy.resolved_profile == ironclaw_host_api::runtime_policy::RuntimeProfile::LocalYolo + }) { + let host_home_root = root.path().join("host-home"); + std::fs::create_dir_all(&host_home_root)?; + ironclaw_reborn_composition::local_runtime_build_input_with_options( + ironclaw_reborn_composition::RebornCompositionProfile::LocalDevYolo, + service_label, + storage_root, + ironclaw_reborn_composition::RebornLocalRuntimeProfileOptions { + confirm_host_access: true, + }, + )? + .with_local_dev_confirmed_host_home_root(host_home_root) + } else { + RebornBuildInput::local_dev(service_label, storage_root) + }; + if let Some(runtime_policy) = runtime_policy { + input = input.with_runtime_policy(runtime_policy); + } + if let Some(egress) = network_http_egress_for_test { + input = input.with_network_http_egress_for_test(egress); + } + let services = build_reborn_services(input).await?; + if seed_extension_credentials { + profiles::extension::seed_extension_lifecycle_credentials(&services, &user_id).await?; + } + // C-JOURNEY: publish bundled WASM packages into the active-extension + // registry directly (see `HostRuntimeHarnessOptions::activate_bundled_extensions_for_test` + // doc) so their capabilities are genuinely dispatchable, not merely + // granted at the harness-authority layer. + for package in &activate_bundled_extensions_for_test { + services + .publish_bundled_extension_for_test(package) + .ok_or( + "local-dev Reborn services missing extension management for test publish", + )??; + } + let approval_parts = services.local_dev_approval_test_parts(); + let auto_approve_settings = services.local_dev_auto_approve_settings_for_test(); + // Capture the profile filesystem + project service + attachment support + // before `services.host_runtime` is moved out below (E-PROFILE / E-PROJ / + // C-ATTACH seams). + let profile_filesystem = services.local_dev_profile_filesystem_for_test(); + // C-SYNTH `project_create` fault-injection seam: wrap the real service + // in `FaultInjectingProjectService` only when the harness opted in + // (`with_project_service_fault_injection`) — every other harness keeps + // the real service unwrapped and behaves exactly as before. + let project_service: Option> = + services.local_dev_project_service_for_test().map(|inner| { + if project_service_fault_injection { + super::project_service_fault::FaultInjectingProjectService::wrapping(inner) + as Arc + } else { + inner + } + }); + // C-JOURNEY: capture product-auth before `services.host_runtime` is + // moved out below, so `seed_github_credential_account` can create a + // real credential account later (auth-gate happy-path resume). + let product_auth = services.product_auth.clone(); + // E-SKILL: build the local-dev skill context source only when this + // harness surfaces the synthetic `skill_activate` capability (i.e. + // `skill_activation_tools`). Built with the caller-supplied tenant + // (`HostRuntimeHarnessOptions::with_skill_activation_tenant`, sourced + // from the group's actual run-scope tenant) so activation visibility + // matches the turn's scope. Must precede the `services.host_runtime` + // move (it borrows `&services`). + let skill_activation_source = if capability_ids.iter().any(|id| { + id.as_str() == ironclaw_reborn_composition::test_support::SKILL_ACTIVATE_CAPABILITY_ID + }) { + let tenant = skill_activation_tenant + .ok_or("skill_activation_tools harness requires with_skill_activation_tenant")?; + ironclaw_reborn_composition::test_support::build_local_dev_skill_context_source_for_test( + &services, &tenant, true, + ) + } else { + None + }; + let attachment_test_support = services.local_dev_attachment_test_support_for_test(); + // W4-ASK-EACH-ONCE: capture the local-dev per-tool permission override + // store unconditionally (mirrors `auto_approve_settings` above), not just + // for `outbound_target_tools()`'s narrower `Some((facade, ..))` arm below + // -- any host-runtime-backed harness/group can now install a per-capability + // `AskEachTime` override via `set_ask_each_time_override_for_test`. + let tool_permission_overrides = services.local_dev_tool_permission_overrides_for_test(); + // C-SYNTH outbound: pair the injected facade double with the local-dev + // settings stores production's `outbound_delivery_capabilities` consumes, + // captured from `RebornServices` before the `host_runtime` move. Only + // `outbound_target_tools()` supplies the facade. + let outbound_target_tools = match outbound_target_facade { + Some((facade, requires_approval)) => { + let tool_permission_overrides = services + .local_dev_tool_permission_overrides_for_test() + .ok_or("outbound_target_tools requires a local-dev tool-override store")?; + let persistent_approval_policies = services + .local_dev_persistent_approval_policies_for_test() + .ok_or("outbound_target_tools requires a local-dev persistent-policy store")?; + Some(OutboundTargetToolsParts { + facade, + requires_approval, + tool_permission_overrides, + persistent_approval_policies, + }) + } + None => None, + }; + let pending_approval_scopes = Arc::new(Mutex::new(HashMap::new())); + let runtime = services + .host_runtime + .ok_or("local-dev Reborn services missing host runtime")?; + let runtime = Arc::new(RecordingHostRuntime::new( + runtime, + Arc::clone(&pending_approval_scopes), + )); + Ok(Self { + runtime, + approval_parts, + auto_approve_settings, + pending_approval_scopes, + io: Arc::new(ProductLiveCapabilityIo::default()), + root, + workspace_root, + mounts, + capability_mount_overrides: Vec::new(), + capability_ids, + runtime_kind: RuntimeKind::FirstParty, + effect_kinds, + network_policy: NetworkPolicy::default(), + secrets, + provider_id, + additional_provider_trust: Vec::new(), + user_id, + invocations: Arc::new(Mutex::new(Vec::new())), + results: Arc::new(Mutex::new(Vec::new())), + http_egress: None, + network_egress: None, + process_port: None, + profile_filesystem, + project_service, + skill_activation_source, + attachment_test_support, + outbound_target_tools, + scope_capability_by_run_owner: false, + product_auth, + tool_permission_overrides, + }) + } + + pub(crate) fn capability_factory( + self: &Arc, + milestone_sink: Arc, + ) -> Arc { + Arc::new(HostRuntimeHarnessCapabilityPortFactory { + harness: Arc::clone(self), + milestone_sink, + }) + } + + pub(crate) fn capability_result_writer( + self: &Arc, + ) -> Arc { + Arc::new(RecordingCapabilityResultWriter { + inner: self.io.clone(), + results: Arc::clone(&self.results), + }) + } + + fn invocations(&self) -> Vec { + self.invocations.lock().unwrap().clone() + } + + pub(crate) fn capability_results(&self) -> Vec { + self.results.lock().unwrap().clone() + } + + pub(crate) fn runtime_http_requests(&self) -> Vec { + self.http_egress + .as_ref() + .map(|egress| egress.requests()) + .unwrap_or_default() + } + + /// Install FIFO response bodies (C-WEBACCESS) onto the recording runtime + /// HTTP egress, consumed in call order ahead of the default body. Mirrors + /// [`install_http_responses`](Self::install_http_responses)'s shape but for + /// the web-access backend's `push_response_body` FIFO queue rather than the + /// keyed matcher — the three-leg Exa MCP handshake (`initialize` → + /// `notifications/initialized` → `tools/call`) all target the same + /// URL/method/capability, so only the FIFO queue can script them + /// independently. Errors if this harness wired no recording egress. Called + /// from `RebornIntegrationHarnessBuilder::build` (build-time only — no + /// post-build mutation). + pub(crate) fn install_web_access_responses( + &self, + bodies: impl IntoIterator>, + ) -> HarnessResult<()> { + let egress = self + .http_egress + .as_ref() + .ok_or("web-access host runtime has no recording egress wired")?; + for body in bodies { + egress.push_response_body(body); + } + Ok(()) + } + + /// Snapshot of every command string recorded by the inert process port. + /// Empty when the harness uses the live `LocalHostProcessPort` + /// (`.with_live_shell()` path). + pub(crate) fn process_commands(&self) -> Vec { + self.process_port + .as_ref() + .map(|port| port.commands()) + .unwrap_or_default() + } + + /// Install URL/method/capability-keyed scripted responses into the recording + /// HTTP egress (§3.6 P1 ergonomics). Errors if this harness wired no + /// recording egress (e.g. the live-HTTP variant). + pub(crate) fn install_http_responses( + &self, + responses: impl IntoIterator, + ) -> HarnessResult<()> { + self.http_egress + .as_ref() + .ok_or("host runtime harness has no recording http egress to script")? + .install_scripted(responses); + Ok(()) + } + + /// W4-AUTHGATE-WIRE: enqueue a FIFO scripted status on the recording + /// **network** HTTP egress (see `RecordingNetworkHttpEgress::push_status`). + /// For `GithubIssueTools`-backed harnesses, the real WASM HTTP call flows + /// through this lane (not `install_http_responses`'s runtime-egress + /// matcher) — see `reborn_integration_secret_injection.rs`'s module doc. + /// Errors if this harness wired no recording network egress. + pub(crate) fn install_network_status_script(&self, status: u16) -> HarnessResult<()> { + self.network_egress + .as_ref() + .ok_or("host runtime harness has no recording network egress to script")? + .push_status(status); + Ok(()) + } + + /// Install a sticky scripted `builtin.shell` process result on the inert + /// recording process port (mirrors `install_http_responses`). Errors if the + /// harness has no recording port (e.g. the `.with_live_shell()` path). + pub(crate) fn install_process_script( + &self, + result: super::process::ScriptedProcessResult, + ) -> HarnessResult<()> { + self.process_port + .as_ref() + .ok_or("host runtime harness has no recording process port to script")? + .set_scripted(result); + Ok(()) + } + + pub(crate) fn network_http_requests(&self) -> Vec { + self.network_egress + .as_ref() + .map(|egress| egress.requests()) + .unwrap_or_default() + } + + pub(crate) fn workspace_file_path(&self, relative: &str) -> PathBuf { + self.workspace_root.join(relative.trim_start_matches('/')) + } + + pub(crate) async fn approve_local_dev_gate(&self, gate_ref: &GateRef) -> HarnessResult<()> { + let approval_parts = self + .approval_parts + .as_ref() + .ok_or("host runtime harness has no local-dev approval stores")?; + let request_id = approval_request_id_from_gate_ref(gate_ref)?; + let scope = self + .pending_approval_scopes + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .get(&request_id) + .cloned() + .ok_or("approval gate was not recorded by the host runtime harness")?; + let record = approval_parts + .approval_requests + .get(&scope, request_id) + .await? + .ok_or("approval request was not persisted")?; + let capability = match record.request.action.as_ref() { + Action::Dispatch { capability, .. } | Action::SpawnCapability { capability, .. } => { + capability.clone() + } + other => return Err(format!("unsupported approval action: {other:?}").into()), + }; + let approval = self.lease_approval_for(&capability); + let resolver = ApprovalResolver::new( + approval_parts.approval_requests.as_ref(), + approval_parts.capability_leases.as_ref(), + ); + match record.request.action.as_ref() { + Action::Dispatch { .. } => { + resolver + .approve_dispatch(&scope, request_id, approval) + .await?; + } + Action::SpawnCapability { .. } => { + resolver.approve_spawn(&scope, request_id, approval).await?; + } + other => return Err(format!("unsupported approval action: {other:?}").into()), + } + Ok(()) + } + + /// Deny a pending local-dev approval gate (the model-declined path). Mirrors + /// [`approve_local_dev_gate`](Self::approve_local_dev_gate) but resolves the + /// persisted request to `Denied` (no lease issued) via `ApprovalResolver::deny`. + /// The caller then resumes the run with `GateResumeDisposition::Denied` so the + /// executor surfaces a non-retryable authorization failure to the model. + pub(crate) async fn deny_local_dev_gate(&self, gate_ref: &GateRef) -> HarnessResult<()> { + let approval_parts = self + .approval_parts + .as_ref() + .ok_or("host runtime harness has no local-dev approval stores")?; + let request_id = approval_request_id_from_gate_ref(gate_ref)?; + let scope = self + .pending_approval_scopes + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .get(&request_id) + .cloned() + .ok_or("approval gate was not recorded by the host runtime harness")?; + let resolver = ApprovalResolver::new( + approval_parts.approval_requests.as_ref(), + approval_parts.capability_leases.as_ref(), + ); + resolver + .deny( + &scope, + request_id, + DenyApproval { + denied_by: Principal::User(scope.user_id.clone()), + }, + ) + .await?; + Ok(()) + } + + /// The persisted approval-request store, when this harness wires the real + /// local-dev approval stores (`file_tools_requiring_approval`). The + /// integration runtime builds an [`ApprovalGateEvidenceStore`] over it so a + /// `BlockedApproval` run is verified at loop exit (mirrors production + /// `runtime.rs:2799`) and genuinely pauses instead of failing. + pub(crate) fn approval_requests_store( + &self, + ) -> Option> { + self.approval_parts + .as_ref() + .map(|parts| Arc::clone(&parts.approval_requests)) + } + + /// The user id this capability harness's first-party tools execute under. + /// The dispatch-time auto-approve check is keyed `(tenant, user)` on THIS + /// user (not the run's binding owner), so the group derives the auto-approve + /// scope from it — see `GroupSharedStorage::auto_approve_scope`. + pub(crate) fn user_id(&self) -> &UserId { + &self.user_id + } + + /// E-PROFILE: the raw local-dev memory filesystem backing the user-profile + /// source, for write→read-back assertions on `context/profile.json`. `Some` + /// only for `new_with_options`-built harnesses. Consumed by the E-PROFILE + /// `profile_tools()` constructor and the `reborn_integration_profile` test. + pub(crate) fn profile_filesystem_for_test(&self) -> Option> { + self.profile_filesystem.clone() + } + + /// E-PROJ: the project service backing the synthetic `project_create` + /// capability, for write→read-back assertions — mirrors + /// `profile_filesystem_for_test`'s role for E-PROFILE. `Some` only for + /// `project_tools()`-built harnesses. Lets a test read a created project + /// back through the SAME `Arc` instance + /// `apply_synthetic_capability_wrappers` dispatches writes through, rather + /// than reconstructing an equivalent (and possibly unwritten) one. + pub(crate) fn project_service_for_test(&self) -> Option> { + self.project_service.clone() + } + + /// C-SYNTH outbound: the injected facade double, for read-back that a + /// `target_set` actually reached the facade seam + /// (`recorded_set_target_ids`). `Some` only for `outbound_target_tools()`. + pub(crate) fn outbound_preferences_facade_for_test( + &self, + ) -> Option> { + self.outbound_target_tools + .as_ref() + .map(|parts| Arc::clone(&parts.facade)) + } + + /// C-SYNTH outbound: persist a `Disabled` per-tool permission override for + /// `outbound_delivery_target_set` under `(tenant, user)`, driving the + /// handler's settings decision to `Deny` → `Failed{policy_denied}`. The + /// scope must be the run's EFFECTIVE dispatch user (the thread binding actor, + /// `harness.binding.actor_user_id`) — the same `(tenant, user)` + /// `StoreApprovalSettingsProvider::tool_override` reads it back under + /// (`PersistentApprovalScope` = tenant+user, invocation-independent). `Some` + /// only for `outbound_target_tools()`. + pub(crate) async fn disable_outbound_target_set_tool( + &self, + tenant_id: TenantId, + user_id: UserId, + ) -> HarnessResult<()> { + let parts = self + .outbound_target_tools + .as_ref() + .ok_or("harness has no outbound_target_tools backing store")?; + let scope = ResourceScope { + tenant_id, + user_id: user_id.clone(), + agent_id: None, + project_id: None, + mission_id: None, + thread_id: None, + invocation_id: InvocationId::new(), + }; + parts + .tool_permission_overrides + .set(ironclaw_approvals::CapabilityPermissionOverrideInput { + scope, + capability_id: CapabilityId::new( + ironclaw_reborn_composition::test_support::OUTBOUND_DELIVERY_TARGET_SET_CAPABILITY_ID, + )?, + state: ironclaw_approvals::CapabilityPermissionOverride::Disabled, + updated_by: Principal::User(user_id), + }) + .await?; + Ok(()) + } + + /// W4-ASK-EACH-ONCE: install a `ToolPermissionOverride::AskEachTime` + /// override for `capability_id` under `(tenant_id, user_id)` via the real + /// local-dev per-tool permission override store — generalizes + /// `disable_outbound_target_set_tool`'s shape to any host-runtime-backed + /// harness/group. Errors if this harness wired no such store (i.e. not + /// built via `new_with_options`). + pub(crate) async fn set_ask_each_time_override_for_test( + &self, + capability_id: &CapabilityId, + tenant_id: TenantId, + user_id: UserId, + ) -> HarnessResult<()> { + let store = self + .tool_permission_overrides + .as_ref() + .ok_or("harness has no local-dev tool-permission-override store")?; + let scope = ResourceScope { + tenant_id, + user_id: user_id.clone(), + agent_id: None, + project_id: None, + mission_id: None, + thread_id: None, + invocation_id: InvocationId::new(), + }; + store + .set(ironclaw_approvals::CapabilityPermissionOverrideInput { + scope, + capability_id: capability_id.clone(), + state: ironclaw_approvals::CapabilityPermissionOverride::AskEachTime, + updated_by: Principal::User(user_id), + }) + .await?; + Ok(()) + } + + /// E-SKILL: the `HostSkillContextSource` to wire as the runtime's + /// `skill_context_source` in `into_group`, so activated-skill instructions + /// inject into the model request. `Some` only for `skill_activation_tools()`. + pub(crate) fn skill_context_source_for_test( + &self, + ) -> Option> { + self.skill_activation_source + .as_ref() + .map(|source| source.context_source()) + } + + /// E-DURABLE: the on-disk local-dev storage root this harness's capability + /// stores persist under (`/local-dev`). Mirrors the `storage_root` + /// computed inline in `new_with_options`. A durability test reopens a fresh, + /// independent store at this path (see + /// `open_local_dev_extension_installation_store_for_test`) to prove capability + /// state survives a reopen, paralleling `assert_reply_persists_after_reopen`. + /// Tests only. + pub(crate) fn storage_root_for_test(&self) -> PathBuf { + self.root.path().join("local-dev") + } + + /// C-DURABLE: resolve `gate_ref` (a `"gate:approval-"` local-dev + /// approval gate) to the `(ApprovalRequestId, ResourceScope)` pair a fresh, + /// independently-reopened `ApprovalRequestStore::get`/`read_versioned` call + /// needs. Reuses the SAME private lookup `approve_local_dev_gate`/ + /// `deny_local_dev_gate` already use (`approval_request_id_from_gate_ref` + + /// `pending_approval_scopes`) so a durability test's scope construction can + /// never drift from the live approve/deny path. Tests only. + pub(crate) fn approval_request_scope_for_test( + &self, + gate_ref: &GateRef, + ) -> HarnessResult<(ApprovalRequestId, ResourceScope)> { + let request_id = approval_request_id_from_gate_ref(gate_ref)?; + let scope = self + .pending_approval_scopes + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .get(&request_id) + .cloned() + .ok_or("approval gate was not recorded by the host runtime harness")?; + Ok((request_id, scope)) + } + + /// E-SKILL: seed a system-scoped skill on this harness's on-disk skill + /// filesystem so the model can activate it (`skill_activate`/`$name`). Writes + /// `/system/skills//SKILL.md` — the system bundle root is + /// always present in the skills extension's roots regardless of the run's + /// tenant/user (`FirstPartySkillsExtensionHandles::bundle_roots`), so both + /// `activate_skills_for_run` (the `skill_activate` capability) and the + /// runtime's `skill_context_source` resolve it deterministically without + /// depending on the harness's run-scope owner resolution. User-scoped skill + /// filesystem resolution is already covered by the runtime.rs suite; this + /// seam only needs the skill to exist so the capability + context wiring can + /// be driven. Mirrors the runtime-test system-skill layout + /// (runtime.rs `system/skills//SKILL.md`). Tests only. + pub(crate) fn seed_system_skill_for_test( + &self, + name: &str, + description: &str, + prompt: &str, + ) -> HarnessResult<()> { + let dir = self + .storage_root_for_test() + .join("system") + .join("skills") + .join(name); + std::fs::create_dir_all(&dir)?; + let body = format!( + "---\nname: {name}\ndescription: {description}\nactivation:\n keywords: [\"{name}\"]\n---\n\n{prompt}" + ); + std::fs::write(dir.join("SKILL.md"), body)?; + Ok(()) + } + + /// C-SYNTH `skill_activate` `AmbiguousSkill` seeding arm: seed a + /// USER-scoped skill (writes + /// `/tenants//users//skills//SKILL.md`) + /// so a name shared with a system-scoped skill + /// (`seed_system_skill_for_test`) resolves to TWO Trusted candidates + /// (`System` and `User` roots both default `Trusted`), triggering + /// `SkillActivationSelectionError::AmbiguousSkill`. `tenant`/`user` must + /// match the driving thread's run scope (`harness.binding`), or the user + /// root never matches the run's own `/skills` mount. Tests only. + pub(crate) fn seed_user_skill_for_test( + &self, + tenant: &TenantId, + user: &UserId, + name: &str, + description: &str, + prompt: &str, + ) -> HarnessResult<()> { + let dir = self + .storage_root_for_test() + .join("tenants") + .join(tenant.as_str()) + .join("users") + .join(user.as_str()) + .join("skills") + .join(name); + std::fs::create_dir_all(&dir)?; + let body = format!( + "---\nname: {name}\ndescription: {description}\nactivation:\n keywords: [\"{name}\"]\n---\n\n{prompt}" + ); + std::fs::write(dir.join("SKILL.md"), body)?; + Ok(()) + } + + /// C-ATTACH: the attachment read port + inbound lander over this harness's + /// local-dev workspace filesystem, for wiring `DefaultPlannedRuntimeParts.attachment_read_port` + /// and `DefaultInboundTurnService::with_inbound_attachments` — mirrors + /// `profile_filesystem_for_test`'s role for E-PROFILE. + pub(crate) fn attachment_test_support_for_test( + &self, + ) -> Option { + self.attachment_test_support.clone() + } + + /// E-PROJ: wrap `port` with the local-dev synthetic capabilities this harness + /// surfaces, in one linear step (keeps the capability-specific knowledge out + /// of `create_capability_port`'s main assembly chain). + /// + /// Partial synthetic wrap: `project_create` (E-PROJ), `skill_activate` + /// (E-SKILL), and the two `outbound_delivery_*` capabilities (C-SYNTH + /// outbound), each layered independently when this harness holds the backing + /// handle, so other local-dev groups are unaffected. See + /// `LocalDevCapabilityPortFactory::build_inner()` for the full production set. + fn apply_synthetic_capability_wrappers( + &self, + port: Arc, + run_context: &LoopRunContext, + input_resolver: Arc, + result_writer: Arc, + ) -> Result, AgentLoopHostError> { + let mut port = port; + // project_create (E-PROJ): wrapped only for `project_tools`. + if let Some(project_service) = &self.project_service + && self.capability_ids.iter().any(|id| { + id.as_str() + == ironclaw_reborn_composition::test_support::PROJECT_CREATE_CAPABILITY_ID + }) + { + port = + ironclaw_reborn_composition::test_support::wrap_project_create_capability_for_test( + port, + Arc::clone(project_service), + self.user_id.clone(), + run_context.clone(), + input_resolver.clone(), + result_writer.clone(), + )?; + } + // outbound_delivery_* (C-SYNTH outbound): wrapped only for + // `outbound_target_tools`. The facade double is injected at the + // production-wired trait seam; the settings/approval stores are the same + // ones `outbound_delivery_capabilities` consumes in production (the + // auto-approve + approval-request/lease stores are reused from the + // sibling `auto_approve_settings` / `approval_parts` harness fields). + if let Some(parts) = &self.outbound_target_tools { + let approval = self.approval_parts.as_ref().ok_or_else(|| { + AgentLoopHostError::new( + AgentLoopHostErrorKind::Internal, + "outbound_target_tools requires local-dev approval stores", + ) + })?; + let auto_approve = self.auto_approve_settings.clone().ok_or_else(|| { + AgentLoopHostError::new( + AgentLoopHostErrorKind::Internal, + "outbound_target_tools requires the local-dev auto-approve store", + ) + })?; + // Record the synthetic capability's approval scope into + // `pending_approval_scopes` (the host-runtime recorder can't see a + // port-level synthetic gate) so `approve_local_dev_gate` / + // `deny_local_dev_gate` resolve it; delegates to the inner store the + // evidence/approve/deny paths read. + let recording_approval_requests: Arc = + Arc::new(RecordingApprovalRequestStore { + inner: Arc::clone(&approval.approval_requests), + pending_approval_scopes: Arc::clone(&self.pending_approval_scopes), + }); + port = + ironclaw_reborn_composition::test_support::wrap_outbound_delivery_capabilities_for_test( + port, + ironclaw_reborn_composition::test_support::OutboundDeliveryCapabilityTestParts { + facade: Arc::clone(&parts.facade) + as Arc, + fallback_user_id: self.user_id.clone(), + approval_requests: recording_approval_requests, + capability_leases: Arc::clone(&approval.capability_leases), + tool_permission_overrides: Arc::clone(&parts.tool_permission_overrides), + auto_approve, + persistent_policies: Arc::clone(&parts.persistent_approval_policies), + target_set_requires_approval: parts.requires_approval, + run_context: run_context.clone(), + input_resolver: input_resolver.clone(), + result_writer: result_writer.clone(), + }, + )?; + } + // skill_activate (E-SKILL): wrapped only for `skill_activation_tools`. + if let Some(skill_source) = &self.skill_activation_source + && self.capability_ids.iter().any(|id| { + id.as_str() + == ironclaw_reborn_composition::test_support::SKILL_ACTIVATE_CAPABILITY_ID + }) + { + port = + ironclaw_reborn_composition::test_support::wrap_skill_activation_capability_for_test( + port, + skill_source, + run_context.clone(), + input_resolver, + result_writer, + )?; + } + Ok(port) + } + + /// Assembles the recording `LoopCapabilityPort` for one run: builds the + /// authority/grant/visible-capability-request chain over this harness's + /// fields, wraps it in the real `HostRuntimeLoopCapabilityPortFactory`, + /// layers the synthetic capability wrappers, and wraps the result in the + /// invocation-recording port. Owned here (rather than in the + /// `HostRuntimeHarnessCapabilityPortFactory` test double) because the + /// assembly reads this harness's fields directly — see + /// `tests/integration/support/doubles/host_runtime_harness_capability_port_factory.rs` + /// for the thin `LoopCapabilityPortFactory` delegating wrapper that calls + /// this method. + pub(crate) async fn create_recording_capability_port( + &self, + run_context: &LoopRunContext, + milestone_sink: &Arc, + ) -> Result, AgentLoopHostError> { + // C-MULTIUSER: resolve the execution user per run (owner/actor) when the + // harness opts in, else the fixed harness user. Both the authority scope + // and the grant grantee MUST use the SAME user so the lease is + // self-consistent (grantee == execution user) — matching production. + let dispatch_user = self.dispatch_user_for_run(run_context); + let mut authority = ProductLiveVisibleCapabilityRequestConfig::new( + dispatch_user.clone(), + self.runtime_kind, + TrustClass::FirstParty, + SurfaceKind::new("agent_loop").map_err(host_runtime_harness_error)?, + CapabilitySurfacePolicy::allow_all(), + ) + .with_mounts(self.mounts.clone()) + .with_grants(capability_grants( + Principal::User(dispatch_user.clone()), + &self.capability_ids, + self.effect_kinds.clone(), + self.mounts.clone(), + &self.capability_mount_overrides, + self.network_policy.clone(), + self.secrets.clone(), + )) + .with_provider_trust_for_effects( + self.provider_id.clone(), + EffectiveTrustClass::user_trusted(), + self.effect_kinds.clone(), + ); + for (provider, effects) in &self.additional_provider_trust { + authority = authority.with_provider_trust_for_effects( + provider.clone(), + EffectiveTrustClass::user_trusted(), + effects.clone(), + ); + } + let execution_mounts = self.mounts.clone(); + let visible_request = visible_capability_request_for_run(run_context, authority) + .map_err(host_runtime_harness_error)?; + let milestone_sink: Arc = milestone_sink.clone(); + let result_writer = Arc::new(RecordingCapabilityResultWriter { + inner: self.io.clone(), + results: Arc::clone(&self.results), + }); + let mut factory = HostRuntimeLoopCapabilityPortFactory::new( + Arc::clone(&self.runtime), + visible_request, + self.io.clone(), + result_writer.clone(), + milestone_sink, + ) + .with_execution_mounts(execution_mounts); + for (capability_id, mounts) in &self.capability_mount_overrides { + factory = + factory.with_capability_execution_mount(capability_id.clone(), mounts.clone()); + } + let port = factory.for_run_context(run_context.clone()); + // E-PROJ: see `apply_synthetic_capability_wrappers`'s doc comment. + let port = self.apply_synthetic_capability_wrappers( + port, + run_context, + self.io.clone(), + result_writer, + )?; + Ok(Arc::new(RecordingDelegatingCapabilityPort { + inner: port, + invocations: Arc::clone(&self.invocations), + })) + } + + /// Override the user this capability harness executes first-party tools + /// under. Dispatch scope, approval persistence, auto-approve keying, and + /// gate-evidence lookup are ALL keyed on this user. The integration harness + /// sets it to the run's binding owner so dispatch and the turn share one + /// `(tenant, user)`, matching production — without this, a + /// `BlockedApproval` gate's evidence lookup uses the turn owner while the + /// request persists under the capability user, and the run never verifies. + pub(crate) fn with_user_id(mut self, user_id: UserId) -> Self { + self.user_id = user_id; + self + } + + /// C-MULTIUSER: opt in to per-actor capability scoping. With this set, + /// [`create_capability_port`] resolves the execution `(tenant, user)` from + /// each run's OWN owner/actor rather than this harness's single fixed + /// `user_id` — so N actors sharing one capability backend dispatch under N + /// distinct scopes. See [`scope_capability_by_run_owner`] and + /// [`dispatch_user_for_run`]. Enabled only by the multiuser group + /// constructors; every other harness keeps the legacy fixed-user behavior. + pub(crate) fn with_run_owner_scoped_capability_dispatch(mut self) -> Self { + self.scope_capability_by_run_owner = true; + self + } + + /// The capability-execution `UserId` for one run. Mirrors production + /// `local_dev_visible_capability_request`'s owner→actor→fallback resolution + /// (`runtime/local_dev.rs`): when [`scope_capability_by_run_owner`] is set, + /// prefer the run scope's explicit owner, then the run actor, then fall back + /// to the fixed harness `user_id`. Without the flag, always the fixed + /// `user_id` (legacy behavior — every existing test unaffected). + fn dispatch_user_for_run(&self, run_context: &LoopRunContext) -> UserId { + if self.scope_capability_by_run_owner { + run_context + .scope + .explicit_owner_user_id() + .cloned() + .or_else(|| run_context.actor().map(|actor| actor.user_id.clone())) + .unwrap_or_else(|| self.user_id.clone()) + } else { + self.user_id.clone() + } + } + + fn lease_approval_for(&self, capability_id: &CapabilityId) -> LeaseApproval { + let mounts = self + .capability_mount_overrides + .iter() + .find(|(override_capability, _)| override_capability == capability_id) + .map(|(_, mounts)| mounts.clone()) + .unwrap_or_else(|| self.mounts.clone()); + LeaseApproval { + issued_by: Principal::HostRuntime, + constraints: GrantConstraints { + allowed_effects: self.effect_kinds.clone(), + mounts, + network: self.network_policy.clone(), + secrets: self.secrets.clone(), + resource_ceiling: None, + expires_at: None, + max_invocations: Some(1), + }, + } + } +} + +fn capability_grants( + grantee: Principal, + capabilities: &[CapabilityId], + allowed_effects: Vec, + mounts: MountView, + mount_overrides: &[(CapabilityId, MountView)], + network: NetworkPolicy, + secrets: Vec, +) -> CapabilitySet { + CapabilitySet { + grants: capabilities + .iter() + .map(|capability| { + let mounts = mount_overrides + .iter() + .find(|(override_capability, _mounts)| override_capability == capability) + .map(|(_capability, mounts)| mounts.clone()) + .unwrap_or_else(|| mounts.clone()); + CapabilityGrant { + id: CapabilityGrantId::new(), + capability: capability.clone(), + grantee: grantee.clone(), + issued_by: Principal::HostRuntime, + constraints: GrantConstraints { + allowed_effects: allowed_effects.clone(), + mounts, + network: network.clone(), + secrets: secrets.clone(), + resource_ceiling: None, + expires_at: None, + max_invocations: None, + }, + } + }) + .collect(), + } +} + +fn host_runtime_harness_error(error: impl std::fmt::Display) -> AgentLoopHostError { + AgentLoopHostError::new(AgentLoopHostErrorKind::InvalidInvocation, error.to_string()) +} + +fn approval_request_id_from_gate_ref(gate_ref: &GateRef) -> HarnessResult { + const APPROVAL_GATE_PREFIX: &str = "gate:approval-"; + let value = gate_ref + .as_str() + .strip_prefix(APPROVAL_GATE_PREFIX) + .ok_or("gate ref is not a local-dev approval gate")?; + Ok(ApprovalRequestId::parse(value)?) +} + +pub(crate) fn product_scope() -> ResourceScope { + test_product_scope("tenant-e2e", "host-user", "agent-e2e", Some("project-e2e")) +} + +pub fn test_product_scope( + tenant_id: &str, + host_user_id: &str, + agent_id: &str, + project_id: Option<&str>, +) -> ResourceScope { + resource_scope( + TenantId::new(tenant_id).expect("valid tenant"), + UserId::new(host_user_id).expect("valid user"), + AgentId::new(agent_id).expect("valid agent"), + project_id.map(|id| ProjectId::new(id).expect("valid project")), + ) +} + +pub(crate) fn scoped_turns_fs( + backend: Arc, + binding: &ResolvedBinding, +) -> HarnessResult>> { + // Include agent_id and project_id in the path when present so that + // distinct agents or projects stored under the same tenant/user + // (e.g. shared-storage multi-harness tests) get isolated turn state + // files and cannot cross-claim each other's queued runs. + // The 4-arm match lives in `super::filesystem::turns_scope_path`; the + // integration tier reuses it with a different prefix via + // `scoped_turns_fs_composite` in builder.rs. + let target = super::filesystem::turns_scope_path("/engine", binding); + let mounts = MountView::new(vec![MountGrant::new( + MountAlias::new("/turns").expect("valid turns alias"), + VirtualPath::new(target).expect("valid turns target"), + MountPermissions::read_write_list_delete(), + )])?; + Ok(Arc::new(ScopedFilesystem::with_fixed_view( + turn_state_root_filesystem(backend)?, + mounts, + ))) +} + +fn turn_state_root_filesystem( + backend: Arc, +) -> HarnessResult> { + let mut root = CompositeRootFilesystem::new(); + root.mount( + local_dev_mount_descriptor( + "/engine", + "reborn-harness-turn-state", + BackendKind::MemoryDocuments, + StorageClass::StructuredRecords, + ContentKind::StructuredRecord, + IndexPolicy::NotIndexed, + backend.capabilities(), + )?, + backend, + )?; + Ok(Arc::new(root)) +} diff --git a/tests/integration/support/harness/options.rs b/tests/integration/support/harness/options.rs new file mode 100644 index 00000000000..61661fecc7e --- /dev/null +++ b/tests/integration/support/harness/options.rs @@ -0,0 +1,295 @@ +use std::path::PathBuf; +use std::sync::Arc; + +use ironclaw_extensions::ExtensionPackage; +use ironclaw_host_api::{ + CapabilityId, EffectKind, ExtensionId, MountView, NetworkPolicy, SecretHandle, TenantId, UserId, +}; +use ironclaw_host_runtime::BUILTIN_FIRST_PARTY_PROVIDER; +use ironclaw_network::NetworkHttpEgress; + +use super::{HarnessResult, HostRuntimeCapabilityHarness}; + +#[derive(Default)] +pub(crate) struct HostRuntimeHarnessOptions { + pub(crate) mounts: MountView, + pub(crate) runtime_policy: Option, + pub(crate) seed_extension_credentials: bool, + /// Tenant the E-SKILL skill context source is constructed under, when this + /// harness surfaces the synthetic `skill_activate` capability. Only + /// `skill_activation_tools()` sets this (via + /// `with_skill_activation_tenant`), passing the SAME tenant the caller's + /// group run scope resolved (`group.rs` `build_base`'s + /// `canonical_binding.tenant_id`) — never a separately hardcoded literal — + /// so `skill_activate` resolves the seeded user skill against the same + /// tenant the turn runs under. `None` for every other harness variant. + pub(crate) skill_activation_tenant: Option, + /// Injected outbound-delivery facade double + `target_set` approval flag, + /// when this harness surfaces the synthetic `outbound_delivery_*` + /// capabilities (C-SYNTH outbound seam). Only `outbound_target_tools()` sets + /// this. `new_with_options` pairs the facade with the local-dev settings + /// stores captured from `RebornServices` to build `OutboundTargetToolsParts`. + pub(crate) outbound_target_facade: Option<( + Arc, + bool, + )>, + /// C-JOURNEY: override the local-dev host network HTTP egress + /// (`RebornBuildInput::with_network_http_egress_for_test`). Without this, + /// `build_local_runtime` defaults to a REAL `ReqwestNetworkTransport` + /// (`factory.rs`), so any harness dispatching a bundled WASM capability + /// that crosses HTTP (e.g. `github.*`) on the `new_with_options` path MUST + /// set this to stay hermetic. `None` for every harness that surfaces no + /// such capability. + pub(crate) network_http_egress_for_test: Option>, + /// C-JOURNEY: bundled first-party WASM packages (e.g. github) to publish + /// directly into the local-dev active-extension registry at construction + /// time, via `RebornServices::publish_bundled_extension_for_test` + /// (reaches the SAME `ActiveExtensionPublisher::publish` step + /// `builtin.extension_activate` calls). Without this, a bundled package's + /// capabilities are granted/trusted at the harness-authority layer + /// (`capability_ids`/`additional_provider_trust`) but NOT present in the + /// runtime's own dispatchable registry, so dispatch silently no-ops (the + /// tool call never reaches `invoke_capability`). Empty for every harness + /// that surfaces no bundled WASM capability. + pub(crate) activate_bundled_extensions_for_test: Vec, + /// C-SYNTH `project_create` fault-injection seam: wrap the real + /// `Arc` (`services.local_dev_project_service_for_test()`) + /// in `FaultInjectingProjectService` before it reaches + /// `wrap_project_create_capability_for_test`, so a `create_project` call + /// naming `FAULT_INJECT_DENIED_PROJECT_NAME` returns + /// `ProjectServiceError::Denied` instead of reaching the real store. + /// Only `project_tools_with_fault_injection()` sets this; every other + /// harness leaves the real service unwrapped. + pub(crate) project_service_fault_injection: bool, +} + +impl HostRuntimeHarnessOptions { + pub(crate) fn new( + mounts: MountView, + runtime_policy: Option, + ) -> Self { + Self { + mounts, + runtime_policy, + seed_extension_credentials: false, + skill_activation_tenant: None, + outbound_target_facade: None, + network_http_egress_for_test: None, + activate_bundled_extensions_for_test: Vec::new(), + project_service_fault_injection: false, + } + } + + pub(crate) fn with_seed_extension_credentials(mut self) -> Self { + self.seed_extension_credentials = true; + self + } + + pub(crate) fn with_skill_activation_tenant(mut self, tenant: TenantId) -> Self { + self.skill_activation_tenant = Some(tenant); + self + } + + pub(crate) fn with_outbound_target_tools( + mut self, + facade: Arc, + target_set_requires_approval: bool, + ) -> Self { + self.outbound_target_facade = Some((facade, target_set_requires_approval)); + self + } + + pub(crate) fn with_network_http_egress_for_test( + mut self, + egress: Arc, + ) -> Self { + self.network_http_egress_for_test = Some(egress); + self + } + + pub(crate) fn with_activated_bundled_extension(mut self, package: ExtensionPackage) -> Self { + self.activate_bundled_extensions_for_test.push(package); + self + } + + pub(crate) fn with_project_service_fault_injection(mut self) -> Self { + self.project_service_fault_injection = true; + self + } +} + +/// Typed capture of a `HostRuntimeCapabilityHarness::new_with_options(..)` call +/// shape plus the post-construct steps a domain constructor applies to the +/// built harness. Each `harness/profiles/.rs` module constructs one +/// `ToolsProfile` and calls `.build()` instead of an inline +/// `new_with_options(..)` + ad-hoc post-construct mutation. +/// +/// Mirrors `new_with_options`'s parameter list plus four ADDITIONAL fields +/// capturing post-construct steps (`network_policy_override`, +/// `provider_trust_override`, `post_construct_asset_copy`, +/// `auto_approve_default`) — see `build()`'s doc comment for the fixed +/// application order. +pub(crate) struct ToolsProfile { + pub(crate) service_label: &'static str, + pub(crate) capability_ids: Vec, + pub(crate) effect_kinds: Vec, + pub(crate) secrets: Vec, + pub(crate) provider_id: ExtensionId, + pub(crate) user_id: UserId, + pub(crate) options: HostRuntimeHarnessOptions, + /// Mirrors constructors that overwrite `harness.network_policy` after + /// `new_with_options` returns (e.g. `extension_lifecycle_tools`, + /// `skill_management_tools`, `trace_commons_tools`). `None` leaves the + /// harness's default `NetworkPolicy::default()`. + pub(crate) network_policy_override: Option, + /// Mirrors constructors that overwrite `harness.additional_provider_trust` + /// (e.g. `extension_lifecycle_tools`, `file_and_github_auth_tools`). + /// `None` leaves the harness's default empty trust list. + pub(crate) provider_trust_override: Option)>>, + /// Mirrors `file_and_github_auth_tools`'s post-construct + /// `copy_dir_recursive(&github_support::asset_root(), &harness.root.path().join(..))` + /// step: `(source_dir, relative_dest_under_harness_root)`. The destination + /// is captured as a path RELATIVE to the harness's tempdir root because the + /// root itself is created inside `new_with_options` and does not exist yet + /// when a `ToolsProfile` is assembled. Plain data (no closure) — the copy + /// is a fixed filesystem operation, not caller-specific logic. + pub(crate) post_construct_asset_copy: Option<(PathBuf, PathBuf)>, + /// Mirrors the `enable_global_auto_approve_for_product_and_harness_users` / + /// `disable_global_auto_approve_for_product_and_harness_users` post-construct + /// calls: `Some(true)` enables, `Some(false)` disables, `None` touches + /// nothing (e.g. `attachment_tools`, `write_only`). + pub(crate) auto_approve_default: Option, +} + +impl ToolsProfile { + /// Neutral baseline: empty capability/effect/secret lists, the universal + /// `BUILTIN_FIRST_PARTY_PROVIDER` provider id (every existing + /// `new_with_options`-based constructor passes this same value — the one + /// "every caller agrees" exception to the empty/None/false default rule), + /// default (empty) harness options, and no post-construct steps. + /// + /// `user_id` is explicit (no placeholder default): every profile has a + /// fixed domain-specific user id, and a silently-valid fallback would let + /// a forgotten override build a harness under the wrong user. + pub(crate) fn new(service_label: &'static str, user_id: &str) -> HarnessResult { + Ok(Self { + service_label, + capability_ids: Vec::new(), + effect_kinds: Vec::new(), + secrets: Vec::new(), + provider_id: ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, + user_id: UserId::new(user_id)?, + options: HostRuntimeHarnessOptions::default(), + network_policy_override: None, + provider_trust_override: None, + post_construct_asset_copy: None, + auto_approve_default: None, + }) + } + + pub(crate) fn with_capability_ids(mut self, capability_ids: Vec) -> Self { + self.capability_ids = capability_ids; + self + } + + pub(crate) fn with_effect_kinds(mut self, effect_kinds: Vec) -> Self { + self.effect_kinds = effect_kinds; + self + } + + pub(crate) fn with_secrets(mut self, secrets: Vec) -> Self { + self.secrets = secrets; + self + } + + pub(crate) fn with_provider_id(mut self, provider_id: ExtensionId) -> Self { + self.provider_id = provider_id; + self + } + + pub(crate) fn with_user_id(mut self, user_id: UserId) -> Self { + self.user_id = user_id; + self + } + + pub(crate) fn with_options(mut self, options: HostRuntimeHarnessOptions) -> Self { + self.options = options; + self + } + + pub(crate) fn with_network_policy_override(mut self, policy: NetworkPolicy) -> Self { + self.network_policy_override = Some(policy); + self + } + + pub(crate) fn with_provider_trust_override( + mut self, + trust: Vec<(ExtensionId, Vec)>, + ) -> Self { + self.provider_trust_override = Some(trust); + self + } + + pub(crate) fn with_post_construct_asset_copy( + mut self, + source_dir: PathBuf, + relative_dest_under_harness_root: PathBuf, + ) -> Self { + self.post_construct_asset_copy = Some((source_dir, relative_dest_under_harness_root)); + self + } + + pub(crate) fn with_auto_approve_default(mut self, enabled: bool) -> Self { + self.auto_approve_default = Some(enabled); + self + } + + /// THE one shared construction path domain profiles build on: calls + /// `HostRuntimeCapabilityHarness::new_with_options(..)` with this profile's + /// core fields, then applies the captured post-construct steps in the SAME + /// fixed order every existing multi-step constructor applies them (verified + /// against `extension_lifecycle_tools` and `file_and_github_auth_tools`, + /// the only two constructors that combine more than one post-construct + /// step): + /// + /// 1. `network_policy_override` (if set) + /// 2. `provider_trust_override` (if set) + /// 3. `post_construct_asset_copy` (if set) + /// 4. `auto_approve_default` (enable/disable/neither) + pub(crate) async fn build(self) -> HarnessResult { + let mut harness = HostRuntimeCapabilityHarness::new_with_options( + self.service_label, + self.capability_ids, + self.effect_kinds, + self.secrets, + self.provider_id, + self.user_id, + self.options, + ) + .await?; + if let Some(policy) = self.network_policy_override { + harness.network_policy = policy; + } + if let Some(trust) = self.provider_trust_override { + harness.additional_provider_trust = trust; + } + if let Some((source_dir, relative_dest)) = self.post_construct_asset_copy { + let dest = harness.root.path().join(relative_dest); + super::copy_dir_recursive(&source_dir, &dest)?; + } + match self.auto_approve_default { + Some(true) => { + harness + .enable_global_auto_approve_for_product_and_harness_users() + .await?; + } + Some(false) => { + harness + .disable_global_auto_approve_for_product_and_harness_users() + .await?; + } + None => {} + } + Ok(harness) + } +} diff --git a/tests/integration/support/harness/profiles/attachment.rs b/tests/integration/support/harness/profiles/attachment.rs new file mode 100644 index 00000000000..5ff88dc5654 --- /dev/null +++ b/tests/integration/support/harness/profiles/attachment.rs @@ -0,0 +1,36 @@ +//! Attachment domain tools profile (`attachment_tools`). + +use ironclaw_host_api::{EffectKind, MountView}; + +use super::super::options::{HostRuntimeHarnessOptions, ToolsProfile}; +use super::super::{HarnessResult, HostRuntimeCapabilityHarness}; + +/// Group with NO first-party capability dispatch — the test drives the +/// C-ATTACH seam purely through the attachment read port + inbound lander, +/// never a tool call. Uses `new_with_options` (mirrors `profile_tools()`), +/// so `attachment_test_support` is populated from +/// `services.local_dev_attachment_test_support_for_test()`. No mounts needed: +/// attachment landing/reading goes through `local_runtime.workspace_filesystem` +/// directly, not the capability-dispatch `MountView` (mirrors +/// `trigger_management_tools()`'s `MountView::default()`, which also has no +/// filesystem capability to gate). +pub(crate) fn attachment_tools_profile() -> HarnessResult { + Ok(ToolsProfile { + effect_kinds: vec![EffectKind::ReadFilesystem, EffectKind::WriteFilesystem], + options: HostRuntimeHarnessOptions::new( + MountView::default(), + Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( + true, + )?), + ), + ..ToolsProfile::new( + "reborn-e2e-attachment-tools", + "reborn-e2e-attachment-tools-user", + )? + }) +} + +/// See [`attachment_tools_profile`]. +pub(crate) async fn attachment_tools() -> HarnessResult { + attachment_tools_profile()?.build().await +} diff --git a/tests/integration/support/harness/profiles/coding_read.rs b/tests/integration/support/harness/profiles/coding_read.rs new file mode 100644 index 00000000000..c297fb27120 --- /dev/null +++ b/tests/integration/support/harness/profiles/coding_read.rs @@ -0,0 +1,36 @@ +//! Coding-read domain tools profile (`coding_read_tools`) — reference example +//! of the `ToolsProfile` pattern (see `harness/options.rs`). + +use ironclaw_host_api::{CapabilityId, EffectKind, MountPermissions}; +use ironclaw_host_runtime::{GLOB_CAPABILITY_ID, GREP_CAPABILITY_ID, LIST_DIR_CAPABILITY_ID}; + +use super::super::options::{HostRuntimeHarnessOptions, ToolsProfile}; +use super::super::{HarnessResult, HostRuntimeCapabilityHarness, workspace_mounts}; + +/// Read-only coding tools (`list_dir`/`glob`/`grep`). Auto-approve is enabled +/// for the product and harness users so the model-visible surface dispatches +/// without a gate. +pub(crate) fn coding_read_tools_profile() -> HarnessResult { + Ok(ToolsProfile { + capability_ids: vec![ + CapabilityId::new(LIST_DIR_CAPABILITY_ID)?, + CapabilityId::new(GLOB_CAPABILITY_ID)?, + CapabilityId::new(GREP_CAPABILITY_ID)?, + ], + effect_kinds: vec![EffectKind::ReadFilesystem], + options: HostRuntimeHarnessOptions::new( + workspace_mounts(MountPermissions::read_write_list_delete())?, + None, + ), + auto_approve_default: Some(true), + ..ToolsProfile::new( + "reborn-e2e-coding-read-tools", + "reborn-e2e-coding-read-user", + )? + }) +} + +/// See [`coding_read_tools_profile`]. +pub(crate) async fn coding_read_tools() -> HarnessResult { + coding_read_tools_profile()?.build().await +} diff --git a/tests/integration/support/harness/profiles/core_builtin.rs b/tests/integration/support/harness/profiles/core_builtin.rs new file mode 100644 index 00000000000..f95c9271fde --- /dev/null +++ b/tests/integration/support/harness/profiles/core_builtin.rs @@ -0,0 +1,225 @@ +//! core_builtin domain tools profile (`core_builtin_tools`). +//! +//! Unlike the other `profiles/*` domains, this harness does NOT flow through +//! `new_with_options`/`RebornServices` — it builds the `HostRuntime` directly +//! via `local_dev_host_runtime_with_http_egress` / +//! `local_dev_host_runtime_with_live_http_egress` and assembles +//! `HostRuntimeCapabilityHarness` by hand (`core_builtin_tools_from_runtime`), +//! so it does not go through `ToolsProfile`/`.build()`. + +use std::collections::HashMap; +use std::path::PathBuf; +use std::sync::{Arc, Mutex}; + +use ironclaw_host_api::{ + CapabilityId, EffectKind, ExtensionId, MountPermissions, NetworkPolicy, RuntimeKind, UserId, +}; +use ironclaw_host_runtime::{ + APPLY_PATCH_CAPABILITY_ID, BUILTIN_FIRST_PARTY_PROVIDER, HTTP_CAPABILITY_ID, + HTTP_SAVE_CAPABILITY_ID, HostRuntime, JSON_CAPABILITY_ID, MEMORY_READ_CAPABILITY_ID, + MEMORY_SEARCH_CAPABILITY_ID, MEMORY_TREE_CAPABILITY_ID, MEMORY_WRITE_CAPABILITY_ID, + PROFILE_SET_CAPABILITY_ID, READ_FILE_CAPABILITY_ID, RuntimeProcessPort, SHELL_CAPABILITY_ID, + TIME_CAPABILITY_ID, +}; +use ironclaw_reborn_composition::ProductLiveCapabilityIo; + +use super::super::{ + HarnessResult, HostRuntimeCapabilityHarness, RecordingRuntimeHttpEgress, + host_runtime_storage_roots, http_test_policy, local_dev_host_runtime_with_http_egress, + local_dev_host_runtime_with_live_http_egress, memory_mounts, workspace_mounts, +}; + +/// Configuration axes for [`core_builtin_tools`]. `Default` matches the +/// zero-arg `core_builtin_tools(CoreBuiltinOptions::default())` call. +pub(crate) struct CoreBuiltinOptions { + /// Network policy the built harness dispatches capabilities under. + /// Defaults to `http_test_policy()`; override via `.with_network_policy(..)`. + pub(crate) network_policy: NetworkPolicy, + /// `true` (default) injects the inert `RecordingProcessPort` so + /// `builtin.shell` invocations in tests never spawn a real OS process. + /// `.with_live_shell()` sets this `false`, which skips injection and lets + /// `HostRuntimeServices` default to the real `LocalHostProcessPort`. Only + /// consulted when `live_http_egress` is `false` — the live-http-egress + /// path never wires a process port either way. + pub(crate) recording_process: bool, + /// `true` selects `local_dev_host_runtime_with_live_http_egress` (a real + /// HTTP egress; no recording `RuntimeHttpEgress`/process port captured on + /// the harness) instead of the default recording-egress runtime + /// construction. Set via `.with_live_http_egress()`. + pub(crate) live_http_egress: bool, +} + +impl Default for CoreBuiltinOptions { + fn default() -> Self { + Self { + network_policy: http_test_policy(), + recording_process: true, + live_http_egress: false, + } + } +} + +impl CoreBuiltinOptions { + pub(crate) fn with_network_policy(mut self, network_policy: NetworkPolicy) -> Self { + self.network_policy = network_policy; + self + } + + /// Opts out of the recording process port so the real `LocalHostProcessPort` + /// executes shell commands on the host. + pub(crate) fn with_live_shell(mut self) -> Self { + self.recording_process = false; + self + } + + pub(crate) fn with_live_http_egress(mut self) -> Self { + self.live_http_egress = true; + self + } +} + +/// Core built-in tools (`time`/`json`/`http`/`memory_*`/`profile_set`/ +/// `read_file`/`apply_patch`/`shell`). See [`CoreBuiltinOptions`] for the axes. +pub(crate) async fn core_builtin_tools( + options: CoreBuiltinOptions, +) -> HarnessResult { + let CoreBuiltinOptions { + network_policy, + recording_process, + live_http_egress, + } = options; + if live_http_egress { + let (root, storage_root, workspace_root) = host_runtime_storage_roots()?; + let runtime = local_dev_host_runtime_with_live_http_egress(storage_root.clone())?; + core_builtin_tools_from_runtime( + root, + workspace_root, + runtime, + network_policy, + UserId::new("reborn-e2e-core-builtins-live-http-user")?, + ) + } else { + let (root, storage_root, workspace_root) = host_runtime_storage_roots()?; + let runtime_http_egress = Arc::new(RecordingRuntimeHttpEgress::with_body( + br#"{"accepted":true}"#.to_vec(), + )); + // Inject the inert recording port by default so `builtin.shell` + // invocations in tests never spawn a real OS process. `.with_live_shell()` + // sets `recording_process = false`, which skips injection and lets + // `HostRuntimeServices` default to the real `LocalHostProcessPort`. + let recording_process_port = if recording_process { + Some(Arc::new( + super::super::super::process::RecordingProcessPort::new(), + )) + } else { + None + }; + let process_port_dyn: Option> = recording_process_port + .as_ref() + .map(|p| Arc::clone(p) as Arc); + let runtime = local_dev_host_runtime_with_http_egress( + storage_root.clone(), + Arc::clone(&runtime_http_egress), + process_port_dyn, + )?; + let mut harness = core_builtin_tools_from_runtime( + root, + workspace_root, + runtime, + network_policy, + UserId::new("reborn-e2e-core-builtins-user")?, + )?; + harness.http_egress = Some(runtime_http_egress); + harness.process_port = recording_process_port; + Ok(harness) + } +} + +/// Zero-arg convenience; most callers want this and never touch +/// `CoreBuiltinOptions`. +pub(crate) async fn core_builtin_tools_default() -> HarnessResult { + core_builtin_tools(CoreBuiltinOptions::default()).await +} + +fn core_builtin_tools_from_runtime( + root: Arc, + workspace_root: PathBuf, + runtime: Arc, + network_policy: NetworkPolicy, + user_id: UserId, +) -> HarnessResult { + let mounts = workspace_mounts(MountPermissions::read_write_list_delete())?; + let memory_mounts = memory_mounts(MountPermissions::read_write_list_delete())?; + let memory_capability_ids = [ + CapabilityId::new(MEMORY_SEARCH_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_WRITE_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_READ_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_TREE_CAPABILITY_ID)?, + // profile_set writes to the memory mount (context/profile.json under + // the user-scoped scope), so it needs the memory mount override just + // like the four memory_* capabilities above. + CapabilityId::new(PROFILE_SET_CAPABILITY_ID)?, + ]; + Ok(HostRuntimeCapabilityHarness { + runtime, + approval_parts: None, + auto_approve_settings: None, + pending_approval_scopes: Arc::new(Mutex::new(HashMap::new())), + io: Arc::new(ProductLiveCapabilityIo::default()), + root, + workspace_root, + mounts, + capability_mount_overrides: memory_capability_ids + .iter() + .cloned() + .map(|capability_id| (capability_id, memory_mounts.clone())) + .collect(), + capability_ids: vec![ + CapabilityId::new(TIME_CAPABILITY_ID)?, + CapabilityId::new(JSON_CAPABILITY_ID)?, + CapabilityId::new(HTTP_CAPABILITY_ID)?, + CapabilityId::new(HTTP_SAVE_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_SEARCH_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_WRITE_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_READ_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_TREE_CAPABILITY_ID)?, + CapabilityId::new(PROFILE_SET_CAPABILITY_ID)?, + CapabilityId::new(READ_FILE_CAPABILITY_ID)?, + CapabilityId::new(APPLY_PATCH_CAPABILITY_ID)?, + // `builtin.shell` on the surface so scripted shell calls route + // through the process port (recording by default, live via + // `.with_live_shell()`). + CapabilityId::new(SHELL_CAPABILITY_ID)?, + ], + runtime_kind: RuntimeKind::FirstParty, + effect_kinds: vec![ + EffectKind::DispatchCapability, + EffectKind::ReadFilesystem, + EffectKind::WriteFilesystem, + EffectKind::Network, + EffectKind::SpawnProcess, + // `builtin.shell` declares ExecuteCode; the grant's allowed_effects + // must include it or the authorizer denies the capability before + // it reaches the process port. + EffectKind::ExecuteCode, + ], + network_policy, + secrets: Vec::new(), + provider_id: ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, + additional_provider_trust: Vec::new(), + user_id, + invocations: Arc::new(Mutex::new(Vec::new())), + results: Arc::new(Mutex::new(Vec::new())), + http_egress: None, + network_egress: None, + process_port: None, + profile_filesystem: None, + project_service: None, + skill_activation_source: None, + attachment_test_support: None, + outbound_target_tools: None, + scope_capability_by_run_owner: false, + product_auth: None, + tool_permission_overrides: None, + }) +} diff --git a/tests/integration/support/harness/profiles/extension.rs b/tests/integration/support/harness/profiles/extension.rs new file mode 100644 index 00000000000..d8fd7d725e5 --- /dev/null +++ b/tests/integration/support/harness/profiles/extension.rs @@ -0,0 +1,139 @@ +//! Extension domain tools profiles. + +use ironclaw_auth::{ + AuthProductScope, AuthProviderId, AuthSurface, CredentialAccountLabel, CredentialAccountStatus, + CredentialOwnership, NewCredentialAccount, ProviderScope, +}; +use ironclaw_host_api::{ + AgentId, InvocationId, MountView, ProjectId, ResourceScope, SecretHandle, TenantId, UserId, +}; + +use super::super::super::extension_surface::{ + BUNDLED_EXTENSION_CAPABILITY_IDS, EXTENSION_LIFECYCLE_CAPABILITY_IDS, +}; +use super::super::options::{HostRuntimeHarnessOptions, ToolsProfile}; +use super::super::{ + HarnessResult, HostRuntimeCapabilityHarness, bundled_extension_provider_trust, + capability_ids_from_strs, local_dev_all_effects, wildcard_test_policy, +}; + +pub(crate) fn extension_lifecycle_tools_profile() -> HarnessResult { + let mut capability_ids = capability_ids_from_strs(EXTENSION_LIFECYCLE_CAPABILITY_IDS)?; + capability_ids.extend(capability_ids_from_strs(BUNDLED_EXTENSION_CAPABILITY_IDS)?); + Ok(ToolsProfile { + capability_ids, + effect_kinds: local_dev_all_effects(), + options: HostRuntimeHarnessOptions::new( + MountView::default(), + Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( + true, + )?), + ) + .with_seed_extension_credentials(), + network_policy_override: Some(wildcard_test_policy()), + provider_trust_override: Some(bundled_extension_provider_trust()?), + auto_approve_default: Some(true), + ..ToolsProfile::new( + "reborn-e2e-extension-lifecycle-tools", + "reborn-e2e-extension-lifecycle-user", + )? + }) +} + +pub(crate) async fn extension_lifecycle_tools() -> HarnessResult { + extension_lifecycle_tools_profile()?.build().await +} + +pub(crate) async fn seed_extension_lifecycle_credentials( + services: &ironclaw_reborn_composition::RebornServices, + user_id: &UserId, +) -> HarnessResult<()> { + let product_auth = services + .product_auth + .as_ref() + .ok_or("extension lifecycle harness missing product auth")?; + let scope = AuthProductScope::credential_owner( + &ResourceScope { + tenant_id: TenantId::new("tenant-e2e")?, + user_id: user_id.clone(), + agent_id: Some(AgentId::new("agent-e2e")?), + project_id: Some(ProjectId::new("project-e2e")?), + mission_id: None, + thread_id: None, + invocation_id: InvocationId::new(), + }, + AuthSurface::Api, + ); + let accounts = product_auth.credential_account_service(); + for seed in extension_lifecycle_credential_seeds() { + accounts + .create_account(NewCredentialAccount { + scope: scope.clone(), + provider: AuthProviderId::new(seed.provider)?, + label: CredentialAccountLabel::new(seed.label)?, + status: CredentialAccountStatus::Configured, + ownership: CredentialOwnership::UserReusable, + owner_extension: None, + granted_extensions: Vec::new(), + access_secret: Some(SecretHandle::new(seed.secret_handle)?), + refresh_secret: None, + scopes: seed + .scopes + .iter() + .map(|scope| ProviderScope::new(*scope)) + .collect::, _>>()?, + }) + .await?; + } + Ok(()) +} + +struct ExtensionLifecycleCredentialSeed { + provider: &'static str, + label: &'static str, + secret_handle: &'static str, + scopes: &'static [&'static str], +} + +fn extension_lifecycle_credential_seeds() -> &'static [ExtensionLifecycleCredentialSeed] { + &[ + ExtensionLifecycleCredentialSeed { + provider: "github", + label: "qa github", + secret_handle: "qa_github_access", + scopes: &[], + }, + ExtensionLifecycleCredentialSeed { + provider: "google", + label: "qa google", + secret_handle: "qa_google_access", + scopes: &[ + "https://www.googleapis.com/auth/calendar.events", + "https://www.googleapis.com/auth/calendar.readonly", + "https://www.googleapis.com/auth/documents", + "https://www.googleapis.com/auth/documents.readonly", + "https://www.googleapis.com/auth/drive", + "https://www.googleapis.com/auth/drive.readonly", + "https://www.googleapis.com/auth/gmail.modify", + "https://www.googleapis.com/auth/gmail.readonly", + "https://www.googleapis.com/auth/gmail.send", + "https://www.googleapis.com/auth/presentations", + "https://www.googleapis.com/auth/presentations.readonly", + "https://www.googleapis.com/auth/spreadsheets", + "https://www.googleapis.com/auth/spreadsheets.readonly", + ], + }, + ExtensionLifecycleCredentialSeed { + provider: "nearai", + label: "qa nearai", + secret_handle: "qa_nearai_access", + scopes: &[], + }, + ExtensionLifecycleCredentialSeed { + provider: "notion", + label: "qa notion", + secret_handle: "qa_notion_access", + scopes: &[], + }, + ] +} diff --git a/tests/integration/support/harness/profiles/file.rs b/tests/integration/support/harness/profiles/file.rs new file mode 100644 index 00000000000..698f5ec0f8a --- /dev/null +++ b/tests/integration/support/harness/profiles/file.rs @@ -0,0 +1,63 @@ +//! File domain tools profiles: `file_tools()` / `file_tools_requiring_approval()` +//! / `write_only()`, sharing `file_tools_with_runtime_policy` as their +//! internal tail. See `harness/options.rs` for the `ToolsProfile` pattern. + +use ironclaw_host_api::{CapabilityId, EffectKind, MountPermissions}; +use ironclaw_host_runtime::{READ_FILE_CAPABILITY_ID, WRITE_FILE_CAPABILITY_ID}; + +use super::super::options::{HostRuntimeHarnessOptions, ToolsProfile}; +use super::super::{HarnessResult, HostRuntimeCapabilityHarness, workspace_mounts}; + +fn file_tools_with_runtime_policy( + runtime_policy: Option, +) -> HarnessResult { + Ok(ToolsProfile { + capability_ids: vec![ + CapabilityId::new(WRITE_FILE_CAPABILITY_ID)?, + CapabilityId::new(READ_FILE_CAPABILITY_ID)?, + ], + effect_kinds: vec![EffectKind::ReadFilesystem, EffectKind::WriteFilesystem], + options: HostRuntimeHarnessOptions::new( + workspace_mounts(MountPermissions::read_write_list_delete())?, + runtime_policy, + ), + ..ToolsProfile::new("reborn-e2e-builtin-tools", "reborn-e2e-builtin-user")? + }) +} + +pub(crate) fn file_tools_profile() -> HarnessResult { + Ok(file_tools_with_runtime_policy(Some( + ironclaw_reborn_composition::local_dev_yolo_runtime_policy(true)?, + ))? + .with_auto_approve_default(true)) +} + +pub(crate) async fn file_tools() -> HarnessResult { + file_tools_profile()?.build().await +} + +pub(crate) fn file_tools_requiring_approval_profile() -> HarnessResult { + // Global auto-approve now defaults ON, so disable it explicitly to keep + // this constructor's per-tool approval gate behavior. + Ok(file_tools_with_runtime_policy(None)?.with_auto_approve_default(false)) +} + +pub(crate) async fn file_tools_requiring_approval() -> HarnessResult { + file_tools_requiring_approval_profile()?.build().await +} + +pub(crate) fn write_only_profile() -> HarnessResult { + Ok(ToolsProfile { + capability_ids: vec![CapabilityId::new(WRITE_FILE_CAPABILITY_ID)?], + effect_kinds: vec![EffectKind::WriteFilesystem], + options: HostRuntimeHarnessOptions::new( + workspace_mounts(MountPermissions::read_write_list_delete())?, + None, + ), + ..ToolsProfile::new("reborn-e2e-write-only", "reborn-e2e-write-only-user")? + }) +} + +pub(crate) async fn write_only() -> HarnessResult { + write_only_profile()?.build().await +} diff --git a/tests/integration/support/harness/profiles/github.rs b/tests/integration/support/harness/profiles/github.rs new file mode 100644 index 00000000000..fb0360ffa75 --- /dev/null +++ b/tests/integration/support/harness/profiles/github.rs @@ -0,0 +1,181 @@ +//! GitHub domain tools profiles. + +use std::collections::HashMap; +use std::sync::{Arc, Mutex}; + +use ironclaw_host_api::{ + CapabilityId, CredentialStageError, MountPermissions, RuntimeKind, SecretHandle, UserId, +}; +use ironclaw_host_runtime::{READ_FILE_CAPABILITY_ID, WRITE_FILE_CAPABILITY_ID}; +use ironclaw_network::NetworkHttpEgress; +use ironclaw_reborn_composition::ProductLiveCapabilityIo; + +use super::super::super::github as github_support; +use super::super::options::{HostRuntimeHarnessOptions, ToolsProfile}; +use super::super::{ + HarnessResult, HostRuntimeCapabilityHarness, RecordingNetworkHttpEgress, + RecordingRuntimeHttpEgress, bundled_extension_provider_trust, local_dev_all_effects, + local_dev_host_runtime_with_registry_and_egress, wildcard_test_policy, workspace_mounts, +}; + +/// C-JOURNEY convergence seam: surfaces the file-tool approval-gate +/// capabilities (`write_file`/`read_file` @ `Ask`) AND a single GitHub +/// capability (`github.get_repo`) on the SAME `build_reborn_services` +/// local-dev runtime (the one wired with the stores both gate classes' +/// resume paths need). Distinct from `github_issue_tools_auth_required` +/// (a separate, lower-level build with a hardcoded credential resolver): +/// here `github.*` resolves through the REAL +/// `ProductAuthRuntimeCredentialResolver`, and no credential account is +/// seeded at construction. +/// +/// **Gate chaining (empirically verified):** the disabled global +/// auto-approve is NOT capability-scoped, so `github.get_repo` first raises +/// `BlockedApproval`; approving re-dispatches the still-uncredentialed +/// capability, which blocks AGAIN at `BlockedAuth`. +/// `RebornIntegrationHarness::resolve_auth_gate` seeds the account and +/// resumes, letting the SAME parked capability complete — see +/// `scenario_auth_then_approval_journey`'s module doc for the full chain. +/// +/// Making `github.*` genuinely dispatchable (not just granted) needs two +/// seams: (1) `RebornServices::publish_bundled_extension_for_test` registers +/// it in the runtime's OWN dispatchable registry (capability_ids/ +/// additional_provider_trust alone only populate the harness-authority grant +/// layer and would silently no-op); (2) `copy_dir_recursive` copies the real +/// github asset directory into this harness's tempdir mount, since +/// `build_local_runtime` otherwise mounts `/system/extensions` empty and WASM +/// compilation fails at dispatch time. +/// +/// Runtime policy is left `None` (not `LocalDevYolo`) so file tools' real +/// `PermissionMode::Ask` gate is preserved. +pub(crate) fn file_and_github_auth_tools_profile() -> HarnessResult { + // Hermetic guard: `new_with_options`'s `build_local_runtime` defaults to + // a REAL `ReqwestNetworkTransport` when no test egress is supplied + // (`factory.rs`). This harness surfaces a `github.*` WASM capability + // that crosses HTTP, so it MUST override the network egress or the + // post-resume dispatch would attempt a live network call. + let github_fixture_response = + br#"{"id":1,"full_name":"octocat/hello-world","private":false}"#.to_vec(); + let network_egress: Arc = Arc::new( + RecordingNetworkHttpEgress::with_body(github_fixture_response), + ); + Ok(ToolsProfile { + capability_ids: vec![ + CapabilityId::new(WRITE_FILE_CAPABILITY_ID)?, + CapabilityId::new(READ_FILE_CAPABILITY_ID)?, + CapabilityId::new("github.get_repo")?, + ], + effect_kinds: local_dev_all_effects(), + options: HostRuntimeHarnessOptions::new( + workspace_mounts(MountPermissions::read_write_list_delete())?, + None, + ) + .with_network_http_egress_for_test(network_egress) + .with_activated_bundled_extension(github_support::extension_package()?), + network_policy_override: Some(wildcard_test_policy()), + provider_trust_override: Some(bundled_extension_provider_trust()?), + post_construct_asset_copy: Some(( + github_support::asset_root(), + std::path::PathBuf::from("local-dev/system/extensions/github"), + )), + auto_approve_default: Some(false), + ..ToolsProfile::new( + "reborn-e2e-file-github-auth-tools", + "reborn-e2e-file-github-auth-user", + )? + }) +} + +/// See [`file_and_github_auth_tools_profile`]. +pub(crate) async fn file_and_github_auth_tools() -> HarnessResult { + file_and_github_auth_tools_profile()?.build().await +} + +/// Wires the GitHub first-party WASM capabilities behind `GithubHarnessAuthorizer`. +/// See `github_issue_tools_with_credential_result` for the credential-injection +/// coupling this relies on (T0-SECRET-INJECT). +pub(crate) async fn github_issue_tools() -> HarnessResult { + // Credential account resolves to a real handle → capability dispatches. + github_issue_tools_with_credential_result(Ok(SecretHandle::new("github_manual_access")?)) +} + +/// E-AUTHGATE: the GitHub extension wired so its credential account resolver +/// returns `AuthRequired`, raising a `TurnStatus::BlockedAuth` gate when a +/// `github.*` capability is dispatched. Used by `RebornIntegrationGroup::live_auth_gate`. +pub(crate) async fn github_issue_tools_auth_required() -> HarnessResult +{ + github_issue_tools_with_credential_result(Err(CredentialStageError::AuthRequired)) +} + +/// Shared GitHub-extension constructor (E-AUTHGATE): the only difference +/// between the happy-path and auth-blocked variants is the credential account +/// resolver result, so the full `Self {..}` literal lives here once. +/// +/// Credential injection runs through two mechanisms, not one: the +/// authorizer's `InjectCredentialAccountOnce` obligation, AND +/// `local_dev_host_runtime_with_registry_and_egress`'s independent +/// `SharedHostWasmRuntimeCredentials` restaging (runs unconditionally on +/// every WASM HTTP call, not gated on the authorizer's `Decision`). So a test +/// asserting the injected header proves the end-to-end wire outcome, not +/// that the obligation alone produced it. +/// +/// Empirically verified: removing the obligation does not fall back to an +/// unauthenticated request — the run hangs and never reaches `Completed`. +/// That's why `reborn_integration_secret_injection.rs`'s mutation-verify +/// flips the secret *value* (a fast, specific failure) rather than removing +/// the obligation (a slow, ambiguous timeout). +fn github_issue_tools_with_credential_result( + credential_account_result: Result, +) -> HarnessResult { + let root = Arc::new(tempfile::tempdir()?); + let storage_root = root.path().join("local-dev"); + let workspace_root = storage_root.join("workspace"); + std::fs::create_dir_all(&workspace_root)?; + let github_fixture_response = + br#"{"object":{"sha":"abc123def4567890abc123def4567890abc123de"},"ok":true}"#.to_vec(); + let runtime_http_egress = Arc::new(RecordingRuntimeHttpEgress::with_body( + github_fixture_response.clone(), + )); + let network_egress = Arc::new(RecordingNetworkHttpEgress::with_body( + github_fixture_response, + )); + let runtime = local_dev_host_runtime_with_registry_and_egress( + storage_root.clone(), + github_support::extension_registry()?, + runtime_http_egress.clone(), + network_egress.clone(), + credential_account_result, + )?; + let mounts = workspace_mounts(MountPermissions::read_write_list_delete())?; + Ok(HostRuntimeCapabilityHarness { + runtime, + approval_parts: None, + auto_approve_settings: None, + pending_approval_scopes: Arc::new(Mutex::new(HashMap::new())), + io: Arc::new(ProductLiveCapabilityIo::default()), + root, + workspace_root, + mounts, + capability_mount_overrides: Vec::new(), + capability_ids: github_support::capability_ids()?, + runtime_kind: RuntimeKind::Wasm, + effect_kinds: github_support::effect_kinds(), + network_policy: github_support::api_policy(), + secrets: github_support::secret_handles()?, + provider_id: github_support::provider_id()?, + additional_provider_trust: Vec::new(), + user_id: UserId::new("reborn-e2e-github-user")?, + invocations: Arc::new(Mutex::new(Vec::new())), + results: Arc::new(Mutex::new(Vec::new())), + http_egress: Some(runtime_http_egress), + network_egress: Some(network_egress), + process_port: None, + profile_filesystem: None, + project_service: None, + skill_activation_source: None, + attachment_test_support: None, + outbound_target_tools: None, + scope_capability_by_run_owner: false, + product_auth: None, + tool_permission_overrides: None, + }) +} diff --git a/tests/integration/support/harness/profiles/mock_mcp.rs b/tests/integration/support/harness/profiles/mock_mcp.rs new file mode 100644 index 00000000000..2238cb8d644 --- /dev/null +++ b/tests/integration/support/harness/profiles/mock_mcp.rs @@ -0,0 +1,98 @@ +//! Mock-MCP domain tools profiles. + +use std::collections::HashMap; +use std::sync::{Arc, Mutex}; + +use ironclaw_extensions::ExtensionRegistry; +use ironclaw_host_api::{ + CapabilityId, EffectKind, ExtensionId, MountPermissions, RuntimeKind, UserId, +}; +use ironclaw_reborn_composition::ProductLiveCapabilityIo; + +use super::super::super::harness_mcp::{ + build_loopback_mcp_runtime, local_dev_host_runtime_with_registry_egress_and_mcp, + mcp_loopback_network_policy, mock_mcp_extension_package, +}; +use super::super::{ + HarnessResult, HostRuntimeCapabilityHarness, RecordingRuntimeHttpEgress, + host_runtime_storage_roots, workspace_mounts, +}; + +/// Wire a single MCP capability backed by the loopback mock server. +/// +/// `mcp_url` — the mock server's MCP endpoint (e.g. `"http://127.0.0.1:PORT/mcp"`). +/// `provider_id` — extension id used in the registry (e.g. `"mock-mcp"`). +/// `capability_id` — capability id surfaced to the model (e.g. `"mock-mcp.search"`). +/// +/// The harness (via the `harness_mcp` scaffolding) builds a loopback MCP +/// egress that makes REAL HTTP connections to the mock server, injecting a +/// fake Bearer token to satisfy the mock's auth gate. Production egress +/// policy, network policy, and credential stores are bypassed — this path is +/// test-only. +pub(crate) async fn mock_mcp_tools( + mcp_url: &str, + provider_id: &str, + capability_id: &str, +) -> HarnessResult { + let (root, storage_root, workspace_root) = host_runtime_storage_roots()?; + // Recording egress for any first-party tool paths (unused in MCP tests, + // but HostRuntimeServices requires it when first_party_capabilities are wired). + let first_party_egress = Arc::new(RecordingRuntimeHttpEgress::with_body( + br#"{"accepted":true}"#.to_vec(), + )); + // Real loopback egress + MCP runtime for the mock MCP server; the + // scaffolding (egress, adapter chain, runtime) lives in `harness_mcp`. + let mcp_runtime = build_loopback_mcp_runtime(mcp_url)?; + let mut registry = ExtensionRegistry::new(); + registry.insert(mock_mcp_extension_package( + provider_id, + mcp_url, + capability_id, + )?)?; + let runtime = local_dev_host_runtime_with_registry_egress_and_mcp( + storage_root, + registry, + Arc::clone(&first_party_egress), + mcp_runtime, + provider_id, + )?; + let mounts = workspace_mounts(MountPermissions::read_write_list_delete())?; + Ok(HostRuntimeCapabilityHarness { + runtime, + approval_parts: None, + auto_approve_settings: None, + pending_approval_scopes: Arc::new(Mutex::new(HashMap::new())), + io: Arc::new(ProductLiveCapabilityIo::default()), + root, + workspace_root, + mounts, + capability_mount_overrides: Vec::new(), + capability_ids: vec![CapabilityId::new(capability_id)?], + runtime_kind: RuntimeKind::Mcp, + effect_kinds: vec![EffectKind::DispatchCapability, EffectKind::Network], + // The MCP capability declares `EffectKind::Network`, so authorization + // attaches an `ApplyNetworkPolicy` obligation that the host runtime + // rejects when `allowed_targets` is empty (a default `NetworkPolicy`). + // The mock server lives at `http://127.0.0.1:/mcp`, so permit the + // loopback host (and disable the private-IP denial that would otherwise + // block 127.0.0.1) so the MCP egress reaches the loopback server. + network_policy: mcp_loopback_network_policy(), + secrets: Vec::new(), + provider_id: ExtensionId::new(provider_id)?, + additional_provider_trust: Vec::new(), + user_id: UserId::new("reborn-itest-mcp-user")?, + invocations: Arc::new(Mutex::new(Vec::new())), + results: Arc::new(Mutex::new(Vec::new())), + http_egress: None, + network_egress: None, + process_port: None, + profile_filesystem: None, + project_service: None, + skill_activation_source: None, + attachment_test_support: None, + outbound_target_tools: None, + scope_capability_by_run_owner: false, + product_auth: None, + tool_permission_overrides: None, + }) +} diff --git a/tests/integration/support/harness/profiles/mod.rs b/tests/integration/support/harness/profiles/mod.rs new file mode 100644 index 00000000000..a172cfe04d4 --- /dev/null +++ b/tests/integration/support/harness/profiles/mod.rs @@ -0,0 +1,16 @@ +pub(crate) mod attachment; +pub(crate) mod coding_read; +pub(crate) mod core_builtin; +pub(crate) mod extension; +pub(crate) mod file; +pub(crate) mod github; +pub(crate) mod mock_mcp; +pub(crate) mod outbound; +pub(crate) mod process; +pub(crate) mod profile; +pub(crate) mod project; +pub(crate) mod qa_smoke; +pub(crate) mod skill; +pub(crate) mod trace_commons; +pub(crate) mod trigger; +pub(crate) mod web_access; diff --git a/tests/integration/support/harness/profiles/outbound.rs b/tests/integration/support/harness/profiles/outbound.rs new file mode 100644 index 00000000000..76686a3a5a8 --- /dev/null +++ b/tests/integration/support/harness/profiles/outbound.rs @@ -0,0 +1,52 @@ +//! Outbound domain tools profile (`outbound_target_tools`). + +use ironclaw_host_api::{CapabilityId, EffectKind, MountView}; + +use super::super::super::outbound_preferences::FakeOutboundPreferencesFacade; +use super::super::options::{HostRuntimeHarnessOptions, ToolsProfile}; +use super::super::{HarnessResult, HostRuntimeCapabilityHarness}; + +/// C-SYNTH outbound: harness surfacing the two local-dev synthetic +/// `outbound_delivery_*` capabilities over an injected +/// [`FakeOutboundPreferencesFacade`] double. +/// `create_capability_port` injects them via +/// `apply_synthetic_capability_wrappers` because +/// `outbound_target_tools` is `Some`. `target_set` runs with +/// `requires_approval = true`, so its settings decision is exercised for +/// real: global auto-approve (default ON) → `Allow`; a `Disabled` tool +/// override (`disable_outbound_target_set_tool`) → `Deny`; auto-approve +/// disabled → `Ask` (approval gate). The RETURNED harness leaves global +/// auto-approve at its default-ON state so the happy/`NotFound` arms +/// dispatch through `Allow`; the gate arm disables it per-test. +pub(crate) fn outbound_target_tools_profile() -> HarnessResult { + let facade = FakeOutboundPreferencesFacade::with_default_targets(); + Ok(ToolsProfile { + capability_ids: vec![ + CapabilityId::new( + ironclaw_reborn_composition::test_support::OUTBOUND_DELIVERY_TARGETS_LIST_CAPABILITY_ID, + )?, + CapabilityId::new( + ironclaw_reborn_composition::test_support::OUTBOUND_DELIVERY_TARGET_SET_CAPABILITY_ID, + )?, + ], + effect_kinds: vec![ + EffectKind::DispatchCapability, + EffectKind::ExternalWrite, + EffectKind::ReadFilesystem, + EffectKind::WriteFilesystem, + ], + options: HostRuntimeHarnessOptions::new( + MountView::default(), + Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( + true, + )?), + ) + .with_outbound_target_tools(facade, true), + ..ToolsProfile::new("reborn-e2e-outbound-target-tools", "reborn-e2e-outbound-target-user")? + }) +} + +/// See [`outbound_target_tools_profile`]. +pub(crate) async fn outbound_target_tools() -> HarnessResult { + outbound_target_tools_profile()?.build().await +} diff --git a/tests/integration/support/harness/profiles/process.rs b/tests/integration/support/harness/profiles/process.rs new file mode 100644 index 00000000000..5daf38a4808 --- /dev/null +++ b/tests/integration/support/harness/profiles/process.rs @@ -0,0 +1,35 @@ +//! Process domain tools profile (`process_tools`) — see `harness/options.rs` +//! for the `ToolsProfile` pattern. + +use ironclaw_host_api::{CapabilityId, EffectKind, MountView}; +use ironclaw_host_runtime::{ + ECHO_CAPABILITY_ID, SHELL_CAPABILITY_ID, SPAWN_SUBAGENT_CAPABILITY_ID, +}; + +use super::super::options::{HostRuntimeHarnessOptions, ToolsProfile}; +use super::super::{HarnessResult, HostRuntimeCapabilityHarness}; + +pub(crate) fn process_tools_profile() -> HarnessResult { + Ok(ToolsProfile { + capability_ids: vec![ + CapabilityId::new(ECHO_CAPABILITY_ID)?, + CapabilityId::new(SHELL_CAPABILITY_ID)?, + CapabilityId::new(SPAWN_SUBAGENT_CAPABILITY_ID)?, + ], + effect_kinds: vec![ + EffectKind::DispatchCapability, + EffectKind::ReadFilesystem, + EffectKind::WriteFilesystem, + EffectKind::Network, + EffectKind::SpawnProcess, + EffectKind::ExecuteCode, + ], + options: HostRuntimeHarnessOptions::new(MountView::default(), None), + auto_approve_default: Some(true), + ..ToolsProfile::new("reborn-e2e-process-tools", "reborn-e2e-process-user")? + }) +} + +pub(crate) async fn process_tools() -> HarnessResult { + process_tools_profile()?.build().await +} diff --git a/tests/integration/support/harness/profiles/profile.rs b/tests/integration/support/harness/profiles/profile.rs new file mode 100644 index 00000000000..321dc7fb8f5 --- /dev/null +++ b/tests/integration/support/harness/profiles/profile.rs @@ -0,0 +1,38 @@ +//! Profile domain tools profile (`profile_tools`). + +use ironclaw_host_api::{CapabilityId, EffectKind, MountPermissions}; +use ironclaw_host_runtime::PROFILE_SET_CAPABILITY_ID; + +use super::super::options::{HostRuntimeHarnessOptions, ToolsProfile}; +use super::super::{HarnessResult, HostRuntimeCapabilityHarness, memory_mounts}; + +/// Group whose ONLY capability is `builtin.profile_set` (E-PROFILE seam). +/// Uses `new_with_options` (not `core_builtin_tools_from_runtime`), so +/// `profile_filesystem` is populated from `services.local_dev_profile_filesystem_for_test()` +/// — the read-back half of the round trip a `RebornIntegrationGroup::profile_tools()` +/// scenario needs. Base mounts are `/memory` directly (this harness's only +/// capability needs it; no per-capability mount override required, unlike +/// `core_builtin_tools_from_runtime`'s multi-capability surface). +pub(crate) fn profile_tools_profile() -> HarnessResult { + Ok(ToolsProfile { + capability_ids: vec![CapabilityId::new(PROFILE_SET_CAPABILITY_ID)?], + effect_kinds: vec![ + EffectKind::DispatchCapability, + EffectKind::ReadFilesystem, + EffectKind::WriteFilesystem, + ], + options: HostRuntimeHarnessOptions::new( + memory_mounts(MountPermissions::read_write_list_delete())?, + Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( + true, + )?), + ), + auto_approve_default: Some(true), + ..ToolsProfile::new("reborn-e2e-profile-tools", "reborn-e2e-profile-tools-user")? + }) +} + +/// See [`profile_tools_profile`]. +pub(crate) async fn profile_tools() -> HarnessResult { + profile_tools_profile()?.build().await +} diff --git a/tests/integration/support/harness/profiles/project.rs b/tests/integration/support/harness/profiles/project.rs new file mode 100644 index 00000000000..165c0768c82 --- /dev/null +++ b/tests/integration/support/harness/profiles/project.rs @@ -0,0 +1,77 @@ +//! Project domain tools profiles (`project_tools`, `project_tools_with_fault_injection`). + +use ironclaw_host_api::{CapabilityId, EffectKind, MountView}; + +use super::super::options::{HostRuntimeHarnessOptions, ToolsProfile}; +use super::super::{HarnessResult, HostRuntimeCapabilityHarness}; + +/// E-PROJ: harness surfacing the local-dev synthetic `project_create` +/// capability. `create_capability_port` injects the synthetic capability via +/// `apply_synthetic_capability_wrappers` because `PROJECT_CREATE_CAPABILITY_ID` +/// is in the allowlist. Auto-approve is enabled so the capability dispatches +/// without a gate. +pub(crate) fn project_tools_profile() -> HarnessResult { + Ok(ToolsProfile { + capability_ids: vec![CapabilityId::new( + ironclaw_reborn_composition::test_support::PROJECT_CREATE_CAPABILITY_ID, + )?], + effect_kinds: vec![ + EffectKind::DispatchCapability, + EffectKind::ReadFilesystem, + EffectKind::WriteFilesystem, + ], + options: HostRuntimeHarnessOptions::new( + MountView::default(), + Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( + true, + )?), + ), + auto_approve_default: Some(true), + ..ToolsProfile::new("reborn-e2e-project-tools", "reborn-e2e-project-tools-user")? + }) +} + +/// See [`project_tools_profile`]. +pub(crate) async fn project_tools() -> HarnessResult { + project_tools_profile()?.build().await +} + +/// C-SYNTH `project_create` fault-injection arm: same surface as +/// `project_tools()`, but the real `Arc` is wrapped in +/// `FaultInjectingProjectService` +/// (`with_project_service_fault_injection`) so a `create_project` call +/// naming `FAULT_INJECT_DENIED_PROJECT_NAME` returns +/// `ProjectServiceError::Denied`/`PolicyDenied` and proves the real +/// capability dispatch's recoverable `Failed` behavior. This is *not* +/// the `project_service_outcome` `Unavailable` / internal-retry path. +/// Any other `create_project` name still reaches the real store. +pub(crate) fn project_tools_with_fault_injection_profile() -> HarnessResult { + Ok(ToolsProfile { + capability_ids: vec![CapabilityId::new( + ironclaw_reborn_composition::test_support::PROJECT_CREATE_CAPABILITY_ID, + )?], + effect_kinds: vec![ + EffectKind::DispatchCapability, + EffectKind::ReadFilesystem, + EffectKind::WriteFilesystem, + ], + options: HostRuntimeHarnessOptions::new( + MountView::default(), + Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( + true, + )?), + ) + .with_project_service_fault_injection(), + auto_approve_default: Some(true), + ..ToolsProfile::new( + "reborn-e2e-project-tools-fault-injection", + "reborn-e2e-project-tools-fault-injection-user", + )? + }) +} + +/// See [`project_tools_with_fault_injection_profile`]. +pub(crate) async fn project_tools_with_fault_injection() +-> HarnessResult { + project_tools_with_fault_injection_profile()?.build().await +} diff --git a/tests/integration/support/harness/profiles/qa_smoke.rs b/tests/integration/support/harness/profiles/qa_smoke.rs new file mode 100644 index 00000000000..01b8a4abe90 --- /dev/null +++ b/tests/integration/support/harness/profiles/qa_smoke.rs @@ -0,0 +1,123 @@ +//! `qa_smoke` domain tools profile. `qa_smoke_tools()` builds the host +//! runtime directly (rather than through `new_with_options`) so it can wire a +//! scripted `RecordingRuntimeHttpEgress` body at construction time — a bespoke +//! full constructor rather than a `ToolsProfile`. + +use std::collections::HashMap; +use std::sync::{Arc, Mutex}; + +use ironclaw_host_api::{ + CapabilityId, EffectKind, ExtensionId, MountPermissions, RuntimeKind, UserId, +}; +use ironclaw_host_runtime::{ + APPLY_PATCH_CAPABILITY_ID, BUILTIN_FIRST_PARTY_PROVIDER, ECHO_CAPABILITY_ID, + GLOB_CAPABILITY_ID, GREP_CAPABILITY_ID, HTTP_CAPABILITY_ID, HTTP_SAVE_CAPABILITY_ID, + JSON_CAPABILITY_ID, LIST_DIR_CAPABILITY_ID, MEMORY_READ_CAPABILITY_ID, + MEMORY_SEARCH_CAPABILITY_ID, MEMORY_TREE_CAPABILITY_ID, MEMORY_WRITE_CAPABILITY_ID, + READ_FILE_CAPABILITY_ID, SHELL_CAPABILITY_ID, SKILL_INSTALL_CAPABILITY_ID, + SKILL_LIST_CAPABILITY_ID, SKILL_REMOVE_CAPABILITY_ID, SPAWN_SUBAGENT_CAPABILITY_ID, + TIME_CAPABILITY_ID, TRIGGER_CREATE_CAPABILITY_ID, TRIGGER_LIST_CAPABILITY_ID, + TRIGGER_PAUSE_CAPABILITY_ID, TRIGGER_REMOVE_CAPABILITY_ID, TRIGGER_RESUME_CAPABILITY_ID, + WRITE_FILE_CAPABILITY_ID, +}; +use ironclaw_reborn_composition::ProductLiveCapabilityIo; + +use super::super::{ + HarnessResult, HostRuntimeCapabilityHarness, RecordingRuntimeHttpEgress, + host_runtime_storage_roots, http_test_policy, local_dev_host_runtime_with_http_egress, + memory_mounts, qa_smoke_mounts, +}; + +pub(crate) async fn qa_smoke_tools() -> HarnessResult { + let (root, storage_root, workspace_root) = host_runtime_storage_roots()?; + std::fs::create_dir_all(storage_root.join("skills"))?; + std::fs::create_dir_all(storage_root.join("system/skills"))?; + let runtime = local_dev_host_runtime_with_http_egress( + storage_root, + Arc::new(RecordingRuntimeHttpEgress::with_body( + br#"{"accepted":true,"source":"qa-smoke"}"#.to_vec(), + )), + // qa_smoke_tools exercises real process execution (SpawnProcess effect); + // leave the default LocalHostProcessPort in place. + None, + )?; + let mounts = qa_smoke_mounts()?; + let memory_mounts = memory_mounts(MountPermissions::read_write_list_delete())?; + let memory_capability_ids = [ + CapabilityId::new(MEMORY_SEARCH_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_WRITE_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_READ_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_TREE_CAPABILITY_ID)?, + ]; + Ok(HostRuntimeCapabilityHarness { + runtime, + approval_parts: None, + auto_approve_settings: None, + pending_approval_scopes: Arc::new(Mutex::new(HashMap::new())), + io: Arc::new(ProductLiveCapabilityIo::default()), + root, + workspace_root, + mounts, + capability_mount_overrides: memory_capability_ids + .iter() + .cloned() + .map(|capability_id| (capability_id, memory_mounts.clone())) + .collect(), + capability_ids: vec![ + CapabilityId::new(ECHO_CAPABILITY_ID)?, + CapabilityId::new(TIME_CAPABILITY_ID)?, + CapabilityId::new(JSON_CAPABILITY_ID)?, + CapabilityId::new(HTTP_CAPABILITY_ID)?, + CapabilityId::new(HTTP_SAVE_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_SEARCH_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_WRITE_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_READ_CAPABILITY_ID)?, + CapabilityId::new(MEMORY_TREE_CAPABILITY_ID)?, + CapabilityId::new(READ_FILE_CAPABILITY_ID)?, + CapabilityId::new(WRITE_FILE_CAPABILITY_ID)?, + CapabilityId::new(LIST_DIR_CAPABILITY_ID)?, + CapabilityId::new(GLOB_CAPABILITY_ID)?, + CapabilityId::new(GREP_CAPABILITY_ID)?, + CapabilityId::new(APPLY_PATCH_CAPABILITY_ID)?, + CapabilityId::new(SHELL_CAPABILITY_ID)?, + CapabilityId::new(SPAWN_SUBAGENT_CAPABILITY_ID)?, + CapabilityId::new(SKILL_LIST_CAPABILITY_ID)?, + CapabilityId::new(SKILL_INSTALL_CAPABILITY_ID)?, + CapabilityId::new(SKILL_REMOVE_CAPABILITY_ID)?, + CapabilityId::new(TRIGGER_CREATE_CAPABILITY_ID)?, + CapabilityId::new(TRIGGER_LIST_CAPABILITY_ID)?, + CapabilityId::new(TRIGGER_PAUSE_CAPABILITY_ID)?, + CapabilityId::new(TRIGGER_RESUME_CAPABILITY_ID)?, + CapabilityId::new(TRIGGER_REMOVE_CAPABILITY_ID)?, + ], + runtime_kind: RuntimeKind::FirstParty, + effect_kinds: vec![ + EffectKind::DispatchCapability, + EffectKind::ReadFilesystem, + EffectKind::WriteFilesystem, + EffectKind::DeleteFilesystem, + EffectKind::Network, + EffectKind::SpawnProcess, + EffectKind::ExecuteCode, + EffectKind::ExternalWrite, + ], + network_policy: http_test_policy(), + secrets: Vec::new(), + provider_id: ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, + additional_provider_trust: Vec::new(), + user_id: UserId::new("reborn-e2e-qa-smoke-user")?, + invocations: Arc::new(Mutex::new(Vec::new())), + results: Arc::new(Mutex::new(Vec::new())), + http_egress: None, + network_egress: None, + process_port: None, + profile_filesystem: None, + project_service: None, + skill_activation_source: None, + attachment_test_support: None, + outbound_target_tools: None, + scope_capability_by_run_owner: false, + product_auth: None, + tool_permission_overrides: None, + }) +} diff --git a/tests/integration/support/harness/profiles/skill.rs b/tests/integration/support/harness/profiles/skill.rs new file mode 100644 index 00000000000..481bdb52590 --- /dev/null +++ b/tests/integration/support/harness/profiles/skill.rs @@ -0,0 +1,90 @@ +//! Skill domain tools profiles. + +use ironclaw_host_api::{CapabilityId, EffectKind, TenantId}; +use ironclaw_host_runtime::{ + SKILL_INSTALL_CAPABILITY_ID, SKILL_LIST_CAPABILITY_ID, SKILL_REMOVE_CAPABILITY_ID, +}; + +use super::super::options::{HostRuntimeHarnessOptions, ToolsProfile}; +use super::super::{HarnessResult, HostRuntimeCapabilityHarness, http_test_policy, skill_mounts}; + +/// `pub(crate)`: also used by `RebornIntegrationGroupBuilder::skill_management_tools` +/// (`group_constructors.rs`, C-SKILL) to wire the SAME preset onto the +/// int-tier group, so the QA/trace-tier smoke test and the int-tier group +/// never drift on capability ids / mounts / policy. +pub(crate) fn skill_management_tools_profile() -> HarnessResult { + Ok(ToolsProfile { + capability_ids: vec![ + CapabilityId::new(SKILL_LIST_CAPABILITY_ID)?, + CapabilityId::new(SKILL_INSTALL_CAPABILITY_ID)?, + CapabilityId::new(SKILL_REMOVE_CAPABILITY_ID)?, + ], + effect_kinds: vec![ + EffectKind::DispatchCapability, + EffectKind::ReadFilesystem, + EffectKind::WriteFilesystem, + EffectKind::DeleteFilesystem, + EffectKind::Network, + ], + options: HostRuntimeHarnessOptions::new( + skill_mounts()?, + Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( + true, + )?), + ), + network_policy_override: Some(http_test_policy()), + auto_approve_default: Some(true), + ..ToolsProfile::new( + "reborn-e2e-skill-management-tools", + "reborn-e2e-skill-management-user", + )? + }) +} + +/// See [`skill_management_tools_profile`]. +pub(crate) async fn skill_management_tools() -> HarnessResult { + skill_management_tools_profile()?.build().await +} + +/// Harness surfacing the local-dev synthetic `skill_activate` capability +/// (E-SKILL seam). `new_with_options` builds the `skill_activation_source` +/// (because `SKILL_ACTIVATE_CAPABILITY_ID` is in the allowlist) under +/// `tenant` — the caller's ACTUAL group run-scope tenant, passed through +/// rather than re-hardcoded here — which `create_capability_port` wraps +/// onto the port and `into_group` wires as the runtime's +/// `skill_context_source`. The skill file the model activates is seeded as +/// a system-scoped skill by `RebornIntegrationGroup::skill_activation_tools`. +/// Mirrors `skill_management_tools`/`project_tools`. +pub(crate) fn skill_activation_tools_profile(tenant: &TenantId) -> HarnessResult { + Ok(ToolsProfile { + capability_ids: vec![CapabilityId::new( + ironclaw_reborn_composition::test_support::SKILL_ACTIVATE_CAPABILITY_ID, + )?], + effect_kinds: vec![ + EffectKind::DispatchCapability, + EffectKind::ReadFilesystem, + EffectKind::WriteFilesystem, + EffectKind::Network, + ], + options: HostRuntimeHarnessOptions::new( + skill_mounts()?, + Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( + true, + )?), + ) + .with_skill_activation_tenant(tenant.clone()), + network_policy_override: Some(http_test_policy()), + auto_approve_default: Some(true), + ..ToolsProfile::new( + "reborn-e2e-skill-activation-tools", + "reborn-e2e-skill-activation-user", + )? + }) +} + +/// See [`skill_activation_tools_profile`]. +pub(crate) async fn skill_activation_tools( + tenant: &TenantId, +) -> HarnessResult { + skill_activation_tools_profile(tenant)?.build().await +} diff --git a/tests/integration/support/harness/profiles/trace_commons.rs b/tests/integration/support/harness/profiles/trace_commons.rs new file mode 100644 index 00000000000..ff41e11c6fe --- /dev/null +++ b/tests/integration/support/harness/profiles/trace_commons.rs @@ -0,0 +1,52 @@ +//! trace_commons domain capability profile. + +use ironclaw_host_api::{CapabilityId, EffectKind, MountView}; +use ironclaw_host_runtime::{ + TRACE_COMMONS_CREDITS_CAPABILITY_ID, TRACE_COMMONS_ONBOARD_CAPABILITY_ID, + TRACE_COMMONS_PROFILE_SET_CAPABILITY_ID, TRACE_COMMONS_PROFILE_TOKEN_CAPABILITY_ID, + TRACE_COMMONS_STATUS_CAPABILITY_ID, +}; + +use super::super::options::{HostRuntimeHarnessOptions, ToolsProfile}; +use super::super::{HarnessResult, HostRuntimeCapabilityHarness, http_test_policy}; + +pub(crate) fn trace_commons_tools_profile() -> HarnessResult { + Ok(ToolsProfile { + capability_ids: vec![ + CapabilityId::new(TRACE_COMMONS_ONBOARD_CAPABILITY_ID)?, + CapabilityId::new(TRACE_COMMONS_STATUS_CAPABILITY_ID)?, + CapabilityId::new(TRACE_COMMONS_CREDITS_CAPABILITY_ID)?, + CapabilityId::new(TRACE_COMMONS_PROFILE_TOKEN_CAPABILITY_ID)?, + CapabilityId::new(TRACE_COMMONS_PROFILE_SET_CAPABILITY_ID)?, + ], + effect_kinds: vec![ + EffectKind::DispatchCapability, + EffectKind::ReadFilesystem, + // onboard/profile_token write device-key material and profile_token.jwt to disk; + // WriteFilesystem must stay in the allow-set or these capabilities get filtered out. + EffectKind::WriteFilesystem, + EffectKind::Network, + EffectKind::ExternalWrite, + ], + // onboard/profile_token/profile_set are PermissionMode::Ask; auto-approve is + // enabled here so the scripted run isn't gated. + options: HostRuntimeHarnessOptions::new( + MountView::default(), + Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( + true, + )?), + ), + // onboard declares EffectKind::Network, so the lease needs a non-empty network + // policy or the obligation check rejects dispatch before the consent gate runs. + network_policy_override: Some(http_test_policy()), + auto_approve_default: Some(true), + ..ToolsProfile::new( + "reborn-e2e-trace-commons-tools", + "reborn-e2e-trace-commons-user", + )? + }) +} + +pub(crate) async fn trace_commons_tools() -> HarnessResult { + trace_commons_tools_profile()?.build().await +} diff --git a/tests/integration/support/harness/profiles/trigger.rs b/tests/integration/support/harness/profiles/trigger.rs new file mode 100644 index 00000000000..5270347c4d4 --- /dev/null +++ b/tests/integration/support/harness/profiles/trigger.rs @@ -0,0 +1,38 @@ +//! trigger domain capability profile. + +use ironclaw_host_api::{CapabilityId, EffectKind, MountView}; +use ironclaw_host_runtime::{ + TRIGGER_CREATE_CAPABILITY_ID, TRIGGER_LIST_CAPABILITY_ID, TRIGGER_PAUSE_CAPABILITY_ID, + TRIGGER_REMOVE_CAPABILITY_ID, TRIGGER_RESUME_CAPABILITY_ID, +}; + +use super::super::options::{HostRuntimeHarnessOptions, ToolsProfile}; +use super::super::{HarnessResult, HostRuntimeCapabilityHarness}; + +pub(crate) fn trigger_management_tools_profile() -> HarnessResult { + Ok(ToolsProfile { + capability_ids: vec![ + CapabilityId::new(TRIGGER_CREATE_CAPABILITY_ID)?, + CapabilityId::new(TRIGGER_LIST_CAPABILITY_ID)?, + CapabilityId::new(TRIGGER_PAUSE_CAPABILITY_ID)?, + CapabilityId::new(TRIGGER_RESUME_CAPABILITY_ID)?, + CapabilityId::new(TRIGGER_REMOVE_CAPABILITY_ID)?, + ], + effect_kinds: vec![EffectKind::DispatchCapability, EffectKind::ExternalWrite], + options: HostRuntimeHarnessOptions::new( + MountView::default(), + Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( + true, + )?), + ), + auto_approve_default: Some(true), + ..ToolsProfile::new( + "reborn-e2e-trigger-management-tools", + "reborn-e2e-trigger-management-user", + )? + }) +} + +pub(crate) async fn trigger_management_tools() -> HarnessResult { + trigger_management_tools_profile()?.build().await +} diff --git a/tests/integration/support/harness/profiles/web_access.rs b/tests/integration/support/harness/profiles/web_access.rs new file mode 100644 index 00000000000..8c65dd0a17f --- /dev/null +++ b/tests/integration/support/harness/profiles/web_access.rs @@ -0,0 +1,72 @@ +//! web_access domain capability profile. + +use std::collections::HashMap; +use std::sync::{Arc, Mutex}; + +use ironclaw_extensions::ExtensionRegistry; +use ironclaw_first_party_extensions::{WEB_GET_CONTENT_CAPABILITY_ID, WEB_SEARCH_CAPABILITY_ID}; +use ironclaw_host_api::{ + CapabilityId, EffectKind, ExtensionId, MountPermissions, RuntimeKind, UserId, +}; +use ironclaw_reborn_composition::ProductLiveCapabilityIo; + +use super::super::super::harness_web_access; +use super::super::{ + HarnessResult, HostRuntimeCapabilityHarness, RecordingRuntimeHttpEgress, + host_runtime_storage_roots, workspace_mounts, +}; + +/// C-WEBACCESS: wires the real first-party web-access capabilities through production's +/// `WebAccessExecutor`; no credential-injecting authorizer needed (declares zero `runtime_credentials`). +/// +/// Exa MCP's three-leg handshake shares one URL, so responses are scripted via +/// `RecordingRuntimeHttpEgress::push_response_body` (FIFO), not the keyed matcher. +pub(crate) async fn web_access_tools() -> HarnessResult { + let (root, storage_root, workspace_root) = host_runtime_storage_roots()?; + let http_egress = Arc::new(RecordingRuntimeHttpEgress::with_body( + br#"{"accepted":true}"#.to_vec(), + )); + let mut registry = ExtensionRegistry::new(); + registry.insert(harness_web_access::web_access_extension_package()?)?; + let runtime = harness_web_access::local_dev_host_runtime_with_web_access( + storage_root, + registry, + Arc::clone(&http_egress), + )?; + let mounts = workspace_mounts(MountPermissions::read_write_list_delete())?; + Ok(HostRuntimeCapabilityHarness { + runtime, + approval_parts: None, + auto_approve_settings: None, + pending_approval_scopes: Arc::new(Mutex::new(HashMap::new())), + io: Arc::new(ProductLiveCapabilityIo::default()), + root, + workspace_root, + mounts, + capability_mount_overrides: Vec::new(), + capability_ids: vec![ + CapabilityId::new(WEB_SEARCH_CAPABILITY_ID)?, + CapabilityId::new(WEB_GET_CONTENT_CAPABILITY_ID)?, + ], + runtime_kind: RuntimeKind::FirstParty, + effect_kinds: vec![EffectKind::DispatchCapability, EffectKind::Network], + network_policy: harness_web_access::exa_mcp_test_network_policy(), + secrets: Vec::new(), + provider_id: ExtensionId::new(harness_web_access::WEB_ACCESS_PROVIDER_ID)?, + additional_provider_trust: Vec::new(), + user_id: UserId::new("reborn-itest-web-access-user")?, + invocations: Arc::new(Mutex::new(Vec::new())), + results: Arc::new(Mutex::new(Vec::new())), + http_egress: Some(http_egress), + network_egress: None, + process_port: None, + profile_filesystem: None, + project_service: None, + skill_activation_source: None, + attachment_test_support: None, + outbound_target_tools: None, + scope_capability_by_run_owner: false, + product_auth: None, + tool_permission_overrides: None, + }) +} diff --git a/tests/integration/support/harness/recorder.rs b/tests/integration/support/harness/recorder.rs new file mode 100644 index 00000000000..fec16954ea9 --- /dev/null +++ b/tests/integration/support/harness/recorder.rs @@ -0,0 +1,143 @@ +use std::{path::PathBuf, sync::Arc}; + +use ironclaw_filesystem::RootFilesystem; +use ironclaw_host_api::{CapabilityId, ResourceScope, RuntimeHttpEgressRequest}; +use ironclaw_network::NetworkHttpRequest; +use ironclaw_turns::{GateRef, run_profile::CapabilityInvocation}; + +use super::super::doubles::RecordingTestCapabilityPort; +use super::{HarnessResult, HostRuntimeCapabilityHarness}; + +#[derive(Debug, Clone)] +pub struct RecordedCapabilityResult { + pub capability_id: CapabilityId, + pub output: serde_json::Value, +} + +#[derive(Clone)] +pub(crate) enum HarnessCapabilityRecorder { + Recording(Arc), + HostRuntime(Arc), +} + +impl HarnessCapabilityRecorder { + pub(crate) fn invocations(&self) -> Vec { + match self { + Self::Recording(port) => port.invocations(), + Self::HostRuntime(harness) => harness.invocations(), + } + } + + pub(crate) fn workspace_file_path(&self, relative: &str) -> Option { + match self { + Self::Recording(_) => None, + Self::HostRuntime(harness) => Some(harness.workspace_file_path(relative)), + } + } + + pub(crate) fn capability_results(&self) -> Vec { + match self { + Self::Recording(_) => Vec::new(), + Self::HostRuntime(harness) => harness.capability_results(), + } + } + + /// E-PROFILE: local-dev memory filesystem backing the user-profile source, if any. + /// `None` for the Echo backend and HostRuntime harnesses without a profile filesystem. + pub(crate) fn profile_filesystem(&self) -> Option> { + match self { + Self::Recording(_) => None, + Self::HostRuntime(harness) => harness.profile_filesystem_for_test(), + } + } + + /// E-SKILL: the `HostSkillContextSource` to wire as the runtime's `skill_context_source`, if any. + /// `None` for the Echo backend and HostRuntime harnesses without skill activation. + pub(crate) fn skill_context_source( + &self, + ) -> Option> { + match self { + Self::Recording(_) => None, + Self::HostRuntime(harness) => harness.skill_context_source_for_test(), + } + } + + /// C-ATTACH: the attachment read port + inbound lander, if any. `None` for the + /// Echo backend and HostRuntime harnesses without a local-dev workspace filesystem. + pub(crate) fn attachment_test_support( + &self, + ) -> Option { + match self { + Self::Recording(_) => None, + Self::HostRuntime(harness) => harness.attachment_test_support_for_test(), + } + } + + pub(crate) fn runtime_http_requests(&self) -> Vec { + match self { + Self::Recording(_) => Vec::new(), + Self::HostRuntime(harness) => harness.runtime_http_requests(), + } + } + + /// See [`HostRuntimeCapabilityHarness::process_commands`]; empty for the + /// Echo recording backend. + pub(crate) fn recorded_process_commands(&self) -> Vec { + match self { + Self::Recording(_) => Vec::new(), + Self::HostRuntime(harness) => harness.process_commands(), + } + } + + pub(crate) fn network_http_requests(&self) -> Vec { + match self { + Self::Recording(_) => Vec::new(), + Self::HostRuntime(harness) => harness.network_http_requests(), + } + } + + pub(crate) async fn approve_local_dev_gate(&self, gate_ref: &GateRef) -> HarnessResult<()> { + match self { + Self::Recording(_) => { + Err("recording capability port has no local-dev approvals".into()) + } + Self::HostRuntime(harness) => harness.approve_local_dev_gate(gate_ref).await, + } + } + + pub(crate) async fn deny_local_dev_gate(&self, gate_ref: &GateRef) -> HarnessResult<()> { + match self { + Self::Recording(_) => { + Err("recording capability port has no local-dev approvals".into()) + } + Self::HostRuntime(harness) => harness.deny_local_dev_gate(gate_ref).await, + } + } + + pub(crate) async fn disable_auto_approve_for(&self, scope: ResourceScope) -> HarnessResult<()> { + match self { + Self::Recording(_) => { + Err("recording capability port has no local-dev auto-approve settings".into()) + } + Self::HostRuntime(harness) => harness.disable_global_auto_approve(scope).await, + } + } + + pub(crate) async fn enable_auto_approve_for(&self, scope: ResourceScope) -> HarnessResult<()> { + match self { + Self::Recording(_) => { + Err("recording capability port has no local-dev auto-approve settings".into()) + } + Self::HostRuntime(harness) => harness.enable_global_auto_approve(scope).await, + } + } + + pub(crate) fn approval_requests_store( + &self, + ) -> Option> { + match self { + Self::Recording(_) => None, + Self::HostRuntime(harness) => harness.approval_requests_store(), + } + } +} diff --git a/tests/support/reborn/harness_mcp.rs b/tests/integration/support/harness_mcp.rs similarity index 93% rename from tests/support/reborn/harness_mcp.rs rename to tests/integration/support/harness_mcp.rs index ccd6b27de93..e840be545a9 100644 --- a/tests/support/reborn/harness_mcp.rs +++ b/tests/integration/support/harness_mcp.rs @@ -1,15 +1,9 @@ -//! Reborn integration-test harness — mock-MCP scaffolding. -//! -//! Extracted from `harness.rs` to keep that file focused: this module owns the -//! loopback mock-MCP wiring — the real `McpRuntime` built over a loopback HTTP -//! egress, the test-only `RuntimeHttpEgress` that talks to the in-process -//! `MockMcpServer`, the mock extension package/registry, and the MCP trust + -//! network policies. -//! -//! The single entry point used by the harness is -//! `HostRuntimeCapabilityHarness::mock_mcp_tools` (in `harness.rs`), which calls -//! the `pub(super)` factories here. Everything in this module is test-only and -//! never ships. +//! Reborn integration-test harness — mock-MCP scaffolding: the real `McpRuntime` +//! built over a loopback HTTP egress, the test-only `RuntimeHttpEgress` that +//! talks to the in-process `MockMcpServer`, the mock extension package/registry, +//! and the MCP trust + network policies. Single entry point: +//! `HostRuntimeCapabilityHarness::mock_mcp_tools` (`harness.rs`), which calls +//! the `pub(super)` factories here. Test-only; never ships. #![allow(dead_code)] // Shared by staged Reborn binary-E2E validation ports. @@ -249,12 +243,9 @@ pub(super) struct LoopbackMcpRuntimeHttpEgress { impl LoopbackMcpRuntimeHttpEgress { fn new(mcp_url: &str) -> HarnessResult { - // Hermetic hardening: refuse any host other than 127.0.0.1 so a typo in - // the mock URL cannot silently turn this test egress into real external - // network I/O. Narrowed to 127.0.0.1 only (not ::1 / localhost) so the - // guard matches `mcp_loopback_network_policy()`, which also only permits - // 127.0.0.1; a caller using "localhost" would otherwise pass this guard - // then fail network authorization — a latent trap. + // Refuse any host other than 127.0.0.1 (not ::1/localhost) so a typo + // can't turn this into real network I/O, and so it matches + // `mcp_loopback_network_policy()` (which only permits 127.0.0.1). let parsed = url::Url::parse(mcp_url) .map_err(|e| format!("invalid mock MCP URL {mcp_url:?}: {e}"))?; let scheme = parsed.scheme(); diff --git a/tests/support/reborn/harness_web_access.rs b/tests/integration/support/harness_web_access.rs similarity index 70% rename from tests/support/reborn/harness_web_access.rs rename to tests/integration/support/harness_web_access.rs index 0f369553696..fdf51f3f064 100644 --- a/tests/support/reborn/harness_web_access.rs +++ b/tests/integration/support/harness_web_access.rs @@ -3,37 +3,23 @@ //! //! `web-access.search` / `web-access.get_content` are `RuntimeKind::FirstParty` //! capabilities, not MCP-extension capabilities — this module does NOT reuse -//! `harness_mcp.rs`'s `McpRuntime` scaffolding. The real dispatch logic is -//! `ironclaw_first_party_extensions::web_access::WebAccessExecutor::dispatch`, -//! which itself speaks MCP JSON-RPC by hand over three sequential -//! `RuntimeHttpEgress` calls (`initialize` → `notifications/initialized` → -//! `tools/call`) to the Exa MCP endpoint. This harness wires the real -//! production handler registration, -//! `ironclaw_reborn_composition::register_bundled_web_access_first_party_handlers` -//! (`crates/ironclaw_reborn_composition/src/web_access.rs`), instead of -//! duplicating the `FirstPartyCapabilityHandler` dispatch/error-mapping glue -//! here. Only the -//! manifest/schema loading below and the trust/network policy further down -//! remain harness-local — those are test-support concerns (reading assets -//! off disk, scoping a test-only trust policy), not business logic that -//! production owns. +//! `harness_mcp.rs`'s `McpRuntime` scaffolding. The real dispatch logic +//! (`WebAccessExecutor::dispatch`, three sequential `RuntimeHttpEgress` calls +//! to the Exa MCP endpoint) lives in production; this harness wires the real +//! production handler registration +//! (`register_bundled_web_access_first_party_handlers`) instead of +//! duplicating the dispatch/error-mapping glue. Only manifest/schema loading +//! and the trust/network policy below are harness-local test-support concerns. //! //! `web_access_extension_package()` mirrors `github.rs`'s -//! `extension_registry()` pattern exactly: it reads the REAL production -//! `crates/ironclaw_first_party_extensions/assets/web-access/manifest.toml` -//! off disk via `ExtensionManifest::parse_with_host_api_contracts` and builds -//! the package via `ExtensionPackage::from_manifest`, instead of hand- -//! authoring a synthetic manifest/schema. That construction is pure in-memory -//! — the capability schema `$ref`s are resolved later, at capability-surface- -//! descriptor build time, against the real schema files on disk mounted at -//! `/system/extensions/web-access` by `LocalDevRootMounts::web_access_assets()` -//! (`harness.rs::local_dev_root_filesystem`). +//! `extension_registry()`: reads the REAL production manifest off disk via +//! `ExtensionManifest::parse_with_host_api_contracts` rather than hand- +//! authoring a synthetic one. Schema `$ref`s resolve later against the real +//! schema files mounted at `/system/extensions/web-access`. //! -//! web-access declares zero `runtime_credentials` and never sets -//! `credential_injections`, so no credential-injecting authorizer is needed — -//! `HostRuntimeCapabilityHarness::web_access_tools` (in `harness.rs`) wires the -//! plain default `GrantAuthorizer`, the same authorizer `core_builtin_tools()` -//! uses for `builtin.http` (also a `Network`-effect capability). +//! web-access declares zero `runtime_credentials`, so no credential-injecting +//! authorizer is needed — `web_access_tools` wires the plain default +//! `GrantAuthorizer`, same as `core_builtin_tools()` for `builtin.http`. #![allow(dead_code)] // Test-only scaffolding; not every consumer exercises every helper. @@ -97,13 +83,9 @@ fn repo_root() -> &'static Path { } /// Trust policy admitting the test-only `web-access` provider as first-party, -/// kept aligned with the shape of `first_party_trust_policy()`/ -/// `github_first_party_trust_policy()` in `harness.rs` — a harness-local -/// `AdminConfig` construction with a `web-access`-specific effect list, not a -/// call into either function (the manifest, by contrast, is asset-backed — -/// see `web_access_extension_package()` above). The manifest path must match -/// the `PackageSource::LocalManifest` key the host runtime derives from -/// `web_access_extension_package()`'s root. +/// kept aligned with `first_party_trust_policy()`/`github_first_party_trust_policy()` +/// in `harness.rs`. The manifest path must match the `PackageSource::LocalManifest` +/// key the host runtime derives from `web_access_extension_package()`'s root. pub(super) fn web_access_first_party_trust_policy() -> HarnessResult { Ok(HostTrustPolicy::new(vec![Box::new( AdminConfig::with_entries(vec![AdminEntry::for_local_manifest( diff --git a/tests/support/reborn/hooks.rs b/tests/integration/support/hooks.rs similarity index 65% rename from tests/support/reborn/hooks.rs rename to tests/integration/support/hooks.rs index 10fb3ce0a58..f8d158a6b70 100644 --- a/tests/support/reborn/hooks.rs +++ b/tests/integration/support/hooks.rs @@ -1,22 +1,8 @@ //! E-HOOK-INFRA: recording hook doubles + per-run `HookDispatcherBuilderFactory` -//! builders, so C-HOOKS can observe hook dispatch on a coordinator-path turn. -//! -//! Wired into a harness/group via `with_hook_factory` / -//! `RebornIntegrationGroupBuilder::hook_dispatcher_builder_factory`. The factory -//! mints a fresh [`HookDispatcherBuilder`] per host build (per-run isolation), -//! installing recording hooks that write every fire into a shared -//! [`RecordingHookLog`]. A test reads that log back after the turn to prove the -//! wired factory actually fires hooks at the expected lifecycle points. -//! -//! These are hand-built first-party hooks (like `ironclaw_reborn`'s -//! `first_party_only_hook_factory` host-plumbing double), NOT composition -//! activation coverage: the production `build_hook_dispatcher_builder_factory` -//! ships an EMPTY first-party catalog, and its activation/projection path is -//! covered at crate tier in `ironclaw_reborn_composition::hooks::tests`. The -//! int-tier gap this fills is the end-to-end turn wire — that a wired -//! `hook_dispatcher_builder_factory` dispatches hooks through the real -//! coordinator → loop → host path — without re-authoring that crate-tier -//! activation coverage. +//! builders so C-HOOKS can observe hook dispatch on a coordinator-path turn. +//! Hand-built test hooks, not composition activation coverage — that's covered +//! at crate tier in `ironclaw_reborn_composition::hooks::tests`; this fills the +//! end-to-end coordinator → loop → host turn-wire gap only. // Shared integration-test support: not every binary that mounts the // `reborn_support` tree consumes this module, so its symbols read as dead there @@ -36,18 +22,15 @@ use ironclaw_hooks::sink::{ }; use ironclaw_reborn::loop_driver_host::{HookDispatcherBuilderFactory, RebornLoopDriverHostError}; -/// Canonical identity paths for the TEST-ONLY recording hooks. Kept distinct so -/// both a `BeforeCapability` gate hook and an `AfterModel` observer can coexist -/// in one dispatcher. +/// Distinct identity paths so a `BeforeCapability` hook and an `AfterModel` +/// observer can coexist in one dispatcher. const RECORDING_OBSERVER_PATH: &str = "tests::reborn::hooks::RecordingObserverHook"; const RECORDING_BEFORE_CAPABILITY_PATH: &str = "tests::reborn::hooks::RecordingBeforeCapabilityHook"; -/// The `&'static` reason a `DenyBeforeCapabilityHook` mints (gate-sink reasons -/// must be `&'static str`, so no `format!`-built string can leak through). +/// Gate-sink reasons must be `&'static str`, so this can't be a `format!`-built string. pub const HOOK_TEST_DENY_REASON: &str = "hook_test_deny"; -/// Shared, cloneable log of hook fires. Each installed hook double clones this -/// handle and appends an entry on every fire; a test reads the entries back. +/// Shared log of hook fires; each installed hook double appends on fire, a test reads it back. #[derive(Clone, Default)] pub struct RecordingHookLog { fires: Arc>>, @@ -79,8 +62,7 @@ impl RecordingHookLog { } } -/// A recording `AfterModel` observer. Observers cannot affect outcomes; this one -/// records the observed kind so a test can prove the observer fired. +/// Records the observed kind so a test can prove the observer fired (observers cannot affect outcomes). struct RecordingObserverHook { log: RecordingHookLog, } @@ -92,9 +74,7 @@ impl ObserverHook for RecordingObserverHook { } } -/// A recording `BeforeCapability` gate hook that records the capability name and -/// then `pass()`es (no opinion) — it observes the point without altering the -/// composed decision, so the capability still runs. +/// Records the capability name then `pass()`es (no opinion) — observes without altering the decision. struct RecordingBeforeCapabilityHook { log: RecordingHookLog, } @@ -108,9 +88,8 @@ impl PrivilegedBeforeCapabilityHook for RecordingBeforeCapabilityHook { } } -/// A `BeforeCapability` gate hook that DENIES `deny_target` (records the fire, -/// then `deny()`s) and `pass()`es every other capability. Drives the C-HOOKS -/// error path: a hook deny must block the capability without wedging the run. +/// DENIES `deny_target`, `pass()`es everything else. Drives the C-HOOKS error +/// path: a hook deny must block the capability without wedging the run. struct DenyBeforeCapabilityHook { log: RecordingHookLog, deny_target: String, @@ -135,11 +114,9 @@ fn hook_install_err(context: &str, error: impl std::fmt::Display) -> RebornLoopD } } -/// A `HookDispatcherBuilderFactory` that installs a recording `AfterModel` -/// observer plus a recording (passing) `BeforeCapability` gate hook, both -/// writing to `log`. The observer fires on every model call; the gate hook fires -/// before every capability invocation — so a turn that invokes a capability -/// records BOTH points. Mints a fresh dispatcher per call (per-run isolation). +/// Installs a recording `AfterModel` observer + passing `BeforeCapability` hook, +/// both writing to `log`; a turn invoking a capability records both points. +/// Mints a fresh dispatcher per call (per-run isolation). pub fn recording_hook_factory(log: RecordingHookLog) -> HookDispatcherBuilderFactory { Arc::new(move || { let log = log.clone(); @@ -163,9 +140,8 @@ pub fn recording_hook_factory(log: RecordingHookLog) -> HookDispatcherBuilderFac }) } -/// A `HookDispatcherBuilderFactory` that installs a recording `AfterModel` -/// observer plus a `BeforeCapability` gate hook that DENIES `deny_target`. Used -/// to prove a hook deny blocks the capability without wedging the run. +/// Installs a recording `AfterModel` observer + a `BeforeCapability` hook that +/// DENIES `deny_target`; proves a hook deny blocks the capability without wedging the run. pub fn denying_hook_factory( log: RecordingHookLog, deny_target: impl Into, diff --git a/tests/support/reborn/http_matcher.rs b/tests/integration/support/http_matcher.rs similarity index 53% rename from tests/support/reborn/http_matcher.rs rename to tests/integration/support/http_matcher.rs index 5eb4215bf27..93950493427 100644 --- a/tests/support/reborn/http_matcher.rs +++ b/tests/integration/support/http_matcher.rs @@ -1,25 +1,11 @@ -//! `ScriptedHttpResponse` — the URL/method/capability-keyed scripting layer over -//! `RecordingRuntimeHttpEgress` (design §3.6 "P1 ergonomics", §3.7 Tier-2). +//! `ScriptedHttpResponse` — the canonical URL/method/capability-keyed scripting +//! layer over `RecordingRuntimeHttpEgress`, letting a multi-step tool-HTTP flow +//! script a different body per request. First-match-wins scripted list, falling +//! back to the FIFO queue then the default body. A match can also script a +//! non-2xx status (`.with_status`, still Completed) or a runtime egress error +//! (`egress_error`, mapping to Failed/Denied — see [`ScriptedHttpOutcome`]). //! -//! Slice 2 shipped the recording egress with a single FIFO body queue. A -//! multi-step tool-HTTP flow (two `builtin.http` calls to different URLs in one -//! turn) needs a DIFFERENT scripted body per request. This module is the -//! canonical keyed-matcher API: a test scripts a list of -//! [`ScriptedHttpResponse`]s; on each `RuntimeHttpEgress::execute` the recording -//! egress returns the body of the FIRST scripted response whose key matches the -//! request, falling back to the FIFO queue, then the default body. -//! -//! A matched response is not always a `200` body: `.with_status(u16)` scripts a -//! non-2xx status (still a successful egress call — `builtin.http` surfaces it -//! as a Completed tool result carrying that status), and -//! `ScriptedHttpResponse::egress_error(url, RuntimeHttpEgressError)` scripts a -//! runtime egress failure (`Err` from `execute`, mapping to a -//! `Failed`/`Denied` capability outcome). See [`ScriptedHttpOutcome`]. -//! -//! Concrete by design (spec §3.7): the `Recording*` structs are deliberately not -//! a premature generic `Recording

` — this is written in the extractable -//! scripted-responses + captured-calls shape so a future rule-of-three lift is -//! mechanical, but no generic is introduced ahead of need. +//! `Recording*` structs are deliberately concrete, not generic, ahead of need. // Shared integration-test support: not every binary that mounts the // `reborn_support` tree consumes this module, so its symbols read as dead there @@ -28,13 +14,10 @@ use ironclaw_host_api::{RuntimeHttpEgressError, RuntimeHttpEgressRequest}; -/// What a matched [`ScriptedHttpResponse`] yields at the runtime HTTP egress -/// boundary: either a successful response (status + body) or a scripted egress -/// error. The error arm lets tests exercise the runtime error paths -/// (`policy_denied`, `response_body_limit_exceeded`, …) that the real -/// `HostHttpEgressService` produces but the recording egress otherwise cannot — -/// the egress *is* the vendor/network seam this tier fakes, so scripting an -/// error here is the same seam as scripting a body. +/// What a matched [`ScriptedHttpResponse`] yields: a successful response +/// (status + body), or a scripted egress error letting tests exercise runtime +/// error paths (`policy_denied`, `response_body_limit_exceeded`) that the real +/// `HostHttpEgressService` produces — the egress is the seam this tier fakes. #[derive(Debug, Clone)] pub enum ScriptedHttpOutcome { /// A successful egress response with the given HTTP status and body. @@ -43,10 +26,9 @@ pub enum ScriptedHttpOutcome { Error(RuntimeHttpEgressError), } -/// One scripted HTTP response keyed by request attributes. Matching is -/// first-match-wins in scripted order. A response with only a URL substring -/// matches any method/capability hitting that URL; adding [`with_method`] or -/// [`with_capability`] narrows the key. +/// One scripted HTTP response keyed by request attributes; first-match-wins in +/// scripted order. URL-substring-only matches any method/capability; narrow +/// with [`with_method`]/[`with_capability`]. /// /// [`with_method`]: ScriptedHttpResponse::with_method /// [`with_capability`]: ScriptedHttpResponse::with_capability @@ -65,12 +47,8 @@ pub struct ScriptedHttpResponse { impl ScriptedHttpResponse { /// Script a `200` response with `body` for any request whose URL contains - /// `url_substr`. Use [`with_status`] to script a non-2xx status (which the - /// `builtin.http` tool surfaces as a *successful* tool result carrying the - /// status), or [`egress_error`] for a runtime egress error. - /// - /// [`with_status`]: ScriptedHttpResponse::with_status - /// [`egress_error`]: ScriptedHttpResponse::egress_error + /// `url_substr`. Use [`with_status`](Self::with_status) for a non-2xx status + /// or [`egress_error`](Self::egress_error) for a runtime egress error. pub fn for_url(url_substr: impl Into, body: impl Into>) -> Self { Self { url_contains: url_substr.into(), @@ -83,10 +61,9 @@ impl ScriptedHttpResponse { } } - /// Script a runtime egress error for any request whose URL contains - /// `url_substr`. The `execute` boundary returns `Err(error)`, driving the - /// `builtin.http` tool's error mapping (e.g. `policy_denied` → `Denied`, - /// `response_body_limit_exceeded` → `Failed{OutputTooLarge}`). + /// Script a runtime egress error (`Err(error)` from `execute`) for requests + /// whose URL contains `url_substr` — drives `builtin.http`'s error mapping + /// (e.g. `policy_denied` → `Denied`, `response_body_limit_exceeded` → `Failed{OutputTooLarge}`). pub fn egress_error(url_substr: impl Into, error: RuntimeHttpEgressError) -> Self { Self { url_contains: url_substr.into(), @@ -96,11 +73,8 @@ impl ScriptedHttpResponse { } } - /// Script a network-layer egress failure (`RuntimeHttpEgressError::Network`) - /// with the given `reason` — e.g. `"policy_denied"`, which the tool's error - /// mapping surfaces as a `Denied` capability outcome. Thin named wrapper over - /// [`egress_error`](Self::egress_error) so test bodies select the scenario by - /// name instead of hand-building the nested error struct. + /// `RuntimeHttpEgressError::Network` with `reason` (e.g. `"policy_denied"` → + /// `Denied`). Named wrapper over [`egress_error`](Self::egress_error). pub fn network_error(url_substr: impl Into, reason: impl Into) -> Self { Self::egress_error( url_substr, @@ -112,11 +86,9 @@ impl ScriptedHttpResponse { ) } - /// Script a response-layer egress failure (`RuntimeHttpEgressError::Response`) - /// with the given `reason` — e.g. `RUNTIME_HTTP_REASON_RESPONSE_BODY_LIMIT_EXCEEDED`, - /// which the tool's error mapping surfaces as `Failed{OutputTooLarge}`. Thin - /// named wrapper over [`egress_error`](Self::egress_error) so test bodies - /// select the scenario by name instead of hand-building the nested error struct. + /// `RuntimeHttpEgressError::Response` with `reason` (e.g. + /// `RUNTIME_HTTP_REASON_RESPONSE_BODY_LIMIT_EXCEEDED` → `Failed{OutputTooLarge}`). + /// Named wrapper over [`egress_error`](Self::egress_error). pub fn response_error(url_substr: impl Into, reason: impl Into) -> Self { Self::egress_error( url_substr, @@ -129,10 +101,8 @@ impl ScriptedHttpResponse { } /// Override the HTTP status of a body response (default `200`). Panics on - /// an [`egress_error`](ScriptedHttpResponse::egress_error) response — the - /// two are mutually exclusive scripted outcomes, and silently no-op'ing - /// would leave the test exercising the egress-error path instead of the - /// status the author intended. + /// an [`egress_error`](ScriptedHttpResponse::egress_error) response — mutually + /// exclusive outcomes; silently no-op'ing would exercise the wrong path. pub fn with_status(mut self, status: u16) -> Self { match &mut self.outcome { ScriptedHttpOutcome::Body { status: s, .. } => *s = status, diff --git a/tests/support/reborn/mod.rs b/tests/integration/support/mod.rs similarity index 80% rename from tests/support/reborn/mod.rs rename to tests/integration/support/mod.rs index 76fe0447d7f..57f1b145766 100644 --- a/tests/support/reborn/mod.rs +++ b/tests/integration/support/mod.rs @@ -1,10 +1,9 @@ -pub mod approval; pub mod assertions; pub mod builder; pub mod capability_backend; pub mod comm_context; pub mod config; -pub mod delivery; +pub mod doubles; pub mod extension_surface; pub mod filesystem; pub mod github; @@ -15,16 +14,11 @@ pub mod harness_mcp; pub mod harness_web_access; pub mod hooks; pub mod http_matcher; -pub mod model_replay; -pub mod network; pub mod oauth_flow; pub mod outbound_preferences; pub mod process; pub mod product_workflow; pub mod project_service_fault; -#[allow(dead_code)] -pub mod qa_scenarios; -pub mod qa_trace; pub mod reply; pub mod scope_gateway; pub mod scripted_provider; diff --git a/tests/support/reborn/oauth_flow.rs b/tests/integration/support/oauth_flow.rs similarity index 82% rename from tests/support/reborn/oauth_flow.rs rename to tests/integration/support/oauth_flow.rs index ae3743c02d0..04c398761a3 100644 --- a/tests/support/reborn/oauth_flow.rs +++ b/tests/integration/support/oauth_flow.rs @@ -1,13 +1,7 @@ -//! Shared Google OAuth connect-flow helper for Reborn integration tests. -//! -//! Provides [`connect_google_account`], the standard OAuth connect flow that -//! drives `create_flow` → `handle_oauth_callback` → `get_account` to produce -//! a connected `CredentialAccount`. Factored out because multiple test files -//! exercise paths that require a pre-connected Google credential account. -//! -//! The function and its dependencies are gated on -//! `any(feature = "libsql", feature = "postgres")` to match the gate on -//! `OAuthProductAuthTestBundle` in `ironclaw_reborn_composition::test_support`. +//! Shared Google OAuth connect-flow helper for Reborn integration tests: +//! [`connect_google_account`] drives `create_flow` → `handle_oauth_callback` → +//! `get_account` to produce a connected `CredentialAccount`. Gated on +//! `any(feature = "libsql", feature = "postgres")` to match `OAuthProductAuthTestBundle`. // Shared support module: not every test binary that mounts the `reborn_support` // tree calls into this helper (e.g. `support_unit_tests` exercises none of it), @@ -38,13 +32,8 @@ fn hex64(fill: u8) -> String { format!("{fill:02x}").repeat(32) } -/// Run the standard Google OAuth connect flow on `bundle` and return the -/// persisted `CredentialAccount`. -/// -/// Drives `create_flow` → `handle_oauth_callback` → `get_account` using -/// `fill` as a seed byte to generate deterministic hex hashes for the OAuth -/// state, PKCE verifier, and authorization code. Call with distinct `fill` -/// values when multiple accounts are needed in the same test. +/// Runs the connect flow on `bundle`, returning the persisted `CredentialAccount`. +/// `fill` seeds deterministic hex hashes; use distinct values for multiple accounts in one test. #[cfg(any(feature = "libsql", feature = "postgres"))] pub async fn connect_google_account( bundle: &OAuthProductAuthTestBundle, diff --git a/tests/support/reborn/outbound_preferences.rs b/tests/integration/support/outbound_preferences.rs similarity index 74% rename from tests/support/reborn/outbound_preferences.rs rename to tests/integration/support/outbound_preferences.rs index 3f770727731..595b9ee262c 100644 --- a/tests/support/reborn/outbound_preferences.rs +++ b/tests/integration/support/outbound_preferences.rs @@ -1,17 +1,9 @@ -//! In-process `OutboundPreferencesProductFacade` double for the C-SYNTH outbound -//! seam. -//! -//! Substitutes ONLY at the production-wired facade trait seam the two synthetic -//! `outbound_delivery_*` capabilities consume (see -//! `ironclaw_reborn_composition::runtime::local_dev::outbound_delivery`). Holds a -//! fixed in-memory target inventory; `set_outbound_preferences` succeeds when the -//! requested target is in the inventory and returns `NotFound` otherwise — so the -//! same double drives BOTH the happy-path (`set` a known target) and the -//! `invalid_input`/`NotFound` reject route (`set` an unknown target) without a -//! per-test facade config. -//! -//! Distinct from `delivery::RecordingOutboundDeliverySink` (the final-reply -//! delivery sink); this is the delivery-*preference* facade. +//! In-process `OutboundPreferencesProductFacade` double for the C-SYNTH seam +//! (`ironclaw_reborn_composition::runtime::local_dev::outbound_delivery`). Fixed +//! in-memory inventory: succeeds for a known target, `NotFound` otherwise — one +//! double drives both the happy path and the reject route without per-test +//! config. Distinct from `delivery::RecordingOutboundDeliverySink` (the +//! final-reply delivery sink; this is the delivery-*preference* facade). #![allow(dead_code)] @@ -27,22 +19,18 @@ use ironclaw_product_workflow::{ RebornServicesErrorKind, RebornSetOutboundPreferencesRequest, WebUiAuthenticatedCaller, }; -/// `set_calls` + `last_accepted` bundled behind ONE mutex (not two) so a -/// concurrent reader can never observe a `set` that recorded its call id but -/// not yet its accepted summary, or vice versa. +/// Bundled behind ONE mutex (not two) so a reader never observes `set_calls` +/// and `last_accepted` out of sync. #[derive(Default)] struct FakeOutboundState { set_calls: Vec, last_accepted: Option, } -/// A fixed in-memory `OutboundPreferencesProductFacade` double. -/// -/// Stateful: `set_outbound_preferences` accepting a known target updates -/// `last_accepted`, and `get_outbound_preferences` reads it back — so a test -/// can prove a `set` actually persisted (not just that the setter's own -/// response echoed the request) by reading it back through a *different* -/// facade method. +/// Fixed in-memory `OutboundPreferencesProductFacade` double. Stateful: +/// `set_outbound_preferences` updates `last_accepted`, and +/// `get_outbound_preferences` reads it back — proves a `set` persisted via a +/// different facade method, not just an echo. pub(crate) struct FakeOutboundPreferencesFacade { targets: Vec, state: Mutex, @@ -61,9 +49,8 @@ impl FakeOutboundPreferencesFacade { }) } - /// The target ids passed to `set_outbound_preferences`, in call order — - /// read-back that a `Completed` outcome actually reached the facade seam (a - /// no-op set that still fabricated a success payload would leave this empty). + /// Target ids passed to `set_outbound_preferences`, in call order — proves a + /// `Completed` outcome reached the facade (a no-op set would leave this empty). pub(crate) fn recorded_set_target_ids(&self) -> Vec { self.state .lock() @@ -161,10 +148,8 @@ fn target_option(target_id: &str, display_name: &str) -> RebornOutboundDeliveryT } } -/// The `NotFound` the production handler maps to -/// `CapabilityOutcome::Failed(InvalidInput)` ("outbound delivery target is not -/// available") — see `OutboundDeliveryTargetSetHandler`'s `NotFound` → -/// `CapabilityFailureKind::InvalidInput` arm in +/// The `NotFound` the production handler maps to `Failed(InvalidInput)` — see +/// `OutboundDeliveryTargetSetHandler`'s `NotFound` arm in /// `runtime/local_dev/outbound_delivery.rs`. fn target_not_found() -> RebornServicesError { RebornServicesError { diff --git a/tests/support/reborn/process.rs b/tests/integration/support/process.rs similarity index 73% rename from tests/support/reborn/process.rs rename to tests/integration/support/process.rs index 0eaa305e59d..bd3a0e12b7b 100644 --- a/tests/support/reborn/process.rs +++ b/tests/integration/support/process.rs @@ -1,10 +1,8 @@ -//! Inert recording process port for the integration-test harness (slice 5). -//! -//! `RecordingProcessPort` implements `RuntimeProcessPort` but never spawns a -//! real OS process. Every `run_command` call records the command string and -//! returns a benign success response (exit 0, empty stdout/stderr). The -//! recorder is the default for the `BuiltinHttpTools` capability backend so -//! that `builtin.shell` test turns are safe to run without any system setup. +//! Inert recording process port for the integration-test harness. +//! `RecordingProcessPort` implements `RuntimeProcessPort` without spawning a +//! real OS process: every `run_command` call is recorded and returns a benign +//! success (exit 0, empty output) by default — the default port for +//! `BuiltinHttpTools` so `builtin.shell` test turns are safe without system setup. // Not every test binary that mounts this support tree exercises the recording // process port — mirrors the `#![allow(dead_code)]` used in sibling modules. @@ -18,10 +16,9 @@ use ironclaw_host_runtime::{ CommandExecutionOutput, CommandExecutionRequest, RuntimeProcessError, RuntimeProcessPort, }; -/// A scripted `run_command` result for the recording process port. Sticky: -/// once set, EVERY `run_command` call returns it (after recording the command), -/// so a retryable failure surfaces consistently across the loop's retry budget -/// instead of being consumed once from a FIFO queue. +/// Sticky scripted `run_command` result: once set, EVERY subsequent call +/// returns it (after recording the command) — a retryable failure surfaces +/// consistently across the loop's retry budget, not just once from a FIFO queue. #[derive(Debug, Clone)] pub enum ScriptedProcessResult { /// Return a benign success with this exit code (non-zero drives the tool's @@ -33,12 +30,8 @@ pub enum ScriptedProcessResult { Timeout, } -/// Inert process port: records every `CommandExecutionRequest` and, by default, -/// returns a benign success (`exit_code = 0`, empty stdout/stderr) without -/// spawning any OS process. A scripted result (via [`set_scripted`]) overrides -/// the default for every subsequent call. -/// -/// [`set_scripted`]: RecordingProcessPort::set_scripted +/// Records every `CommandExecutionRequest`; returns a benign success by default +/// (no OS process spawned). [`set_scripted`](Self::set_scripted) overrides it. #[derive(Debug, Clone, Default)] pub struct RecordingProcessPort { commands: Arc>>, diff --git a/tests/support/reborn/product_workflow.rs b/tests/integration/support/product_workflow.rs similarity index 100% rename from tests/support/reborn/product_workflow.rs rename to tests/integration/support/product_workflow.rs diff --git a/tests/support/reborn/project_service_fault.rs b/tests/integration/support/project_service_fault.rs similarity index 53% rename from tests/support/reborn/project_service_fault.rs rename to tests/integration/support/project_service_fault.rs index 6fe9de01f24..0680623bc06 100644 --- a/tests/support/reborn/project_service_fault.rs +++ b/tests/integration/support/project_service_fault.rs @@ -1,42 +1,13 @@ //! In-process `ProjectService` fault-injecting decorator for the C-SYNTH -//! `project_create` seam. +//! `project_create` seam: wraps the real inner `ProjectService`; a sentinel +//! `create_project` name triggers a scripted fault, any other call passes +//! straight through. //! -//! Substitutes ONLY at the production-wired `Arc` seam the -//! `builtin.project_create` synthetic capability consumes -//! (`wrap_project_create_capability_for_test` -> -//! `ProjectCreateHandler::project_service`, see -//! `crates/ironclaw_reborn_composition/src/runtime/local_dev/project_create.rs`). -//! Delegates every call to a REAL inner `ProjectService` (the same -//! `local_dev_project_service_for_test()` instance every other project-tools -//! harness uses) except `create_project`, where a caller-chosen sentinel name -//! triggers a scripted `ProjectServiceError` instead of reaching the real -//! store — the same "double at the trait seam production already uses" -//! pattern as `FakeOutboundPreferencesFacade` (`outbound_preferences.rs`) and -//! `ScriptedHttpResponse::egress_error`. Any other name passes straight -//! through to the real service, so the same double can drive both a -//! fault-injection arm and an ordinary happy-path project_create in the same -//! group. -//! -//! Deliberately forces `ProjectServiceError::Denied` (`PolicyDenied`), NOT -//! `Unavailable`/`Internal`: those two `CapabilityFailureKind`s route through -//! `DefaultRecoveryStrategy`'s capability-retry branch -//! (`crates/ironclaw_agent_loop/src/strategies/recovery.rs`), which -//! re-dispatches via `capability_invocation_from_candidate` reusing the -//! ORIGINAL `input_ref`. For a provider-tool-call-originated invocation under -//! local-dev composition, that retry hits a real, independently confirmed -//! production bug — `LocalDevCapabilityIo::resolve_capability_input` -//! (`crates/ironclaw_reborn_composition/src/runtime/local_dev.rs`) rejects the -//! SAME `input_ref` on the retry with `InvalidInvocation`/"capability input -//! ref was not staged for this loop run" (the first attempt's input resolves -//! through a different, staging-bypassing path — see -//! `LocalDevCapabilityIo`'s own doc comment), collapsing the documented -//! "retry twice, then a model-visible `Failed`" contract into an immediate -//! terminal `driver_unavailable`. See issue #5608 for the full -//! repro. `Denied` avoids that retry path entirely -//! (`capability_error_is_model_visible_tool_failure` surfaces `PolicyDenied` -//! straight to the model on the FIRST attempt), so this double still proves a -//! genuine, distinct `project_service_outcome` arm end-to-end without -//! tripping the unrelated retry bug. +//! Deliberately forces `ProjectServiceError::Denied`, not +//! `Unavailable`/`Internal`: those retry via `DefaultRecoveryStrategy` and hit +//! a real `LocalDevCapabilityIo` input-ref restaging bug (issue #5608), +//! collapsing the retry contract into an immediate `driver_unavailable`. +//! `Denied` surfaces to the model on the first attempt, avoiding the bug. #![allow(dead_code)] @@ -51,16 +22,12 @@ use ironclaw_product_workflow::{ RebornRemoveMemberRequest, RebornUpdateMemberRoleRequest, RebornUpdateProjectRequest, }; -/// Sentinel `create_project` name that triggers the injected fault instead of -/// reaching the real project store. Kept distinct from ordinary test project -/// names (`"My Project"` etc.) so the same double never accidentally -/// intercepts an unrelated happy-path create. +/// Sentinel `create_project` name that triggers the injected fault; distinct +/// from ordinary test project names so it never intercepts a real create. pub(crate) const FAULT_INJECT_DENIED_PROJECT_NAME: &str = "FAULT_INJECT_DENIED_PROJECT"; -/// Decorator around a real `Arc` that forces -/// `ProjectServiceError::Denied` on a `create_project` call naming the -/// sentinel, and delegates everything else (including non-sentinel -/// `create_project` calls) to the wrapped real service. +/// Decorator around a real `Arc`: forces `Denied` on the +/// sentinel `create_project` name, delegates everything else to the inner service. pub(crate) struct FaultInjectingProjectService { inner: Arc, } diff --git a/tests/integration/support/reply.rs b/tests/integration/support/reply.rs new file mode 100644 index 00000000000..d17f50f4333 --- /dev/null +++ b/tests/integration/support/reply.rs @@ -0,0 +1,97 @@ +//! `RebornScriptedReply` — terse façade for scripting one model turn in a +//! Reborn integration test; each reply maps 1:1 to a `TraceStep`. Raw +//! `TraceStep`/`LlmTrace`/`TraceResponse` construction is forbidden in new +//! Reborn integration tests (design §4.2) — use this. + +// dead_code: `support_unit_tests.rs` mounts `reborn_support` without +// consuming this module, so symbols read unused there under `-D warnings` +// (matches sibling support modules). +#![allow(dead_code)] + +use crate::support::trace_llm::{TraceResponse, TraceStep, TraceToolCall}; +use std::sync::atomic::{AtomicU64, Ordering}; + +static NEXT_TOOL_CALL_ID: AtomicU64 = AtomicU64::new(1); + +/// One scripted model turn. +pub struct RebornScriptedReply { + step: TraceStep, +} + +impl RebornScriptedReply { + /// A plain assistant text reply. + pub fn text(content: impl Into) -> Self { + Self { + step: TraceStep { + request_hint: None, + response: TraceResponse::Text { + content: content.into(), + input_tokens: 0, + output_tokens: 0, + }, + expected_tool_results: Vec::new(), + }, + } + } + + /// Scripts one model tool-call turn (CapabilityId, e.g. `"builtin.http"`). + /// Applies the `'.' → "__"` encoding `ProviderToolName::new` requires (it + /// rejects dots); `model_replay.rs`'s `trace_provider_tool_name` has an + /// identical, independent encoder for the fixture-replay seam — keep both + /// in sync if the mapping changes. NOT collision-safe (`.` vs `__` can + /// collide) and not length-truncated — fine for single-capability tests + /// only. The `id` auto-fills from a process-scoped counter, canonicalized + /// per trace by `scripted_provider::scripted_trace_llm`. + pub fn tool_call(capability_id: &str, arguments: serde_json::Value) -> Self { + let name = capability_id.replace('.', "__"); + let id = format!("call-{}", NEXT_TOOL_CALL_ID.fetch_add(1, Ordering::Relaxed)); + Self { + step: TraceStep { + request_hint: None, + response: TraceResponse::ToolCalls { + tool_calls: vec![TraceToolCall { + id, + name, + arguments, + }], + input_tokens: 0, + output_tokens: 0, + }, + expected_tool_results: Vec::new(), + }, + } + } + + /// Scripts one model turn carrying MULTIPLE tool calls (a "parallel" + /// tool-calls turn from ONE model call). Each pair gets `tool_call`'s + /// same encoding/id treatment. Still counts as exactly ONE script entry — + /// the caller must follow with one more entry (the post-execution model + /// call reacting to the tool results). + pub fn tool_calls<'a>(calls: impl IntoIterator) -> Self { + let tool_calls = calls + .into_iter() + .map(|(capability_id, arguments)| TraceToolCall { + id: format!("call-{}", NEXT_TOOL_CALL_ID.fetch_add(1, Ordering::Relaxed)), + name: capability_id.replace('.', "__"), + arguments, + }) + .collect(); + Self { + step: TraceStep { + request_hint: None, + response: TraceResponse::ToolCalls { + tool_calls, + input_tokens: 0, + output_tokens: 0, + }, + expected_tool_results: Vec::new(), + }, + } + } + + /// Consume into the underlying replay step (crate-internal seam used by + /// `scripted_provider::scripted_trace_llm`). + pub(crate) fn into_step(self) -> TraceStep { + self.step + } +} diff --git a/tests/support/reborn/scope_gateway.rs b/tests/integration/support/scope_gateway.rs similarity index 61% rename from tests/support/reborn/scope_gateway.rs rename to tests/integration/support/scope_gateway.rs index 96882094397..a89056345db 100644 --- a/tests/support/reborn/scope_gateway.rs +++ b/tests/integration/support/scope_gateway.rs @@ -1,21 +1,9 @@ -//! Scope-routing gateway for Reborn group integration tests. -//! -//! `ScopeRegistryGateway` multiplexes model calls to per-thread scripted -//! gateways by `TurnScope`. The loop-driver host calls -//! `resolve_for_scope(&scope)` at host-construction time (not on the model -//! hot path) and wraps the returned gateway in -//! `ThreadResolvingLoopModelGateway`. The registry's own `stream_model` is -//! therefore **never called** when routing succeeds; it exists only to satisfy -//! the trait contract and fails **loudly** so a routing miss (unregistered -//! scope) surfaces as `ConfigurationError` rather than masking as the original -//! flake. -//! -//! ## Invariant -//! This dispatcher sits at the `HostManagedModelGateway` seam but routes to -//! REAL `LlmProviderModelGateway` instances over the `ironclaw_llm` decorator -//! chain. The single-fake-at-the-vendor-SDK-seam invariant (CLAUDE.md §5–8, -//! §28) is preserved: the only fake is the `TraceLlm` at the bottom of each -//! registered gateway's chain, not this dispatcher. +//! Scope-routing gateway for Reborn group integration tests: routes model calls to +//! per-thread scripted gateways by `TurnScope` via `resolve_for_scope`, called at +//! host-construction time (off the model hot path). Its own `stream_model` is +//! unreachable on success — fails loudly with `ConfigurationError` on an unregistered +//! scope, preserving the single-fake-at-vendor-SDK-seam invariant (CLAUDE.md §5-8, §28): +//! the only fake is the `TraceLlm` at the bottom of each registered gateway's chain. // Shared by all group test binaries; symbols read as dead when a binary // does not exercise every variant. @@ -33,34 +21,22 @@ use ironclaw_turns::TurnScope; /// Scope-keyed gateway registry for Reborn group integration tests. /// -/// Call [`register`](Self::register) (with `&self`) before submitting any -/// turns; the loop-driver calls -/// [`resolve_for_scope`][HostManagedModelGateway::resolve_for_scope] at -/// host-construction time to obtain the per-thread scripted gateway. -/// -/// The registry's own `stream_model` is a deliberate sentinel: it always -/// returns [`HostManagedModelErrorKind::ConfigurationError`] so a routing -/// miss fails legibly and cannot be confused with `TraceLlm` deque exhaustion -/// (which surfaces as `Unavailable`) or `driver_protocol_violation`. +/// `stream_model` is a sentinel: always returns `ConfigurationError` on a routing miss so +/// it can't be confused with `TraceLlm` exhaustion (`Unavailable`) or `driver_protocol_violation`. #[derive(Default)] pub struct ScopeRegistryGateway { map: Mutex>>, } impl ScopeRegistryGateway { - /// Construct an empty registry. pub fn new() -> Self { Self { map: Mutex::new(HashMap::new()), } } - /// Register `gateway` for `scope`. - /// - /// Interior-mutable (`&self`) so callers can hold an - /// `Arc` and register threads one-by-one before any - /// turn is submitted. This is called from `.thread(conv).script().build()` - /// before any turn reaches the scheduler. + /// Register `gateway` for `scope`. `&self` (not `&mut self`) so callers holding an + /// `Arc` can register threads one-by-one before any turn is submitted. pub fn register(&self, scope: TurnScope, gateway: Arc) { let replaced = self .map @@ -76,14 +52,9 @@ impl ScopeRegistryGateway { #[async_trait] impl HostManagedModelGateway for ScopeRegistryGateway { - /// Sentinel — never reached when routing succeeds. - /// - /// The loop-driver host resolves the per-scope gateway via - /// [`resolve_for_scope`](Self::resolve_for_scope) and calls `stream_model` - /// on **that** gateway. If execution reaches here it means a scope was not - /// registered — failing with `ConfigurationError` makes the miss impossible - /// to confuse with model exhaustion (`Unavailable`) or - /// `driver_protocol_violation`. + /// Sentinel — never reached when routing succeeds; execution here means a scope was not + /// registered. Fails with `ConfigurationError`, distinct from `Unavailable` (model + /// exhaustion) or `driver_protocol_violation`. async fn stream_model( &self, request: HostManagedModelRequest, @@ -101,12 +72,9 @@ impl HostManagedModelGateway for ScopeRegistryGateway { )) } - /// Return the gateway registered for `scope`, or `None` if no match. - /// - /// The loop-driver host calls this at host-construction time (off the - /// model hot path). `None` → the host falls back to `Arc::clone(self)`, - /// causing the next `stream_model` call to emit the sentinel error above, - /// making the routing miss immediately visible. + /// Returns the gateway registered for `scope`, or `None`. Called at host-construction + /// time (off the model hot path); `None` makes the host fall back to `Arc::clone(self)`, + /// so the next `stream_model` call emits the sentinel error above. fn resolve_for_scope(&self, scope: &TurnScope) -> Option> { self.map .lock() @@ -120,9 +88,7 @@ impl HostManagedModelGateway for ScopeRegistryGateway { mod tests { use super::*; - /// Trivial gateway stub: another `ScopeRegistryGateway` satisfies the trait - /// bound and lets us check `Arc::ptr_eq` identity without touching any model - /// logic. + /// Stub gateway (reuses `ScopeRegistryGateway`) for `Arc::ptr_eq` identity checks. fn stub_gateway() -> Arc { Arc::new(ScopeRegistryGateway::new()) } @@ -136,10 +102,8 @@ mod tests { ) } - /// Mutation guard (a): if `resolve_for_scope` is changed to always return - /// `None`, this test goes RED on the `is_some()` assertion. - /// Mutation guard (b): if it returns the wrong entry or ignores scope, - /// `two_scopes_route_to_distinct_gateways` goes RED on `ptr_eq`. + /// Mutation guards: always-None `resolve_for_scope` fails `is_some()` here; wrong-entry + /// resolution fails `ptr_eq` in `two_scopes_route_to_distinct_gateways`. #[test] fn registered_scope_returns_some() { let registry = ScopeRegistryGateway::new(); @@ -169,9 +133,8 @@ mod tests { assert!(resolved.is_none(), "unregistered scope must return None"); } - /// Fail-loud guard: re-registering the same `TurnScope` must panic rather - /// than silently re-pointing the first registration's callers at the - /// second gateway (the bug this assert prevents — see module docs). + /// Fail-loud guard: re-registering a scope must panic, not silently repoint the first + /// registration's callers at the second gateway. #[test] #[should_panic(expected = "duplicate scope registration")] fn duplicate_register_for_same_scope_panics() { @@ -184,10 +147,8 @@ mod tests { registry.register(scope, gw_second); } - /// Construct a minimal `HostManagedModelRequest` for exercising the - /// `stream_model` sentinel directly; field contents beyond `run_id`/ - /// `turn_id` are irrelevant since the sentinel never inspects them for - /// anything but the error message. + /// Minimal request for exercising the `stream_model` sentinel; only `run_id`/`turn_id` + /// matter (the sentinel only echoes them into the error message). fn make_request() -> HostManagedModelRequest { HostManagedModelRequest { model_profile_id: ironclaw_turns::run_profile::ModelProfileId::new("interactive_model") @@ -200,13 +161,8 @@ mod tests { } } - /// De-mask guard: a routing miss (no scope registered) must fail as the - /// distinct `ConfigurationError` sentinel, not silently resemble model - /// exhaustion (`Unavailable`) or `driver_protocol_violation`. If this - /// sentinel regresses to a different kind or loses its diagnostic text, - /// a real routing miss in a group scenario would become ambiguous with - /// those other failure categories — exactly the masking this gateway - /// exists to prevent (see module docs). + /// De-mask guard: a routing miss must fail as the distinct `ConfigurationError` sentinel, + /// not resemble model exhaustion (`Unavailable`) or `driver_protocol_violation`. #[tokio::test] async fn stream_model_sentinel_reports_configuration_error_on_routing_miss() { let registry = ScopeRegistryGateway::new(); diff --git a/tests/support/reborn/scripted_provider.rs b/tests/integration/support/scripted_provider.rs similarity index 63% rename from tests/support/reborn/scripted_provider.rs rename to tests/integration/support/scripted_provider.rs index c427d476454..9f7cfe98f2d 100644 --- a/tests/support/reborn/scripted_provider.rs +++ b/tests/integration/support/scripted_provider.rs @@ -1,8 +1,7 @@ -//! Scripted raw-provider seam for Reborn integration tests. Reuses `TraceLlm`'s -//! replay engine (no new provider): builds an in-memory `LlmTrace` from the -//! `RebornScriptedReply` façade, canonicalizes harness-local tool-call ids per -//! trace, and returns a `TraceLlm` to sit at the bottom of the real -//! `ironclaw_llm` decorator chain (design §3.1/§3.3). +//! Scripted raw-provider seam for Reborn integration tests. Reuses `TraceLlm`'s replay +//! engine: builds an in-memory `LlmTrace` from `RebornScriptedReply`, canonicalizes +//! tool-call ids, and sits at the bottom of the real `ironclaw_llm` decorator chain +//! (design §3.1/§3.3). // The parking provider (`ParkingModelGate`/`ParkingLlm`) is consumed only by the // `reborn_integration_cancel` test binary, so it reads as dead in the @@ -85,23 +84,9 @@ impl ScriptedToolCallIds { // Parking model provider (E-GATEWAY seam) — mid-turn cancel coverage. // --------------------------------------------------------------------------- -/// Synchronization handle for a [`ParkingLlm`]: the test waits until the model -/// call parks, then releases it. Cloneable (shares one [`ParkingState`] over an -/// `Arc`), so the test keeps a handle while a clone lives inside the provider. -/// -/// Uses `oneshot` channels rather than `Notify` so signalling is lost-wakeup -/// free: `oneshot::Sender::send` stores the value regardless of whether the -/// receiver is already awaiting, so `release()` may run before or after the -/// provider reaches its `await` without racing. The first model call parks; a -/// second call (if any) delegates immediately. -/// -/// The `take()`-based single-shot design is deliberate and idempotent: a second -/// `park()` (e.g. from a retry/failover hop in the real `ironclaw_llm` decorator -/// chain this provider sits under) finds its channel already consumed and -/// returns immediately rather than blocking. A plain `Notify` pair would -/// *deadlock* that second `park()` — `Notify` stores only one permit, so once -/// the single `release` permit is consumed there is nothing left to wake a -/// second waiter. +/// Synchronization handle for a [`ParkingLlm`]: the test waits until the model call parks, +/// then releases it. Uses `oneshot` (not `Notify`) so release-before-park and a second +/// `park()` call are both lost-wakeup-free — `Notify`'s single permit would deadlock the latter. #[derive(Clone)] pub struct ParkingModelGate(Arc); @@ -134,8 +119,7 @@ impl ParkingModelGate { } } - /// Release the parked model call so it delegates to the inner trace and the - /// turn proceeds. + /// Release the parked model call so it delegates to the inner trace. pub fn release(&self) { if let Some(tx) = lock(&self.0.release_tx).take() { let _ = tx.send(()); @@ -165,21 +149,16 @@ fn lock(m: &Mutex) -> std::sync::MutexGuard<'_, T> { m.lock().unwrap_or_else(std::sync::PoisonError::into_inner) } -/// A raw `LlmProvider` that parks the first model call until the test releases -/// it, then delegates to the inner scripted [`TraceLlm`]. Sits at the same -/// vendor-SDK seam `scripted_trace_llm` fills, preserving the tier's -/// single-fake invariant (the real `ironclaw_llm` decorator chain still runs -/// on top). +/// A raw `LlmProvider` that parks the first model call until the test releases it, then +/// delegates to the inner scripted `TraceLlm`. Same vendor-SDK seam as `scripted_trace_llm`, +/// preserving the single-fake invariant. pub struct ParkingLlm { inner: Arc, gate: ParkingModelGate, } -/// Build a parking provider wrapping the already-built `inner` trace, released -/// via `gate`. Takes an `Arc` (rather than building its own from raw -/// replies) so the caller retains the SAME trace handle the parked provider -/// replays through — parking mode is only a wrapper around the same scripted -/// provider, not a separate trace. +/// Wraps `inner` (an `Arc`, not fresh-built) so the caller retains the same +/// trace handle the parked provider replays through. pub fn parking_trace_llm(gate: ParkingModelGate, inner: Arc) -> ParkingLlm { ParkingLlm { inner, gate } } @@ -213,16 +192,10 @@ impl LlmProvider for ParkingLlm { // category coverage (C-ERRORS). // --------------------------------------------------------------------------- -/// A raw `LlmProvider` that always fails with a fixed, NON-retryable -/// `LlmError::ContextLengthExceeded`. Deliberately NOT the `LlmError::RequestFailed` -/// a naturally-exhausted `TraceLlm` returns (see `next_step` above) — `RequestFailed` -/// IS retryable (`ironclaw_llm::retry::is_retryable`), so scripting it would drive -/// several seconds of real exponential backoff (1s/2s/4s) before the run finally -/// failed. `ContextLengthExceeded` is excluded from `is_retryable`, so the run -/// fails on the first model call — fast and deterministic. Sits at the same -/// vendor-SDK seam `scripted_trace_llm`/`ParkingLlm` fill; the real `ironclaw_llm` -/// decorator chain still runs on top, so this proves the chain's non-retryable-error -/// mapping through to a terminal `TurnStatus::Failed`, not just the seam itself. +/// A raw `LlmProvider` that always fails with non-retryable `LlmError::ContextLengthExceeded` +/// — deliberately not `RequestFailed` (retryable, would add several seconds of real backoff). +/// Same vendor-SDK seam as `scripted_trace_llm`/`ParkingLlm`; proves the real decorator +/// chain's non-retryable-error mapping through to `TurnStatus::Failed`. pub struct ErrLlm; #[async_trait] @@ -253,30 +226,14 @@ mod tests { use super::*; - /// Enforces both concurrency guarantees documented on `ParkingModelGate` - /// (module docs above) that the committed `reborn_integration_cancel` - /// integration test does not exercise directly, since it always calls - /// `release()` *after* the provider has already parked. - /// - /// Guarantee 1 — release-before-await ordering: `release()` sends on a - /// `oneshot::Sender` and `oneshot::Sender::send` buffers the value - /// regardless of whether a receiver is already awaiting, so calling - /// `release()` before `park()` has ever run must not be a lost wakeup — - /// the eventual `park()` call must still resolve promptly rather than - /// hanging forever waiting on `release_rx`. - /// - /// Guarantee 2 — second call does not block: `parked_tx`/`release_tx` are - /// `Mutex>` `take()`-based single-shot channels, so once the - /// first `park()` call has consumed them, a second `park()` call (e.g. - /// simulating a retry/failover hop in the real decorator chain) must - /// return immediately instead of blocking on an already-consumed - /// channel. + /// Covers two `ParkingModelGate` guarantees the committed cancel test doesn't exercise + /// (it always releases after parking): (1) release-before-park is not a lost wakeup; + /// (2) a second `park()` call after the channels are consumed returns immediately. #[tokio::test] async fn parking_llm_release_before_await_and_second_call_do_not_block() { let gate = ParkingModelGate::new(); - // Guarantee 1: release fires before any `park()` call exists to - // receive it. + // Guarantee 1: release fires before park() exists to receive it. gate.release(); tokio::time::timeout(Duration::from_secs(5), gate.park()) .await @@ -285,9 +242,7 @@ mod tests { (oneshot send is lost-wakeup free)", ); - // Guarantee 2: a second park() call, after the first full - // park+release cycle already consumed both channels, must not - // block. + // Guarantee 2: second park() call, after both channels are consumed. tokio::time::timeout(Duration::from_secs(5), gate.park()) .await .expect("second park() call must return immediately, not block"); diff --git a/tests/support/reborn/session_thread.rs b/tests/integration/support/session_thread.rs similarity index 56% rename from tests/support/reborn/session_thread.rs rename to tests/integration/support/session_thread.rs index 066019c164f..83ecaf750ad 100644 --- a/tests/support/reborn/session_thread.rs +++ b/tests/integration/support/session_thread.rs @@ -22,21 +22,12 @@ pub enum RebornThreadHarnessError { MissingFinalReply(String), } -/// Thin harness over a `FilesystemSessionThreadService` for asserting thread -/// history in integration and binary-tier tests. +/// Thin harness over `FilesystemSessionThreadService` for asserting thread history. /// -/// The type parameter `F` defaults to `InMemoryBackend` so that all existing -/// callers that write `RebornThreadHarness` (no type parameter) continue to -/// compile as `RebornThreadHarness` without modification. -/// `InMemoryBackend` (not `LocalFilesystem`) is the default because it is -/// CAS-capable and models the production database-backed filesystem that these -/// stores are mounted on in real deployments; `LocalFilesystem` is a byte-only -/// backend that production never uses for record-shaped CAS stores (see -/// e3e155803). -/// -/// The integration tier uses `RebornThreadHarness` via -/// `filesystem_shared_composite`, mounting the thread service directly on the -/// per-`build()` production-path composite (threads at `/tenants/{t}/users/{u}/threads`). +/// Defaults to `InMemoryBackend` (CAS-capable, models the production DB-backed filesystem) +/// rather than the byte-only `LocalFilesystem` (see e3e155803). The integration tier uses +/// `RebornThreadHarness` via `filesystem_shared_composite`, mounted +/// on the per-`build()` production-path composite. pub struct RebornThreadHarness where F: RootFilesystem, @@ -44,19 +35,14 @@ where pub scope: ThreadScope, pub service: Arc>, backend: Arc, - /// Backing `TempDir` to keep alive for tiers whose backend persists to - /// disk (e.g. the `CompositeRootFilesystem` integration tier). `None` for - /// in-memory tiers, which have nothing to keep alive. + /// Backing `TempDir` for disk-persisting tiers (e.g. `CompositeRootFilesystem`); + /// `None` for in-memory tiers. root: Option>, - /// Path prefix inserted before `/tenants/...` when constructing the thread - /// scoped filesystem. The default `InMemoryBackend` tier uses `"/engine"` - /// (preserving the historical `/engine/tenants/...` layout); the - /// integration-tier `CompositeRootFilesystem` harness uses `""` so threads - /// land at `/tenants/...` inside the production composite. + /// Path prefix before `/tenants/...`. Default `InMemoryBackend` tier uses `"/engine"`; + /// integration-tier `CompositeRootFilesystem` uses `""` (production composite layout). root_prefix: String, } -/// Shared methods: work for any `F: RootFilesystem`. impl RebornThreadHarness { pub fn reopened(&self) -> Result { let scoped = scoped_threads_fs_at(&self.root_prefix, Arc::clone(&self.backend))?; @@ -120,8 +106,7 @@ impl RebornThreadHarness { } } -/// `InMemoryBackend`-specific constructors (default tier); see the type doc -/// above for why this is the default. +/// `InMemoryBackend`-specific constructors (default tier). impl RebornThreadHarness { pub fn filesystem_temp(scope: ThreadScope) -> Result { let backend = Arc::new(InMemoryBackend::new()); @@ -146,15 +131,10 @@ impl RebornThreadHarness { /// `CompositeRootFilesystem`-specific constructor (integration tier). impl RebornThreadHarness { - /// Create a harness backed by a shared production-path composite. - /// - /// Threads land at `/tenants/{tenant}/users/{user}/threads` inside the - /// composite (no `/engine` prefix), so they are visible through the - /// `/tenants` mount that `mount_local_dev_database_roots` installs. - /// `root` keeps the composite's `TempDir` alive; the same `Arc` is also - /// held by `GroupSharedStorage::turn_root` so on-disk libsql data persists - /// across calls to `reopened()`, which rebuilds the scoped service from - /// the same composite backend. + /// Harness backed by a shared production-path composite; threads land at + /// `/tenants/{tenant}/users/{user}/threads` (visible via `mount_local_dev_database_roots`). + /// `root` (also held by `GroupSharedStorage::turn_root`) keeps the `TempDir` alive so + /// on-disk libsql data persists across `reopened()` calls. pub fn filesystem_shared_composite( scope: ThreadScope, backend: Arc, @@ -174,23 +154,13 @@ impl RebornThreadHarness { /// Build the scoped thread filesystem. /// -/// The `/threads` mount is resolved **per filesystem operation** from that -/// operation's own `ResourceScope` (production's `invocation_mount_view` -/// shape: `ScopedFilesystem::new` + resolver) — NOT fixed once at -/// construction. `FilesystemSessionThreadService` derives each op's -/// `ResourceScope` from the request's `ThreadScope` (owner included, via -/// `ThreadScope::to_resource_scope`), so ONE service instance serves every -/// owner's `/tenants/{t}/users/{owner}/threads` subtree. This is what lets a -/// group's ONE shared runtime resolve a second actor's thread (issue #5479): -/// the runtime's per-turn owner rewrite -/// (`ThreadScopeResolver::resolve_for_turn`) now lands on the right physical -/// root instead of a mount pinned to the group's canonical actor. For any -/// single owner the resolved path is byte-identical to the previous fixed -/// view, so single-actor tests are unaffected. +/// The `/threads` mount resolves **per filesystem operation** from that op's `ResourceScope` +/// (via `ThreadScope::to_resource_scope`), not fixed at construction — so one service instance +/// serves every owner's subtree, letting a group's shared runtime resolve a second actor's +/// thread (issue #5479). Single-owner tests are unaffected (path is byte-identical). /// -/// `root_prefix` is prepended before `/tenants/...`: -/// - Default `InMemoryBackend` tier: `"/engine"` → `/engine/tenants/{t}/users/{u}/threads` -/// - Integration tier: `""` → `/tenants/{t}/users/{u}/threads` +/// `root_prefix` precedes `/tenants/...`: `"/engine"` for the default `InMemoryBackend` tier, +/// `""` for the integration tier. fn scoped_threads_fs_at( root_prefix: &str, backend: Arc, @@ -206,15 +176,11 @@ where /// The single `/threads` mount grant for one operation's `ResourceScope`. /// -/// System-scoped operations (e.g. `find_idempotency_record`, which routes -/// through `ResourceScope::system()`) and owner-less thread scopes carry the -/// `SYSTEM_RESERVED_ID` sentinel — control bytes, not path-safe — in the -/// tenant and/or user segment. Map it to the harness's historical `_system` -/// segment: this mirrors production's `resource_scope_path_segment` -/// (`invocation_mount_view`, ironclaw_reborn_composition) only in SHAPE -/// (sentinel-in, path-safe-segment-out) — production's actual segment value -/// is `__system__`, not `_system`. Deliberately NOT switched to match, to -/// avoid rewriting every existing harness fixture path; see `path_segment`. +/// System-scoped ops carry the `SYSTEM_RESERVED_ID` sentinel (control bytes, not path-safe) +/// in the tenant/user segment; mapped to the harness's `_system` segment — matching +/// production's `resource_scope_path_segment` only in SHAPE, not value (prod uses +/// `__system__`). Deliberately not switched, to avoid rewriting every existing fixture +/// path; see `path_segment`. pub(crate) fn threads_mount_view( root_prefix: &str, scope: &ironclaw_host_api::ResourceScope, @@ -229,10 +195,9 @@ pub(crate) fn threads_mount_view( )]) } -/// Path-safe segment for one scope axis: the `SYSTEM_RESERVED_ID` sentinel -/// becomes the harness's historical `_system` segment (NOT production's -/// `__system__` value — see `threads_mount_view`); everything else is used -/// verbatim, matching production's `resource_scope_path_segment` shape. +/// Path-safe segment for one scope axis: `SYSTEM_RESERVED_ID` becomes `_system` (not +/// production's `__system__` — see `threads_mount_view`); everything else passes through +/// verbatim. pub(crate) fn path_segment(value: &str) -> &str { if value == ironclaw_host_api::SYSTEM_RESERVED_ID { "_system" diff --git a/tests/support/reborn/test_adapter.rs b/tests/integration/support/test_adapter.rs similarity index 100% rename from tests/support/reborn/test_adapter.rs rename to tests/integration/support/test_adapter.rs diff --git a/tests/support/reborn/triggered_submit.rs b/tests/integration/support/triggered_submit.rs similarity index 59% rename from tests/support/reborn/triggered_submit.rs rename to tests/integration/support/triggered_submit.rs index 24135be11b9..83b5f653c5d 100644 --- a/tests/support/reborn/triggered_submit.rs +++ b/tests/integration/support/triggered_submit.rs @@ -1,28 +1,13 @@ -//! E-TRIGGERED-SUBMIT enabler seam: submit a turn through the REAL -//! `TrustedTriggerFireSubmitter` so it carries a genuine -//! `TurnOriginKind::ScheduledTrigger` origin, end to end. +//! E-TRIGGERED-SUBMIT seam: submits turns through the REAL +//! `trusted_trigger_fire_submitter` (not a fake) over the harness's shared +//! `coordinator`, so runs carry a genuine `TurnOriginKind::ScheduledTrigger` +//! origin and land in the same turn store/scheduler as any other turn. //! -//! [`RebornIntegrationHarness::submit_triggered_turn`] builds a synthetic -//! `TriggerFire` + `TriggerMaterializedPrompt::for_fire` (test ctor) and hands -//! them to the production `trusted_trigger_fire_submitter` — it does NOT fake -//! or re-implement origin tagging itself. That submitter runs over a fresh, -//! per-call `InMemoryConversationServices` dedicated to the trigger path (the -//! same conversation-services type production's own local-dev build wires for -//! this exact purpose), while the harness's REAL shared `coordinator` is -//! passed through unchanged, so the submitted run lands in the same turn -//! store/scheduler as every other harness turn. -//! -//! Two variants: -//! - [`RebornIntegrationHarness::submit_triggered_turn`] — submit only; the -//! run then fails benignly on the scope-miss sentinel (asserts submit-time -//! state, e.g. the persisted origin). -//! - [`RebornIntegrationHarness::submit_triggered_turn_scripted`] — materializes -//! through the REAL production trusted-trigger pipeline -//! (`ironclaw_reborn_composition::test_support::materialize_trigger_prompt_for_test`: -//! authorize, validate, resolve binding, thread-recorded prompt, real -//! content ref) and registers a scripted gateway for the fire's exact -//! scope, so the triggered run can be driven to completion or through -//! mid-fire gates (scope-aware `wait`/`approve`/`deny` helpers below). +//! Two variants: [`RebornIntegrationHarness::submit_triggered_turn`] +//! (submit-only; asserts submit-time state) and +//! [`RebornIntegrationHarness::submit_triggered_turn_scripted`] (full +//! production materialization + scripted gateway, drivable to completion or +//! a mid-fire gate). // Shared integration-test support: not every binary that mounts the // `reborn_support` tree consumes this module — its symbols read as dead there @@ -59,13 +44,11 @@ use super::reply::RebornScriptedReply; use super::scripted_provider::{SCRIPTED_MODEL_NAME, scripted_trace_llm}; use crate::support::trace_llm::TraceLlm; -// `builder.rs`'s `HarnessResult` is module-private; every sibling file that -// needs the alias (`assertions.rs`, `harness.rs`, `harness_mcp.rs`) declares -// its own identical copy rather than reaching across the module boundary. +// `builder.rs`'s `HarnessResult` is module-private, so each sibling file +// declares its own identical copy rather than reaching across the boundary. type HarnessResult = Result>; -/// Far-future, deterministic fire slot — no wall-clock flake, matches the -/// style already used in `tests/reborn_group_triggers/scenario_verbs_lifecycle.rs`. +/// Far-future, deterministic fire slot — avoids wall-clock flake. fn triggered_fire_slot() -> chrono::DateTime { Utc.with_ymd_and_hms(2999, 1, 1, 0, 0, 0).unwrap() } @@ -73,17 +56,15 @@ fn triggered_fire_slot() -> chrono::DateTime { /// Result of a successful `submit_triggered_turn` call. pub(crate) struct TriggeredSubmission { pub(crate) run_id: TurnRunId, - /// The trigger's OWN resolved scope, as returned by the real submitter. - /// The trusted-submit contract forbids re-deriving binding keys from a - /// `TriggerFire` (see `ironclaw_triggers::trusted_submit` docs), so callers - /// must read this back rather than reconstruct it. + /// The trigger's OWN resolved scope. The trusted-submit contract forbids + /// re-deriving binding keys from a `TriggerFire`, so callers must read + /// this back rather than reconstruct it. pub(crate) turn_scope: TurnScope, } impl RebornIntegrationHarness { - /// Synthetic `TriggerFire` for this harness's binding — the shared fire - /// construction both triggered-submit seams hand to the production - /// submitter. + /// Synthetic `TriggerFire` shared by both triggered-submit seams, handed + /// to the production submitter. fn triggered_fire(&self, prompt: &str, fire_slot: chrono::DateTime) -> TriggerFire { TriggerFire { identity: TriggerFireIdentity::new( @@ -98,15 +79,11 @@ impl RebornIntegrationHarness { } } - /// Fresh `InMemoryConversationServices` for the trigger path with the - /// trigger's canonical external actor pre-paired — mirrors - /// `TriggerTrustedInboundBinding::for_fire`'s own derivation exactly and - /// mirrors production's pre-seed requirement (`resolve_actor` hard-fails - /// `BindingRequired` without it, on both trusted and untrusted resolve - /// paths). Uses `try_pair_external_actor` (not the infallible - /// `pair_external_actor` wrapper) so a pairing failure surfaces here, at - /// the seam boundary, instead of resurfacing later as an indirect - /// binding-resolution error. + /// Fresh `InMemoryConversationServices` with the trigger's actor + /// pre-paired (mirrors `TriggerTrustedInboundBinding::for_fire`; + /// production's `resolve_actor` hard-fails `BindingRequired` without it). + /// Uses `try_pair_external_actor` so a pairing failure surfaces here, not + /// as an indirect binding-resolution error downstream. async fn trigger_conversations_with_paired_actor( &self, ) -> HarnessResult { @@ -126,18 +103,12 @@ impl RebornIntegrationHarness { Ok(conversations) } - /// Submit a turn through the REAL `TrustedTriggerFireSubmitter` so it carries - /// a genuine `TurnOriginKind::ScheduledTrigger` origin, end to end (E-TRIGGERED-SUBMIT). - /// See the module docs for how the fire/prompt/conversation-services are built. - /// - /// The submitted run then executes autonomously on the background scheduler; no - /// scripted model gateway is registered for the trigger's own resolved scope, so it - /// fails benignly on a scope-miss (`ScopeRegistryGateway`'s `ConfigurationError` - /// sentinel). `product_context` (carrying the origin) is persisted synchronously at - /// submit time, before that later failure — this seam and its driving test assert - /// only on that submit-time state, not on anything after. Driving a triggered run - /// to model completion, or asserting behavior across that later failure, is - /// C-TRIGGERED-DELIVERY, not this seam. + /// Submits through the REAL `TrustedTriggerFireSubmitter` for a genuine + /// `TurnOriginKind::ScheduledTrigger` origin (E-TRIGGERED-SUBMIT). No + /// scripted gateway is registered, so the run fails benignly on + /// scope-miss after submit-time state (e.g. persisted origin) is + /// recorded — this seam asserts only that state; driving to model + /// completion is C-TRIGGERED-DELIVERY. pub(crate) async fn submit_triggered_turn( &self, prompt: &str, @@ -169,35 +140,21 @@ impl RebornIntegrationHarness { } } - /// [`submit_triggered_turn`](Self::submit_triggered_turn), with the - /// triggered run's model calls scripted and the trigger prompt recorded - /// into the harness's REAL thread service — so the run can be driven past - /// submission (to completion, or to a mid-fire gate) instead of failing - /// benignly on the scope-miss sentinel. + /// [`submit_triggered_turn`](Self::submit_triggered_turn) with model + /// calls scripted and the prompt recorded into the harness's REAL thread + /// service, so the run can be driven to completion or a mid-fire gate + /// instead of failing on scope-miss. /// - /// Materializes through the REAL production trusted-trigger pipeline via - /// `ironclaw_reborn_composition::test_support::materialize_trigger_prompt_for_test` - /// (authorize, validate, resolve-or-create binding, record the prompt as - /// a real inbound thread message, build the real `thread-message:` - /// content ref, AND return the resolved `TurnScope`) rather than - /// hand-mirroring `trigger_resolve_request` + `record_trigger_prompt` + - /// the content-ref shape field-by-field — flagged as a drift trap on PR - /// #5584 (trusted-trigger materialization is an ownership boundary, - /// `AGENTS.md:61`) and extracted as the agreed fast-follow. This - /// mirrors the production trigger poller's OWN materializer - /// (`ConversationContentRefMaterializer::materialize_prompt`) exactly, - /// since it now calls the SAME code, not a copy of it — which the plain - /// `submit_triggered_turn`'s `TriggerMaterializedPrompt::for_fire` - /// shortcut still skips. + /// Materializes via `materialize_trigger_prompt_for_test`, which calls + /// the SAME production code as the trigger poller's own materializer — + /// trusted-trigger materialization is an ownership boundary (see + /// `AGENTS.md`), so this must not hand-mirror it field-by-field. /// - /// After materialization: register the scripted gateway for the EXACT - /// resolved scope on the group's `ScopeRegistryGateway` (the same - /// real-chain construction thread gateways use: `scripted_trace_llm` → + /// After materialization, registers the scripted gateway for the exact + /// resolved scope (real-chain construction: `scripted_trace_llm` → /// `provider_chain_over` → `LlmProviderModelGateway`, preserving the - /// one-fake-at-the-vendor-SDK-seam invariant), then submit through the - /// production submitter over the SAME conversation services — its own - /// resolve reuses the just-created binding, so the run executes under the - /// pre-registered scope with no race. + /// one-fake-at-the-vendor-SDK-seam invariant) before submitting, so the + /// run executes under the pre-registered scope with no race. pub(crate) async fn submit_triggered_turn_scripted( &self, prompt: &str, @@ -283,11 +240,9 @@ impl RebornIntegrationHarness { } } - /// `wait_for_status` for a run living in a scope OTHER than this harness - /// thread's own (`builder.rs::wait_for_status` polls `self.turn_scope`) — - /// today only triggered runs, whose scope is minted at submit time and - /// returned on [`TriggeredSubmission`]. Same poll loop, deadline, and - /// fail-fast-on-wrong-terminal semantics. + /// Scope-parameterized twin of `builder.rs::wait_for_status` (which polls + /// `self.turn_scope`), for runs in another scope (e.g. triggered runs). + /// Same poll loop, deadline, and fail-fast-on-wrong-terminal semantics. pub(crate) async fn wait_for_status_in_scope( &self, scope: &TurnScope, @@ -324,10 +279,9 @@ impl RebornIntegrationHarness { } } - /// `approve_gate` for a run in a non-thread scope (triggered runs). Same - /// two steps as `builder.rs::approve_gate` — resolve the persisted approval - /// request to an issued lease, then resume with the production - /// `BlockedApprovalGate` precondition — but resuming in the given scope. + /// `builder.rs::approve_gate` for a non-thread scope (triggered runs): + /// resolve the persisted approval to an issued lease, then resume with + /// `BlockedApprovalGate` precondition in the given scope. pub(crate) async fn approve_gate_in_scope( &self, scope: &TurnScope, @@ -347,10 +301,10 @@ impl RebornIntegrationHarness { .await } - /// `deny_gate` for a run in a non-thread scope (triggered runs). Mirrors - /// `builder.rs::deny_gate`: resolve the persisted request to `Denied`, then - /// resume with `GateResumeDisposition::Denied` so the executor surfaces a - /// non-retryable authorization failure to the model. + /// `builder.rs::deny_gate` for a non-thread scope (triggered runs): + /// resolve the persisted request to `Denied`, then resume with + /// `GateResumeDisposition::Denied` so the executor surfaces a + /// non-retryable authorization failure. pub(crate) async fn deny_gate_in_scope( &self, scope: &TurnScope, @@ -370,17 +324,13 @@ impl RebornIntegrationHarness { .await } - /// Scope-parameterised twin of `builder.rs::resume_run` (which resumes in - /// `self.turn_scope`). The actor stays the harness actor: a trigger fire's - /// creator IS `binding.actor_user_id` (`submit_triggered_turn` builds the - /// fire from it), so the approving user and the trigger creator coincide, - /// as in production's approval-resolution path. + /// Scope-parameterized twin of `builder.rs::resume_run` (resumes in + /// `self.turn_scope`). The approving actor is the harness actor, matching + /// the trigger creator (`binding.actor_user_id`), as in production. /// - /// `pub(crate)` (not just a private tail of `approve_gate_in_scope`/ - /// `deny_gate_in_scope`) so deny-edge scenarios can drive a resume with a - /// deliberately WRONG scope without rebuilding the `ResumeTurnRequest` by - /// hand — a coordinator-level rejection (`TurnError`) propagates out - /// unchanged for the caller to pin. + /// `pub(crate)` so deny-edge scenarios can drive a resume with a + /// deliberately WRONG scope without hand-rebuilding `ResumeTurnRequest` — + /// a coordinator-level rejection propagates out for the caller to pin. pub(crate) async fn resume_run_in_scope( &self, scope: &TurnScope, diff --git a/tests/reborn_integration_tool_call.rs b/tests/integration/tool_call.rs similarity index 52% rename from tests/reborn_integration_tool_call.rs rename to tests/integration/tool_call.rs index 934661b768d..3456e662b15 100644 --- a/tests/reborn_integration_tool_call.rs +++ b/tests/integration/tool_call.rs @@ -1,16 +1,12 @@ -//! Reborn integration-test framework — tool-calling turn. -//! -//! Proves the tool path + the §3.7 two-tier egress design end-to-end: the -//! scripted model emits a `builtin.http` tool call → the real first-party tool -//! runtime executes it through `RuntimeHttpEgress` → the call is captured by the -//! recording egress (Tier-2) → the model finalizes a text reply. Uses the same -//! scripted `TraceLlm` seam beneath the real decorator chain as other harness -//! tests; no network, no services, no keys, no Docker, no `integration` feature. +//! Tool-calling turn: proves the §3.7 two-tier egress design end-to-end — +//! scripted `builtin.http` call → real `RuntimeHttpEgress` → recording egress +//! (Tier-2) → finalized reply. Same scripted `TraceLlm` seam as other harness tests. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::builder::RebornIntegrationHarness; @@ -42,10 +38,8 @@ async fn runs_http_tool_call_through_recorded_egress() { const HTTP_TOOL_URL: &str = "https://api.example.test/v1/items"; -/// Guards the assertion helpers against silently passing: when the scripted tool -/// call is *absent* (a plain text turn on the default echo backend, which runs no -/// tool and captures no egress), both `assert_tool_invoked` and -/// `assert_egress_request_matching` must return `Err`. +/// Guards against vacuous pass: with no scripted tool call, both +/// `assert_tool_invoked` and `assert_egress_request_matching` must return `Err`. #[tokio::test] async fn assertions_fail_when_tool_did_not_run() { let h = RebornIntegrationHarness::test_default() @@ -62,10 +56,9 @@ async fn assertions_fail_when_tool_did_not_run() { ); } -/// Proves the assertion helpers discriminate when the invocation + egress lists -/// are NON-empty: a real `builtin.http` call runs, but assertions for a -/// *different* capability / host must return `Err` — exercising the -/// "present but no match" branch, not the empty-list branch (builder.rs:331). +/// Proves the assertions discriminate when the invocation + egress lists are +/// NON-empty: a real `builtin.http` call runs, but assertions for a *different* +/// capability/host must still return `Err` (the "present but no match" branch). #[tokio::test] async fn assertions_fail_when_tool_present_but_requested_tool_or_url_does_not_match() { let h = RebornIntegrationHarness::test_default() @@ -78,8 +71,8 @@ async fn assertions_fail_when_tool_present_but_requested_tool_or_url_does_not_ma .await .expect("harness builds"); h.submit_turn("fetch items").await.expect("turn completes"); - // Prove the capture lists are NON-empty first, so the negative checks below - // exercise the mismatch branch rather than passing vacuously on empty lists. + // Prove capture lists are NON-empty first, so the checks below exercise the + // mismatch branch, not the empty-list branch. h.assert_tool_invoked("builtin.http") .await .expect("http tool ran before mismatch assertions"); @@ -100,14 +93,9 @@ async fn assertions_fail_when_tool_present_but_requested_tool_or_url_does_not_ma ); } -/// Proves the multi-segment `builtin.http.save` capability id — whose -/// `.`→`__` encoding produces `builtin__http__save` at the provider seam -/// (reply.rs:33) — resolves end-to-end through the real runtime. -/// -/// Args: `url` (same constant the existing test uses) + `save_to` under the -/// `/workspace` mount that `core_builtin_tools` provides with read-write -/// permissions. The recording egress returns a fixed `{"accepted":true}` body; -/// no network is touched. +/// Proves the multi-segment `builtin.http.save` capability id (`.`→`__` +/// encoding to `builtin__http__save` at the provider seam) resolves end-to-end, +/// writing to the `/workspace` mount `core_builtin_tools` provides read-write. #[tokio::test] async fn runs_http_save_tool_call_through_recorded_egress() { let h = RebornIntegrationHarness::test_default() @@ -128,8 +116,7 @@ async fn runs_http_save_tool_call_through_recorded_egress() { h.assert_tool_invoked("builtin.http.save") .await .expect("http.save tool ran"); - // The save path must reach the real `RuntimeHttpEgress`; assert the recorded - // egress so a regression that bypasses it cannot pass this test. + // The save path must reach the real `RuntimeHttpEgress`. h.assert_egress_request_matching("api.example.test") .await .expect("http.save egress captured"); @@ -139,22 +126,11 @@ async fn runs_http_save_tool_call_through_recorded_egress() { } /// The globally-disabled `builtin.spawn_subagent` capability -/// (`ironclaw_reborn::runtime::DISABLED_CAPABILITY_IDS`, applied as the -/// OUTERMOST `PerSurfaceCapabilityDenyDecorator` in -/// `build_default_planned_runtime_inner` — see that function's doc comments) -/// must never reach the model-facing tool list, whichever port would -/// otherwise have surfaced it: the flavor-aware `SubagentSpawnCapabilityDecorator` -/// (always wired, independent of any harness extension registry) or the -/// host-runtime first-party manifest stub (`builtin_first_party_package()` in -/// `crates/ironclaw_host_runtime/src/first_party_tools/mod.rs`, included in -/// `core_builtin_tools()`'s registry unconditionally). -/// -/// Non-vacuity: confirmed by direct inspection that `core_builtin_tools()`'s -/// capability port surfaces `builtin__spawn_subagent` when the deny decorator -/// is bypassed (i.e. `spawn_decorator` runs before the outermost deny filter -/// in composition order) — so this assertion is pinning a real strip, not -/// asserting absence from an already-empty surface. `builtin__http` is -/// asserted present as the non-vacuity control for THIS test's own capture. +/// (`ironclaw_reborn::runtime::DISABLED_CAPABILITY_IDS`, stripped by the +/// outermost `PerSurfaceCapabilityDenyDecorator`) must never reach the +/// model-facing tool list, from either the flavor-aware spawn decorator or the +/// host-runtime first-party manifest stub. `builtin__http` is asserted present +/// as the non-vacuity control. #[tokio::test] async fn disabled_spawn_subagent_capability_is_stripped_from_model_surface() { let h = RebornIntegrationHarness::test_default() @@ -173,9 +149,7 @@ async fn disabled_spawn_subagent_capability_is_stripped_from_model_surface() { .collect(); // Neither encoding of the disabled capability id may appear in what the - // model was shown (the `.`→`__` provider-seam encoding, or the raw - // dotted capability id — structurally impossible as a provider tool name - // since `ProviderToolName` rejects dots, but checked defensively). + // model was shown (provider-seam `__` encoding, or the raw dotted id). assert!( !names.contains(&"builtin__spawn_subagent"), "disabled capability's provider seam name must not be advertised: {names:?}" @@ -192,37 +166,13 @@ async fn disabled_spawn_subagent_capability_is_stripped_from_model_surface() { ); } -/// A model that hallucinates a call to the disabled `builtin.spawn_subagent` -/// capability anyway — the deny filter's `CapabilitySurfaceDenyFilter` strips -/// the id from `tool_definitions()`, so `builtin__spawn_subagent` is not in -/// `advertised_tool_names` at the model gateway -/// (`crates/ironclaw_reborn/src/model_gateway.rs::tool_response_to_host`). -/// The gateway falls back to `provider_calls_are_advertised_or_resolvable`, -/// which resolves the id via `provider_tool_call_capability_ids` — the deny -/// filter's inner port still resolves it structurally, but the deny filter's -/// own scope check then rejects it (`"provider tool call targets a disabled -/// capability"`, `AgentLoopHostErrorKind::InvalidInvocation`) — so the whole -/// provider response (not just the one call) is rejected with -/// `HostManagedModelErrorKind::InvalidOutput` before any -/// `CapabilityCallCandidate` is ever registered. -/// -/// Observed contract (pinned as-is, no production changes): `InvalidOutput` -/// maps to `AgentLoopHostErrorKind::Unavailable` (`model_gateway_error` in -/// `crates/ironclaw_loop_support/src/lib.rs`), which the executor's -/// `ModelStage` classifies as `ModelErrorClass::Unavailable` and hands to the -/// recovery strategy — which aborts without a further model call, so the -/// run reaches a terminal `TurnStatus::Failed` with -/// `SanitizedFailure::category() == "model_error"` (`LoopFailureKind::ModelError`) -/// after consuming exactly the ONE scripted model turn (mirrors the sibling -/// gateway-error contract pinned by -/// `reborn_integration_cancel.rs::mid_turn_provider_error_reaches_failed_with_model_error_category`, -/// reached here via a distinct root cause: deny-filter rejection at -/// registration, not a raw provider error). This is a run-level failure, not -/// a model-visible `Failed`/`Denied` tool result — no `ToolResultReference` -/// is ever persisted for the call, because the deny filter rejects it before -/// `register_provider_tool_call` ever stages an invocation. The no-side-effect -/// proof is `assert_tool_invoked` returning `Err` — the capability was never -/// dispatched. +/// A model that calls the disabled `builtin.spawn_subagent` anyway is rejected +/// at the gateway (`CapabilitySurfaceDenyFilter`, before +/// `register_provider_tool_call` ever stages an invocation) — the whole +/// provider response fails with `InvalidOutput` → `Unavailable`, reaching a +/// terminal `TurnStatus::Failed`/`"model_error"` after exactly one scripted +/// turn. No `ToolResultReference` is persisted; `assert_tool_invoked` +/// returning `Err` proves the capability was never dispatched. #[tokio::test] async fn disabled_spawn_subagent_capability_call_anyway_fails_the_run() { let h = RebornIntegrationHarness::test_default() diff --git a/tests/reborn_integration_tracecap.rs b/tests/integration/tracecap.rs similarity index 56% rename from tests/reborn_integration_tracecap.rs rename to tests/integration/tracecap.rs index cbcd3a4a9a8..5c94b2ce6d1 100644 --- a/tests/reborn_integration_tracecap.rs +++ b/tests/integration/tracecap.rs @@ -1,26 +1,16 @@ -//! C-TRACECAP: `turn_event_sink` int-tier coverage (rev-3 Tier-2, A1 audit). +//! C-TRACECAP: `turn_event_sink` int-tier coverage. //! -//! Production wires a best-effort turn-lifecycle sink via -//! `lifecycle_bus.subscribe_best_effort(sink)` in -//! `build_default_planned_runtime_inner` (`crates/ironclaw_reborn/src/runtime.rs:613-619`), -//! fed in real deployments by `CompositeTurnEventSink` over -//! `[TraceCaptureTurnEventSink, ..]` (`crates/ironclaw_reborn_composition/src/runtime.rs:3229-3290`) -//! — the entry point to the 0%-covered `ironclaw_reborn_traces` crate. That -//! seam was never exercised by any Reborn test: `DefaultPlannedRuntimeParts.turn_event_sink` -//! was `None` in every harness/group construction. -//! -//! This test wires `ironclaw_turns::InMemoryTurnEventSink` — a real, already-shipped -//! production `TurnEventSink` impl with zero callers anywhere in the codebase -//! today — into the harness's planned runtime via `.with_turn_event_sink()`, and -//! proves `subscribe_best_effort` actually publishes to it for a real completed -//! turn. Distinct from T0-SYSPROMPT's `TraceLlm` captured model requests (a -//! different seam: what the model saw, not what the turn coordinator published) -//! and from `reborn_recorded_trace_parity.rs` (recorded-response replay). +//! Wires `ironclaw_turns::InMemoryTurnEventSink` into the harness's planned +//! runtime via `.with_turn_event_sink()`, proving production's +//! `lifecycle_bus.subscribe_best_effort` seam actually publishes turn-lifecycle +//! events. Distinct from T0-SYSPROMPT (captured model requests) and +//! `reborn_recorded_trace_parity.rs` (recorded-response replay). #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use ironclaw_turns::TurnEventKind; @@ -48,10 +38,8 @@ async fn turn_event_sink_receives_completed_event_for_a_finished_turn() { .expect("turn-lifecycle sink recorded the Completed event for the finished turn"); } -/// Negative control: a harness that never calls `.with_turn_event_sink()` has no -/// sink installed, so `DefaultPlannedRuntimeParts.turn_event_sink` stays `None` -/// (matching every pre-existing reborn test) and no events are recorded. Proves -/// the assertion is discriminating on real wiring, not a tautology. +/// Negative control: without `.with_turn_event_sink()`, no sink is installed and +/// no events are recorded — proves the assertion discriminates on real wiring. #[tokio::test] async fn no_events_recorded_without_opting_in() { let harness = RebornIntegrationHarness::test_default() @@ -75,13 +63,9 @@ async fn no_events_recorded_without_opting_in() { ); } -/// Regression: the sink is shared across every thread in a `.with_turn_event_sink()` -/// group (it has no per-thread channel), so `assert_turn_event_recorded` must slice -/// `[baseline_turn_event_count..]` to see only THIS thread's events. Drives a first -/// thread to Completed, then builds a second thread AFTER that (so its baseline -/// already includes the first thread's event) and never submits a turn on it. If -/// the baseline slice were missing, the second thread's assertion would incorrectly -/// pass on the first thread's Completed event. +/// Regression: the sink is shared across every thread in a group, so +/// `assert_turn_event_recorded` must slice `[baseline_turn_event_count..]` — +/// a second thread built after the first's Completed event must not see it. #[tokio::test] async fn group_thread_does_not_see_a_sibling_threads_turn_event() { let group = RebornIntegrationGroup::builder() @@ -105,7 +89,7 @@ async fn group_thread_does_not_see_a_sibling_threads_turn_event() { .await .expect("first thread recorded its own Completed event"); - // Built after `first` completed, so the shared sink already holds `first`'s + // Built after `first` completed, so the shared sink already holds its // Completed event ahead of this thread's baseline. let second = group .thread("conv-tracecap-second") diff --git a/tests/reborn_integration_triggered_submit.rs b/tests/integration/triggered_submit.rs similarity index 57% rename from tests/reborn_integration_triggered_submit.rs rename to tests/integration/triggered_submit.rs index fab56108bb4..7dd3ff42544 100644 --- a/tests/reborn_integration_triggered_submit.rs +++ b/tests/integration/triggered_submit.rs @@ -1,31 +1,14 @@ -//! Reborn integration-test framework — E-TRIGGERED-SUBMIT driving test. -//! -//! Proves the trusted-trigger submission seam end-to-end: this exercises the -//! real `TrustedTriggerFireSubmitter` (via -//! `RebornIntegrationHarness::submit_triggered_turn`), proving that -//! `TurnOriginKind::ScheduledTrigger` propagates all the way into the -//! persisted run state observable at the coordinator boundary -//! (`TurnCoordinator::get_run_state`). -//! -//! Three slices, one wire: -//! - `triggered_submit_carries_scheduled_trigger_origin` — submission accepted, -//! run state carries the scheduled-trigger origin (unscripted: the run then -//! fails benignly on the scope-miss sentinel, which this slice never reads). -//! - `interactive_submit_carries_inbound_origin_not_scheduled_trigger` — the -//! discriminating contrast arm (C-TRIGGERED-ORIGIN). -//! - `triggered_run_completes_and_persists_reply_in_trigger_thread` — drives a -//! triggered run to completion via `submit_triggered_turn_scripted` and pins -//! the int-tier-observable delivery contract (reply persisted in the -//! trigger's own thread; see that test's docs for why the outbound push leg -//! is out of reach at this tier). -//! -//! The Slack push/delivery-routing matrix stays with the services-shell spike -//! (C-TRIGGERED-DELIVERY defer). +//! E-TRIGGERED-SUBMIT: proves the trusted-trigger submission seam end-to-end — +//! `TrustedTriggerFireSubmitter` propagates `TurnOriginKind::ScheduledTrigger` +//! into persisted run state at `TurnCoordinator::get_run_state`. Contrast arm: +//! C-TRIGGERED-ORIGIN. Delivery/push routing stays with the Slack +//! services-shell spike (C-TRIGGERED-DELIVERY defer). #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::builder::RebornIntegrationHarness; @@ -61,20 +44,10 @@ async fn triggered_submit_carries_scheduled_trigger_origin() { ); } -/// C-TRIGGERED-ORIGIN contrast arm: a normal interactive user turn (through the -/// same `submit_turn` → `accept_inbound` → coordinator wire this harness always -/// uses) must record `TurnOriginKind::Inbound`, NOT `ScheduledTrigger`. -/// -/// This is what makes the `triggered_submit_carries_scheduled_trigger_origin` -/// assertion above *discriminating*: without a contrasting turn on the same wire, -/// a `ScheduledTrigger` assertion could pass even if the origin were hardcoded -/// everywhere. Both origins are read through the identical -/// `coordinator.get_run_state(...).product_context.origin` boundary. -/// -/// (`TurnOriginKind` has no distinct "interactive" variant — the enum is -/// `WebUi | Inbound | ScheduledTrigger`, `crates/ironclaw_turns/src/origin.rs`. -/// The harness's `submit_turn` classifies as `ProductTriggerReason::DirectChat` -/// on the Untrusted inbound path, which resolves to `Inbound`.) +/// C-TRIGGERED-ORIGIN contrast arm: an interactive turn on the same +/// `submit_turn` → coordinator wire must record `TurnOriginKind::Inbound`, not +/// `ScheduledTrigger` — the control proving the trigger path's origin above +/// isn't hardcoded. #[tokio::test] async fn interactive_submit_carries_inbound_origin_not_scheduled_trigger() { let harness = RebornIntegrationHarness::test_default() @@ -106,19 +79,9 @@ async fn interactive_submit_carries_inbound_origin_not_scheduled_trigger() { ); } -/// Post-fire delivery semantics at int tier: a triggered run driven to -/// completion persists its final reply in the TRIGGER's own thread, readable -/// through the same thread-history boundary interactive replies use. -/// -/// At this tier a completed run's final reply does NOT route through -/// `ProductAdapter::render_outbound` (and therefore never reaches an -/// `OutboundDeliverySink`) — the only production constructor of -/// `ProductOutboundDeliveryRequest` is the Slack delivery services-shell -/// (`slack_delivery.rs`, feature-gated), which no harness composition wires. -/// The int-tier-observable delivery contract is therefore: reply finalized + -/// persisted in the trigger's own thread — the same state production's -/// `deliver_triggered_run` reads before pushing. The push leg itself stays -/// with the services-shell spike (C-TRIGGERED-DELIVERY defer). +/// Int-tier delivery contract: a completed triggered run persists its reply in +/// the trigger's own thread. No `OutboundDeliverySink` push happens at this +/// tier — that's the Slack services-shell (C-TRIGGERED-DELIVERY defer). #[tokio::test] async fn triggered_run_completes_and_persists_reply_in_trigger_thread() { let harness = RebornIntegrationHarness::test_default() diff --git a/tests/reborn_integration_web_access.rs b/tests/integration/web_access.rs similarity index 74% rename from tests/reborn_integration_web_access.rs rename to tests/integration/web_access.rs index 30231b51485..d0fd8ce573e 100644 --- a/tests/reborn_integration_web_access.rs +++ b/tests/integration/web_access.rs @@ -1,20 +1,15 @@ -//! Reborn integration-test framework — first-party `web-access.*` coverage -//! (C-WEBACCESS). -//! -//! `web-access.search` / `web-access.get_content` are `RuntimeKind::FirstParty` -//! capabilities dispatched through the real `WebAccessExecutor`, which speaks -//! MCP JSON-RPC by hand over three sequential `RuntimeHttpEgress` calls -//! (`initialize` → `notifications/initialized` → `tools/call`) to the Exa MCP -//! endpoint. `.with_web_access_tools([..])` wires the real executor behind a -//! thin test adapter and scripts the three-leg handshake onto the recording -//! egress's FIFO queue at build time (all three legs share one URL, so the -//! keyed HTTP matcher can't tell them apart). Same single LLM seam as every -//! other Reborn integration test; no real network, services, keys, or Docker. +//! C-WEBACCESS: first-party `web-access.*` capabilities dispatched through the +//! real `WebAccessExecutor`, which speaks MCP JSON-RPC over three sequential +//! `RuntimeHttpEgress` calls (initialize → notifications/initialized → +//! tools/call) to the Exa MCP endpoint. `.with_web_access_tools([..])` scripts +//! the three-leg handshake onto the recording egress's FIFO queue (all three +//! legs share one URL, so the keyed HTTP matcher can't tell them apart). #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/mod.rs"] mod reborn_support; #[allow(dead_code)] +#[path = "../support/mod.rs"] mod support; use reborn_support::assertions::ToolErrorClass; @@ -26,12 +21,10 @@ const MCP_INIT_BODY: &[u8] = br#"{"jsonrpc":"2.0","id":1,"result":{"protocolVersion":"2024-11-05","capabilities":{}}}"#; const MCP_NOTIF_BODY: &[u8] = br#"{"accepted":true}"#; -/// SSE-framed `initialize` result body with a leading keepalive `ping` event -/// (`event: ping\ndata:\n`) before the real `message` event. Streamable-HTTP -/// MCP servers may legally answer any leg this way because `web_access` sends -/// `Accept: application/json, text/event-stream` on every request. Hand-authored -/// (no live-captured MCP bodies exist under `tests/fixtures/`); the shape mirrors -/// the production unit fixture in `web_access.rs::extracts_text_from_sse_mcp_response`. +/// SSE-framed `initialize` result with a leading keepalive `ping` event before +/// the real `message` event — legal because `web_access` sends `Accept: +/// application/json, text/event-stream`. Mirrors the unit fixture in +/// `web_access.rs::extracts_text_from_sse_mcp_response`. const MCP_INIT_BODY_SSE: &[u8] = b"event: ping\ndata:\n\nevent: message\ndata: {\"jsonrpc\":\"2.0\",\"id\":1,\"result\":{\"protocolVersion\":\"2024-11-05\",\"capabilities\":{}}}\n\n"; /// SSE-framed `tools/call` result wrapping the same `{"result":{"content":..}}` @@ -48,12 +41,8 @@ fn mcp_tool_call_result_body_sse(content_text: &str) -> Vec { format!("event: message\ndata: {json}\n\n").into_bytes() } -/// Builds the `tools/call` JSON-RPC result body the scripted Exa MCP -/// handshake's third leg returns: -/// `{"result":{"content":[{"type":"text","text":""}]}}`. Both -/// `web-access.search` and `web-access.get_content` MCP responses share this -/// shape — only the text content differs — so every test scripts its third -/// leg with `mcp_tool_call_result_body(..)` instead of hand-writing the JSON. +/// Builds the `tools/call` JSON-RPC result body shared by `web-access.search` +/// and `web-access.get_content` responses (only the text content differs). fn mcp_tool_call_result_body(content_text: &str) -> Vec { json!({ "result": { @@ -175,15 +164,10 @@ async fn get_content_dispatches_through_scripted_exa_mcp() { .expect("scripted MCP fetch result surfaced back to the model"); } -/// Format-matrix regression (C-WIREFMT) through the REAL Reborn handler: the -/// Exa MCP server answers `initialize` with SSE framing. `web-access.search` -/// must still round-trip end-to-end (model -> capability -> `WebAccessExecutor` -/// -> egress) and surface the result. Before the sibling-parity fix in -/// `is_valid_mcp_initialize_response`, the JSON-only init parser rejected the -/// SSE body and the handshake aborted before `tools/call`, so the search result -/// never reached the model. This is the int-tier twin of the crate-tier -/// `search_accepts_sse_framed_initialize_response`, exercising the full dispatch -/// pipeline rather than the executor in isolation. +/// Format-matrix regression (C-WIREFMT): the Exa MCP server answers +/// `initialize` with SSE framing; `web-access.search` must still round-trip +/// end-to-end. Int-tier twin of the crate-tier +/// `search_accepts_sse_framed_initialize_response`. #[tokio::test] async fn web_search_over_sse_framed_initialize_dispatches_through_exa_mcp() { let harness = RebornIntegrationHarness::test_default() @@ -227,11 +211,9 @@ async fn web_search_over_sse_framed_initialize_dispatches_through_exa_mcp() { .expect("SSE-framed initialize accepted so the handshake reached tools/call"); } -/// Sibling parity (C-WIREFMT): BOTH body-parsing legs — `initialize` -/// (`is_valid_mcp_initialize_response`) and `tools/call` (`extract_mcp_text`) — -/// are SSE-framed on the same handshake, proving each leg inherits the other's -/// framing matrix rather than only the leg its author happened to test. Uses the -/// richest framing (multi-event with a keepalive `ping` prelude) for both legs. +/// Sibling parity (C-WIREFMT): both body-parsing legs (`initialize` and +/// `tools/call`) are SSE-framed on the same handshake, proving each leg +/// inherits the other's framing matrix rather than only the tested one. #[tokio::test] async fn web_access_handshake_over_sse_framed_both_legs() { let harness = RebornIntegrationHarness::test_default() @@ -316,10 +298,9 @@ async fn assert_egress_body_contains_any_fails_when_substring_absent() { ); } -/// Guards `assert_egress_body_contains_any` against its OTHER error branch — -/// when no captured egress request's URL matches `url_substr` at all (as -/// opposed to matching the URL but missing the body substring, covered by -/// `assert_egress_body_contains_any_fails_when_substring_absent` above). +/// Guards the OTHER error branch of `assert_egress_body_contains_any`: no +/// captured request's URL matches `url_substr` at all (vs. matching the URL +/// but missing the body substring, covered by the sibling test above). #[tokio::test] async fn assert_egress_body_contains_any_fails_when_url_absent() { let harness = RebornIntegrationHarness::test_default() @@ -354,10 +335,8 @@ async fn assert_egress_body_contains_any_fails_when_url_absent() { "assert_egress_body_contains_any must fail when no captured egress request's URL \ matches url_substr at all", ); - // Pins the specific "no matching URL" branch (not just any `Err`) — the - // sibling substring-absent branch below it in the same function returns a - // differently worded `Err`, so this message check is what actually - // distinguishes the two failure paths. + // Pins the specific "no matching URL" branch, not just any `Err` — the + // sibling substring-absent branch returns a differently worded message. assert!( err.to_string() .contains("no captured egress request matching url"), @@ -365,17 +344,9 @@ async fn assert_egress_body_contains_any_fails_when_url_absent() { ); } -/// Error path — `web-access.get_content` requests a URL the scripted Exa MCP -/// response reports as failed to fetch. `WebAccessExecutor::fetch_content` -/// treats an `"Error fetching "` line in the MCP tool-call -/// text as a hard failure (`parse_fetch_results` -> `operation_error()` -> -/// `RuntimeDispatchErrorKind::OperationFailed`), proving the production -/// `register_bundled_web_access_first_party_handlers` error-mapping path -/// surfaces a real `WebAccessDispatchError` as a model-visible `Failed` -/// tool error — the run -/// is expected to reach `Completed` rather than a terminal `driver_unavailable` -/// (implied here by the presence of a final reply) — rather than a dropped -/// `Err` or a mis-mapped error class. +/// Error path: a scripted Exa fetch-error response surfaces as a model-visible +/// `Failed`/`operation_failed` tool error, and the run still reaches +/// `Completed` with a final reply rather than terminal `driver_unavailable`. #[tokio::test] async fn get_content_fetch_error_surfaces_recoverable_failed() { let harness = RebornIntegrationHarness::test_default() diff --git a/tests/reborn_adapter_installation_scope_isolation_parity.rs b/tests/reborn_adapter_installation_scope_isolation_parity.rs index 2b760f0abe8..002201aa37e 100644 --- a/tests/reborn_adapter_installation_scope_isolation_parity.rs +++ b/tests/reborn_adapter_installation_scope_isolation_parity.rs @@ -1,16 +1,17 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; use ironclaw_loop_support::HostManagedModelResponse; use ironclaw_threads::{MessageKind, MessageStatus, ThreadMessageRecord}; use ironclaw_turns::TurnStatus; -use reborn_support::harness::{ - RebornBinaryE2EHarness, RebornHarnessSharedStorage, RecordingTestCapabilityPort, - test_product_scope, -}; -use reborn_support::model_replay::RebornTraceReplayModelGateway; +use parity_qa_support::binary_e2e::{RebornBinaryE2EHarness, RebornHarnessSharedStorage}; +use parity_qa_support::model_replay::RebornTraceReplayModelGateway; +use reborn_support::harness::{RecordingTestCapabilityPort, test_product_scope}; #[tokio::test] async fn reborn_adapter_installation_scope_isolation_parity() { diff --git a/tests/reborn_agent_scope_isolation_parity.rs b/tests/reborn_agent_scope_isolation_parity.rs index 301d9b2a997..223ad905074 100644 --- a/tests/reborn_agent_scope_isolation_parity.rs +++ b/tests/reborn_agent_scope_isolation_parity.rs @@ -1,16 +1,17 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; use ironclaw_loop_support::HostManagedModelResponse; use ironclaw_threads::{MessageKind, MessageStatus, ThreadMessageRecord}; use ironclaw_turns::TurnStatus; -use reborn_support::harness::{ - RebornBinaryE2EHarness, RebornHarnessSharedStorage, RecordingTestCapabilityPort, - test_product_scope, -}; -use reborn_support::model_replay::RebornTraceReplayModelGateway; +use parity_qa_support::binary_e2e::{RebornBinaryE2EHarness, RebornHarnessSharedStorage}; +use parity_qa_support::model_replay::RebornTraceReplayModelGateway; +use reborn_support::harness::{RecordingTestCapabilityPort, test_product_scope}; #[tokio::test] async fn reborn_agent_scope_isolation_parity() { diff --git a/tests/reborn_approval_traces_parity.rs b/tests/reborn_approval_traces_parity.rs index 0e3d92d82b6..2d75f656670 100644 --- a/tests/reborn_approval_traces_parity.rs +++ b/tests/reborn_approval_traces_parity.rs @@ -1,17 +1,18 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; use ironclaw_loop_support::HostManagedModelResponse; use ironclaw_turns::{TurnStatus, run_profile::LoopHostMilestoneKind}; -use reborn_support::{ - harness::{ - RebornBinaryE2EHarness, RecordingTestCapabilityPort, assert_milestone_order, - trace_tool_call_response, - }, +use parity_qa_support::{ + binary_e2e::{RebornBinaryE2EHarness, assert_milestone_order, trace_tool_call_response}, model_replay::RebornTraceReplayModelGateway, }; +use reborn_support::harness::RecordingTestCapabilityPort; #[tokio::test] async fn reborn_approval_traces_parity() { diff --git a/tests/reborn_direct_chat_user_scope_isolation_parity.rs b/tests/reborn_direct_chat_user_scope_isolation_parity.rs index fba9308e777..e1f1ab1ba3b 100644 --- a/tests/reborn_direct_chat_user_scope_isolation_parity.rs +++ b/tests/reborn_direct_chat_user_scope_isolation_parity.rs @@ -1,16 +1,17 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; use ironclaw_loop_support::HostManagedModelResponse; use ironclaw_threads::{MessageKind, MessageStatus, ThreadMessageRecord}; use ironclaw_turns::TurnStatus; -use reborn_support::harness::{ - RebornBinaryE2EHarness, RebornHarnessSharedStorage, RecordingTestCapabilityPort, - test_product_scope, -}; -use reborn_support::model_replay::RebornTraceReplayModelGateway; +use parity_qa_support::binary_e2e::{RebornBinaryE2EHarness, RebornHarnessSharedStorage}; +use parity_qa_support::model_replay::RebornTraceReplayModelGateway; +use reborn_support::harness::{RecordingTestCapabilityPort, test_product_scope}; #[tokio::test] async fn reborn_direct_chat_user_scope_isolation_parity() { diff --git a/tests/reborn_group_approvals/main.rs b/tests/reborn_group_approvals/main.rs deleted file mode 100644 index e9faca3d930..00000000000 --- a/tests/reborn_group_approvals/main.rs +++ /dev/null @@ -1,182 +0,0 @@ -//! Group integration tests for the Reborn approval flow — the real gate path. -//! -//! One sequential `#[tokio::test]` drives eight scenarios over a shared -//! [`RebornIntegrationGroup::live_approvals`] group (one approval-request store, -//! one capability-lease store, one `(tenant, user)` auto-approve toggle, all -//! shared across threads). See `tests/support/reborn/CLAUDE.md` §"Group tests". -//! -//! Every scenario drives the REAL gate path end-to-end: the scripted model emits -//! a `builtin.write_file` tool call → the real first-party runtime raises a real -//! `TurnStatus::BlockedApproval` gate (auto-approve is disabled for the group at -//! construction) → the test resolves it through the real `ApprovalResolver` -//! (`approve_gate`/`deny_gate`) and `coordinator.resume_turn`. Nothing is faked -//! except the model at the vendor-SDK seam. The exception is -//! `failure_category_demasked`, which drives a genuinely-FAILED run (no gate -//! involved) to prove the group's loop-exit de-mask wiring. -//! -//! ## Scenario ordering (a state machine over the shared auto-approve store) -//! -//! 1. `gate_then_approve` — gate fires (auto-approve OFF), approve → Completed. -//! 2. `gate_then_deny` — gate fires, deny → the model sees an authorization -//! failure, not a hang. -//! 3. `concurrent_dual_gate_resume` (HEADLINE, Option P) — two threads parked -//! on `BlockedApproval` SIMULTANEOUSLY on the group's one shared -//! `TurnCoordinator`, resolved independently (approve one, deny the other) -//! — proves resume dispatch is keyed by `run_id` with zero cross-resume. -//! Must run while auto-approve is still OFF (same control window as 1–2). -//! 4. `failure_category_demasked` — an empty-scripted thread drives a run to a -//! genuine `TurnStatus::Failed` and asserts the TRUE failure category -//! (`"model_error"`) survives instead of being rewritten to the masking -//! `"driver_protocol_violation"` sentinel. Independent of the auto-approve -//! toggle (no gate involved); ordered alongside the other independent -//! scenarios, before the toggle is flipped. -//! 5. `gate_ref_edge_cases::stale_gate_ref_resume` (C-DENYEDGE row 7) — the -//! local-dev approval resolve succeeds with the run's REAL gate_ref, but -//! the coordinator resume is issued with a different, STALE gate_ref, -//! reaching the `TurnError::InvalidRequest { reason: "gate resolution -//! reference mismatch" }` path `approve_gate` alone cannot reach; then a -//! non-vacuity resume with the real ref completes the run. -//! 6. `gate_ref_edge_cases::missing_gate_bare_resolve` (C-DENYEDGE row 10) — a -//! syntactically well-formed but never-issued gate_ref is resolved on a -//! thread that never raised any gate; pins the harness's own -//! request-not-found rejection. Independent of the auto-approve toggle (no -//! real gate involved). -//! 7. `approval_request_persists_after_reopen` (C-DURABLE) — reopens a FRESH -//! `ApprovalRequestStore` at the same on-disk root and confirms the -//! `Pending` request survives, independent of the auto-approve toggle -//! (its own gate, resolved before returning). -//! 8. `approve_always_persists_cross_thread` (HEADLINE) — thread A flips -//! auto-approve ON; a DIFFERENT thread B then writes with NO gate. Proves the -//! setting persists across thread boundaries. MUST run before scenario 9 (it -//! flips the toggle ON for the whole group), so the gate scenarios above are -//! the control proving the gate was real before the flip. -//! 9. `ask_each_time_resumes_once` (W4-ASK-EACH-ONCE, #5306 class) — installs a -//! persistent, group-wide `ToolPermissionOverride::AskEachTime` override on -//! `builtin.write_file`, so it MUST run LAST: every scenario above assumes -//! the plain default (no override) Ask-mode gate for that capability, and -//! this override is a separate CAS store from `AutoApproveSettingStore` that -//! force-gates fresh invocations regardless of auto-approve. Proves an -//! approved AskEachTime-gated resume completes in ONE round trip, not an -//! unresumable re-gate loop. - -#[allow(dead_code)] -#[path = "../support/reborn/mod.rs"] -mod reborn_support; -#[allow(dead_code)] -#[path = "../support/mod.rs"] -mod support; - -mod scenario_approval_request_persists_after_reopen; -mod scenario_approve_always_persists_cross_thread; -mod scenario_ask_each_time_resumes_once; -mod scenario_concurrent_dual_gate_resume; -mod scenario_failure_category_demasked; -mod scenario_gate_ref_edge_cases; -mod scenario_gate_then_approve; -mod scenario_gate_then_deny; - -use reborn_support::builder::StorageMode; -use reborn_support::group::{RebornIntegrationGroup, ScenarioReport}; - -#[tokio::test] -async fn approvals_group_e2e() { - let g = RebornIntegrationGroup::live_approvals() - .await - .expect("group builds"); - - let mut report = ScenarioReport::new(); - // Independent gate scenarios (run while auto-approve is still OFF — they are - // the control proving the gate is real before scenario 4 flips it ON). - report.record( - "gate_then_approve", - scenario_gate_then_approve::run(&g).await, - ); - report.record("gate_then_deny", scenario_gate_then_deny::run(&g).await); - report.record( - "concurrent_dual_gate_resume", - scenario_concurrent_dual_gate_resume::run(&g).await, - ); - report.record( - "failure_category_demasked", - scenario_failure_category_demasked::run(&g).await, - ); - report.record( - "stale_gate_ref_resume", - scenario_gate_ref_edge_cases::stale_gate_ref_resume(&g).await, - ); - report.record( - "missing_gate_bare_resolve", - scenario_gate_ref_edge_cases::missing_gate_bare_resolve(&g).await, - ); - // C-DURABLE: independent of the auto-approve toggle (its own gate, resolved - // before returning) — the approval-request store is always on-disk - // regardless of the group's `StorageMode` (a separate capability-harness - // filesystem), so this needs no `StorageMode::LibSql` variant. - report.record( - "approval_request_persists_after_reopen", - scenario_approval_request_persists_after_reopen::run(&g).await, - ); - // Dependent: must run last (flips the (tenant, user) auto-approve toggle ON). - scenario_approve_always_persists_cross_thread::run(&g) - .await - .expect("approve-always persists cross-thread"); - // W4-ASK-EACH-ONCE: MUST run after every other `builtin.write_file` - // scenario above -- it installs a persistent, group-wide - // `ToolPermissionOverride::AskEachTime` override for `builtin.write_file` - // (the same shared per-`(tenant, user)` CAS store `disable_outbound_target_set_tool` - // uses), which would otherwise force-gate their plain-Ask-mode writes too. - // Independent of the auto-approve toggle's value (`ask_each_time` always - // gates fresh invocations regardless of auto-approve -- see the scenario's - // module doc), so running it after the toggle flip is a stronger proof, - // not a weaker one. - report.record( - "ask_each_time_resumes_once", - scenario_ask_each_time_resumes_once::run(&g).await, - ); - report.assert_all_passed(); -} - -#[tokio::test] -async fn approvals_group_libsql_e2e() { - let g = RebornIntegrationGroup::builder() - .storage(StorageMode::LibSql) - .live_approvals() - .await - .expect("group builds"); - - let mut report = ScenarioReport::new(); - // Independent gate scenarios (run while auto-approve is still OFF — they are - // the control proving the gate is real before scenario 4 flips it ON). - report.record( - "gate_then_approve", - scenario_gate_then_approve::run(&g).await, - ); - report.record("gate_then_deny", scenario_gate_then_deny::run(&g).await); - report.record( - "concurrent_dual_gate_resume", - scenario_concurrent_dual_gate_resume::run(&g).await, - ); - report.record( - "failure_category_demasked", - scenario_failure_category_demasked::run(&g).await, - ); - report.record( - "stale_gate_ref_resume", - scenario_gate_ref_edge_cases::stale_gate_ref_resume(&g).await, - ); - report.record( - "missing_gate_bare_resolve", - scenario_gate_ref_edge_cases::missing_gate_bare_resolve(&g).await, - ); - // Dependent: must run last (flips the (tenant, user) auto-approve toggle ON). - scenario_approve_always_persists_cross_thread::run(&g) - .await - .expect("approve-always persists cross-thread"); - // W4-ASK-EACH-ONCE: MUST run after every other `builtin.write_file` - // scenario above -- see the non-libsql variant's comment above for why. - report.record( - "ask_each_time_resumes_once", - scenario_ask_each_time_resumes_once::run(&g).await, - ); - report.assert_all_passed(); -} diff --git a/tests/reborn_group_approvals/scenario_concurrent_dual_gate_resume.rs b/tests/reborn_group_approvals/scenario_concurrent_dual_gate_resume.rs deleted file mode 100644 index 743074961be..00000000000 --- a/tests/reborn_group_approvals/scenario_concurrent_dual_gate_resume.rs +++ /dev/null @@ -1,155 +0,0 @@ -//! HEADLINE scenario for Option P: two threads simultaneously parked on the -//! group's ONE shared `TurnCoordinator`, resolved independently with opposite -//! dispositions, asserting each run's own gate disposition was applied and -//! neither run's resume disturbed the other's pending gate or terminal state. -//! -//! Note on scope: the on-disk assertions below prove resume is dispatched -//! correctly by `run_id` (A's approval landed A's write; B's denial never -//! produced B's write). They do **not** prove per-thread workspace -//! isolation — `GroupSharedStorage` builds ONE `capability_recorder` / -//! workspace root for the whole group (`into_group`, -//! `tests/support/reborn/group.rs`), so `thread_a` and `thread_b` read and -//! write the *same* on-disk workspace. A file written under one thread's -//! handle is trivially visible from the other's, regardless of resume -//! correctness — that was confirmed empirically: an added -//! `thread_b.assert_workspace_file_absent("concurrent_a.txt")` check failed -//! because A's approved write is visible through B's handle on the shared -//! workspace. Cross-thread workspace isolation is therefore not a property -//! this group harness enforces or this scenario can test. -//! -//! ## Why this needs a new scenario (consolidate-don't-proliferate, CLAUDE.md -//! "Testing Discipline") -//! -//! `scenario_gate_then_approve` / `scenario_gate_then_deny` / -//! `scenario_approve_always_persists_cross_thread` each drive ONE thread's gate -//! to resolution *before* the next thread submits a turn — fully sequential. -//! All group threads share ONE coordinator/scheduler -//! (`GroupSharedStorage::coordinator`, built once in -//! `RebornIntegrationGroupBuilder::into_group`, `tests/support/reborn/group.rs`), -//! so a worker can in principle pick up the wrong run for the wrong thread. -//! The only way to prove resume dispatch is keyed correctly (by `run_id`, not -//! by registration order or a shared scope) is to put TWO runs on the shared -//! coordinator in `Blocked` state AT THE SAME TIME — both turns in flight, -//! both parked on a gate, before either is resolved — then resolve them with -//! DIFFERENT, divergent dispositions and assert each thread's own real side -//! effect. No existing scenario can absorb this: doing it sequentially (as -//! the three existing scenarios do) cannot distinguish "resume keys on -//! run_id" from "resume keys on registration order" or "resume keys on a -//! constant", because there is never more than one blocked run on the -//! coordinator at a time to confuse. - -use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; -use super::reborn_support::reply::RebornScriptedReply; -use ironclaw_turns::TurnStatus; -use serde_json::json; - -pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { - // Two distinct threads, two distinct gated writes, different target files - // so each run's own disposition (A approved, B denied) is independently - // verifiable on disk — not just from in-process return values. Both - // threads share one on-disk workspace (see module note above), so this - // proves resume is dispatched by `run_id` rather than proving per-thread - // workspace isolation. - let thread_a = g - .thread("conv-concurrent-dual-gate-a") - .script([ - RebornScriptedReply::tool_call( - "builtin.write_file", - json!({"path": "/workspace/concurrent_a.txt", "content": "thread A approved"}), - ), - RebornScriptedReply::text("A: write approved"), - ]) - .build() - .await?; - let thread_b = g - .thread("conv-concurrent-dual-gate-b") - .script([ - RebornScriptedReply::tool_call( - "builtin.write_file", - json!({"path": "/workspace/concurrent_b.txt", "content": "thread B should not persist"}), - ), - RebornScriptedReply::text("B: write was not authorized"), - ]) - .build() - .await?; - - // Submit both turns and drive both to `BlockedApproval`, then resolve them - // one at a time. The essential coverage is COEXISTENCE: two distinct runs - // are simultaneously parked on the ONE shared coordinator (A stays blocked - // while B is submitted and blocked), and each gate is then resolved by its - // own `run_id`. We deliberately do NOT `tokio::join!` the submit/resume - // calls: hammering the shared CAS-over-libsql turn-state store with two - // *truly parallel* read-modify-write turns is a separate, prod-relevant - // concurrency concern (see issue: parallel same-tenant runs vs the - // `FilesystemTurnStateStore` CAS loop / libsql backend) that this - // test-framework refactor is not the place to fix — and it is orthogonal to - // what this scenario proves (run_id-keyed resume on a shared coordinator). - let (run_a, gate_a) = thread_a - .submit_turn_until_blocked("write the concurrent A file") - .await?; - let (run_b, gate_b) = thread_b - .submit_turn_until_blocked("write the concurrent B file") - .await?; - - // Non-vacuity: two genuinely distinct runs/gates are actually in flight — - // if the harness collapsed both threads onto one run this would catch it - // before the resolution step even starts. - if run_a == run_b { - return Err(format!("expected distinct run ids, both runs were {run_a}").into()); - } - if gate_a.as_str() == gate_b.as_str() { - return Err(format!("expected distinct gate refs, both were {gate_a:?}").into()); - } - - // Resolve independently with OPPOSITE dispositions. APPROVE A first and - // drive it to completion while B is STILL blocked — if resume keyed on - // anything other than `run_id` (a constant, or "most recently blocked - // run"), approving A would disturb B's still-pending gate. - thread_a.approve_gate(run_a, &gate_a).await?; - let state_a = thread_a - .wait_for_status(run_a, TurnStatus::Completed) - .await?; - - // Now DENY B. Its gate must still be intact and independently resolvable. - thread_b.deny_gate(run_b, &gate_b).await?; - let state_b = thread_b - .wait_for_status(run_b, TurnStatus::Completed) - .await?; - - // No error-category leakage: a genuinely-correct concurrent resume reaches - // `Completed` with no recorded failure on either run. Cross-resume bugs in - // this shared-coordinator shape historically surface as - // `driver_protocol_violation` (the masked failure category the de-mask fix - // addresses) or `TraceLlm exhausted` (a worker draining the wrong thread's - // scripted-reply deque) — assert neither leaked onto either run. - for (label, state) in [("A", &state_a), ("B", &state_b)] { - if let Some(failure) = &state.failure { - return Err(format!( - "thread {label} run reached Completed but recorded a failure \ - (no error category should leak on a clean concurrent resume): {failure:?}" - ) - .into()); - } - } - - // A's gate was APPROVED: the gated capability re-dispatched and the real - // write landed on disk under A's own path. - thread_a - .assert_workspace_file_contains("concurrent_a.txt", "thread A approved") - .await?; - // B's gate was DENIED: the gated capability was never re-dispatched — B's - // file must be absent. If A's approval had cross-bled onto B's run (e.g. - // resume dispatch ignoring `run_id`), this file would exist. - thread_b - .assert_workspace_file_absent("concurrent_b.txt") - .await?; - // Same on-disk fact, re-checked via `thread_a`'s handle (both handles - // read the same shared workspace — see module note above): rules out a - // resume path that writes to "whichever gate resolved last" regardless - // of which run it came from. - thread_a - .assert_workspace_file_absent("concurrent_b.txt") - .await?; - - Ok(()) -} diff --git a/tests/reborn_group_approvals/scenario_failure_category_demasked.rs b/tests/reborn_group_approvals/scenario_failure_category_demasked.rs deleted file mode 100644 index 98c70f814b9..00000000000 --- a/tests/reborn_group_approvals/scenario_failure_category_demasked.rs +++ /dev/null @@ -1,102 +0,0 @@ -//! Scenario: a genuinely-FAILED group run reports its TRUE failure category, -//! not the masking `driver_protocol_violation`. -//! -//! `RebornIntegrationGroupBuilder::into_group` (`tests/support/reborn/group.rs`, -//! ~line 472) wires `.with_checkpoint_state_store(checkpoint_state_store.clone())` -//! onto the group-level `ThreadCheckpointLoopExitEvidencePort` — that is the -//! de-mask fix. Without it, `ThreadCheckpointLoopExitEvidencePort::verify_failure_evidence` -//! (`crates/ironclaw_reborn/src/loop_exit_applier.rs`) returns `Ok(false)` -//! unconditionally (it short-circuits on `self.checkpoint_state_store` being -//! `None`), so `LoopExitApplier::apply`'s `validate_failed_exit` -//! (`crates/ironclaw_turns/src/loop_exit.rs`) treats every `Failed` exit as -//! `LoopExitViolationKind::UnverifiedFailureEvidence` and rewrites it to the -//! opaque `"driver_protocol_violation"` category, discarding the loop's real -//! failure reason. With the store wired, a failed exit whose checkpoint -//! actually recorded the claimed failure kind verifies and the run's TRUE -//! category survives onto `TurnRunState::failure`. -//! -//! ## Why this needs a NEW scenario (consolidate-don't-proliferate, root -//! CLAUDE.md "Testing Discipline") -//! -//! Every other scenario in this binary (`scenario_gate_then_approve`, -//! `scenario_gate_then_deny`, `scenario_concurrent_dual_gate_resume`, -//! `scenario_approve_always_persists_cross_thread`) asserts CLEAN completion -//! or gate resolution — none of them drives a run all the way to -//! `TurnStatus::Failed`, so none of them can observe the de-mask wiring at -//! all: `verify_failure_evidence` is only ever called on `LoopExit::Failed`. -//! `concurrent_dual_gate_resume`'s module doc explicitly calls out -//! `driver_protocol_violation` as a failure mode it asserts AGAINST (no -//! failure leaks on a clean run) — it cannot also be the test that proves the -//! masked category is correctly de-masked on a genuinely failed run, because -//! it never produces one. The `loop_exit_applier` unit tests -//! (`crates/ironclaw_reborn/src/loop_exit_applier/tests/mod.rs`) exercise -//! `verify_failure_evidence` directly against a hand-built fixture; they do -//! NOT prove the group composition in `group.rs` actually wires the store -//! through `into_group` end-to-end over the real scheduler/coordinator. This -//! scenario is the only one that closes that gap. -//! -//! ## How the failure is produced -//! -//! The thread is built with an EMPTY scripted-reply list -//! (`.script([])`). `TraceLlm::next_step` (`tests/support/trace_llm.rs`) -//! returns `LlmError::RequestFailed` ("TraceLlm exhausted: served 0 call(s), -//! no steps left") on the very first model call — deterministic, no tool -//! dispatch or timing race involved. That model error flows through the -//! bounded-retry `RecoveryStrategy` (`crates/ironclaw_agent_loop/src/strategies/recovery.rs`, -//! default `max_attempts_per_class: 2`) until the retry budget is exhausted, -//! at which point the loop emits `LoopExit::Failed` with -//! `LoopFailureKind::ModelError` and the run reaches `TurnStatus::Failed`. -//! This is exactly the original flake's failure shape (model/driver failure -//! racing checkpoint evidence), now correctly surfaced through the de-masked -//! path instead of being swallowed into `driver_protocol_violation`. - -use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; -use ironclaw_turns::TurnStatus; - -pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { - // No scripted replies at all: the very first model call exhausts the - // scripted provider, deterministically driving the run to a genuine - // `Failed` terminal state (see module doc). - let h = g - .thread("conv-failure-category-demasked") - .script([]) - .build() - .await?; - - let run_id = h.submit_turn_async("trigger a model failure").await?; - let state = h.wait_for_status(run_id, TurnStatus::Failed).await?; - - let failure = state - .failure - .as_ref() - .ok_or("run reached Failed but TurnRunState::failure was None")?; - - // The de-mask fix's entire point (see module doc): the TRUE category - // must survive, not the masking sentinel. - if failure.category() == "driver_protocol_violation" { - return Err(format!( - "failure category was the masking sentinel \"driver_protocol_violation\"; \ - the group-level checkpoint_state_store wiring (group.rs ~line 472) is not \ - de-masking the real failure category (got: {failure:?})" - ) - .into()); - } - - // Empirically discovered: a `TraceLlm` exhaustion on the very first model - // call surfaces, after the bounded model-error retry budget is spent, as - // `LoopFailureKind::ModelError` → category `"model_error"` - // (`crates/ironclaw_turns/src/loop_exit.rs`, - // `LoopFailureKind::category` -> `Self::ModelError => "model_error"`). - // Asserting the exact value (not just `!= driver_protocol_violation`) - // proves the de-masked path produces the loop's REAL reason, not some - // other incidental category. - if failure.category() != "model_error" { - return Err(format!( - "expected de-masked failure category \"model_error\" (TraceLlm exhaustion -> \ - LoopFailureKind::ModelError), got {failure:?}" - ) - .into()); - } - - Ok(()) -} diff --git a/tests/reborn_group_extensions/scenario_activate_then_active_cross_thread.rs b/tests/reborn_group_extensions/scenario_activate_then_active_cross_thread.rs deleted file mode 100644 index f06300fe2aa..00000000000 --- a/tests/reborn_group_extensions/scenario_activate_then_active_cross_thread.rs +++ /dev/null @@ -1,155 +0,0 @@ -//! Scenario 3 (HEADLINE): install an extension in thread A, ACTIVATE it in -//! thread B (a DIFFERENT conversation), and confirm thread C (yet another -//! conversation) observes the extension as ACTIVE — not merely installed — over -//! the shared store. This closes the `extension_activate` int-tier gap: install, -//! search, and remove already have cross-thread coverage; activation did not. -//! -//! Uses "web-access" (NOT "github"/"notion") for two reasons. First, it is the -//! only bundled extension that activates WITHOUT credentials / without raising an -//! auth gate, so activation reaches a SUCCESS result (`activated:true`) in this -//! harness rather than blocking on a credential gate. "github" et al. require -//! credentials and would return an auth gate (confirmed by the unit test -//! `local_dev_extension_activate_returns_auth_gate_for_missing_extension_credentials` -//! in `extension_lifecycle_capabilities.rs`). Second, it is untouched by Scenario -//! 1 ("github") and Scenario 2 ("notion"), so a fresh install→activate cycle here -//! observes a real lifecycle transition over the shared store rather than an -//! already-installed/activated no-op. -//! -//! Key behaviours asserted (from `extension_lifecycle.rs::commit_activation` and -//! `search_installation_phase`): a successful activate yields `"activated":true` -//! plus a `visible_capability_ids` array of the now-published capability ids; and -//! `extension_search` renders an ACTIVE extension as `"installation_phase":"active"` -//! versus `"installed"` for an installed-but-not-yet-activated one. -//! -//! Because all three conversations use different conversation IDs but the same -//! `Arc`, asserting that thread C sees -//! `installation_phase:active` proves cross-thread ACTIVATION persistence: an -//! activate in thread B is durably visible to thread C, and the extension's -//! capability surface (e.g. `web-access.search`) has come online. - -use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; -use super::reborn_support::reply::RebornScriptedReply; -use serde_json::json; - -pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { - // ── Thread A: installer ───────────────────────────────────────────────── - // Install "web-access" so there is an installed-but-inactive extension to - // activate. The install persists to the shared HostRuntimeCapabilityHarness - // filesystem so the activator thread sees it immediately. - let installer = g - .thread("ext-activate-phase-install") - .script([ - RebornScriptedReply::tool_call( - "builtin.extension_install", - json!({"extension_id": "web-access"}), - ), - RebornScriptedReply::text("installed"), - ]) - .build() - .await?; - installer.submit_turn("install web-access").await?; - installer - .assert_tool_invoked("builtin.extension_install") - .await?; - // Confirm the install succeeded: output carries `"installed":true`. - installer - .assert_tool_result_contains("\"installed\":true") - .await?; - - // ── Thread B: activator (DIFFERENT conversation, SAME shared store) ────── - // A distinct conversation_id → distinct binding/thread scope, but the same - // `HostRuntimeCapabilityHarness`, so the activator can see and activate the - // installation Thread A just wrote. "web-access" needs no credentials, so - // activation reaches a SUCCESS result instead of an auth gate. - let activator = g - .thread("ext-activate-phase-activate") - .script([ - RebornScriptedReply::tool_call( - "builtin.extension_activate", - json!({"extension_id": "web-access"}), - ), - RebornScriptedReply::text("activated"), - ]) - .build() - .await?; - activator.submit_turn("activate web-access").await?; - activator - .assert_tool_invoked("builtin.extension_activate") - .await?; - // Direct effect: activation succeeded → output carries `"activated":true`. - // Assert the VALUE, not just the key, so an `activated:false` / auth-gate - // outcome cannot satisfy this. - activator - .assert_tool_result_contains("\"activated\":true") - .await?; - // Capability surfaces: the activate payload's `visible_capability_ids` array - // carries the now-published capability ids. `web-access.search` coming online - // is the observable proof that activation published the extension's tool - // surface (mere install does NOT publish capabilities). - activator - .assert_tool_result_contains(r#""web-access.search""#) - .await?; - - // ── Thread C: viewer (DIFFERENT conversation, SAME shared store) ───────── - // A third distinct conversation_id over the Arc-cloned store. Searching for - // "web-access" must now report it as ACTIVE — observing Thread B's activation - // across threads. - let viewer = g - .thread("ext-activate-phase-viewer") - .script([ - RebornScriptedReply::tool_call( - "builtin.extension_search", - json!({"query": "web-access"}), - ), - RebornScriptedReply::text("searched"), - ]) - .build() - .await?; - viewer - .submit_turn("search web-access after activation") - .await?; - viewer - .assert_tool_invoked("builtin.extension_search") - .await?; - // Cross-thread activation persistence: the search result carries - // `installation_phase:"active"` only because Thread B's activation enabled the - // installation in the shared store. Assert the VALUE so a still-`installed` - // phase cannot satisfy this. - viewer - .assert_tool_result_contains(r#""installation_phase":"active""#) - .await?; - - // Discriminating guard: an extension that was installed but NOT activated - // renders `installation_phase:"installed"`. Its ABSENCE here proves the phase - // genuinely advanced past install — a no-op activate (leaving the extension - // inactive) would surface `"installed"` and trip this guard. - if viewer - .assert_tool_result_contains(r#""installation_phase":"installed""#) - .await - .is_ok() - { - return Err( - "web-access still shows installation_phase:installed after a cross-thread activate; \ - builtin.extension_activate did not advance the lifecycle through the shared store" - .into(), - ); - } - - // Non-vacuity guard: "web-access" must still appear in the catalog search - // result, proving the search actually ran and returned a catalog entry. The - // presence of `installation_phase:active` is therefore meaningful — not a - // symptom of an empty or errored result. - if viewer - .assert_tool_result_contains("\"web-access\"") - .await - .is_err() - { - return Err( - "non-vacuity guard failed: web-access catalog entry must appear in search results \ - after activation; the active-phase assertion would otherwise be vacuous" - .into(), - ); - } - - Ok(()) -} diff --git a/tests/reborn_group_extensions/scenario_install_unknown_extension_id_fails_safely.rs b/tests/reborn_group_extensions/scenario_install_unknown_extension_id_fails_safely.rs deleted file mode 100644 index 847c8882333..00000000000 --- a/tests/reborn_group_extensions/scenario_install_unknown_extension_id_fails_safely.rs +++ /dev/null @@ -1,58 +0,0 @@ -//! Scenario 4 (W4-EXT-MANIFEST-ERR, narrowed): `builtin.extension_install` -//! with an `extension_id` that is not in the bundled catalog fails safely with -//! a model-visible tool error, instead of panicking or silently no-oping. -//! -//! `builtin.extension_install`'s only input is `extension_id: String`, resolved -//! against a FIXED, compile-time-embedded catalog -//! (`AvailableExtensionCatalog::resolve`, `available_extensions.rs`) — there is -//! no live path for a model/user to submit raw manifest TOML through this -//! capability (every bundled manifest is asset-embedded and always valid), so -//! the originally-scoped "schema mismatch / reserved id / forbidden trust -//! level" arms (`ironclaw_extensions::v2::ManifestV2Error` variants) are not -//! reachable through `extension_install` in production. The one genuinely -//! reachable, wired error arm through this capability is an unknown -//! `extension_id`: `catalog.resolve` returns -//! `ProductWorkflowError::InvalidBindingRequest`, which -//! `extension_lifecycle_capabilities.rs::lifecycle_error` maps to -//! `RuntimeDispatchErrorKind::InputEncode`, rendered by the executor as the -//! `"invalid_input"` reason token — a `Failed` (not `Denied`) capability -//! outcome, distinct from the `"installed":true` success path Scenario 1 -//! already covers. - -use super::reborn_support::assertions::ToolErrorClass; -use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; -use super::reborn_support::reply::RebornScriptedReply; -use serde_json::json; - -pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { - let h = g - .thread("ext-install-unknown-id") - .script([ - RebornScriptedReply::tool_call( - "builtin.extension_install", - json!({"extension_id": "not-a-real-bundled-extension"}), - ), - RebornScriptedReply::text("could not install that extension"), - ]) - .build() - .await?; - h.submit_turn("install the not-a-real-bundled-extension extension") - .await?; - h.assert_tool_invoked("builtin.extension_install").await?; - - // The capability outcome is `Failed{invalid_input}` — proved as a *class* - // (not a needle-prefix convention): the same reason string can in - // principle render under either class, so asserting the class - // discriminates structurally, not just by text convention. - h.assert_tool_error(ToolErrorClass::Failed, "invalid_input") - .await?; - - // Discriminating negative arm: the SAME reason token under the OTHER - // class must be absent, proving `assert_tool_error`'s class argument is - // load-bearing here rather than a convention `assert_tool_result_contains` - // would also satisfy. - h.assert_no_tool_error(ToolErrorClass::Denied, "invalid_input") - .await?; - - Ok(()) -} diff --git a/tests/reborn_group_journeys/main.rs b/tests/reborn_group_journeys/main.rs deleted file mode 100644 index 248abbf490a..00000000000 --- a/tests/reborn_group_journeys/main.rs +++ /dev/null @@ -1,145 +0,0 @@ -//! C-JOURNEY — multi-turn Reborn journeys: the deterministic twins of the live -//! canary use cases that chain a gate → resume → next turn on ONE -//! conversation/harness. Distinct from `reborn_group_approvals` / -//! `reborn_integration_auth_gate` (single-gate mechanics): the value here is the -//! CHAINING across turns and the cross-permutation matrix below. -//! -//! ## Permutation matrix (inbound source × gate class × outcome) -//! -//! | scenario | inbound | gate(s) | outcome | -//! |---------------------------------------|---------------|-------------------------------|----------------| -//! | interactive_approval_journey | interactive | approval → approval | approve, deny, follow-up | -//! | auth_then_approval_journey | interactive | (approval→auth) → approval | approve+resolve, approve, follow-up | -//! | auth_deny_then_retry_journey | interactive | (approval→auth) → (approval→auth) | approve+deny, approve+resolve | -//! | multi_actor_gate_isolation | interactive×2 | approval (A) / approval (B) | per-actor gate + resume isolation | -//! -//! Shared turn-script helpers keep permutations from fanning out into N -//! near-identical fully-expanded scenarios (see each scenario module). -//! -//! Gate-arm discipline: a gated tool-call turn consumes exactly TWO script -//! entries (the call + one post-resume model call) regardless of approve/deny; a -//! plain follow-up turn consumes ONE. -//! -//! ## Auth→approval convergence (C-JOURNEY enabler) -//! -//! `auth_then_approval_journey` and `auth_deny_then_retry_journey` run on a -//! SECOND group, `RebornIntegrationGroup::live_auth_and_approval()` (built -//! from `HostRuntimeCapabilityHarness::file_and_github_auth_tools`), NOT -//! `live_approvals` — do not add them to the `live_approvals` group above. -//! The enabler: converge the auth gate onto the SAME `build_reborn_services` -//! runtime `live_approvals` already uses (unlike `live_auth_gate`, a separate, -//! lower-level `HostRuntimeServices` build with a hardcoded credential -//! resolver and no `run_state`/`approval_requests`/`capability_leases` stores -//! — see `reborn_integration_auth_gate.rs`'s deferred-arm note, now -//! superseded by this group for the happy-resume case). No GitHub credential -//! account is seeded at construction, so `github.get_repo` first raises a -//! real `BlockedApproval` (this harness's global auto-approve is disabled for -//! the file-tool arm, and that toggle is not capability-scoped, so it also -//! gates the WASM github capability); approving re-dispatches the -//! still-uncredentialed capability, which blocks AGAIN at a real -//! `TurnStatus::BlockedAuth`. `resolve_auth_gate` seeds a credential through -//! the REAL `ProductAuthRuntimeCredentialResolver` and resumes, letting the -//! SAME parked capability re-dispatch and complete. Making `github.*` -//! genuinely dispatchable on this runtime (not just granted/trusted) required -//! two additive `#[cfg(feature = "test-support")]` composition seams — see -//! `HostRuntimeCapabilityHarness::file_and_github_auth_tools`'s doc comment -//! (`tests/support/reborn/harness.rs`) for the mechanism (active-registry -//! publish + real asset-directory mount). -//! -//! ## Deferred / blocked permutations -//! -//! - **Triggered-origin chained journey** (trigger fire → gate → resume → -//! follow-up): the scripted-gateway seam this needs -//! (`RebornIntegrationHarness::submit_triggered_turn_scripted`) mints the -//! trigger's own scope and registers a scripted gateway for it, so a -//! triggered run can be driven to completion (see -//! `reborn_integration_triggered_submit`). A CHAINED triggered journey that -//! additionally parks on a real approval/auth gate is a FOLLOW-UP: resolving -//! such a gate must reconcile the trigger's minted owner scope with the -//! journey approval helpers' (`approve_gate`/`resolve_auth_gate`) binding -//! scope. Single-turn triggered-origin coverage already exists. -//! - **Multi-actor GATED journey** (`multi_actor_gate_isolation`): runs on -//! `RebornIntegrationGroup::multiuser_approvals()`, whose per-actor -//! capability dispatch (the C-MULTIUSER `scope_capability_by_run_owner` -//! harness seam) scopes each actor's gated write to ITS OWN run owner — so -//! actor B no longer dies with `driver_protocol_violation` under actor A's -//! user. The scenario disables auto-approve for each owner (that group -//! defaults auto-approve ON per owner) so both actors raise a real -//! `BlockedApproval`, then pins that gate resolution + resume state stay -//! bound to the raising actor. (Plain — non-gated — distinct-actor isolation -//! is covered by `reborn_group_multiuser::two_actors_own_threads`.) - -#[allow(dead_code)] -#[path = "../support/reborn/mod.rs"] -mod reborn_support; -#[allow(dead_code)] -#[path = "../support/mod.rs"] -mod support; - -mod scenario_auth_deny_then_retry_journey; -mod scenario_auth_then_approval_journey; -mod scenario_interactive_approval_journey; -mod scenario_multi_actor_gate_isolation; - -use reborn_support::group::{RebornIntegrationGroup, ScenarioReport}; - -#[tokio::test] -async fn journeys_group_e2e() { - let g = RebornIntegrationGroup::live_approvals() - .await - .expect("group builds"); - - let mut report = ScenarioReport::new(); - report.record( - "interactive_approval_journey", - scenario_interactive_approval_journey::run(&g).await, - ); - report.assert_all_passed(); -} - -/// C-JOURNEY: auth→approval convergence journeys, on SEPARATE -/// `live_auth_and_approval` groups (not the `live_approvals` group above — -/// see the module-level "Auth→approval convergence" note). -/// -/// ONE GROUP PER SCENARIO (not shared): `resolve_auth_gate` seeds a -/// `UserReusable` GitHub credential account under the group's canonical -/// `(tenant, user, agent, project)` scope — exactly like production, where a -/// submitted token persists for the user. On a SHARED group, scenario 1's -/// seeded credential would make scenario 2's `github.get_repo` resolve -/// immediately instead of raising the fresh `BlockedAuth` gate the scenario -/// pins (verified: the shared-group variant fails with "expected BlockedAuth -/// but run reached terminal status Completed"). Each journey needs a -/// pristine no-credential runtime, so each builds its own group. -#[tokio::test] -async fn journeys_group_auth_convergence_e2e() { - let mut report = ScenarioReport::new(); - - let g = RebornIntegrationGroup::live_auth_and_approval() - .await - .expect("auth+approval group builds"); - report.record( - "auth_then_approval_journey", - scenario_auth_then_approval_journey::run(&g).await, - ); - - let g_deny = RebornIntegrationGroup::live_auth_and_approval() - .await - .expect("auth+approval deny group builds"); - report.record( - "auth_deny_then_retry_journey", - scenario_auth_deny_then_retry_journey::run(&g_deny).await, - ); - report.assert_all_passed(); -} - -/// The multi-actor GATED journey — see the module doc's "Multi-actor GATED -/// journey" note for the `scope_capability_by_run_owner` seam this requires. -#[tokio::test] -async fn multi_actor_gate_isolation() { - let g = RebornIntegrationGroup::multiuser_approvals() - .await - .expect("group builds"); - scenario_multi_actor_gate_isolation::run(&g) - .await - .expect("multi-actor gated journey"); -} diff --git a/tests/reborn_group_journeys/scenario_auth_deny_then_retry_journey.rs b/tests/reborn_group_journeys/scenario_auth_deny_then_retry_journey.rs deleted file mode 100644 index 0a5f83b7cf6..00000000000 --- a/tests/reborn_group_journeys/scenario_auth_deny_then_retry_journey.rs +++ /dev/null @@ -1,106 +0,0 @@ -//! C-JOURNEY convergence scenario (the "mixed arm" companion to -//! `scenario_auth_then_approval_journey`): a single conversation whose FIRST -//! auth gate is DENIED, then a SECOND auth gate on the SAME thread is -//! RESOLVED — proving a denied auth gate does not poison a later successful -//! resolve on the same run/thread. Complementary to -//! `reborn_integration_auth_gate.rs` (which already covers single-turn -//! auth-deny in isolation): the value here is the CHAIN across two turns. -//! Each turn also chains an APPROVAL gate before the auth gate (see -//! `scenario_auth_then_approval_journey`'s module doc for why `github.get_repo` -//! raises both in sequence on this harness). -//! -//! turn 1: `github.get_repo` -> `BlockedApproval` -> APPROVE -> `BlockedAuth` -//! -> DENY -> resumes, the model sees a non-retryable authorization -//! failure, no re-dispatch -> `Completed`; -//! turn 2: `github.get_repo` again (still no credential seeded by turn 1's -//! deny) -> a FRESH `BlockedApproval` -> APPROVE -> a FRESH -//! `BlockedAuth` -> RESOLVE (seed a real credential + resume) -> -//! the SAME parked capability re-dispatches for real -> `Completed`. - -use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; -use super::reborn_support::reply::RebornScriptedReply; -use ironclaw_turns::TurnStatus; -use serde_json::json; - -pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { - let h = g - .thread("conv-journey-auth-deny-then-retry") - .script([ - // turn 1 (2 entries: approval+auth-gated call + post-deny reply) - RebornScriptedReply::tool_call( - "github.get_repo", - json!({"owner": "octocat", "repo": "hello-world"}), - ), - RebornScriptedReply::text("could not look up the repo without authorization"), - // turn 2 (2 entries: approval+auth-gated call + post-resolve reply) - RebornScriptedReply::tool_call( - "github.get_repo", - json!({"owner": "octocat", "repo": "hello-world"}), - ), - RebornScriptedReply::text( - "AUTHDENYRETRY_TURN2 repo info retrieved after connecting github", - ), - ]) - .build() - .await?; - - // --- turn 1: approve the action, then DENY the auth gate --- - let (run1, approval_gate1) = h - .submit_turn_until_blocked("AUTHDENYRETRY_TURN1 look up the repo the first time") - .await?; - h.approve_gate(run1, &approval_gate1).await?; - let auth_state1 = h.wait_for_status(run1, TurnStatus::BlockedAuth).await?; - let auth_gate1 = auth_state1 - .gate_ref - .ok_or("blocked auth run missing gate ref")?; - h.deny_auth_gate(run1, &auth_gate1).await?; - h.wait_for_status(run1, TurnStatus::Completed).await?; - // Pin WHAT the deny path produced, not just that it reached a terminal - // status: the finalized reply must be turn 1's own scripted post-deny - // text (never turn 2's), and zero network egress must have escaped - // despite the model still being told about the (declined) capability — - // this also anchors the "turn 2's tool result can't be turn 1 residue" - // reasoning in the `assert_tool_result_contains` pin below. - h.assert_reply_contains("could not look up the repo without authorization") - .await?; - h.assert_egress_count(0).await?; - - // --- turn 2: SAME conversation, github call raises BOTH gates AGAIN - // (turn 1's deny did not seed a credential or leave a stale approval; - // a fresh dispatch still needs a fresh approval + still resolves - // AuthRequired) -> approve, then RESOLVE (seed credential + resume) --- - let (run2, approval_gate2) = h - .submit_turn_until_blocked("AUTHDENYRETRY_TURN2 look up the repo the second time") - .await?; - if run2 == run1 { - return Err("turn 2 reused turn 1's run id -- turns did not chain".into()); - } - h.approve_gate(run2, &approval_gate2).await?; - let auth_state2 = h.wait_for_status(run2, TurnStatus::BlockedAuth).await?; - let auth_gate2 = auth_state2 - .gate_ref - .ok_or("blocked auth run missing gate ref")?; - h.resolve_auth_gate(run2, &auth_gate2).await?; - h.wait_for_status(run2, TurnStatus::Completed).await?; - h.assert_reply_contains("AUTHDENYRETRY_TURN2").await?; - // The SECOND github dispatch (post-resolve) actually EXECUTED through the - // real capability path: the scripted network fixture body surfaced back as - // a recorded Completed-path result — proving turn 1's deny did not poison - // this run's later successful resolve. (`assert_tool_invoked` alone is not - // discriminating — see `scenario_auth_then_approval_journey`; turn 1's - // denied dispatch never produces a result, so the recorded result here can - // only come from turn 2's credential-backed re-dispatch.) - // - // `assert_tool_result_contains` itself scans ALL captured results since - // thread baseline, so on its own it can't rule out turn-1 residue. There - // is no `*_since`-scoped variant for successful tool results (only - // `assert_tool_error_since` exists, for the error family — see - // `tests/support/reborn/assertions.rs`), so the discriminating power here - // comes from PAIRING with the turn-1 negative arm above: turn 1's deny - // path is now pinned to have produced zero egress - // (`assert_egress_count(0)`) and its own distinct scripted reply, so no - // successful "octocat/hello-world" result could have originated there — - // this assertion can only be satisfied by turn 2's own dispatch. - h.assert_tool_result_contains("octocat/hello-world").await?; - Ok(()) -} diff --git a/tests/reborn_group_triggers/scenario_trigger_self_create_denied.rs b/tests/reborn_group_triggers/scenario_trigger_self_create_denied.rs deleted file mode 100644 index 7a8e5659e0f..00000000000 --- a/tests/reborn_group_triggers/scenario_trigger_self_create_denied.rs +++ /dev/null @@ -1,153 +0,0 @@ -//! C-DENYEDGE (row 4): a scheduled-trigger fire must not be able to create -//! (or remove/pause/resume) triggers of its own — the int-tier twin of the -//! `ironclaw_reborn::runtime` unit coverage for issue #5505 -//! (`SCHEDULED_TRIGGER_DENIED_CAPABILITY_IDS`, PR #5515). -//! -//! Drives a REAL triggered-origin run (`submit_triggered_turn_scripted`, -//! `TurnOriginKind::ScheduledTrigger`) that scripts a `builtin.trigger_create` -//! call. The trusted-trigger submit path (`ironclaw_conversations::inbound`) -//! sets `requested_run_profile: RunProfileId::scheduled_trigger()` on the -//! `SubmitTurnRequest`, so the run resolves under the dedicated -//! `scheduled_trigger` run profile -//! (`crate::planned_driver_factory::scheduled_trigger_planned_profile_definition`), -//! whose `capability_surface_profile_id` is -//! `SCHEDULED_TRIGGER_CAPABILITY_SURFACE_PROFILE_ID`. The host's -//! `PerSurfaceCapabilityDenyDecorator` — wired unconditionally in -//! `ironclaw_reborn::runtime::build_default_planned_runtime`, keyed on that -//! profile id, resolving to a `CapabilitySurfaceDenyFilter` — strips -//! `builtin.trigger_create` (and remove/pause/resume; `trigger_list` stays -//! visible) from the fire's model-visible capability surface. -//! -//! ## Where the denial actually happens (traced, not assumed) -//! -//! A model tool call targeting a denied capability is rejected earlier than -//! `CapabilityStage`'s per-call `denied_calls` bucket -//! (`crates/ironclaw_agent_loop/src/executor/capabilities.rs`, the -//! `"capability is not visible in the filtered surface"` literal — that arm -//! is for a capability that WAS registered as a `CapabilityCallCandidate` -//! but fell outside the surface between advertisement and dispatch, e.g. a -//! stale-surface race). For a capability denied from the start (ours), -//! `ironclaw_reborn::model_gateway`'s response classification calls -//! `capabilities.validate_provider_tool_call(provider_call)` on every raw -//! provider tool call BEFORE registering it -//! (`crates/ironclaw_reborn/src/model_gateway.rs` ~line 1226); the deny -//! filter's `CapabilitySurfaceDenyFilter::validate_provider_tool_call` -//! (`crates/ironclaw_loop_support/src/capability_surface_filter.rs`) returns -//! `AgentLoopHostErrorKind::InvalidInvocation` with the fixed message -//! `"provider tool call targets a disabled capability"`, and -//! `map_provider_tool_output_error` maps that to -//! `HostManagedModelError::safe(HostManagedModelErrorKind::InvalidOutput, ..)`. -//! The whole model response is rejected at the GATEWAY seam — no -//! `CapabilityCallCandidate` is ever constructed, so `CapabilityStage` never -//! runs for this call, and nothing is appended via -//! `append_capability_result_ref`/`append_tool_result_reference` (confirmed -//! empirically: the triggered thread's persisted history is exactly -//! `[User, Assistant]` — no `ToolResultReference` message at all). The -//! executor transparently re-issues the model call (consuming the second -//! scripted `text(..)` reply) rather than gating or failing the run. -//! -//! Because nothing is persisted to thread history or the in-process capability -//! recorder for this seam, `assert_tool_error`/`assert_tool_error_summary_contains` -//! (which both read persisted `ToolResultReference` envelopes) cannot observe -//! it — there is no host-authored summary string to pin here, unlike the -//! gate-declined/filtered-surface-race families those assertions were built -//! for. This scenario instead asserts the actual security property directly: -//! (1) `builtin.trigger_create` was never dispatched (`assert_tool_invoked` -//! returns `Err`), (2) the run completes cleanly (no hang, no terminal -//! failure), and (3) no trigger with the attempted name exists in the shared -//! repository afterward, verified by dispatching a real (non-triggered) -//! `builtin.trigger_list` call, whose surface is unaffected by the -//! scheduled-trigger deny map. -//! -//! Distinct from `scenario_verbs_lifecycle`'s `trigger_create` coverage in -//! this same binary: that scenario submits through the plain (non-triggered) -//! `submit_turn` wire, which resolves the default/interactive run profile — -//! the trigger-mutator surface is fully visible there, so it never exercises -//! this deny map at all. - -use super::reborn_support::group::{HarnessResult, RebornIntegrationGroup}; -use super::reborn_support::reply::RebornScriptedReply; -use ironclaw_turns::TurnStatus; -use serde_json::json; - -/// Distinctive enough that a false-positive match against another scenario's -/// trigger name (e.g. `scenario_verbs_lifecycle`'s `"t0-triggers-once"`) is -/// not a concern. -const SELF_CREATE_ATTEMPT_TRIGGER_NAME: &str = "self-created-follow-up-should-not-exist"; - -pub async fn run(g: &RebornIntegrationGroup) -> HarnessResult<()> { - let h = g.thread("conv-trigger-self-create-denied").build().await?; - - let submission = h - .submit_triggered_turn_scripted( - "create a follow-up reminder", - [ - RebornScriptedReply::tool_call( - "builtin.trigger_create", - json!({ - "name": SELF_CREATE_ATTEMPT_TRIGGER_NAME, - "prompt": "remind me again", - "schedule": {"kind": "once", "at": "2999-01-01T00:00:00", "timezone": "UTC"}, - }), - ), - RebornScriptedReply::text("understood, I can't schedule that myself"), - ], - ) - .await?; - - // Must NOT hang or fail: the denial is a model-recoverable outcome, not - // a gate or a terminal failure. - h.wait_for_status_in_scope( - &submission.turn_scope, - submission.run_id, - TurnStatus::Completed, - ) - .await?; - - // The capability must never have reached dispatch at all — a - // discriminating negative check against the real in-process invocation - // recorder (not a bare `.is_err()` on something unrelated: this reads - // the SAME recorder `assert_tool_invoked` uses for the positive case - // in `scenario_verbs_lifecycle`). - if h.assert_tool_invoked("builtin.trigger_create") - .await - .is_ok() - { - return Err( - "expected builtin.trigger_create to be denied for a scheduled-trigger fire, \ - but the capability recorder shows it was invoked" - .into(), - ); - } - - // Strongest proof of the security property: no trigger with the - // attempted name actually exists in the shared repository. Verified - // through a genuine INTERACTIVE (non-triggered) `trigger_list` call on a - // fresh thread in the SAME group — `trigger_list` is read-only and stays - // visible on every profile, so this reads the real, current repository - // state rather than anything scoped to the denied run. - let verifier = g - .thread("conv-trigger-self-create-denied-verify") - .script([ - RebornScriptedReply::tool_call("builtin.trigger_list", json!({})), - RebornScriptedReply::text("listed"), - ]) - .build() - .await?; - verifier.submit_turn("list my triggers").await?; - let listed = verifier.tool_result_output("builtin.trigger_list").await?; - let triggers = listed["triggers"] - .as_array() - .ok_or("trigger_list output missing triggers array")?; - if triggers - .iter() - .any(|t| t["name"] == json!(SELF_CREATE_ATTEMPT_TRIGGER_NAME)) - { - return Err(format!( - "expected no trigger named {SELF_CREATE_ATTEMPT_TRIGGER_NAME:?} to exist, \ - but trigger_list returned {listed}" - ) - .into()); - } - Ok(()) -} diff --git a/tests/reborn_http_network_scope_isolation_parity.rs b/tests/reborn_http_network_scope_isolation_parity.rs index 127827be0d0..a24925e8914 100644 --- a/tests/reborn_http_network_scope_isolation_parity.rs +++ b/tests/reborn_http_network_scope_isolation_parity.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -7,8 +10,8 @@ use ironclaw_host_api::{CapabilityId, NetworkPolicy, NetworkScheme, NetworkTarge use ironclaw_host_runtime::HTTP_CAPABILITY_ID; use ironclaw_loop_support::{HostManagedModelMessageRole, HostManagedModelResponse}; use ironclaw_turns::TurnStatus; -use reborn_support::{ - harness::RebornBinaryE2EHarness, +use parity_qa_support::{ + binary_e2e::RebornBinaryE2EHarness, model_replay::{ RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, }, diff --git a/tests/reborn_identity_project_scope_isolation_parity.rs b/tests/reborn_identity_project_scope_isolation_parity.rs index 677dd4a171b..da444708552 100644 --- a/tests/reborn_identity_project_scope_isolation_parity.rs +++ b/tests/reborn_identity_project_scope_isolation_parity.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -15,11 +18,9 @@ use ironclaw_turns::{ LoopMessageRef, TurnStatus, run_profile::{LoopRunContext, PromptMode}, }; -use reborn_support::harness::{ - RebornBinaryE2EHarness, RebornHarnessSharedStorage, RecordingTestCapabilityPort, - test_product_scope, -}; -use reborn_support::model_replay::RebornTraceReplayModelGateway; +use parity_qa_support::binary_e2e::{RebornBinaryE2EHarness, RebornHarnessSharedStorage}; +use parity_qa_support::model_replay::RebornTraceReplayModelGateway; +use reborn_support::harness::{RecordingTestCapabilityPort, test_product_scope}; use tokio::sync::{RwLock, watch}; const PROJECT_ALPHA_IDENTITY: &str = "Alice project alpha identity: carries amber notebook."; @@ -175,7 +176,7 @@ struct ProjectIdentityKey { } impl ProjectIdentityKey { - fn from_turn(turn: &reborn_support::harness::SubmittedTurn) -> Self { + fn from_turn(turn: &parity_qa_support::binary_e2e::SubmittedTurn) -> Self { Self { tenant_id: turn.scope.tenant_id.as_str().to_string(), user_id: turn.actor.user_id.as_str().to_string(), diff --git a/tests/reborn_identity_prompt_scope_isolation_parity.rs b/tests/reborn_identity_prompt_scope_isolation_parity.rs index 1db3af6edfe..8c3997bf440 100644 --- a/tests/reborn_identity_prompt_scope_isolation_parity.rs +++ b/tests/reborn_identity_prompt_scope_isolation_parity.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -19,8 +22,9 @@ use ironclaw_turns::{ LoopMessageRef, TurnStatus, run_profile::{LoopRunContext, PromptMode}, }; -use reborn_support::harness::{RebornBinaryE2EHarness, RecordingTestCapabilityPort}; -use reborn_support::model_replay::RebornTraceReplayModelGateway; +use parity_qa_support::binary_e2e::RebornBinaryE2EHarness; +use parity_qa_support::model_replay::RebornTraceReplayModelGateway; +use reborn_support::harness::RecordingTestCapabilityPort; const ALICE_IDENTITY: &str = "Alice is a software engineer who lives in Seattle."; const BOB_IDENTITY: &str = "Bob is a marine biologist who lives in Miami."; diff --git a/tests/reborn_identity_tenant_scope_isolation_parity.rs b/tests/reborn_identity_tenant_scope_isolation_parity.rs index 50254904f07..6aec2bb58cb 100644 --- a/tests/reborn_identity_tenant_scope_isolation_parity.rs +++ b/tests/reborn_identity_tenant_scope_isolation_parity.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -18,11 +21,9 @@ use ironclaw_turns::{ LoopMessageRef, TurnStatus, run_profile::{LoopRunContext, PromptMode}, }; -use reborn_support::harness::{ - RebornBinaryE2EHarness, RebornHarnessSharedStorage, RecordingTestCapabilityPort, - test_product_scope, -}; -use reborn_support::model_replay::RebornTraceReplayModelGateway; +use parity_qa_support::binary_e2e::{RebornBinaryE2EHarness, RebornHarnessSharedStorage}; +use parity_qa_support::model_replay::RebornTraceReplayModelGateway; +use reborn_support::harness::{RecordingTestCapabilityPort, test_product_scope}; const TENANT_ALPHA_IDENTITY: &str = "Alice alpha tenant identity: likes rust ferris."; const TENANT_BETA_IDENTITY: &str = "Alice beta tenant identity: likes neon orchids."; @@ -180,7 +181,7 @@ struct TenantIdentityKey { } impl TenantIdentityKey { - fn from_turn(turn: &reborn_support::harness::SubmittedTurn) -> Self { + fn from_turn(turn: &parity_qa_support::binary_e2e::SubmittedTurn) -> Self { Self { tenant_id: turn.scope.tenant_id.as_str().to_string(), user_id: turn.actor.user_id.as_str().to_string(), diff --git a/tests/reborn_minimal_dispatch_parity.rs b/tests/reborn_minimal_dispatch_parity.rs index 37aa15578fd..36b0de3b526 100644 --- a/tests/reborn_minimal_dispatch_parity.rs +++ b/tests/reborn_minimal_dispatch_parity.rs @@ -1,12 +1,15 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; use ironclaw_product_adapters::ProductInboundAck; use ironclaw_threads::{MessageKind, MessageStatus}; use ironclaw_turns::TurnStatus; -use reborn_support::harness::RebornBinaryE2EHarness; +use parity_qa_support::binary_e2e::RebornBinaryE2EHarness; #[tokio::test] async fn reborn_minimal_dispatch_parity() { diff --git a/tests/reborn_outbound_reply_target_scope_isolation_parity.rs b/tests/reborn_outbound_reply_target_scope_isolation_parity.rs index 66a19d91bd8..0579b14ea97 100644 --- a/tests/reborn_outbound_reply_target_scope_isolation_parity.rs +++ b/tests/reborn_outbound_reply_target_scope_isolation_parity.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -9,9 +12,8 @@ use ironclaw_product_adapters::{ ProductOutboundTarget, ProductRenderOutcome, ProjectionCursor, }; use ironclaw_turns::{ReplyTargetBindingRef, TurnRunId}; -use reborn_support::{ - delivery::RecordingOutboundDeliverySink, test_adapter::RebornTestProductAdapter, -}; +use parity_qa_support::delivery::RecordingOutboundDeliverySink; +use reborn_support::test_adapter::RebornTestProductAdapter; #[tokio::test] async fn reborn_outbound_reply_target_scope_isolation_parity() { diff --git a/tests/reborn_project_scope_isolation_parity.rs b/tests/reborn_project_scope_isolation_parity.rs index 6932fcf06c0..f85bcde890b 100644 --- a/tests/reborn_project_scope_isolation_parity.rs +++ b/tests/reborn_project_scope_isolation_parity.rs @@ -1,16 +1,17 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; use ironclaw_loop_support::HostManagedModelResponse; use ironclaw_threads::{MessageKind, MessageStatus, ThreadMessageRecord}; use ironclaw_turns::TurnStatus; -use reborn_support::harness::{ - RebornBinaryE2EHarness, RebornHarnessSharedStorage, RecordingTestCapabilityPort, - test_product_scope, -}; -use reborn_support::model_replay::RebornTraceReplayModelGateway; +use parity_qa_support::binary_e2e::{RebornBinaryE2EHarness, RebornHarnessSharedStorage}; +use parity_qa_support::model_replay::RebornTraceReplayModelGateway; +use reborn_support::harness::{RecordingTestCapabilityPort, test_product_scope}; #[tokio::test] async fn reborn_project_scope_isolation_parity() { diff --git a/tests/reborn_qa_channel_delivery.rs b/tests/reborn_qa_channel_delivery.rs index 53986b5a439..4a5561ae94c 100644 --- a/tests/reborn_qa_channel_delivery.rs +++ b/tests/reborn_qa_channel_delivery.rs @@ -12,7 +12,10 @@ //! outbound delivery is asserted through the recording delivery sink. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -24,13 +27,13 @@ use ironclaw_product_adapters::{ }; use ironclaw_threads::{MessageKind, MessageStatus}; use ironclaw_turns::{ReplyTargetBindingRef, TurnRunId, TurnStatus}; +use parity_qa_support::binary_e2e::{ + RebornBinaryE2EHarness, RebornHarnessSharedStorage, trace_tool_call_response, +}; +use parity_qa_support::delivery::RecordingOutboundDeliverySink; +use parity_qa_support::model_replay::RebornTraceReplayModelGateway; use reborn_support::{ - delivery::RecordingOutboundDeliverySink, - harness::{ - RebornBinaryE2EHarness, RebornHarnessSharedStorage, RecordingTestCapabilityPort, - test_product_scope, trace_tool_call_response, - }, - model_replay::RebornTraceReplayModelGateway, + harness::{RecordingTestCapabilityPort, test_product_scope}, test_adapter::RebornTestProductAdapter, }; diff --git a/tests/reborn_qa_connect_flows.rs b/tests/reborn_qa_connect_flows.rs index e57168ba745..aadbb197bec 100644 --- a/tests/reborn_qa_connect_flows.rs +++ b/tests/reborn_qa_connect_flows.rs @@ -9,7 +9,10 @@ //! exercised; the gate raise/resume path and credential persistence are. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -17,11 +20,9 @@ use ironclaw_host_api::CapabilityId; use ironclaw_host_runtime::WRITE_FILE_CAPABILITY_ID; use ironclaw_loop_support::HostManagedModelResponse; use ironclaw_turns::{TurnStatus, run_profile::LoopHostMilestoneKind}; -use reborn_support::{ - harness::{RebornBinaryE2EHarness, assert_milestone_order}, - model_replay::{ - RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, - }, +use parity_qa_support::binary_e2e::{RebornBinaryE2EHarness, assert_milestone_order}; +use parity_qa_support::model_replay::{ + RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, }; struct ConnectFlowCase { diff --git a/tests/reborn_qa_doc_grounding.rs b/tests/reborn_qa_doc_grounding.rs index 8adf10eace7..81394840678 100644 --- a/tests/reborn_qa_doc_grounding.rs +++ b/tests/reborn_qa_doc_grounding.rs @@ -12,7 +12,10 @@ //! real `builtin.http` capability against a live loopback server. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -27,12 +30,12 @@ use ironclaw_host_api::CapabilityId; use ironclaw_host_runtime::{HTTP_CAPABILITY_ID, READ_FILE_CAPABILITY_ID}; use ironclaw_loop_support::HostManagedModelResponse; use ironclaw_turns::TurnStatus; -use reborn_support::{ - harness::RebornBinaryE2EHarness, - model_replay::{ - RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, - }, - network::{LiveLoopbackHttpServer, LiveLoopbackHttpState, loopback_http_policy}, +use parity_qa_support::binary_e2e::RebornBinaryE2EHarness; +use parity_qa_support::model_replay::{ + RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, +}; +use parity_qa_support::network::{ + LiveLoopbackHttpServer, LiveLoopbackHttpState, loopback_http_policy, }; const STRATEGY_DOC_CONTENT: &str = "NEAR AI Strategy: user-owned agents are the core pillar; users keep custody of credentials and data."; diff --git a/tests/reborn_qa_recorded_behavior.rs b/tests/reborn_qa_recorded_behavior.rs index deaade8abd6..af884d12169 100644 --- a/tests/reborn_qa_recorded_behavior.rs +++ b/tests/reborn_qa_recorded_behavior.rs @@ -52,7 +52,10 @@ //! `#[ignore]` because they spend tokens and may import live credentials. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -64,14 +67,12 @@ use std::{ use chrono::Utc; use ironclaw_host_api::TenantId; use ironclaw_triggers::{TriggerRunStatus, TriggerState}; -use reborn_support::{ - model_replay::RebornTraceReplayModelGateway, - qa_trace::{ - build_qa_trace_runtime_with_http_exchanges, - build_qa_trace_runtime_with_http_exchanges_and_trigger_poller, load_qa_trace, - qa_trace_tenant_id, record_qa_phrase, recorded_tool_calls, send_qa_phrase, - strip_expected_tool_results, - }, +use parity_qa_support::model_replay::RebornTraceReplayModelGateway; +use parity_qa_support::qa_trace::{ + build_qa_trace_runtime_with_http_exchanges, + build_qa_trace_runtime_with_http_exchanges_and_trigger_poller, load_qa_trace, + qa_trace_tenant_id, record_qa_phrase, recorded_tool_calls, send_qa_phrase, + strip_expected_tool_results, }; use support::trace_llm::{LlmTrace, TraceExpects, TraceResponse, TraceStep, TraceTurn}; diff --git a/tests/reborn_qa_routines.rs b/tests/reborn_qa_routines.rs index 5e4e0936323..1ec5b4d5135 100644 --- a/tests/reborn_qa_routines.rs +++ b/tests/reborn_qa_routines.rs @@ -18,7 +18,10 @@ //! carrying the routine prompt. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -49,11 +52,9 @@ use ironclaw_reborn_composition::{ use ironclaw_triggers::{TriggerId, TriggerPollerWorkerConfig, TriggerRunStatus, TriggerState}; use ironclaw_trust::{AuthorityCeiling, EffectiveTrustClass, TrustDecision, TrustProvenance}; use ironclaw_turns::TurnStatus; -use reborn_support::{ - harness::RebornBinaryE2EHarness, - model_replay::{ - RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, - }, +use parity_qa_support::binary_e2e::RebornBinaryE2EHarness; +use parity_qa_support::model_replay::{ + RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, }; use serde_json::{Value, json}; use tokio::sync::Mutex as TokioMutex; diff --git a/tests/reborn_qa_smoke_scenarios_e2e.rs b/tests/reborn_qa_smoke_scenarios_e2e.rs index 9dd39fe84a7..c33d9fb2a18 100644 --- a/tests/reborn_qa_smoke_scenarios_e2e.rs +++ b/tests/reborn_qa_smoke_scenarios_e2e.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -16,6 +19,10 @@ use ironclaw_loop_support::{ DEFAULT_SPAWN_SUBAGENT_CAPABILITY_ID, HostManagedModelMessageRole, HostManagedModelResponse, }; use ironclaw_turns::TurnStatus; +use parity_qa_support::binary_e2e::RebornBinaryE2EHarness; +use parity_qa_support::model_replay::{ + RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, +}; use reborn_support::{ config::WaitConfig, extension_surface::{ @@ -24,10 +31,7 @@ use reborn_support::{ EXTENSION_REMOVE_CAPABILITY_ID, EXTENSION_SEARCH_CAPABILITY_ID, }, github as github_support, - harness::{RebornBinaryE2EHarness, RecordingTestCapabilityPort}, - model_replay::{ - RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, - }, + harness::RecordingTestCapabilityPort, }; const COVERED_QA_SCENARIOS: &[&str] = &[ @@ -57,7 +61,7 @@ const COVERED_QA_SCENARIOS: &[&str] = &[ #[test] fn every_pasted_qa_scenario_has_reborn_e2e_coverage() { - reborn_support::qa_scenarios::assert_all_covered(COVERED_QA_SCENARIOS); + parity_qa_support::qa_scenarios::assert_all_covered(COVERED_QA_SCENARIOS); } #[tokio::test] diff --git a/tests/reborn_qa_web_fetch.rs b/tests/reborn_qa_web_fetch.rs index 6227d80f7a2..75a461f8c3c 100644 --- a/tests/reborn_qa_web_fetch.rs +++ b/tests/reborn_qa_web_fetch.rs @@ -12,7 +12,10 @@ //! exercised deterministically. #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -27,12 +30,12 @@ use ironclaw_host_api::CapabilityId; use ironclaw_host_runtime::HTTP_CAPABILITY_ID; use ironclaw_loop_support::HostManagedModelResponse; use ironclaw_turns::TurnStatus; -use reborn_support::{ - harness::RebornBinaryE2EHarness, - model_replay::{ - RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, - }, - network::{LiveLoopbackHttpServer, LiveLoopbackHttpState, loopback_http_policy}, +use parity_qa_support::binary_e2e::RebornBinaryE2EHarness; +use parity_qa_support::model_replay::{ + RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, +}; +use parity_qa_support::network::{ + LiveLoopbackHttpServer, LiveLoopbackHttpState, loopback_http_policy, }; #[tokio::test] diff --git a/tests/reborn_recorded_trace_parity.rs b/tests/reborn_recorded_trace_parity.rs index c6bd18dd734..d8b35a9b5f6 100644 --- a/tests/reborn_recorded_trace_parity.rs +++ b/tests/reborn_recorded_trace_parity.rs @@ -1,17 +1,18 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; use ironclaw_loop_support::{HostManagedModelMessageRole, HostManagedModelResponse}; use ironclaw_turns::{TurnStatus, run_profile::LoopHostMilestoneKind}; -use reborn_support::{ - harness::{ - RebornBinaryE2EHarness, RecordingTestCapabilityPort, assert_milestone_order, - trace_tool_call_response, - }, - model_replay::RebornTraceReplayModelGateway, +use parity_qa_support::binary_e2e::{ + RebornBinaryE2EHarness, assert_milestone_order, trace_tool_call_response, }; +use parity_qa_support::model_replay::RebornTraceReplayModelGateway; +use reborn_support::harness::RecordingTestCapabilityPort; #[tokio::test] async fn reborn_recorded_trace_parity() { diff --git a/tests/reborn_response_order_parity.rs b/tests/reborn_response_order_parity.rs index ab78f251fd8..449d494fc8c 100644 --- a/tests/reborn_response_order_parity.rs +++ b/tests/reborn_response_order_parity.rs @@ -1,10 +1,13 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; use ironclaw_turns::{TurnStatus, run_profile::LoopHostMilestoneKind}; -use reborn_support::harness::{RebornBinaryE2EHarness, assert_milestone_order}; +use parity_qa_support::binary_e2e::{RebornBinaryE2EHarness, assert_milestone_order}; #[tokio::test] async fn reborn_response_order_parity() { diff --git a/tests/reborn_subagent_spawn_e2e.rs b/tests/reborn_subagent_spawn_e2e.rs index 27dbf6c51f5..bd1e40ac009 100644 --- a/tests/reborn_subagent_spawn_e2e.rs +++ b/tests/reborn_subagent_spawn_e2e.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -11,13 +14,11 @@ use ironclaw_loop_support::{ DEFAULT_SPAWN_SUBAGENT_CAPABILITY_ID, HostManagedModelMessageRole, HostManagedModelResponse, }; use ironclaw_turns::TurnStatus; -use reborn_support::{ - config::WaitConfig, - harness::{RebornBinaryE2EHarness, RecordingTestCapabilityPort, SubmittedTurn}, - model_replay::{ - RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, - }, +use parity_qa_support::binary_e2e::{RebornBinaryE2EHarness, SubmittedTurn}; +use parity_qa_support::model_replay::{ + RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, }; +use reborn_support::{config::WaitConfig, harness::RecordingTestCapabilityPort}; #[tokio::test] #[ignore = "TEMP(disable-spawn-subagents): spawn_subagent temporarily disabled via capability deny filter; re-enable by emptying DISABLED_CAPABILITY_IDS"] diff --git a/tests/reborn_tenant_binding_scope_isolation_parity.rs b/tests/reborn_tenant_binding_scope_isolation_parity.rs index 14d33daeb56..c7e0ca0acae 100644 --- a/tests/reborn_tenant_binding_scope_isolation_parity.rs +++ b/tests/reborn_tenant_binding_scope_isolation_parity.rs @@ -1,16 +1,17 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; use ironclaw_loop_support::HostManagedModelResponse; use ironclaw_threads::{MessageKind, MessageStatus, ThreadMessageRecord}; use ironclaw_turns::TurnStatus; -use reborn_support::harness::{ - RebornBinaryE2EHarness, RebornHarnessSharedStorage, RecordingTestCapabilityPort, - test_product_scope, -}; -use reborn_support::model_replay::RebornTraceReplayModelGateway; +use parity_qa_support::binary_e2e::{RebornBinaryE2EHarness, RebornHarnessSharedStorage}; +use parity_qa_support::model_replay::RebornTraceReplayModelGateway; +use reborn_support::harness::{RecordingTestCapabilityPort, test_product_scope}; #[tokio::test] async fn reborn_tenant_binding_scope_isolation_parity() { diff --git a/tests/reborn_thread_binding_isolation_parity.rs b/tests/reborn_thread_binding_isolation_parity.rs index 8ab147b5b4a..7b9b01de9dd 100644 --- a/tests/reborn_thread_binding_isolation_parity.rs +++ b/tests/reborn_thread_binding_isolation_parity.rs @@ -1,13 +1,17 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; use ironclaw_loop_support::HostManagedModelResponse; use ironclaw_threads::{MessageKind, MessageStatus, ThreadMessageRecord}; use ironclaw_turns::TurnStatus; -use reborn_support::harness::{RebornBinaryE2EHarness, RecordingTestCapabilityPort}; -use reborn_support::model_replay::RebornTraceReplayModelGateway; +use parity_qa_support::binary_e2e::RebornBinaryE2EHarness; +use parity_qa_support::model_replay::RebornTraceReplayModelGateway; +use reborn_support::harness::RecordingTestCapabilityPort; #[tokio::test] async fn reborn_thread_binding_isolation_parity() { diff --git a/tests/reborn_tool_param_coercion_parity.rs b/tests/reborn_tool_param_coercion_parity.rs index 1f40f81a8ca..049a1c73c57 100644 --- a/tests/reborn_tool_param_coercion_parity.rs +++ b/tests/reborn_tool_param_coercion_parity.rs @@ -1,7 +1,10 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; -// Required by reborn_support::model_replay through crate::support::trace_llm. +// Required by parity_qa_support::model_replay through crate::support::trace_llm. mod support; use ironclaw_host_api::{ @@ -12,8 +15,8 @@ use ironclaw_host_runtime::{ }; use ironclaw_loop_support::{HostManagedModelMessageRole, HostManagedModelResponse}; use ironclaw_turns::TurnStatus; -use reborn_support::{ - harness::RebornBinaryE2EHarness, +use parity_qa_support::{ + binary_e2e::RebornBinaryE2EHarness, model_replay::{ RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, }, diff --git a/tests/reborn_trace_coding_read_tools_parity.rs b/tests/reborn_trace_coding_read_tools_parity.rs index 69771678e9e..81d922e1f18 100644 --- a/tests/reborn_trace_coding_read_tools_parity.rs +++ b/tests/reborn_trace_coding_read_tools_parity.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -7,8 +10,8 @@ use ironclaw_host_api::CapabilityId; use ironclaw_host_runtime::{GLOB_CAPABILITY_ID, GREP_CAPABILITY_ID, LIST_DIR_CAPABILITY_ID}; use ironclaw_loop_support::{HostManagedModelMessageRole, HostManagedModelResponse}; use ironclaw_turns::{TurnStatus, run_profile::LoopHostMilestoneKind}; -use reborn_support::{ - harness::{RebornBinaryE2EHarness, assert_milestone_order}, +use parity_qa_support::{ + binary_e2e::{RebornBinaryE2EHarness, assert_milestone_order}, model_replay::{ RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, }, diff --git a/tests/reborn_trace_core_builtin_tools_parity.rs b/tests/reborn_trace_core_builtin_tools_parity.rs index 89110b19726..e29b9d845ef 100644 --- a/tests/reborn_trace_core_builtin_tools_parity.rs +++ b/tests/reborn_trace_core_builtin_tools_parity.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -21,8 +24,8 @@ use ironclaw_host_runtime::{ }; use ironclaw_loop_support::{HostManagedModelMessageRole, HostManagedModelResponse}; use ironclaw_turns::{TurnStatus, run_profile::LoopHostMilestoneKind}; -use reborn_support::{ - harness::{HarnessWaitConfig, RebornBinaryE2EHarness, assert_milestone_order}, +use parity_qa_support::{ + binary_e2e::{HarnessWaitConfig, RebornBinaryE2EHarness, assert_milestone_order}, model_replay::{ RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, }, diff --git a/tests/reborn_trace_error_path_parity.rs b/tests/reborn_trace_error_path_parity.rs index b33676db08d..41c5575e886 100644 --- a/tests/reborn_trace_error_path_parity.rs +++ b/tests/reborn_trace_error_path_parity.rs @@ -1,13 +1,16 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; use ironclaw_host_api::CapabilityId; use ironclaw_host_runtime::READ_FILE_CAPABILITY_ID; use ironclaw_turns::{TurnStatus, run_profile::LoopHostMilestoneKind}; -use reborn_support::{ - harness::RebornBinaryE2EHarness, +use parity_qa_support::{ + binary_e2e::RebornBinaryE2EHarness, model_replay::{ RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, }, diff --git a/tests/reborn_trace_file_tools_parity.rs b/tests/reborn_trace_file_tools_parity.rs index 05971ed2b5a..898edab186c 100644 --- a/tests/reborn_trace_file_tools_parity.rs +++ b/tests/reborn_trace_file_tools_parity.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -7,8 +10,8 @@ use ironclaw_host_api::CapabilityId; use ironclaw_host_runtime::{READ_FILE_CAPABILITY_ID, WRITE_FILE_CAPABILITY_ID}; use ironclaw_loop_support::{HostManagedModelMessageRole, HostManagedModelResponse}; use ironclaw_turns::{TurnStatus, run_profile::LoopHostMilestoneKind}; -use reborn_support::{ - harness::{RebornBinaryE2EHarness, assert_milestone_order}, +use parity_qa_support::{ + binary_e2e::{RebornBinaryE2EHarness, assert_milestone_order}, model_replay::{ RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, }, diff --git a/tests/reborn_trace_first_party_tool_coverage.rs b/tests/reborn_trace_first_party_tool_coverage.rs index c8e7df35db7..e9ed756b2ec 100644 --- a/tests/reborn_trace_first_party_tool_coverage.rs +++ b/tests/reborn_trace_first_party_tool_coverage.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -21,8 +24,8 @@ use ironclaw_host_runtime::{ }; use ironclaw_loop_support::{HostManagedModelMessageRole, HostManagedModelResponse}; use ironclaw_turns::{TurnStatus, run_profile::LoopHostMilestoneKind}; -use reborn_support::{ - harness::{HarnessWaitConfig, RebornBinaryE2EHarness, assert_milestone_order}, +use parity_qa_support::{ + binary_e2e::{HarnessWaitConfig, RebornBinaryE2EHarness, assert_milestone_order}, model_replay::{ RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, }, diff --git a/tests/reborn_trace_wasm_github_fixture_parity.rs b/tests/reborn_trace_wasm_github_fixture_parity.rs index 7d68923cf71..abb6b747420 100644 --- a/tests/reborn_trace_wasm_github_fixture_parity.rs +++ b/tests/reborn_trace_wasm_github_fixture_parity.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -9,8 +12,8 @@ use ironclaw_host_api::{CapabilityId, NetworkMethod}; use ironclaw_loop_support::{HostManagedModelMessageRole, HostManagedModelResponse}; use ironclaw_network::NetworkHttpRequest; use ironclaw_turns::TurnStatus; -use reborn_support::{ - harness::{HarnessWaitConfig, RebornBinaryE2EHarness}, +use parity_qa_support::{ + binary_e2e::{HarnessWaitConfig, RebornBinaryE2EHarness}, model_replay::{ RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayModelGateway, }, diff --git a/tests/reborn_turn_state_lock_free_submit_parity.rs b/tests/reborn_turn_state_lock_free_submit_parity.rs index 7d362b10ee3..570714be31c 100644 --- a/tests/reborn_turn_state_lock_free_submit_parity.rs +++ b/tests/reborn_turn_state_lock_free_submit_parity.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -8,11 +11,9 @@ use std::time::Duration; use ironclaw_loop_support::HostManagedModelResponse; use ironclaw_product_adapters::ProductInboundAck; use ironclaw_turns::TurnStatus; -use reborn_support::harness::{ - RebornBinaryE2EHarness, RebornHarnessSharedStorage, RecordingTestCapabilityPort, - test_product_scope, -}; -use reborn_support::model_replay::RebornTraceReplayModelGateway; +use parity_qa_support::binary_e2e::{RebornBinaryE2EHarness, RebornHarnessSharedStorage}; +use parity_qa_support::model_replay::RebornTraceReplayModelGateway; +use reborn_support::harness::{RecordingTestCapabilityPort, test_product_scope}; #[tokio::test] async fn reborn_user_submit_completes_while_another_turn_state_write_is_blocked() { diff --git a/tests/reborn_wrong_scope_access_isolation_parity.rs b/tests/reborn_wrong_scope_access_isolation_parity.rs index 92748b46953..c095848fea4 100644 --- a/tests/reborn_wrong_scope_access_isolation_parity.rs +++ b/tests/reborn_wrong_scope_access_isolation_parity.rs @@ -1,5 +1,8 @@ #[allow(dead_code)] -#[path = "support/reborn/mod.rs"] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -7,10 +10,11 @@ use ironclaw_host_api::{TenantId, UserId}; use ironclaw_loop_support::HostManagedModelResponse; use ironclaw_threads::ThreadScope; use ironclaw_turns::{TurnActor, TurnScope, TurnStatus}; -use reborn_support::{ - harness::{RebornBinaryE2EHarness, RecordingTestCapabilityPort, trace_tool_call_response}, +use parity_qa_support::{ + binary_e2e::{RebornBinaryE2EHarness, trace_tool_call_response}, model_replay::RebornTraceReplayModelGateway, }; +use reborn_support::harness::RecordingTestCapabilityPort; #[tokio::test] async fn reborn_wrong_scope_access_isolation_parity() { diff --git a/tests/snapshots/golden_payload__context_surfacing.snap b/tests/snapshots/golden_payload__context_surfacing.snap index 53be835e86c..7eccecb2d86 100644 --- a/tests/snapshots/golden_payload__context_surfacing.snap +++ b/tests/snapshots/golden_payload__context_surfacing.snap @@ -1,5 +1,5 @@ --- -source: tests/support/reborn/golden.rs +source: tests/integration/support/golden.rs assertion_line: 128 --- ===== inference call 0 ===== diff --git a/tests/snapshots/golden_payload__gated_turn_approve.snap b/tests/snapshots/golden_payload__gated_turn_approve.snap index 197b8c60d08..568400dcd0f 100644 --- a/tests/snapshots/golden_payload__gated_turn_approve.snap +++ b/tests/snapshots/golden_payload__gated_turn_approve.snap @@ -1,5 +1,5 @@ --- -source: tests/support/reborn/golden.rs +source: tests/integration/support/golden.rs --- ===== inference call 0 ===== { diff --git a/tests/snapshots/golden_payload__greeting.snap b/tests/snapshots/golden_payload__greeting.snap index dd90915bedc..e6dead257e2 100644 --- a/tests/snapshots/golden_payload__greeting.snap +++ b/tests/snapshots/golden_payload__greeting.snap @@ -1,5 +1,5 @@ --- -source: tests/support/reborn/golden.rs +source: tests/integration/support/golden.rs --- ===== inference call 0 ===== { diff --git a/tests/snapshots/golden_payload__image_attachment.snap b/tests/snapshots/golden_payload__image_attachment.snap index 71ba730c887..81fa52e86e9 100644 --- a/tests/snapshots/golden_payload__image_attachment.snap +++ b/tests/snapshots/golden_payload__image_attachment.snap @@ -1,5 +1,5 @@ --- -source: tests/support/reborn/golden.rs +source: tests/integration/support/golden.rs --- ===== inference call 0 ===== { diff --git a/tests/snapshots/golden_payload__multi_turn.snap b/tests/snapshots/golden_payload__multi_turn.snap index 8cfc20a8120..5a31537858f 100644 --- a/tests/snapshots/golden_payload__multi_turn.snap +++ b/tests/snapshots/golden_payload__multi_turn.snap @@ -1,5 +1,5 @@ --- -source: tests/support/reborn/golden.rs +source: tests/integration/support/golden.rs --- ===== inference call 0 ===== { diff --git a/tests/snapshots/golden_payload__parallel_tool_calls.snap b/tests/snapshots/golden_payload__parallel_tool_calls.snap index 9d790696b16..7e57da850be 100644 --- a/tests/snapshots/golden_payload__parallel_tool_calls.snap +++ b/tests/snapshots/golden_payload__parallel_tool_calls.snap @@ -1,5 +1,5 @@ --- -source: tests/support/reborn/golden.rs +source: tests/integration/support/golden.rs --- ===== inference call 0 ===== { diff --git a/tests/snapshots/golden_payload__tool_call.snap b/tests/snapshots/golden_payload__tool_call.snap index e6b14042fe1..8cf133eddc0 100644 --- a/tests/snapshots/golden_payload__tool_call.snap +++ b/tests/snapshots/golden_payload__tool_call.snap @@ -1,5 +1,5 @@ --- -source: tests/support/reborn/golden.rs +source: tests/integration/support/golden.rs assertion_line: 128 --- ===== inference call 0 ===== diff --git a/tests/support/reborn/approval.rs b/tests/support/reborn/approval.rs deleted file mode 100644 index a72ab80805a..00000000000 --- a/tests/support/reborn/approval.rs +++ /dev/null @@ -1,21 +0,0 @@ -//! Approval helpers for Reborn parity harnesses. -//! -//! This module intentionally does not replace run state, gate persistence, or -//! authorization stores. Full approval helpers are added with the runtime -//! harness that drives the real blocked/resume path. - -#![allow(dead_code)] // External-boundary shims consumed by future binary-E2E tests. - -use super::config::WaitConfig; - -pub type ApprovalWaitConfig = WaitConfig; - -/// Re-export of the canonical gate reference so approval tests name a single -/// `GateRef` type. The gate-resolution *logic* lives on the harness (where the -/// approval-store fields live, `builder.rs` + `harness.rs`); this module is -/// types-only. -// Not every test binary that mounts the support tree consumes this re-export, -// so it reads as unused there under `-D warnings`; the module-level -// `#![allow(dead_code)]` does not cover `unused_imports` for a `pub use`. -#[allow(unused_imports)] -pub use ironclaw_turns::GateRef; diff --git a/tests/support/reborn/golden.rs b/tests/support/reborn/golden.rs deleted file mode 100644 index 6e0b12b03ae..00000000000 --- a/tests/support/reborn/golden.rs +++ /dev/null @@ -1,181 +0,0 @@ -//! Golden-payload assertions for [`RebornIntegrationHarness`] — exact-match of -//! the FULL model-visible inference payload (system prompt, conversation turns, -//! and tool-call/tool-result messages) per inference iteration, plus a compact -//! ordered tool surface and the exact final user-visible reply. -//! -//! ## Why exact-match, and what is normalized -//! -//! The `assert_system_prompt_contains` / `assert_model_request_contains` family -//! (assertions.rs) proves a *substring* reached the model. This module proves the -//! WHOLE payload — every byte the model saw — is constructed exactly as pinned, -//! catching silent drift in prompt assembly, turn/history accumulation, and -//! tool-result feed-back that a substring check cannot see. -//! -//! The captured payload is rendered to canonical JSON (via `serde_json::Value`, -//! whose object keys are BTree-sorted, so key order is deterministic across -//! runs and builds) and snapshotted with `insta` — the repo's established -//! snapshot tool (`assert_replay_snapshot!` in `replay_outcome.rs`). Review / -//! regenerate drift with `cargo insta review` (or `INSTA_UPDATE=always`); on a -//! mismatch insta prints the full expected-vs-actual diff. -//! -//! ### Messages exact, tool surface compact — deliberately -//! -//! Per call we render `messages` in FULL (the crux: system prompt, every turn, -//! and the assistant-tool-call/tool-result pair) but the tool surface only as -//! the ORDERED LIST OF PROVIDER-SEAM TOOL NAMES (`tool_surface`), not the full -//! JSON parameter schemas. Reasons: -//! - The named target of this coverage is prompt/turn construction (system -//! prompt + conversation turns + tool results) — the messages. -//! - The full 14-tool builtin schema is ~1.2k lines and would couple this -//! golden to every unrelated edit of any builtin tool's help text/schema. -//! - The surface's *content* is already pinned inside the exact-matched system -//! prompt: the `surface sha256:…` line is a content hash over the capability -//! surface, and the capability id/name/description list is rendered inline. -//! Per-tool parameter schemas are pinned by each tool's own tests. -//! -//! The name list still catches a tool appearing / disappearing / reordering and -//! the `.`→`__` provider-seam encoding (`builtin.http` → `builtin__http`). -//! -//! ### Normalization set -//! -//! Two values in the payload are genuinely nondeterministic; both are -//! anchored on an exact literal prefix so nothing else is touched: -//! -//! - The runtime context's model-visible wall clock, rendered as -//! `Current date/time at loop start: Z`. Production has no -//! clock seam the harness substitutes (per the NO-WIRE rule we do not add -//! one), so the real minute leaks in — rewritten to ``. -//! - An image-attachment scenario's landed project path, which embeds -//! today's real UTC date (`chrono::Utc::now()` at -//! `crates/ironclaw_reborn_composition/src/attachment_landing.rs`, no test -//! seam either): `.../attachments//...` — rewritten to -//! `.../attachments//...`. Only scenarios that land an attachment -//! (`RebornIntegrationGroup::attachment_tools()`) ever contain this -//! substring; every other golden test is a no-op match. -//! -//! Everything else stays EXACT on purpose: -//! - Tool-call ids (`call-1`, `call-2`, …) come from `RebornScriptedReply`'s -//! `NEXT_TOOL_CALL_ID` — a counter shared by every test in this ONE -//! compiled binary (`reply.rs` is per-binary, not per-test), so its raw -//! value depends on which sibling golden test's `tool_call`/`tool_calls` -//! happened to run concurrently first (`cargo test` runs a binary's tests -//! on a thread pool by default) — not reproducible across runs once more -//! than one golden scenario scripts a tool call. `scripted_trace_llm` -//! canonicalizes the scripted trace to stable per-trace `call-1`, -//! `call-2`, … ids before the model sees it, so the golden pins the actual -//! materialized ids without depending on the specific, racy raw counter -//! value. -//! - The `surface sha256:…` line is a content hash of the capability surface: -//! deterministic given the surface, and a surface change SHOULD ripple into -//! the golden (that is the point). - -#![allow(dead_code)] - -use ironclaw_llm::{ChatMessage, ToolDefinition}; - -use super::builder::RebornIntegrationHarness; - -type HarnessResult = Result>; - -/// Render every captured inference request as one canonical, human-readable -/// block per call: `===== inference call {i} =====` followed by the pretty -/// JSON of `{ "messages": [...], "tool_surface": ["name", ...] }`. `messages` -/// is rendered in full; `tool_surface` is the ordered provider-seam tool names -/// only (see the module docs for why schemas are excluded). Key order is -/// deterministic (BTree-sorted through `serde_json::Value`); no volatile -/// normalization happens here — that is the caller's `insta` filter. -fn render_inference_payloads( - requests: &[Vec], - tool_definitions: &[Vec], -) -> String { - let empty = Vec::new(); - let mut out = String::new(); - for (index, messages) in requests.iter().enumerate() { - let tool_surface: Vec<&str> = tool_definitions - .get(index) - .unwrap_or(&empty) - .iter() - .map(|tool| tool.name.as_str()) - .collect(); - let payload = serde_json::json!({ "messages": messages, "tool_surface": tool_surface }); - let pretty = serde_json::to_string_pretty(&payload) - .expect("captured inference payload serializes to JSON"); - out.push_str(&format!("===== inference call {index} =====\n{pretty}\n")); - } - out -} - -/// Replace the two nondeterministic values in the payload — the runtime -/// context's model-visible wall clock (`Current date/time at loop start: -/// Z`) and, for attachment-landing scenarios, today's real -/// UTC date embedded in the landed project path (`.../attachments//...`) -/// — with ``/`` respectively. Each is anchored on an exact -/// literal prefix so no other field is touched; tool-call ids and the -/// `surface sha256:` hash stay exact after scripted-provider materialization -/// (see module docs). -fn normalize_volatile(rendered: &str) -> String { - let clock = - regex::Regex::new(r"Current date/time at loop start: \d{4}-\d{2}-\d{2}T\d{2}:\d{2}Z") - .expect("valid loop-start-clock regex"); - let rendered = clock.replace_all(rendered, "Current date/time at loop start: "); - let attachment_date = regex::Regex::new(r"/attachments/\d{4}-\d{2}-\d{2}/") - .expect("valid attachment-landing-date regex"); - attachment_date - .replace_all(&rendered, "/attachments//") - .into_owned() -} - -impl RebornIntegrationHarness { - /// Assert the FULL model-visible inference payload for this thread (every - /// captured inference call: system prompt + turns + tool messages + tool - /// surface) matches the committed golden snapshot `golden_payload__{name}`. - /// - /// Reads the retained scripted `TraceLlm`'s `captured_requests()` / - /// `captured_tool_definitions()` (the same capture source - /// `assert_system_prompt_contains` reads), renders them canonically, and - /// snapshots via `insta` with loop-start-clock/date normalization applied. - /// Panics (like the sibling `assert_replay_snapshot!`) on mismatch; run - /// `cargo insta review` to inspect and accept drift. - pub fn assert_golden_payload(&self, name: &str) { - let rendered = normalize_volatile(&render_inference_payloads( - &self.scripted_llm.captured_requests(), - &self.scripted_llm.captured_tool_definitions(), - )); - let mut settings = insta::Settings::clone_current(); - settings.set_snapshot_path(concat!(env!("CARGO_MANIFEST_DIR"), "/tests/snapshots")); - settings.set_prepend_module_to_snapshot(false); - settings.set_omit_expression(true); - settings.bind(|| { - insta::assert_snapshot!(format!("golden_payload__{name}"), rendered); - }); - } - - /// Assert the finalized assistant reply on this thread is EXACTLY `expected` - /// (not a substring — the output-seam counterpart to `assert_golden_payload`, - /// pinning that the model's final text reaches the user verbatim). - pub async fn assert_reply_eq(&self, expected: &str) -> HarnessResult<()> { - let actual = self.final_reply_text().await?; - if actual == expected { - return Ok(()); - } - Err(format!("finalized reply {actual:?} does not exactly equal {expected:?}").into()) - } - - /// The exact finalized assistant reply text on this thread (last finalized - /// `Assistant` message). Errors if none is present. - async fn final_reply_text(&self) -> HarnessResult { - let history = self - .thread_harness - .history(self.binding.thread_id.clone()) - .await?; - history - .iter() - .rev() - .find(|message| { - message.kind == ironclaw_threads::MessageKind::Assistant - && message.status == ironclaw_threads::MessageStatus::Finalized - }) - .and_then(|message| message.content.clone()) - .ok_or_else(|| "no finalized assistant reply on thread".into()) - } -} diff --git a/tests/support/reborn/harness.rs b/tests/support/reborn/harness.rs deleted file mode 100644 index 5ff68bb9026..00000000000 --- a/tests/support/reborn/harness.rs +++ /dev/null @@ -1,5530 +0,0 @@ -//! Reborn binary-E2E harness. -//! -//! This harness drives the product caller path used by the #3702 validation -//! ports: -//! -//! inbound bytes -> ProductAdapter -> DefaultProductWorkflow -> -//! DefaultInboundTurnService -> DefaultTurnCoordinator -> TurnRunScheduler -> -//! Reborn planned agent loop -> model/capability/transcript evidence. -//! -//! Documented test-support substitutions: -//! - the model gateway is scripted trace replay; -//! - the capability port is a local recording echo/approval port; -//! - external internet, delivery, and OAuth are not exercised by this harness. - -#![allow(dead_code)] // Shared by staged Reborn binary-E2E validation ports. - -// arch-exempt: large_file, Reborn binary-E2E + host-runtime capability harness; the -// mock-MCP scaffolding has been split into `harness_mcp.rs`, further focused splits -// (auth, hooks) are tracked in `tests/support/reborn/CLAUDE.md`. - -use std::{ - collections::{HashMap, VecDeque}, - path::{Path, PathBuf}, - sync::{ - Arc, Mutex, - atomic::{AtomicUsize, Ordering}, - }, - time::Duration, -}; - -use async_trait::async_trait; -use ironclaw_approvals::{ApprovalResolver, AutoApproveSettingInput, DenyApproval, LeaseApproval}; -use ironclaw_auth::{ - AuthProductScope, AuthProviderId, AuthSurface, CredentialAccountLabel, CredentialAccountStatus, - CredentialOwnership, NewCredentialAccount, ProviderScope, -}; -use ironclaw_authorization::{GrantAuthorizer, TrustAwareCapabilityDispatchAuthorizer}; -use ironclaw_extensions::{ExtensionPackage, ExtensionRegistry}; -use ironclaw_filesystem::{ - BackendCapabilities, BackendId, BackendKind, CompositeRootFilesystem, ContentKind, - InMemoryBackend, IndexPolicy, LocalFilesystem, MountDescriptor, RootFilesystem, - ScopedFilesystem, StorageClass, -}; -use ironclaw_first_party_extensions::{WEB_GET_CONTENT_CAPABILITY_ID, WEB_SEARCH_CAPABILITY_ID}; -use ironclaw_host_api::{ - Action, AgentId, ApprovalRequestId, CapabilityDescriptor, CapabilityGrant, CapabilityGrantId, - CapabilityId, CapabilitySet, CredentialStageError, Decision, EffectKind, ExecutionContext, - ExtensionId, GrantConstraints, HostPath, InvocationId, MountAlias, MountGrant, - MountPermissions, MountView, NetworkPolicy, NetworkScheme, NetworkTargetPattern, Obligation, - Obligations, PackageId, Principal, ProjectId, ProviderToolName, ResourceEstimate, - ResourceScope, RuntimeCredentialAccountProviderId, RuntimeHttpEgress, RuntimeHttpEgressError, - RuntimeHttpEgressRequest, RuntimeHttpEgressResponse, RuntimeKind, SecretHandle, TenantId, - ThreadId, TrustClass, UserId, VirtualPath, -}; -use ironclaw_host_runtime::{ - APPLY_PATCH_CAPABILITY_ID, BUILTIN_FIRST_PARTY_PROVIDER, CancelRuntimeWorkOutcome, - CancelRuntimeWorkRequest, CapabilitySurfacePolicy, - CapabilitySurfaceVersion as HostRuntimeCapabilitySurfaceVersion, ECHO_CAPABILITY_ID, - GLOB_CAPABILITY_ID, GREP_CAPABILITY_ID, HTTP_CAPABILITY_ID, HTTP_SAVE_CAPABILITY_ID, - HostRuntime, HostRuntimeError, HostRuntimeHealth, HostRuntimeServices, HostRuntimeStatus, - JSON_CAPABILITY_ID, LIST_DIR_CAPABILITY_ID, MEMORY_READ_CAPABILITY_ID, - MEMORY_SEARCH_CAPABILITY_ID, MEMORY_TREE_CAPABILITY_ID, MEMORY_WRITE_CAPABILITY_ID, - PROFILE_SET_CAPABILITY_ID, READ_FILE_CAPABILITY_ID, RuntimeCapabilityOutcome, - RuntimeCapabilityRequest, RuntimeCapabilityResumeRequest, RuntimeCredentialAccessSecret, - RuntimeCredentialAccountRequest, RuntimeCredentialAccountResolver, RuntimeProcessPort, - RuntimeStatusRequest, SHELL_CAPABILITY_ID, SKILL_INSTALL_CAPABILITY_ID, - SKILL_LIST_CAPABILITY_ID, SKILL_REMOVE_CAPABILITY_ID, SPAWN_SUBAGENT_CAPABILITY_ID, - SurfaceKind, TIME_CAPABILITY_ID, TRACE_COMMONS_CREDITS_CAPABILITY_ID, - TRACE_COMMONS_ONBOARD_CAPABILITY_ID, TRACE_COMMONS_PROFILE_SET_CAPABILITY_ID, - TRACE_COMMONS_PROFILE_TOKEN_CAPABILITY_ID, TRACE_COMMONS_STATUS_CAPABILITY_ID, - TRIGGER_CREATE_CAPABILITY_ID, TRIGGER_LIST_CAPABILITY_ID, TRIGGER_PAUSE_CAPABILITY_ID, - TRIGGER_REMOVE_CAPABILITY_ID, TRIGGER_RESUME_CAPABILITY_ID, - VisibleCapabilityRequest as RuntimeVisibleCapabilityRequest, - VisibleCapabilitySurface as RuntimeVisibleCapabilitySurface, WRITE_FILE_CAPABILITY_ID, - builtin_first_party_handlers, builtin_first_party_package, -}; -use ironclaw_host_runtime::{SchedulerTurnRunWakeNotifier, TurnRunSchedulerHandle}; -use ironclaw_loop_support::{ - CapabilityAllowSet, CapabilityResolveError, CapabilityResultWrite, - CapabilitySurfaceProfileResolver, CapabilityWriteResult, DEFAULT_SPAWN_SUBAGENT_CAPABILITY_ID, - EmptyUserProfileSource, HostIdentityContextBuildError, HostIdentityContextCandidate, - HostIdentityContextSource, HostManagedModelRequest, HostRuntimeLoopCapabilityPortFactory, - JsonSpawnSubagentInputCodec, LoopCapabilityPortFactory, LoopCapabilityResultWriter, -}; -use ironclaw_network::{ - NetworkHttpEgress, NetworkHttpError, NetworkHttpRequest, NetworkHttpResponse, NetworkUsage, - PolicyNetworkHttpEgress, ReqwestNetworkTransport, -}; -use ironclaw_product_adapters::{ - ProductInboundAck, ProductInboundEnvelope, ProductInboundPayload, ProductTriggerReason, - ProductWorkflow, -}; -use ironclaw_product_workflow::{ - ConversationBindingService, DefaultInboundTurnService, DefaultProductWorkflow, - IdempotencyLedger, InboundTurnService, ProductConversationRouteKind, ProjectService, - ResolveBindingRequest, ResolvedBinding, -}; -use ironclaw_reborn::subagent::{ - flavors::StaticSubagentDefinitionResolver, gate_resolution::BoundedSubagentGateResolutionStore, - goal_store::InMemoryBoundedSubagentGoalStore, -}; -use ironclaw_reborn::{ - loop_exit_applier::{ - BlockedEvidenceRequest, CompletionEvidenceRequest, FailureEvidenceRequest, - FinalCheckpointEvidenceRequest, LoopExitEvidencePort, ThreadCheckpointLoopExitEvidencePort, - }, - runtime::{ - DefaultPlannedRuntimeConfig, DefaultPlannedRuntimeParts, RebornRuntimeLoopComposition, - RuntimeTurnStateStore, build_default_planned_runtime, - }, -}; -use ironclaw_reborn_composition::test_support::SkillActivationTestSource; -use ironclaw_reborn_composition::{ - ProductLiveCapabilityIo, ProductLiveVisibleCapabilityRequestConfig, RebornBuildInput, - RebornLocalDevApprovalTestParts, RebornProductAuthServices, build_reborn_services, - visible_capability_request_for_run, -}; -use ironclaw_resources::InMemoryResourceGovernor; -use ironclaw_secrets::{ - InMemorySecretStore, SecretLease, SecretLeaseId, SecretLeaseStatus, SecretMaterial, - SecretMetadata, SecretStore, SecretStoreError, -}; -use ironclaw_threads::{ - FilesystemSessionThreadService, SessionThreadService, ThreadHistoryRequest, - ThreadMessageRecord, ThreadScope, -}; -use ironclaw_trust::{AdminConfig, AdminEntry, HostTrustAssignment, HostTrustPolicy}; -use ironclaw_trust::{EffectiveTrustClass, TrustDecision}; -use ironclaw_turns::{ - CancelRunRequest, FilesystemTurnStateStore, GateRef, GetLoopCheckpointRequest, - GetRunStateRequest, IdempotencyKey, InMemoryCheckpointStateStore, LoopBlockedKind, - LoopCheckpointKind, LoopCheckpointStore, LoopGateRef, LoopResultRef, ReplyTargetBindingRef, - ResumeTurnRequest, SanitizedCancelReason, SourceBindingRef, TurnActor, TurnCoordinator, - TurnError, TurnRunId, TurnRunRecord, TurnRunState, TurnScope, TurnSpawnTreeStateStore, - TurnStateStore, TurnStatus, - run_profile::{ - AgentLoopHostError, AgentLoopHostErrorKind, CapabilityBatchInvocation, - CapabilityBatchOutcome, CapabilityCallCandidate, CapabilityDescriptorView, - CapabilityInputRef, CapabilityInvocation, CapabilityOutcome, CapabilityResultMessage, - CapabilitySurfaceVersion, ConcurrencyHint, LoopCapabilityPort, LoopHostMilestone, - LoopHostMilestoneKind, LoopHostMilestoneSink, LoopRunContext, ParentLoopOutput, PromptMode, - ProviderToolCall, ProviderToolCallReplay, ProviderToolDefinition, VisibleCapabilityRequest, - VisibleCapabilitySurface, - }, -}; -use ironclaw_wasm::{WitToolHost, WitToolRuntimeConfig}; -use serde_json::json; - -use super::{ - config::WaitConfig, - extension_surface::{ - BUNDLED_EXTENSION_CAPABILITY_IDS, BUNDLED_EXTENSION_IDS, EXTENSION_LIFECYCLE_CAPABILITY_IDS, - }, - filesystem::{BlockingTurnStatePutFilesystem, local_filesystem}, - github as github_support, - harness_mcp::{ - build_loopback_mcp_runtime, local_dev_host_runtime_with_registry_egress_and_mcp, - mcp_loopback_network_policy, mock_mcp_extension_package, - }, - harness_web_access, - model_replay::RebornTraceReplayModelGateway, - product_workflow::{RebornProductWorkflowHarness, resource_scope}, - session_thread::RebornThreadHarness, - test_adapter::{RebornTestIngress, RebornTestProductAdapter}, -}; - -pub type HarnessWaitConfig = WaitConfig; - -const TEST_CAPABILITY_ID: &str = "test.echo"; -const TEST_CAPABILITY_SURFACE_VERSION: &str = "trace_replay_v1"; -const SUBAGENT_ALLOWED_TEST_TOOL_NAME: &str = "test_read_file"; - -type HarnessResult = Result>; -pub(crate) type HarnessCapabilityParts = ( - Arc, - Arc, - Arc, - Arc, - HarnessCapabilityRecorder, -); -pub(crate) type HarnessTurnStorageBackend = BlockingTurnStatePutFilesystem; -pub(crate) type HarnessTurnBackend = CompositeRootFilesystem; - -pub struct RebornBinaryE2EHarness { - ingress: RebornTestIngress, - workflow: DefaultProductWorkflow, - external_conversation_id: String, - binding: ResolvedBinding, - thread_scope: ThreadScope, - turn_scope: TurnScope, - turn_store: Arc>, - coordinator: Arc, - _product_harness: RebornProductWorkflowHarness, - thread_harness: RebornThreadHarness, - model_gateway: RebornTraceReplayModelGateway, - capability_recorder: HarnessCapabilityRecorder, - milestone_sink: Arc, - scheduler_handle: Option, - scheduler_notifier: Arc, - _turn_root: Arc, -} - -pub struct SubmittedTurn { - pub ack: ProductInboundAck, - pub run_id: TurnRunId, - pub thread_id: ThreadId, - pub thread_scope: ThreadScope, - pub scope: TurnScope, - pub actor: TurnActor, -} - -#[derive(Clone)] -pub struct RebornHarnessSharedStorage { - product_backend: Arc, - product_root: Arc, - thread_backend: Arc, - turn_backend: Arc, - turn_root: Arc, -} - -impl RebornHarnessSharedStorage { - pub fn new() -> HarnessResult { - let product_root = Arc::new(tempfile::tempdir()?); - let turn_root = Arc::new(tempfile::tempdir()?); - Ok(Self { - product_backend: Arc::new(local_filesystem(product_root.path())?), - product_root, - thread_backend: Arc::new(InMemoryBackend::new()), - turn_backend: Arc::new(BlockingTurnStatePutFilesystem::new(InMemoryBackend::new())), - turn_root, - }) - } - - pub fn block_next_turn_state_put(&self) { - self.turn_backend.block_next_put(); - } - - pub async fn wait_for_blocked_turn_state_put(&self) { - self.turn_backend.wait_for_blocked_put().await; - } - - pub fn release_blocked_turn_state_put(&self) { - self.turn_backend.release_blocked_put(); - } -} - -#[derive(Debug, Clone)] -pub struct RecordedCapabilityResult { - pub capability_id: CapabilityId, - pub output: serde_json::Value, -} - -pub(crate) enum HarnessCapabilityMode { - Recording(RecordingTestCapabilityPort), - HostRuntime(Arc), -} - -#[derive(Clone)] -pub(crate) enum HarnessCapabilityRecorder { - Recording(Arc), - HostRuntime(Arc), -} - -impl HarnessCapabilityRecorder { - pub(crate) fn invocations(&self) -> Vec { - match self { - Self::Recording(port) => port.invocations(), - Self::HostRuntime(harness) => harness.invocations(), - } - } - - pub(crate) fn workspace_file_path(&self, relative: &str) -> Option { - match self { - Self::Recording(_) => None, - Self::HostRuntime(harness) => Some(harness.workspace_file_path(relative)), - } - } - - pub(crate) fn capability_results(&self) -> Vec { - match self { - Self::Recording(_) => Vec::new(), - Self::HostRuntime(harness) => harness.capability_results(), - } - } - - /// E-PROFILE: the local-dev memory filesystem backing the user-profile - /// source for this backend, if any. `None` for the Echo backend and for - /// HostRuntime harnesses without a profile filesystem. - pub(crate) fn profile_filesystem(&self) -> Option> { - match self { - Self::Recording(_) => None, - Self::HostRuntime(harness) => harness.profile_filesystem_for_test(), - } - } - - /// E-SKILL: the `HostSkillContextSource` to wire as the runtime's - /// `skill_context_source` for this backend, if any. `None` for the Echo - /// backend and for HostRuntime harnesses without skill activation. - pub(crate) fn skill_context_source( - &self, - ) -> Option> { - match self { - Self::Recording(_) => None, - Self::HostRuntime(harness) => harness.skill_context_source_for_test(), - } - } - - /// C-ATTACH: the attachment read port + inbound lander for this backend, if - /// any. `None` for the Echo backend and for HostRuntime harnesses without a - /// local-dev workspace filesystem. - pub(crate) fn attachment_test_support( - &self, - ) -> Option { - match self { - Self::Recording(_) => None, - Self::HostRuntime(harness) => harness.attachment_test_support_for_test(), - } - } - - pub(crate) fn runtime_http_requests(&self) -> Vec { - match self { - Self::Recording(_) => Vec::new(), - Self::HostRuntime(harness) => harness.runtime_http_requests(), - } - } - - /// See [`HostRuntimeCapabilityHarness::process_commands`]; empty for the - /// Echo recording backend. - pub(crate) fn recorded_process_commands(&self) -> Vec { - match self { - Self::Recording(_) => Vec::new(), - Self::HostRuntime(harness) => harness.process_commands(), - } - } - - pub(crate) fn network_http_requests(&self) -> Vec { - match self { - Self::Recording(_) => Vec::new(), - Self::HostRuntime(harness) => harness.network_http_requests(), - } - } - - pub(crate) async fn approve_local_dev_gate(&self, gate_ref: &GateRef) -> HarnessResult<()> { - match self { - Self::Recording(_) => { - Err("recording capability port has no local-dev approvals".into()) - } - Self::HostRuntime(harness) => harness.approve_local_dev_gate(gate_ref).await, - } - } - - pub(crate) async fn deny_local_dev_gate(&self, gate_ref: &GateRef) -> HarnessResult<()> { - match self { - Self::Recording(_) => { - Err("recording capability port has no local-dev approvals".into()) - } - Self::HostRuntime(harness) => harness.deny_local_dev_gate(gate_ref).await, - } - } - - pub(crate) async fn disable_auto_approve_for(&self, scope: ResourceScope) -> HarnessResult<()> { - match self { - Self::Recording(_) => { - Err("recording capability port has no local-dev auto-approve settings".into()) - } - Self::HostRuntime(harness) => harness.disable_global_auto_approve(scope).await, - } - } - - pub(crate) async fn enable_auto_approve_for(&self, scope: ResourceScope) -> HarnessResult<()> { - match self { - Self::Recording(_) => { - Err("recording capability port has no local-dev auto-approve settings".into()) - } - Self::HostRuntime(harness) => harness.enable_global_auto_approve(scope).await, - } - } - - pub(crate) fn approval_requests_store( - &self, - ) -> Option> { - match self { - Self::Recording(_) => None, - Self::HostRuntime(harness) => harness.approval_requests_store(), - } - } -} - -impl RebornBinaryE2EHarness { - pub async fn reply_only( - conversation_id: &str, - reply: impl Into, - ) -> HarnessResult { - Self::with_model_gateway( - conversation_id, - RebornTraceReplayModelGateway::with_responses([ - ironclaw_loop_support::HostManagedModelResponse::assistant_reply(reply), - ]), - RecordingTestCapabilityPort::echo(), - ) - .await - } - - pub async fn with_model_gateway( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_port: RecordingTestCapabilityPort, - ) -> HarnessResult { - Self::with_model_gateway_options(conversation_id, model_gateway, capability_port, false) - .await - } - - pub async fn with_model_gateway_scope_shared_storage( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_port: RecordingTestCapabilityPort, - scope: ResourceScope, - shared_storage: RebornHarnessSharedStorage, - ) -> HarnessResult { - Self::with_model_gateway_scope_identity_source_trigger_installation_shared_storage( - conversation_id, - model_gateway, - capability_port, - scope, - Arc::new(EmptyIdentityContextSource), - ProductTriggerReason::DirectChat, - "reborn-test", - "install-1", - "alice", - shared_storage, - ) - .await - } - - pub async fn with_model_gateway_scope_installation_shared_storage( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_port: RecordingTestCapabilityPort, - scope: ResourceScope, - adapter_id: &str, - installation_id: &str, - shared_storage: RebornHarnessSharedStorage, - ) -> HarnessResult { - Self::with_model_gateway_scope_initial_actor_installation_shared_storage( - conversation_id, - "alice", - model_gateway, - capability_port, - scope, - adapter_id, - installation_id, - shared_storage, - ) - .await - } - - #[allow(clippy::too_many_arguments)] - pub async fn with_model_gateway_scope_initial_actor_installation_shared_storage( - conversation_id: &str, - initial_actor_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_port: RecordingTestCapabilityPort, - scope: ResourceScope, - adapter_id: &str, - installation_id: &str, - shared_storage: RebornHarnessSharedStorage, - ) -> HarnessResult { - Self::with_model_gateway_scope_identity_source_trigger_installation_shared_storage( - conversation_id, - model_gateway, - capability_port, - scope, - Arc::new(EmptyIdentityContextSource), - ProductTriggerReason::DirectChat, - adapter_id, - installation_id, - initial_actor_id, - shared_storage, - ) - .await - } - - #[allow(clippy::too_many_arguments)] - pub async fn with_model_gateway_scope_identity_source_trigger_installation_shared_storage( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_port: RecordingTestCapabilityPort, - scope: ResourceScope, - identity_context_source: Arc, - initial_trigger: ProductTriggerReason, - adapter_id: &str, - installation_id: &str, - initial_actor_id: &str, - shared_storage: RebornHarnessSharedStorage, - ) -> HarnessResult { - Self::with_model_gateway_capability_mode_identity_source_trigger_storage_and_adapter( - conversation_id, - model_gateway, - HarnessCapabilityMode::Recording(capability_port), - false, - initial_trigger, - identity_context_source, - scope, - Some(shared_storage), - adapter_id, - installation_id, - initial_actor_id, - ) - .await - } - - pub async fn with_model_gateway_identity_source_shared( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_port: RecordingTestCapabilityPort, - identity_context_source: Arc, - ) -> HarnessResult { - Self::with_model_gateway_options_identity_source_trigger( - conversation_id, - model_gateway, - capability_port, - false, - ProductTriggerReason::BotMention, - identity_context_source, - ) - .await - } - - pub async fn with_host_runtime_file_capabilities( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - ) -> HarnessResult { - let host_runtime = Arc::new(HostRuntimeCapabilityHarness::file_tools().await?); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - true, - ) - .await - } - - pub async fn with_host_runtime_file_capabilities_requiring_approval( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - ) -> HarnessResult { - let host_runtime = - Arc::new(HostRuntimeCapabilityHarness::file_tools_requiring_approval().await?); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - true, - ) - .await - } - - pub async fn with_host_runtime_write_only( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - ) -> HarnessResult { - let host_runtime = Arc::new(HostRuntimeCapabilityHarness::write_only().await?); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - false, - ) - .await - } - - pub async fn with_host_runtime_coding_read_capabilities( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - ) -> HarnessResult { - let host_runtime = Arc::new(HostRuntimeCapabilityHarness::coding_read_tools().await?); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - false, - ) - .await - } - - pub async fn with_host_runtime_core_builtin_capabilities( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - ) -> HarnessResult { - let host_runtime = Arc::new(HostRuntimeCapabilityHarness::core_builtin_tools().await?); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - false, - ) - .await - } - - pub async fn with_host_runtime_process_capabilities( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - ) -> HarnessResult { - let host_runtime = Arc::new(HostRuntimeCapabilityHarness::process_tools().await?); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - false, - ) - .await - } - - pub async fn with_host_runtime_qa_smoke_capabilities( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - ) -> HarnessResult { - let host_runtime = Arc::new(HostRuntimeCapabilityHarness::qa_smoke_tools().await?); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - false, - ) - .await - } - - pub async fn with_host_runtime_extension_lifecycle_capabilities( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - ) -> HarnessResult { - let host_runtime = - Arc::new(HostRuntimeCapabilityHarness::extension_lifecycle_tools().await?); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - false, - ) - .await - } - - pub async fn with_host_runtime_skill_management_capabilities( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - ) -> HarnessResult { - let host_runtime = Arc::new(HostRuntimeCapabilityHarness::skill_management_tools().await?); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - false, - ) - .await - } - - pub async fn with_host_runtime_trigger_management_capabilities( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - ) -> HarnessResult { - let host_runtime = - Arc::new(HostRuntimeCapabilityHarness::trigger_management_tools().await?); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - false, - ) - .await - } - - pub async fn with_host_runtime_trace_commons_capabilities( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - ) -> HarnessResult { - let host_runtime = Arc::new(HostRuntimeCapabilityHarness::trace_commons_tools().await?); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - false, - ) - .await - } - - pub async fn with_host_runtime_core_builtin_capabilities_network_policy( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - network_policy: NetworkPolicy, - ) -> HarnessResult { - let host_runtime = Arc::new( - HostRuntimeCapabilityHarness::core_builtin_tools_with_network_policy(network_policy) - .await?, - ); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - false, - ) - .await - } - - pub async fn with_host_runtime_core_builtin_capabilities_live_http_egress( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - network_policy: NetworkPolicy, - ) -> HarnessResult { - let host_runtime = Arc::new( - HostRuntimeCapabilityHarness::core_builtin_tools_with_live_http_egress(network_policy) - .await?, - ); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - false, - ) - .await - } - - pub async fn with_host_runtime_github_issue_capabilities( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - ) -> HarnessResult { - let host_runtime = Arc::new(HostRuntimeCapabilityHarness::github_issue_tools().await?); - Self::with_model_gateway_capability_mode( - conversation_id, - model_gateway, - HarnessCapabilityMode::HostRuntime(host_runtime), - false, - ) - .await - } - - pub async fn with_harness_blocked_evidence( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_port: RecordingTestCapabilityPort, - ) -> HarnessResult { - Self::with_model_gateway_options(conversation_id, model_gateway, capability_port, true) - .await - } - - async fn with_model_gateway_options( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_port: RecordingTestCapabilityPort, - accept_harness_blocked_evidence: bool, - ) -> HarnessResult { - Self::with_model_gateway_options_identity_source( - conversation_id, - model_gateway, - capability_port, - accept_harness_blocked_evidence, - Arc::new(EmptyIdentityContextSource), - ) - .await - } - - async fn with_model_gateway_options_identity_source( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_port: RecordingTestCapabilityPort, - accept_harness_blocked_evidence: bool, - identity_context_source: Arc, - ) -> HarnessResult { - Self::with_model_gateway_options_identity_source_trigger( - conversation_id, - model_gateway, - capability_port, - accept_harness_blocked_evidence, - ProductTriggerReason::DirectChat, - identity_context_source, - ) - .await - } - - async fn with_model_gateway_options_identity_source_trigger( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_port: RecordingTestCapabilityPort, - accept_harness_blocked_evidence: bool, - initial_trigger: ProductTriggerReason, - identity_context_source: Arc, - ) -> HarnessResult { - Self::with_model_gateway_capability_mode_identity_source_trigger( - conversation_id, - model_gateway, - HarnessCapabilityMode::Recording(capability_port), - accept_harness_blocked_evidence, - initial_trigger, - identity_context_source, - ) - .await - } - - async fn with_model_gateway_capability_mode( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_mode: HarnessCapabilityMode, - accept_harness_blocked_evidence: bool, - ) -> HarnessResult { - Self::with_model_gateway_capability_mode_identity_source( - conversation_id, - model_gateway, - capability_mode, - accept_harness_blocked_evidence, - Arc::new(EmptyIdentityContextSource), - ) - .await - } - - async fn with_model_gateway_capability_mode_identity_source( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_mode: HarnessCapabilityMode, - accept_harness_blocked_evidence: bool, - identity_context_source: Arc, - ) -> HarnessResult { - Self::with_model_gateway_capability_mode_identity_source_trigger( - conversation_id, - model_gateway, - capability_mode, - accept_harness_blocked_evidence, - ProductTriggerReason::DirectChat, - identity_context_source, - ) - .await - } - - async fn with_model_gateway_capability_mode_identity_source_trigger( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_mode: HarnessCapabilityMode, - accept_harness_blocked_evidence: bool, - initial_trigger: ProductTriggerReason, - identity_context_source: Arc, - ) -> HarnessResult { - Self::with_model_gateway_capability_mode_identity_source_trigger_storage_and_adapter( - conversation_id, - model_gateway, - capability_mode, - accept_harness_blocked_evidence, - initial_trigger, - identity_context_source, - product_scope(), - None, - "reborn-test", - "install-1", - "alice", - ) - .await - } - - #[allow(clippy::too_many_arguments)] - async fn with_model_gateway_capability_mode_identity_source_trigger_storage_and_adapter( - conversation_id: &str, - model_gateway: RebornTraceReplayModelGateway, - capability_mode: HarnessCapabilityMode, - accept_harness_blocked_evidence: bool, - initial_trigger: ProductTriggerReason, - identity_context_source: Arc, - product_scope: ResourceScope, - shared_storage: Option, - adapter_id: &str, - installation_id: &str, - initial_actor_id: &str, - ) -> HarnessResult { - let adapter = RebornTestProductAdapter::new(adapter_id, installation_id)?; - let ingress = RebornTestIngress::new(adapter); - let product_harness = if let Some(storage) = shared_storage.as_ref() { - RebornProductWorkflowHarness::filesystem_shared_backend( - product_scope.clone(), - Arc::clone(&storage.product_backend), - Arc::clone(&storage.product_root), - )? - } else { - RebornProductWorkflowHarness::filesystem_temp(product_scope)? - }; - let binding = product_harness - .binding_service()? - .resolve_binding(binding_request_with_trigger_and_actor( - &ingress, - conversation_id, - initial_actor_id, - initial_trigger, - )?) - .await?; - let thread_scope = thread_scope_from_binding_with_route_kind( - &binding, - route_kind_for_trigger(initial_trigger), - )?; - let turn_scope = TurnScope::new_with_owner( - binding.tenant_id.clone(), - binding.agent_id.clone(), - binding.project_id.clone(), - binding.thread_id.clone(), - binding.subject_user_id.clone(), - ); - let thread_harness = if let Some(storage) = shared_storage.as_ref() { - RebornThreadHarness::filesystem_shared_backend( - thread_scope.clone(), - Arc::clone(&storage.thread_backend), - )? - } else { - RebornThreadHarness::filesystem_temp(thread_scope.clone())? - }; - let (turn_backend, turn_root) = if let Some(storage) = shared_storage.as_ref() { - ( - Arc::clone(&storage.turn_backend), - Arc::clone(&storage.turn_root), - ) - } else { - let turn_root = Arc::new(tempfile::tempdir()?); - ( - Arc::new(BlockingTurnStatePutFilesystem::new(InMemoryBackend::new())), - turn_root, - ) - }; - let turn_store = Arc::new(FilesystemTurnStateStore::new(scoped_turns_fs( - turn_backend, - &binding, - )?)); - let checkpoint_state_store = Arc::new(InMemoryCheckpointStateStore::default()); - let loop_checkpoint_store: Arc = turn_store.clone(); - let milestone_sink = - Arc::new(ironclaw_turns::run_profile::InMemoryLoopHostMilestoneSink::default()); - let ( - capability_factory, - capability_surface_resolver, - capability_input_resolver, - capability_result_writer, - capability_recorder, - ) = capability_mode.into_parts(milestone_sink.clone())?; - let turn_state_for_evidence: Arc = turn_store.clone(); - let evidence = Arc::new(HarnessLoopExitEvidencePort { - inner: ThreadCheckpointLoopExitEvidencePort::new_with_thread_scope( - thread_harness.service.clone(), - turn_state_for_evidence, - Arc::clone(&loop_checkpoint_store), - thread_scope.clone(), - ), - loop_checkpoint_store: Arc::clone(&loop_checkpoint_store), - accept_harness_blocked_evidence, - }); - let turn_state_for_runtime: Arc = turn_store.clone(); - let composition = build_default_planned_runtime(DefaultPlannedRuntimeParts { - turn_state: turn_state_for_runtime, - thread_service: thread_harness.service.clone() - as Arc, - thread_scope: thread_scope.clone(), - model_gateway: Arc::new(model_gateway.clone()), - checkpoint_state_store, - loop_checkpoint_store, - milestone_sink: milestone_sink.clone(), - capability_factory, - capability_surface_resolver, - capability_result_writer, - subagent_goal_store: Arc::new(InMemoryBoundedSubagentGoalStore::new()), - subagent_gate_store: Arc::new(BoundedSubagentGateResolutionStore::new()), - subagent_definition_resolver: Arc::new(StaticSubagentDefinitionResolver), - subagent_spawn_input_codec: Arc::new(JsonSpawnSubagentInputCodec::new( - capability_input_resolver, - )), - subagent_spawn_limits: ironclaw_loop_support::SubagentSpawnLimits::default(), - loop_exit_evidence: evidence, - config: DefaultPlannedRuntimeConfig { - // Keep the durable runner heartbeat at its production default; - // test responsiveness comes from fast scheduler polling below. - poll_interval: Duration::from_millis(10), - ..DefaultPlannedRuntimeConfig::default() - }, - model_route_resolver: None, - cancellation_factory: None, - skill_context_source: None, - input_queue: None, - identity_context_source, - user_profile_source: Arc::new(EmptyUserProfileSource), - model_policy_guard: None, - model_budget_accountant: None, - safety_context: None, - hook_dispatcher_builder_factory: None, - communication_context_provider: None, - hook_security_audit_sink: None, - turn_event_sink: None, - attachment_read_port: None, - scheduler_wake_wiring: None, - })?; - let binding_service: Arc = - Arc::new(product_harness.binding_service()?); - let inbound: Arc = Arc::new(DefaultInboundTurnService::new( - Arc::clone(&binding_service), - thread_harness.service_instance()?, - composition.coordinator.clone(), - )); - let ledger: Arc = Arc::new(product_harness.idempotency_ledger()); - let workflow = DefaultProductWorkflow::new(inbound, ledger, binding_service); - - Ok(Self::from_composition( - ingress, - workflow, - conversation_id.to_string(), - binding, - thread_scope, - turn_scope, - turn_store, - product_harness, - thread_harness, - model_gateway, - capability_recorder, - milestone_sink, - composition, - turn_root, - )) - } - - #[allow(clippy::too_many_arguments)] - fn from_composition( - ingress: RebornTestIngress, - workflow: DefaultProductWorkflow, - external_conversation_id: String, - binding: ResolvedBinding, - thread_scope: ThreadScope, - turn_scope: TurnScope, - turn_store: Arc>, - product_harness: RebornProductWorkflowHarness, - thread_harness: RebornThreadHarness, - model_gateway: RebornTraceReplayModelGateway, - capability_recorder: HarnessCapabilityRecorder, - milestone_sink: Arc, - composition: RebornRuntimeLoopComposition< - dyn SessionThreadService, - RebornTraceReplayModelGateway, - >, - turn_root: Arc, - ) -> Self { - let coordinator = Arc::clone(&composition.coordinator); - let scheduler_notifier = composition.scheduler_handle.wake_notifier(); - Self { - ingress, - workflow, - external_conversation_id, - binding, - thread_scope, - turn_scope, - turn_store, - coordinator, - _product_harness: product_harness, - thread_harness, - model_gateway, - capability_recorder, - milestone_sink, - scheduler_handle: Some(composition.scheduler_handle), - scheduler_notifier, - _turn_root: turn_root, - } - } - - pub fn start(&mut self) { - // The scheduler is started automatically inside build_default_planned_runtime. - // This method is kept for API compatibility. - } - - pub fn start_workers(&mut self, _count: usize) { - // The scheduler is started automatically inside build_default_planned_runtime. - // Worker count is configured via DefaultPlannedRuntimeConfig.worker_count. - } - - pub async fn shutdown(&mut self) { - if let Some(scheduler) = self.scheduler_handle.take() { - scheduler.shutdown().await; - } - } - - pub async fn submit_text(&self, event_id: &str, text: &str) -> HarnessResult { - self.submit_text_for(&self.external_conversation_id, "alice", event_id, text) - .await - } - - pub async fn submit_text_for( - &self, - conversation_id: &str, - actor_id: &str, - event_id: &str, - text: &str, - ) -> HarnessResult { - self.submit_text_for_with_trigger( - conversation_id, - actor_id, - event_id, - text, - ProductTriggerReason::DirectChat, - ) - .await - } - - pub async fn submit_text_for_with_trigger( - &self, - conversation_id: &str, - actor_id: &str, - event_id: &str, - text: &str, - trigger: ProductTriggerReason, - ) -> HarnessResult { - let envelope = self.ingress.verified_text_envelope_with_trigger( - event_id, - actor_id, - conversation_id, - text, - trigger, - )?; - let binding_request = binding_request_from_envelope(&envelope); - let route_kind = binding_request.route_kind; - let binding = self - ._product_harness - .binding_service()? - .resolve_binding(binding_request) - .await?; - let thread_scope = thread_scope_from_binding_with_route_kind(&binding, route_kind)?; - let turn_scope = TurnScope::new_with_owner( - binding.tenant_id.clone(), - binding.agent_id.clone(), - binding.project_id.clone(), - binding.thread_id.clone(), - binding.subject_user_id.clone(), - ); - let actor = TurnActor::new(binding.actor_user_id.clone()); - let ack = self.workflow.accept_inbound(envelope).await?; - let run_id = match &ack { - ProductInboundAck::Accepted { - submitted_run_id, .. - } => *submitted_run_id, - other => { - return Err(format!("expected accepted inbound ack, got {other:?}").into()); - } - }; - Ok(SubmittedTurn { - ack, - run_id, - thread_id: binding.thread_id, - thread_scope, - scope: turn_scope, - actor, - }) - } - - pub async fn resume_blocked_turn(&self, run_id: TurnRunId) -> HarnessResult<()> { - let blocked = self - .run_state(run_id) - .await? - .gate_ref - .ok_or("blocked run missing gate ref")?; - self.resume_with_gate(run_id, blocked).await - } - - pub async fn approve_and_resume_local_dev_gate( - &self, - run_id: TurnRunId, - ) -> HarnessResult { - let blocked = self - .run_state(run_id) - .await? - .gate_ref - .ok_or("blocked run missing gate ref")?; - self.capability_recorder - .approve_local_dev_gate(&blocked) - .await?; - self.resume_with_gate(run_id, blocked.clone()).await?; - Ok(blocked) - } - - pub async fn resume_blocked_turn_in_scope( - &self, - scope: TurnScope, - actor: TurnActor, - run_id: TurnRunId, - ) -> HarnessResult<()> { - let blocked = self - .run_state_in_scope(scope.clone(), run_id) - .await? - .gate_ref - .ok_or("blocked run missing gate ref")?; - self.resume_with_gate_as(scope, actor, run_id, blocked, format!("resume-{run_id}")) - .await - } - - pub async fn resume_with_gate( - &self, - run_id: TurnRunId, - gate_ref: GateRef, - ) -> HarnessResult<()> { - self.resume_with_gate_as( - self.turn_scope.clone(), - TurnActor::new(self.binding.actor_user_id.clone()), - run_id, - gate_ref, - format!("resume-{run_id}"), - ) - .await - } - - pub async fn resume_with_gate_as( - &self, - scope: TurnScope, - actor: TurnActor, - run_id: TurnRunId, - gate_ref: GateRef, - idempotency_key: impl Into, - ) -> HarnessResult<()> { - let response = self - .coordinator - .resume_turn(ResumeTurnRequest { - scope, - actor, - run_id, - gate_resolution_ref: gate_ref, - precondition: ironclaw_turns::ResumeTurnPrecondition::AnyBlockedGate, - source_binding_ref: SourceBindingRef::new("src:resume")?, - reply_target_binding_ref: ReplyTargetBindingRef::new("reply:resume")?, - idempotency_key: IdempotencyKey::new(idempotency_key.into())?, - resume_disposition: None, - }) - .await?; - if response.status != TurnStatus::Queued { - return Err(format!("expected resumed run to queue, got {:?}", response.status).into()); - } - Ok(()) - } - - pub async fn cancel_blocked_turn(&self, run_id: TurnRunId) -> HarnessResult<()> { - self.cancel_run_as( - self.turn_scope.clone(), - TurnActor::new(self.binding.actor_user_id.clone()), - run_id, - format!("cancel-{run_id}"), - ) - .await - } - - pub async fn cancel_run_as( - &self, - scope: TurnScope, - actor: TurnActor, - run_id: TurnRunId, - idempotency_key: impl Into, - ) -> HarnessResult<()> { - let response = self - .coordinator - .cancel_run(CancelRunRequest { - scope, - actor, - run_id, - reason: SanitizedCancelReason::UserRequested, - idempotency_key: IdempotencyKey::new(idempotency_key.into())?, - }) - .await?; - if !matches!( - response.status, - TurnStatus::Cancelled | TurnStatus::CancelRequested - ) { - return Err(format!( - "expected run to be cancelled or cancel-requested, got {:?}", - response.status - ) - .into()); - } - Ok(()) - } - - pub async fn wait_for_status( - &self, - run_id: TurnRunId, - expected: TurnStatus, - ) -> HarnessResult { - self.wait_for_status_with_config(run_id, expected, WaitConfig::default()) - .await - } - - pub async fn wait_for_status_with_config( - &self, - run_id: TurnRunId, - expected: TurnStatus, - wait: WaitConfig, - ) -> HarnessResult { - self.wait_for_status_in_scope_with_config(self.turn_scope.clone(), run_id, expected, wait) - .await - } - - pub async fn wait_for_submitted_status( - &self, - submitted: &SubmittedTurn, - expected: TurnStatus, - ) -> HarnessResult { - self.wait_for_status_in_scope(submitted.scope.clone(), submitted.run_id, expected) - .await - } - - pub async fn wait_for_status_in_scope( - &self, - scope: TurnScope, - run_id: TurnRunId, - expected: TurnStatus, - ) -> HarnessResult { - self.wait_for_status_in_scope_with_config(scope, run_id, expected, WaitConfig::default()) - .await - } - - pub async fn wait_for_status_in_scope_with_config( - &self, - scope: TurnScope, - run_id: TurnRunId, - expected: TurnStatus, - wait: WaitConfig, - ) -> HarnessResult { - let deadline = tokio::time::Instant::now() + wait.timeout; - loop { - let state = self.run_state_in_scope(scope.clone(), run_id).await?; - if state.status == expected { - return Ok(state); - } - // A terminal status (Completed/Failed/Cancelled/RecoveryRequired) is - // never left, so once we observe one that is not the target the run - // can never reach `expected`. Fail fast instead of polling to the - // deadline — otherwise a run that fails early (e.g. a spawn capability - // returning a terminal `driver_unavailable`) burns the whole timeout - // and buries the real failure category. - if state.status.is_terminal() { - return Err(format!( - "expected {expected:?} but run reached terminal status {:?}; failure={:?}", - state.status, state.failure - ) - .into()); - } - if tokio::time::Instant::now() >= deadline { - return Err(format!( - "timed out waiting for {expected:?}; last status={:?} failure={:?}", - state.status, state.failure - ) - .into()); - } - tokio::time::sleep(wait.poll_interval).await; - } - } - - pub async fn run_state(&self, run_id: TurnRunId) -> HarnessResult { - self.run_state_in_scope(self.turn_scope.clone(), run_id) - .await - } - - pub async fn run_state_in_scope( - &self, - scope: TurnScope, - run_id: TurnRunId, - ) -> HarnessResult { - Ok(self - .turn_store - .get_run_state(GetRunStateRequest { scope, run_id }) - .await?) - } - - pub async fn assert_final_reply(&self, text: &str) -> HarnessResult<()> { - Ok(self - .thread_harness - .assert_final_reply(self.binding.thread_id.clone(), text) - .await?) - } - - pub async fn history(&self) -> HarnessResult> { - self.history_for_thread(self.binding.thread_id.clone()) - .await - } - - pub async fn history_for_submitted_thread( - &self, - submitted: &SubmittedTurn, - ) -> HarnessResult> { - self.history_for_thread_in_scope( - submitted.thread_scope.clone(), - submitted.thread_id.clone(), - ) - .await - } - - pub async fn history_for_thread( - &self, - thread_id: ThreadId, - ) -> HarnessResult> { - self.history_for_thread_in_scope(self.thread_scope.clone(), thread_id) - .await - } - - pub async fn history_for_thread_in_scope( - &self, - scope: ThreadScope, - thread_id: ThreadId, - ) -> HarnessResult> { - Ok(self - .thread_harness - .service - .list_thread_history(ThreadHistoryRequest { scope, thread_id }) - .await? - .messages) - } - - pub async fn children_of( - &self, - scope: &TurnScope, - run_id: TurnRunId, - ) -> HarnessResult> { - Ok(self.turn_store.children_of(scope, run_id).await?) - } - - pub fn model_requests(&self) -> Vec { - self.model_gateway.requests() - } - - pub fn remaining_model_responses(&self) -> usize { - self.model_gateway.remaining_responses() - } - - pub fn assert_model_exhausted(&self) { - self.model_gateway.assert_exhausted(); - } - - pub fn capability_invocations(&self) -> Vec { - self.capability_recorder.invocations() - } - - pub fn capability_results(&self) -> Vec { - self.capability_recorder.capability_results() - } - - pub fn runtime_http_requests(&self) -> Vec { - self.capability_recorder.runtime_http_requests() - } - - pub fn network_http_requests(&self) -> Vec { - self.capability_recorder.network_http_requests() - } - - pub fn host_workspace_file_path(&self, relative: &str) -> HarnessResult { - self.capability_recorder - .workspace_file_path(relative) - .ok_or_else(|| "harness is not using host-runtime capabilities".into()) - } - - pub fn milestones(&self) -> Vec { - self.milestone_sink.milestones() - } -} - -impl Drop for RebornBinaryE2EHarness { - fn drop(&mut self) { - // Scheduler handle is Option; shutdown is async - // and cannot be called from Drop. The handle is taken in shutdown() and - // here we just let it drop. The scheduler supervisor task exits when the - // command channel closes on drop. - let _ = self.scheduler_handle.take(); - } -} - -struct HarnessLoopExitEvidencePort { - inner: ThreadCheckpointLoopExitEvidencePort>, - loop_checkpoint_store: Arc, - accept_harness_blocked_evidence: bool, -} - -#[async_trait] -impl LoopExitEvidencePort for HarnessLoopExitEvidencePort { - async fn verify_completion_refs( - &self, - request: CompletionEvidenceRequest<'_>, - ) -> Result { - self.inner.verify_completion_refs(request).await - } - - async fn verify_final_checkpoint( - &self, - request: FinalCheckpointEvidenceRequest<'_>, - ) -> Result { - self.inner.verify_final_checkpoint(request).await - } - - async fn verify_blocked_evidence( - &self, - request: BlockedEvidenceRequest<'_>, - ) -> Result { - if self.inner.verify_blocked_evidence(request.clone()).await? { - return Ok(true); - } - if !self.accept_harness_blocked_evidence { - return Ok(false); - } - if !matches!( - request.blocked.kind, - LoopBlockedKind::Approval | LoopBlockedKind::AwaitDependentRun - ) || GateRef::new(request.blocked.gate_ref.as_str()).is_err() - { - return Ok(false); - } - let checkpoint = self - .loop_checkpoint_store - .get_loop_checkpoint(GetLoopCheckpointRequest { - scope: request.scope.clone(), - turn_id: request.turn_id, - run_id: request.run_id, - checkpoint_id: request.blocked.checkpoint_id, - }) - .await?; - Ok(checkpoint - .map(|record| { - record.kind == LoopCheckpointKind::BeforeBlock - && record.state_ref == request.blocked.state_ref - }) - .unwrap_or(false)) - } - - async fn verify_failure_evidence( - &self, - request: FailureEvidenceRequest<'_>, - ) -> Result { - self.inner.verify_failure_evidence(request).await - } - - async fn is_cancellation_observed( - &self, - scope: &TurnScope, - turn_id: ironclaw_turns::TurnId, - run_id: TurnRunId, - ) -> Result { - self.inner - .is_cancellation_observed(scope, turn_id, run_id) - .await - } - - async fn latest_checkpoint_kind( - &self, - scope: &TurnScope, - turn_id: ironclaw_turns::TurnId, - run_id: TurnRunId, - ) -> Result, TurnError> { - self.inner - .latest_checkpoint_kind(scope, turn_id, run_id) - .await - } -} - -impl HarnessCapabilityMode { - pub(crate) fn into_parts( - self, - milestone_sink: Arc, - ) -> HarnessResult { - match self { - Self::Recording(port) => { - let port = Arc::new(port); - let capability_io = Arc::new(ProductLiveCapabilityIo::default()); - Ok(( - Arc::new(HarnessCapabilityPortFactory { - port: Arc::clone(&port), - }), - Arc::new(StaticCapabilitySurfaceProfileResolver { - allow_set: CapabilityAllowSet::allowlist(port.capability_allowlist()), - }), - capability_io.clone(), - capability_io, - HarnessCapabilityRecorder::Recording(port), - )) - } - Self::HostRuntime(harness) => Ok(( - harness.capability_factory(milestone_sink), - Arc::new(StaticCapabilitySurfaceProfileResolver { - allow_set: CapabilityAllowSet::allowlist(harness.capability_ids.clone()), - }), - harness.io.clone(), - harness.capability_result_writer(), - HarnessCapabilityRecorder::HostRuntime(harness), - )), - } - } -} - -/// Backing handles for the two synthetic `outbound_delivery_*` capabilities -/// (C-SYNTH outbound seam). `Some` only for `outbound_target_tools()`. Bundles -/// the injected facade double + the settings stores the production -/// `outbound_delivery_capabilities` wiring consumes, so the harness struct -/// widens by ONE field instead of four. The auto-approve store and -/// approval-request/lease stores are already held as sibling harness fields -/// (`auto_approve_settings` / `approval_parts`) and re-used, not duplicated here. -struct OutboundTargetToolsParts { - /// Concrete double (not the trait object) so tests can read `set` calls back; - /// upcast to `Arc` at wrap time. - facade: Arc, - requires_approval: bool, - tool_permission_overrides: Arc, - persistent_approval_policies: Arc, -} - -pub(crate) struct HostRuntimeCapabilityHarness { - runtime: Arc, - approval_parts: Option, - auto_approve_settings: Option>, - pending_approval_scopes: Arc>>, - io: Arc, - root: Arc, - workspace_root: PathBuf, - mounts: MountView, - capability_mount_overrides: Vec<(CapabilityId, MountView)>, - capability_ids: Vec, - runtime_kind: RuntimeKind, - effect_kinds: Vec, - network_policy: NetworkPolicy, - secrets: Vec, - provider_id: ExtensionId, - additional_provider_trust: Vec<(ExtensionId, Vec)>, - user_id: UserId, - invocations: Arc>>, - results: Arc>>, - http_egress: Option>, - network_egress: Option>, - /// Inert recording process port (slice 5). `Some` when the harness injected - /// a `RecordingProcessPort`; `None` when the live `LocalHostProcessPort` was - /// used (`.with_live_shell()` path) or the harness predates slice 5. - process_port: Option>, - /// Raw local-dev memory filesystem backing the user-profile source - /// (E-PROFILE seam). `Some` only for `new_with_options`-built harnesses (which - /// flow through `RebornServices`); `None` for the lower-level constructors and - /// the Echo backend. Read back via `profile_filesystem_for_test`. - profile_filesystem: Option>, - /// Project service for the local-dev synthetic `project_create` capability - /// (E-PROJ seam). `Some` for every `new_with_options`-built local-dev harness; - /// `None` for the lower-level constructors and the Echo backend. - /// `create_capability_port` wraps the port with the synthetic project-create - /// capability ONLY when `PROJECT_CREATE_CAPABILITY_ID` is also in - /// `capability_ids` (i.e. only `project_tools()` surfaces it), so other - /// local-dev groups are unaffected. Tests read projects back via `project_service`. - project_service: Option>, - /// Local-dev skill context source for the synthetic `skill_activate` - /// capability and runtime prompt injection (E-SKILL seam). `Some` only for - /// `skill_activation_tools()`; `None` otherwise. `create_capability_port` - /// wraps the port with the synthetic `skill_activate` capability ONLY when - /// `SKILL_ACTIVATE_CAPABILITY_ID` is also in `capability_ids`, and its - /// `context_source()` is wired as the runtime's `skill_context_source` in - /// `into_group`. Held as the opaque test-support handle so this crate never - /// names the crate-private source type. - skill_activation_source: Option, - /// Attachment read port + inbound lander backing the C-ATTACH seam. `Some` - /// only for `new_with_options`-built harnesses (which flow through - /// `RebornServices`, and thus have a local-dev workspace filesystem to build - /// both over); `None` for the lower-level constructors and the Echo backend. - /// Read back via `attachment_test_support_for_test`. - attachment_test_support: Option, - /// Backing handles for the synthetic `outbound_delivery_*` capabilities - /// (C-SYNTH outbound seam). `Some` only for `outbound_target_tools()`; - /// `create_capability_port` wraps the port with the two capabilities via - /// `apply_synthetic_capability_wrappers` when this is `Some`. - outbound_target_tools: Option, - /// C-MULTIUSER seam: when `true`, [`create_capability_port`] resolves the - /// capability-execution user from the RUN's owner/actor (mirroring - /// production `local_dev_visible_capability_request`, - /// `crates/ironclaw_reborn_composition/src/runtime/local_dev.rs`) instead of - /// this harness's single fixed `user_id`. That is what lets two distinct - /// actors dispatching over the group's ONE shared capability backend run - /// under DISTINCT `(tenant, user)` scopes, so memory, auto-approve, and - /// approval-settings isolate per actor — the real production behavior. - /// Defaults `false` so every existing fixed-user harness is byte-identical; - /// only the multiuser group constructors flip it on via - /// [`with_run_owner_scoped_capability_dispatch`]. - scope_capability_by_run_owner: bool, - /// Local-dev product-auth services (C-JOURNEY convergence seam). `Some` - /// only for `new_with_options`-built harnesses (which flow through - /// `RebornServices`); `None` for the lower-level constructors and the Echo - /// backend. `seed_github_credential_account` reads this to create a real - /// credential account through `credential_account_service()`, letting a - /// parked `github.*` auth gate's `ProductAuthRuntimeCredentialResolver` - /// lookup resolve on re-dispatch. - product_auth: Option>, - /// W4-ASK-EACH-ONCE: local-dev per-tool permission override store (mirrors - /// `auto_approve_settings`). `Some` only for `new_with_options`-built - /// harnesses (which flow through `RebornServices`); `None` for the - /// lower-level constructors and the Echo backend. Lets a test install a - /// dynamic `ToolPermissionOverride::AskEachTime` override on any capability - /// via `set_ask_each_time_override_for_test`, independent of the - /// `outbound_target_tools()`-only `OutboundTargetToolsParts` copy. - tool_permission_overrides: Option>, -} - -struct HostRuntimeHarnessOptions { - mounts: MountView, - runtime_policy: Option, - seed_extension_credentials: bool, - /// Tenant the E-SKILL skill context source is constructed under, when this - /// harness surfaces the synthetic `skill_activate` capability. Only - /// `skill_activation_tools()` sets this (via - /// `with_skill_activation_tenant`), passing the SAME tenant the caller's - /// group run scope resolved (`group.rs` `build_base`'s - /// `canonical_binding.tenant_id`) — never a separately hardcoded literal — - /// so `skill_activate` resolves the seeded user skill against the same - /// tenant the turn runs under. `None` for every other harness variant. - skill_activation_tenant: Option, - /// Injected outbound-delivery facade double + `target_set` approval flag, - /// when this harness surfaces the synthetic `outbound_delivery_*` - /// capabilities (C-SYNTH outbound seam). Only `outbound_target_tools()` sets - /// this. `new_with_options` pairs the facade with the local-dev settings - /// stores captured from `RebornServices` to build `OutboundTargetToolsParts`. - outbound_target_facade: Option<( - Arc, - bool, - )>, - /// C-JOURNEY: override the local-dev host network HTTP egress - /// (`RebornBuildInput::with_network_http_egress_for_test`). Without this, - /// `build_local_runtime` defaults to a REAL `ReqwestNetworkTransport` - /// (`factory.rs`), so any harness dispatching a bundled WASM capability - /// that crosses HTTP (e.g. `github.*`) on the `new_with_options` path MUST - /// set this to stay hermetic. `None` for every harness that surfaces no - /// such capability. - network_http_egress_for_test: Option>, - /// C-JOURNEY: bundled first-party WASM packages (e.g. github) to publish - /// directly into the local-dev active-extension registry at construction - /// time, via `RebornServices::publish_bundled_extension_for_test` - /// (reaches the SAME `ActiveExtensionPublisher::publish` step - /// `builtin.extension_activate` calls). Without this, a bundled package's - /// capabilities are granted/trusted at the harness-authority layer - /// (`capability_ids`/`additional_provider_trust`) but NOT present in the - /// runtime's own dispatchable registry, so dispatch silently no-ops (the - /// tool call never reaches `invoke_capability`). Empty for every harness - /// that surfaces no bundled WASM capability. - activate_bundled_extensions_for_test: Vec, - /// C-SYNTH `project_create` fault-injection seam: wrap the real - /// `Arc` (`services.local_dev_project_service_for_test()`) - /// in `FaultInjectingProjectService` before it reaches - /// `wrap_project_create_capability_for_test`, so a `create_project` call - /// naming `FAULT_INJECT_DENIED_PROJECT_NAME` returns - /// `ProjectServiceError::Denied` instead of reaching the real store. - /// Only `project_tools_with_fault_injection()` sets this; every other - /// harness leaves the real service unwrapped. - project_service_fault_injection: bool, -} - -impl HostRuntimeHarnessOptions { - fn new( - mounts: MountView, - runtime_policy: Option, - ) -> Self { - Self { - mounts, - runtime_policy, - seed_extension_credentials: false, - skill_activation_tenant: None, - outbound_target_facade: None, - network_http_egress_for_test: None, - activate_bundled_extensions_for_test: Vec::new(), - project_service_fault_injection: false, - } - } - - fn with_seed_extension_credentials(mut self) -> Self { - self.seed_extension_credentials = true; - self - } - - fn with_skill_activation_tenant(mut self, tenant: TenantId) -> Self { - self.skill_activation_tenant = Some(tenant); - self - } - - fn with_outbound_target_tools( - mut self, - facade: Arc, - target_set_requires_approval: bool, - ) -> Self { - self.outbound_target_facade = Some((facade, target_set_requires_approval)); - self - } - - fn with_network_http_egress_for_test(mut self, egress: Arc) -> Self { - self.network_http_egress_for_test = Some(egress); - self - } - - fn with_activated_bundled_extension(mut self, package: ExtensionPackage) -> Self { - self.activate_bundled_extensions_for_test.push(package); - self - } - - fn with_project_service_fault_injection(mut self) -> Self { - self.project_service_fault_injection = true; - self - } -} - -impl HostRuntimeCapabilityHarness { - async fn file_tools() -> HarnessResult { - let harness = Self::file_tools_with_runtime_policy(Some( - ironclaw_reborn_composition::local_dev_yolo_runtime_policy(true)?, - )) - .await?; - harness - .enable_global_auto_approve_for_product_and_harness_users() - .await?; - Ok(harness) - } - - pub(crate) async fn file_tools_requiring_approval() -> HarnessResult { - let harness = Self::file_tools_with_runtime_policy(None).await?; - // Global auto-approve now defaults ON, so disable it explicitly to keep - // this constructor's per-tool approval gate behavior. - harness - .disable_global_auto_approve_for_product_and_harness_users() - .await?; - Ok(harness) - } - - async fn file_tools_with_runtime_policy( - runtime_policy: Option, - ) -> HarnessResult { - Self::new( - "reborn-e2e-builtin-tools", - vec![ - CapabilityId::new(WRITE_FILE_CAPABILITY_ID)?, - CapabilityId::new(READ_FILE_CAPABILITY_ID)?, - ], - vec![EffectKind::ReadFilesystem, EffectKind::WriteFilesystem], - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-builtin-user")?, - runtime_policy, - ) - .await - } - - async fn write_only() -> HarnessResult { - Self::new( - "reborn-e2e-write-only", - vec![CapabilityId::new(WRITE_FILE_CAPABILITY_ID)?], - vec![EffectKind::WriteFilesystem], - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-write-only-user")?, - None, - ) - .await - } - - async fn coding_read_tools() -> HarnessResult { - let harness = Self::new( - "reborn-e2e-coding-read-tools", - vec![ - CapabilityId::new(LIST_DIR_CAPABILITY_ID)?, - CapabilityId::new(GLOB_CAPABILITY_ID)?, - CapabilityId::new(GREP_CAPABILITY_ID)?, - ], - vec![EffectKind::ReadFilesystem], - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-coding-read-user")?, - None, - ) - .await?; - harness - .enable_global_auto_approve_for_product_and_harness_users() - .await?; - Ok(harness) - } - - async fn process_tools() -> HarnessResult { - let harness = Self::new_with_options( - "reborn-e2e-process-tools", - vec![ - CapabilityId::new(ECHO_CAPABILITY_ID)?, - CapabilityId::new(SHELL_CAPABILITY_ID)?, - CapabilityId::new(SPAWN_SUBAGENT_CAPABILITY_ID)?, - ], - vec![ - EffectKind::DispatchCapability, - EffectKind::ReadFilesystem, - EffectKind::WriteFilesystem, - EffectKind::Network, - EffectKind::SpawnProcess, - EffectKind::ExecuteCode, - ], - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-process-user")?, - HostRuntimeHarnessOptions::new(MountView::default(), None), - ) - .await?; - harness - .enable_global_auto_approve_for_product_and_harness_users() - .await?; - Ok(harness) - } - - async fn qa_smoke_tools() -> HarnessResult { - let (root, storage_root, workspace_root) = host_runtime_storage_roots()?; - std::fs::create_dir_all(storage_root.join("skills"))?; - std::fs::create_dir_all(storage_root.join("system/skills"))?; - let runtime = local_dev_host_runtime_with_http_egress( - storage_root, - Arc::new(RecordingRuntimeHttpEgress::with_body( - br#"{"accepted":true,"source":"qa-smoke"}"#.to_vec(), - )), - // qa_smoke_tools exercises real process execution (SpawnProcess effect); - // leave the default LocalHostProcessPort in place. - None, - )?; - let mounts = qa_smoke_mounts()?; - let memory_mounts = memory_mounts(MountPermissions::read_write_list_delete())?; - let memory_capability_ids = [ - CapabilityId::new(MEMORY_SEARCH_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_WRITE_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_READ_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_TREE_CAPABILITY_ID)?, - ]; - Ok(Self { - runtime, - approval_parts: None, - auto_approve_settings: None, - pending_approval_scopes: Arc::new(Mutex::new(HashMap::new())), - io: Arc::new(ProductLiveCapabilityIo::default()), - root, - workspace_root, - mounts, - capability_mount_overrides: memory_capability_ids - .iter() - .cloned() - .map(|capability_id| (capability_id, memory_mounts.clone())) - .collect(), - capability_ids: vec![ - CapabilityId::new(ECHO_CAPABILITY_ID)?, - CapabilityId::new(TIME_CAPABILITY_ID)?, - CapabilityId::new(JSON_CAPABILITY_ID)?, - CapabilityId::new(HTTP_CAPABILITY_ID)?, - CapabilityId::new(HTTP_SAVE_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_SEARCH_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_WRITE_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_READ_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_TREE_CAPABILITY_ID)?, - CapabilityId::new(READ_FILE_CAPABILITY_ID)?, - CapabilityId::new(WRITE_FILE_CAPABILITY_ID)?, - CapabilityId::new(LIST_DIR_CAPABILITY_ID)?, - CapabilityId::new(GLOB_CAPABILITY_ID)?, - CapabilityId::new(GREP_CAPABILITY_ID)?, - CapabilityId::new(APPLY_PATCH_CAPABILITY_ID)?, - CapabilityId::new(SHELL_CAPABILITY_ID)?, - CapabilityId::new(SPAWN_SUBAGENT_CAPABILITY_ID)?, - CapabilityId::new(SKILL_LIST_CAPABILITY_ID)?, - CapabilityId::new(SKILL_INSTALL_CAPABILITY_ID)?, - CapabilityId::new(SKILL_REMOVE_CAPABILITY_ID)?, - CapabilityId::new(TRIGGER_CREATE_CAPABILITY_ID)?, - CapabilityId::new(TRIGGER_LIST_CAPABILITY_ID)?, - CapabilityId::new(TRIGGER_PAUSE_CAPABILITY_ID)?, - CapabilityId::new(TRIGGER_RESUME_CAPABILITY_ID)?, - CapabilityId::new(TRIGGER_REMOVE_CAPABILITY_ID)?, - ], - runtime_kind: RuntimeKind::FirstParty, - effect_kinds: vec![ - EffectKind::DispatchCapability, - EffectKind::ReadFilesystem, - EffectKind::WriteFilesystem, - EffectKind::DeleteFilesystem, - EffectKind::Network, - EffectKind::SpawnProcess, - EffectKind::ExecuteCode, - EffectKind::ExternalWrite, - ], - network_policy: http_test_policy(), - secrets: Vec::new(), - provider_id: ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - additional_provider_trust: Vec::new(), - user_id: UserId::new("reborn-e2e-qa-smoke-user")?, - invocations: Arc::new(Mutex::new(Vec::new())), - results: Arc::new(Mutex::new(Vec::new())), - http_egress: None, - network_egress: None, - process_port: None, - profile_filesystem: None, - project_service: None, - skill_activation_source: None, - attachment_test_support: None, - outbound_target_tools: None, - scope_capability_by_run_owner: false, - product_auth: None, - tool_permission_overrides: None, - }) - } - - pub(crate) async fn extension_lifecycle_tools() -> HarnessResult { - let mut capability_ids = capability_ids_from_strs(EXTENSION_LIFECYCLE_CAPABILITY_IDS)?; - capability_ids.extend(capability_ids_from_strs(BUNDLED_EXTENSION_CAPABILITY_IDS)?); - let mut harness = Self::new_with_options( - "reborn-e2e-extension-lifecycle-tools", - capability_ids, - local_dev_all_effects(), - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-extension-lifecycle-user")?, - HostRuntimeHarnessOptions::new( - MountView::default(), - Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( - true, - )?), - ) - .with_seed_extension_credentials(), - ) - .await?; - harness.network_policy = wildcard_test_policy(); - harness.additional_provider_trust = bundled_extension_provider_trust()?; - harness - .enable_global_auto_approve_for_product_and_harness_users() - .await?; - Ok(harness) - } - - /// C-JOURNEY convergence seam: surfaces the file-tool approval-gate - /// capabilities (`write_file`/`read_file`, `PermissionMode::Ask` — same - /// grant shape as `file_tools_requiring_approval`) AND a single GitHub - /// capability (`github.get_repo`) on the SAME `build_reborn_services` - /// local-dev runtime — the one wired with the - /// `run_state`/`approval_requests`/`capability_leases` stores BOTH gate - /// classes' resume paths need (`new_with_options` -> `build_reborn_services`). - /// - /// Distinct from `github_issue_tools_auth_required` (a separate, - /// lower-level `HostRuntimeServices` build with a hardcoded - /// `FixedRuntimeCredentialAccountResolver` and no run_state store — see - /// that constructor's doc comment): this harness's `github.*` credential - /// resolves through the REAL `ProductAuthRuntimeCredentialResolver` - /// (`factory.rs`, wired unconditionally by `build_reborn_services`). No - /// GitHub credential account is seeded at construction (unlike - /// `extension_lifecycle_tools`, which seeds all four bundled providers via - /// `.with_seed_extension_credentials()`). - /// - /// **Gate chaining (empirically verified, not assumed):** the global - /// auto-approve toggle this harness disables (for the file-tool arm) is - /// NOT capability-scoped, so `github.get_repo` first raises a real - /// `TurnStatus::BlockedApproval` too. Approving re-dispatches the - /// still-uncredentialed capability, which blocks AGAIN at a real - /// `TurnStatus::BlockedAuth` (`CredentialStageError::AuthRequired`). - /// `RebornIntegrationHarness::resolve_auth_gate` seeds the account - /// (`seed_github_credential_account`) and resumes, letting the SAME - /// parked capability re-dispatch and complete — the happy-path auth - /// resume the `github_issue_tools_auth_required` fixture cannot do. See - /// `scenario_auth_then_approval_journey`'s module doc for the full - /// approval->auth chain a caller must drive. - /// - /// **Making `github.*` genuinely dispatchable (not just granted) needed - /// two additive test-support seams, both required together:** - /// 1. `capability_ids`/`additional_provider_trust` alone are NOT enough — - /// they only populate the harness-authority grant layer. The runtime's - /// OWN dispatchable registry (`build_local_runtime`'s - /// `local_dev_builtin_extension_registry()`) contains only first-party - /// builtins + the four lifecycle capabilities; bundled packages - /// (github, gmail, …) live in a SEPARATE `AvailableExtensionCatalog` - /// used for search only. Without registry presence, a scripted - /// `github.*` call silently never reaches `invoke_capability` (the run - /// completes with zero recorded invocations). Fixed via - /// `RebornServices::publish_bundled_extension_for_test` - /// (`factory.rs`, new `#[cfg(feature = "test-support")]` accessor) — - /// reaches the SAME `ActiveExtensionPublisher::publish` step - /// `builtin.extension_activate` calls, called directly at harness - /// construction instead of via a scripted install/activate handshake. - /// 2. Registry presence alone still isn't sufficient: `build_local_runtime` - /// mounts `/system/extensions` at an EMPTY per-harness tempdir, so the - /// runtime fails to compile `wasm/github_tool.wasm` at dispatch time - /// (`Failed{host_creation_failed}`) even once the package metadata is - /// registered. Fixed by copying the REAL asset directory - /// (`github_support::asset_root()`, already used by the - /// `github_issue_tools_*` harnesses) into this harness's own tempdir - /// mount (`copy_dir_recursive`) — no new fixtures, reuses the existing - /// on-disk asset tree. - /// - /// Runtime policy is left at `None` (like `file_tools_requiring_approval`, - /// NOT the `LocalDevYolo` policy `extension_lifecycle_tools` uses) so the - /// file tools' real `PermissionMode::Ask` gate is preserved; the two seams - /// above are independent of the runtime-policy profile. - pub(crate) async fn file_and_github_auth_tools() -> HarnessResult { - // Hermetic guard: `new_with_options`'s `build_local_runtime` defaults to - // a REAL `ReqwestNetworkTransport` when no test egress is supplied - // (`factory.rs`). This harness surfaces a `github.*` WASM capability - // that crosses HTTP, so it MUST override the network egress or the - // post-resume dispatch would attempt a live network call. - let github_fixture_response = - br#"{"id":1,"full_name":"octocat/hello-world","private":false}"#.to_vec(); - let network_egress: Arc = Arc::new( - RecordingNetworkHttpEgress::with_body(github_fixture_response), - ); - let mut harness = Self::new_with_options( - "reborn-e2e-file-github-auth-tools", - vec![ - CapabilityId::new(WRITE_FILE_CAPABILITY_ID)?, - CapabilityId::new(READ_FILE_CAPABILITY_ID)?, - CapabilityId::new("github.get_repo")?, - ], - local_dev_all_effects(), - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-file-github-auth-user")?, - HostRuntimeHarnessOptions::new( - workspace_mounts(MountPermissions::read_write_list_delete())?, - None, - ) - .with_network_http_egress_for_test(network_egress) - .with_activated_bundled_extension(github_support::extension_package()?), - ) - .await?; - harness.network_policy = wildcard_test_policy(); - harness.additional_provider_trust = bundled_extension_provider_trust()?; - // See point 2 of this constructor's doc comment: registry presence - // alone isn't enough, the WASM asset bytes must be copied into this - // harness's own tempdir mount too. - copy_dir_recursive( - &github_support::asset_root(), - &harness - .root - .path() - .join("local-dev/system/extensions/github"), - )?; - // Global auto-approve now defaults ON; disable it so write_file/read_file - // raise real `BlockedApproval` gates (mirrors `file_tools_requiring_approval`). - // The GitHub auth gate is a separate mechanism (credential resolution, not - // approval mode) and is unaffected by this toggle. - harness - .disable_global_auto_approve_for_product_and_harness_users() - .await?; - Ok(harness) - } - - /// C-JOURNEY: seed a real GitHub credential account — WITH real secret - /// material — through the PRODUCTION manual-token flow - /// (`request_manual_token_setup` → `submit_manual_token`, the same - /// two-step path the real "user pastes a token in the UI" flow drives), so - /// a parked `github.*` auth gate's `ProductAuthRuntimeCredentialResolver` - /// lookup resolves on re-dispatch AND the re-dispatched WASM capability's - /// credential obligation can actually stage the token. A bare - /// `credential_account_service().create_account(..)` carrying a dangling - /// `SecretHandle` is NOT enough: it clears the auth gate, but the - /// re-dispatched execution then fails at `stage_credential_material` - /// (no material behind the handle) and the run still completes with a - /// model-visible failure — caught by `assert_tool_result_contains`. - /// - /// `scope` MUST be the run's actual dispatch-time `(tenant, user, agent, - /// project)` — `ProductAuthRuntimeCredentialResolver::resolve_access_secret`'s - /// `account_visible_from_runtime_scope` check matches on all four, so a - /// mismatched scope (e.g. a fixed literal that doesn't match the calling - /// group/harness's real run scope) silently seeds an account the - /// dispatch-time lookup never finds, leaving the run stuck at - /// `BlockedAuth`. Callers build this from their own resolved run scope - /// (see `RebornIntegrationHarness::resolve_auth_gate`, which uses - /// `self.turn_scope` + `self.binding.actor_user_id` — the SAME fields - /// `resume_run` uses). - /// - /// Continuation is `AuthContinuationRef::SetupOnly`: the harness's - /// `resolve_auth_gate` performs the run resume itself (mirroring the - /// approval-gate helpers), so the flow must not ALSO dispatch a - /// `TurnGateResume` continuation. - pub(crate) async fn seed_github_credential_account( - &self, - scope: &ResourceScope, - ) -> HarnessResult<()> { - let product_auth = self - .product_auth - .as_ref() - .ok_or("harness missing local-dev product auth (not built via new_with_options)")?; - let scope = AuthProductScope::credential_owner(scope, AuthSurface::Api); - let challenge = product_auth - .request_manual_token_setup( - ironclaw_reborn_composition::RebornManualTokenSetupRequest::new( - scope.clone(), - AuthProviderId::new("github")?, - CredentialAccountLabel::new("journey github")?, - ironclaw_auth::AuthContinuationRef::SetupOnly, - chrono::Utc::now() + chrono::Duration::minutes(10), - ), - ) - .await - .map_err(|error| format!("manual token setup failed: {error:?}"))?; - product_auth - .submit_manual_token( - ironclaw_reborn_composition::RebornManualTokenSubmitRequest::new( - scope.clone(), - challenge.interaction_id, - secrecy::SecretString::from("journey-github-token"), - ), - ) - .await - .map_err(|error| format!("manual token submit failed: {error:?}"))?; - Ok(()) - } - - /// E-PROJ: harness surfacing the local-dev synthetic `project_create` - /// capability. `create_capability_port` injects the synthetic capability via - /// `apply_synthetic_capability_wrappers` because `PROJECT_CREATE_CAPABILITY_ID` - /// is in the allowlist. Auto-approve is enabled so the capability dispatches - /// without a gate. - pub(crate) async fn project_tools() -> HarnessResult { - let harness = Self::new_with_options( - "reborn-e2e-project-tools", - vec![CapabilityId::new( - ironclaw_reborn_composition::test_support::PROJECT_CREATE_CAPABILITY_ID, - )?], - vec![ - EffectKind::DispatchCapability, - EffectKind::ReadFilesystem, - EffectKind::WriteFilesystem, - ], - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-project-tools-user")?, - HostRuntimeHarnessOptions::new( - MountView::default(), - Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( - true, - )?), - ), - ) - .await?; - harness - .enable_global_auto_approve_for_product_and_harness_users() - .await?; - Ok(harness) - } - - /// C-SYNTH `project_create` fault-injection arm: same surface as - /// `project_tools()`, but the real `Arc` is wrapped in - /// `FaultInjectingProjectService` - /// (`with_project_service_fault_injection`) so a `create_project` call - /// naming `FAULT_INJECT_DENIED_PROJECT_NAME` returns - /// `ProjectServiceError::Denied`/`PolicyDenied` and proves the real - /// capability dispatch's recoverable `Failed` behavior. This is *not* - /// the `project_service_outcome` `Unavailable` / internal-retry path. - /// Any other `create_project` name still reaches the real store. - pub(crate) async fn project_tools_with_fault_injection() -> HarnessResult { - let harness = Self::new_with_options( - "reborn-e2e-project-tools-fault-injection", - vec![CapabilityId::new( - ironclaw_reborn_composition::test_support::PROJECT_CREATE_CAPABILITY_ID, - )?], - vec![ - EffectKind::DispatchCapability, - EffectKind::ReadFilesystem, - EffectKind::WriteFilesystem, - ], - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-project-tools-fault-injection-user")?, - HostRuntimeHarnessOptions::new( - MountView::default(), - Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( - true, - )?), - ) - .with_project_service_fault_injection(), - ) - .await?; - harness - .enable_global_auto_approve_for_product_and_harness_users() - .await?; - Ok(harness) - } - - /// C-SYNTH outbound: harness surfacing the two local-dev synthetic - /// `outbound_delivery_*` capabilities over an injected - /// [`FakeOutboundPreferencesFacade`] double. - /// `create_capability_port` injects them via - /// `apply_synthetic_capability_wrappers` because - /// `outbound_target_tools` is `Some`. `target_set` runs with - /// `requires_approval = true`, so its settings decision is exercised for - /// real: global auto-approve (default ON) → `Allow`; a `Disabled` tool - /// override (`disable_outbound_target_set_tool`) → `Deny`; auto-approve - /// disabled → `Ask` (approval gate). The RETURNED harness leaves global - /// auto-approve at its default-ON state so the happy/`NotFound` arms - /// dispatch through `Allow`; the gate arm disables it per-test. - pub(crate) async fn outbound_target_tools() -> HarnessResult { - let facade = - super::outbound_preferences::FakeOutboundPreferencesFacade::with_default_targets(); - Self::new_with_options( - "reborn-e2e-outbound-target-tools", - vec![ - CapabilityId::new( - ironclaw_reborn_composition::test_support::OUTBOUND_DELIVERY_TARGETS_LIST_CAPABILITY_ID, - )?, - CapabilityId::new( - ironclaw_reborn_composition::test_support::OUTBOUND_DELIVERY_TARGET_SET_CAPABILITY_ID, - )?, - ], - vec![ - EffectKind::DispatchCapability, - EffectKind::ExternalWrite, - EffectKind::ReadFilesystem, - EffectKind::WriteFilesystem, - ], - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-outbound-target-user")?, - HostRuntimeHarnessOptions::new( - MountView::default(), - Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( - true, - )?), - ) - .with_outbound_target_tools(facade, true), - ) - .await - } - - /// Group whose ONLY capability is `builtin.profile_set` (E-PROFILE seam). - /// Uses `new_with_options` (not `core_builtin_tools_from_runtime`), so - /// `profile_filesystem` is populated from `services.local_dev_profile_filesystem_for_test()` - /// — the read-back half of the round trip a `RebornIntegrationGroup::profile_tools()` - /// scenario needs. Base mounts are `/memory` directly (this harness's only - /// capability needs it; no per-capability mount override required, unlike - /// `core_builtin_tools_from_runtime`'s multi-capability surface). - pub(crate) async fn profile_tools() -> HarnessResult { - let harness = Self::new_with_options( - "reborn-e2e-profile-tools", - vec![CapabilityId::new(PROFILE_SET_CAPABILITY_ID)?], - vec![ - EffectKind::DispatchCapability, - EffectKind::ReadFilesystem, - EffectKind::WriteFilesystem, - ], - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-profile-tools-user")?, - HostRuntimeHarnessOptions::new( - memory_mounts(MountPermissions::read_write_list_delete())?, - Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( - true, - )?), - ), - ) - .await?; - harness - .enable_global_auto_approve_for_product_and_harness_users() - .await?; - Ok(harness) - } - - /// Group with NO first-party capability dispatch — the test drives the - /// C-ATTACH seam purely through the attachment read port + inbound lander, - /// never a tool call. Uses `new_with_options` (mirrors `profile_tools()`), - /// so `attachment_test_support` is populated from - /// `services.local_dev_attachment_test_support_for_test()`. No mounts needed: - /// attachment landing/reading goes through `local_runtime.workspace_filesystem` - /// directly, not the capability-dispatch `MountView` (mirrors - /// `trigger_management_tools()`'s `MountView::default()`, which also has no - /// filesystem capability to gate). - pub(crate) async fn attachment_tools() -> HarnessResult { - Self::new_with_options( - "reborn-e2e-attachment-tools", - Vec::new(), - vec![EffectKind::ReadFilesystem, EffectKind::WriteFilesystem], - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-attachment-tools-user")?, - HostRuntimeHarnessOptions::new( - MountView::default(), - Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( - true, - )?), - ), - ) - .await - } - - /// `pub(crate)`: also used by `RebornIntegrationGroupBuilder::skill_management_tools` - /// (`group_constructors.rs`, C-SKILL) to wire the SAME preset onto the - /// int-tier group, so the QA/trace-tier smoke test and the int-tier group - /// never drift on capability ids / mounts / policy. - pub(crate) async fn skill_management_tools() -> HarnessResult { - let mut harness = Self::new_with_options( - "reborn-e2e-skill-management-tools", - vec![ - CapabilityId::new(SKILL_LIST_CAPABILITY_ID)?, - CapabilityId::new(SKILL_INSTALL_CAPABILITY_ID)?, - CapabilityId::new(SKILL_REMOVE_CAPABILITY_ID)?, - ], - vec![ - EffectKind::DispatchCapability, - EffectKind::ReadFilesystem, - EffectKind::WriteFilesystem, - EffectKind::DeleteFilesystem, - EffectKind::Network, - ], - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-skill-management-user")?, - HostRuntimeHarnessOptions::new( - skill_mounts()?, - Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( - true, - )?), - ), - ) - .await?; - harness.network_policy = http_test_policy(); - harness - .enable_global_auto_approve_for_product_and_harness_users() - .await?; - Ok(harness) - } - - /// Harness surfacing the local-dev synthetic `skill_activate` capability - /// (E-SKILL seam). `new_with_options` builds the `skill_activation_source` - /// (because `SKILL_ACTIVATE_CAPABILITY_ID` is in the allowlist) under - /// `tenant` — the caller's ACTUAL group run-scope tenant, passed through - /// rather than re-hardcoded here — which `create_capability_port` wraps - /// onto the port and `into_group` wires as the runtime's - /// `skill_context_source`. The skill file the model activates is seeded as - /// a system-scoped skill by `RebornIntegrationGroup::skill_activation_tools`. - /// Mirrors `skill_management_tools`/`project_tools`. - pub(crate) async fn skill_activation_tools(tenant: &TenantId) -> HarnessResult { - let mut harness = Self::new_with_options( - "reborn-e2e-skill-activation-tools", - vec![CapabilityId::new( - ironclaw_reborn_composition::test_support::SKILL_ACTIVATE_CAPABILITY_ID, - )?], - vec![ - EffectKind::DispatchCapability, - EffectKind::ReadFilesystem, - EffectKind::WriteFilesystem, - EffectKind::Network, - ], - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-skill-activation-user")?, - HostRuntimeHarnessOptions::new( - skill_mounts()?, - Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( - true, - )?), - ) - .with_skill_activation_tenant(tenant.clone()), - ) - .await?; - harness.network_policy = http_test_policy(); - harness - .enable_global_auto_approve_for_product_and_harness_users() - .await?; - Ok(harness) - } - - pub(crate) async fn trigger_management_tools() -> HarnessResult { - let harness = Self::new_with_options( - "reborn-e2e-trigger-management-tools", - vec![ - CapabilityId::new(TRIGGER_CREATE_CAPABILITY_ID)?, - CapabilityId::new(TRIGGER_LIST_CAPABILITY_ID)?, - CapabilityId::new(TRIGGER_PAUSE_CAPABILITY_ID)?, - CapabilityId::new(TRIGGER_RESUME_CAPABILITY_ID)?, - CapabilityId::new(TRIGGER_REMOVE_CAPABILITY_ID)?, - ], - vec![EffectKind::DispatchCapability, EffectKind::ExternalWrite], - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-trigger-management-user")?, - HostRuntimeHarnessOptions::new( - MountView::default(), - Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( - true, - )?), - ), - ) - .await?; - harness - .enable_global_auto_approve_for_product_and_harness_users() - .await?; - Ok(harness) - } - - async fn enable_global_auto_approve_for_product_and_harness_users(&self) -> HarnessResult<()> { - let product_scope = product_scope(); - self.enable_global_auto_approve(product_scope.clone()) - .await?; - let mut harness_user_scope = product_scope; - harness_user_scope.user_id = self.user_id.clone(); - self.enable_global_auto_approve(harness_user_scope).await?; - Ok(()) - } - - pub(crate) async fn enable_global_auto_approve( - &self, - scope: ResourceScope, - ) -> HarnessResult<()> { - let store = self - .auto_approve_settings - .as_ref() - .ok_or("host runtime harness missing local-dev auto-approve settings")?; - store - .set(AutoApproveSettingInput { - updated_by: Principal::User(scope.user_id.clone()), - scope, - enabled: true, - }) - .await?; - Ok(()) - } - - /// Global auto-approve now defaults ON. A test that needs to exercise the - /// per-tool approval gate must flip it OFF for the product and harness-user - /// scopes the run authorizes against, as an explicit precondition. - pub async fn disable_global_auto_approve_for_product_and_harness_users( - &self, - ) -> HarnessResult<()> { - let product_scope = product_scope(); - self.disable_global_auto_approve(product_scope.clone()) - .await?; - let mut harness_user_scope = product_scope; - harness_user_scope.user_id = self.user_id.clone(); - self.disable_global_auto_approve(harness_user_scope).await?; - Ok(()) - } - - pub(crate) async fn disable_global_auto_approve( - &self, - scope: ResourceScope, - ) -> HarnessResult<()> { - let store = self - .auto_approve_settings - .as_ref() - .ok_or("host runtime harness missing local-dev auto-approve settings")?; - store - .set(AutoApproveSettingInput { - updated_by: Principal::User(scope.user_id.clone()), - scope, - enabled: false, - }) - .await?; - Ok(()) - } - - async fn trace_commons_tools() -> HarnessResult { - let mut harness = Self::new_with_options( - "reborn-e2e-trace-commons-tools", - vec![ - CapabilityId::new(TRACE_COMMONS_ONBOARD_CAPABILITY_ID)?, - CapabilityId::new(TRACE_COMMONS_STATUS_CAPABILITY_ID)?, - CapabilityId::new(TRACE_COMMONS_CREDITS_CAPABILITY_ID)?, - CapabilityId::new(TRACE_COMMONS_PROFILE_TOKEN_CAPABILITY_ID)?, - CapabilityId::new(TRACE_COMMONS_PROFILE_SET_CAPABILITY_ID)?, - ], - vec![ - EffectKind::DispatchCapability, - EffectKind::ReadFilesystem, - // onboard persists device-key material (Ed25519 keypair + - // policy.json) and profile_token writes profile_token.jwt, so - // the harness allow-set must grant WriteFilesystem or those - // capabilities are filtered out of the model-visible surface. - EffectKind::WriteFilesystem, - EffectKind::Network, - EffectKind::ExternalWrite, - ], - Vec::new(), - ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - UserId::new("reborn-e2e-trace-commons-user")?, - // The Trace Commons write/network capabilities are - // PermissionMode::Ask (onboard, profile_token, profile_set) — like - // the skill/trigger harnesses, the scripted run enables global - // auto-approve so it is not gated. - HostRuntimeHarnessOptions::new( - MountView::default(), - Some(ironclaw_reborn_composition::local_dev_yolo_runtime_policy( - true, - )?), - ), - ) - .await?; - // onboard declares EffectKind::Network, so the lease must carry a - // non-empty network policy or the obligation check rejects dispatch - // before the consent gate runs. - harness.network_policy = http_test_policy(); - harness - .enable_global_auto_approve_for_product_and_harness_users() - .await?; - Ok(harness) - } - - async fn new( - service_label: &'static str, - capability_ids: Vec, - effect_kinds: Vec, - secrets: Vec, - provider_id: ExtensionId, - user_id: UserId, - runtime_policy: Option, - ) -> HarnessResult { - Self::new_with_options( - service_label, - capability_ids, - effect_kinds, - secrets, - provider_id, - user_id, - HostRuntimeHarnessOptions::new( - workspace_mounts(MountPermissions::read_write_list_delete())?, - runtime_policy, - ), - ) - .await - } - - async fn new_with_options( - service_label: &'static str, - capability_ids: Vec, - effect_kinds: Vec, - secrets: Vec, - provider_id: ExtensionId, - user_id: UserId, - options: HostRuntimeHarnessOptions, - ) -> HarnessResult { - let HostRuntimeHarnessOptions { - mounts, - runtime_policy, - seed_extension_credentials, - skill_activation_tenant, - outbound_target_facade, - network_http_egress_for_test, - activate_bundled_extensions_for_test, - project_service_fault_injection, - } = options; - let root = Arc::new(tempfile::tempdir()?); - let storage_root = root.path().join("local-dev"); - let workspace_root = storage_root.join("workspace"); - std::fs::create_dir_all(&workspace_root)?; - let mut input = if runtime_policy.as_ref().is_some_and(|policy| { - policy.resolved_profile == ironclaw_host_api::runtime_policy::RuntimeProfile::LocalYolo - }) { - let host_home_root = root.path().join("host-home"); - std::fs::create_dir_all(&host_home_root)?; - ironclaw_reborn_composition::local_runtime_build_input_with_options( - ironclaw_reborn_composition::RebornCompositionProfile::LocalDevYolo, - service_label, - storage_root, - ironclaw_reborn_composition::RebornLocalRuntimeProfileOptions { - confirm_host_access: true, - }, - )? - .with_local_dev_confirmed_host_home_root(host_home_root) - } else { - RebornBuildInput::local_dev(service_label, storage_root) - }; - if let Some(runtime_policy) = runtime_policy { - input = input.with_runtime_policy(runtime_policy); - } - if let Some(egress) = network_http_egress_for_test { - input = input.with_network_http_egress_for_test(egress); - } - let services = build_reborn_services(input).await?; - if seed_extension_credentials { - seed_extension_lifecycle_credentials(&services, &user_id).await?; - } - // C-JOURNEY: publish bundled WASM packages into the active-extension - // registry directly (see `HostRuntimeHarnessOptions::activate_bundled_extensions_for_test` - // doc) so their capabilities are genuinely dispatchable, not merely - // granted at the harness-authority layer. - for package in &activate_bundled_extensions_for_test { - services - .publish_bundled_extension_for_test(package) - .ok_or( - "local-dev Reborn services missing extension management for test publish", - )??; - } - let approval_parts = services.local_dev_approval_test_parts(); - let auto_approve_settings = services.local_dev_auto_approve_settings_for_test(); - // Capture the profile filesystem + project service + attachment support - // before `services.host_runtime` is moved out below (E-PROFILE / E-PROJ / - // C-ATTACH seams). - let profile_filesystem = services.local_dev_profile_filesystem_for_test(); - // C-SYNTH `project_create` fault-injection seam: wrap the real service - // in `FaultInjectingProjectService` only when the harness opted in - // (`with_project_service_fault_injection`) — every other harness keeps - // the real service unwrapped and behaves exactly as before. - let project_service: Option> = - services.local_dev_project_service_for_test().map(|inner| { - if project_service_fault_injection { - super::project_service_fault::FaultInjectingProjectService::wrapping(inner) - as Arc - } else { - inner - } - }); - // C-JOURNEY: capture product-auth before `services.host_runtime` is - // moved out below, so `seed_github_credential_account` can create a - // real credential account later (auth-gate happy-path resume). - let product_auth = services.product_auth.clone(); - // E-SKILL: build the local-dev skill context source only when this - // harness surfaces the synthetic `skill_activate` capability (i.e. - // `skill_activation_tools`). Built with the caller-supplied tenant - // (`HostRuntimeHarnessOptions::with_skill_activation_tenant`, sourced - // from the group's actual run-scope tenant) so activation visibility - // matches the turn's scope. Must precede the `services.host_runtime` - // move (it borrows `&services`). - let skill_activation_source = if capability_ids.iter().any(|id| { - id.as_str() == ironclaw_reborn_composition::test_support::SKILL_ACTIVATE_CAPABILITY_ID - }) { - let tenant = skill_activation_tenant - .ok_or("skill_activation_tools harness requires with_skill_activation_tenant")?; - ironclaw_reborn_composition::test_support::build_local_dev_skill_context_source_for_test( - &services, &tenant, true, - ) - } else { - None - }; - let attachment_test_support = services.local_dev_attachment_test_support_for_test(); - // W4-ASK-EACH-ONCE: capture the local-dev per-tool permission override - // store unconditionally (mirrors `auto_approve_settings` above), not just - // for `outbound_target_tools()`'s narrower `Some((facade, ..))` arm below - // -- any host-runtime-backed harness/group can now install a per-capability - // `AskEachTime` override via `set_ask_each_time_override_for_test`. - let tool_permission_overrides = services.local_dev_tool_permission_overrides_for_test(); - // C-SYNTH outbound: pair the injected facade double with the local-dev - // settings stores production's `outbound_delivery_capabilities` consumes, - // captured from `RebornServices` before the `host_runtime` move. Only - // `outbound_target_tools()` supplies the facade. - let outbound_target_tools = match outbound_target_facade { - Some((facade, requires_approval)) => { - let tool_permission_overrides = services - .local_dev_tool_permission_overrides_for_test() - .ok_or("outbound_target_tools requires a local-dev tool-override store")?; - let persistent_approval_policies = services - .local_dev_persistent_approval_policies_for_test() - .ok_or("outbound_target_tools requires a local-dev persistent-policy store")?; - Some(OutboundTargetToolsParts { - facade, - requires_approval, - tool_permission_overrides, - persistent_approval_policies, - }) - } - None => None, - }; - let pending_approval_scopes = Arc::new(Mutex::new(HashMap::new())); - let runtime = services - .host_runtime - .ok_or("local-dev Reborn services missing host runtime")?; - let runtime = Arc::new(RecordingHostRuntime::new( - runtime, - Arc::clone(&pending_approval_scopes), - )); - Ok(Self { - runtime, - approval_parts, - auto_approve_settings, - pending_approval_scopes, - io: Arc::new(ProductLiveCapabilityIo::default()), - root, - workspace_root, - mounts, - capability_mount_overrides: Vec::new(), - capability_ids, - runtime_kind: RuntimeKind::FirstParty, - effect_kinds, - network_policy: NetworkPolicy::default(), - secrets, - provider_id, - additional_provider_trust: Vec::new(), - user_id, - invocations: Arc::new(Mutex::new(Vec::new())), - results: Arc::new(Mutex::new(Vec::new())), - http_egress: None, - network_egress: None, - process_port: None, - profile_filesystem, - project_service, - skill_activation_source, - attachment_test_support, - outbound_target_tools, - scope_capability_by_run_owner: false, - product_auth, - tool_permission_overrides, - }) - } - - pub(crate) async fn core_builtin_tools() -> HarnessResult { - Self::core_builtin_tools_with_network_policy(http_test_policy()).await - } - - async fn core_builtin_tools_with_network_policy( - network_policy: NetworkPolicy, - ) -> HarnessResult { - Self::core_builtin_tools_with_network_policy_and_process_port(network_policy, true).await - } - - /// Variant used by `.with_live_shell()`: same as `core_builtin_tools_with_network_policy` - /// but opts out of the recording process port so the real `LocalHostProcessPort` - /// executes shell commands on the host. - pub(crate) async fn core_builtin_tools_with_live_shell() -> HarnessResult { - Self::core_builtin_tools_with_network_policy_and_process_port(http_test_policy(), false) - .await - } - - async fn core_builtin_tools_with_network_policy_and_process_port( - network_policy: NetworkPolicy, - recording_process: bool, - ) -> HarnessResult { - let (root, storage_root, workspace_root) = host_runtime_storage_roots()?; - let runtime_http_egress = Arc::new(RecordingRuntimeHttpEgress::with_body( - br#"{"accepted":true}"#.to_vec(), - )); - // Slice 5: inject the inert recording port by default so `builtin.shell` - // invocations in tests never spawn a real OS process. The `.with_live_shell()` - // opt-in passes `recording_process = false`, which skips injection and lets - // `HostRuntimeServices` default to the real `LocalHostProcessPort`. - let recording_process_port = if recording_process { - Some(Arc::new(super::process::RecordingProcessPort::new())) - } else { - None - }; - let process_port_dyn: Option> = recording_process_port - .as_ref() - .map(|p| Arc::clone(p) as Arc); - let runtime = local_dev_host_runtime_with_http_egress( - storage_root.clone(), - Arc::clone(&runtime_http_egress), - process_port_dyn, - )?; - let mut harness = Self::core_builtin_tools_from_runtime( - root, - workspace_root, - runtime, - network_policy, - UserId::new("reborn-e2e-core-builtins-user")?, - )?; - harness.http_egress = Some(runtime_http_egress); - harness.process_port = recording_process_port; - Ok(harness) - } - - async fn core_builtin_tools_with_live_http_egress( - network_policy: NetworkPolicy, - ) -> HarnessResult { - let (root, storage_root, workspace_root) = host_runtime_storage_roots()?; - let runtime = local_dev_host_runtime_with_live_http_egress(storage_root.clone())?; - Self::core_builtin_tools_from_runtime( - root, - workspace_root, - runtime, - network_policy, - UserId::new("reborn-e2e-core-builtins-live-http-user")?, - ) - } - - fn core_builtin_tools_from_runtime( - root: Arc, - workspace_root: PathBuf, - runtime: Arc, - network_policy: NetworkPolicy, - user_id: UserId, - ) -> HarnessResult { - let mounts = workspace_mounts(MountPermissions::read_write_list_delete())?; - let memory_mounts = memory_mounts(MountPermissions::read_write_list_delete())?; - let memory_capability_ids = [ - CapabilityId::new(MEMORY_SEARCH_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_WRITE_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_READ_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_TREE_CAPABILITY_ID)?, - // profile_set writes to the memory mount (context/profile.json under - // the user-scoped scope), so it needs the memory mount override just - // like the four memory_* capabilities above. - CapabilityId::new(PROFILE_SET_CAPABILITY_ID)?, - ]; - Ok(Self { - runtime, - approval_parts: None, - auto_approve_settings: None, - pending_approval_scopes: Arc::new(Mutex::new(HashMap::new())), - io: Arc::new(ProductLiveCapabilityIo::default()), - root, - workspace_root, - mounts, - capability_mount_overrides: memory_capability_ids - .iter() - .cloned() - .map(|capability_id| (capability_id, memory_mounts.clone())) - .collect(), - capability_ids: vec![ - CapabilityId::new(TIME_CAPABILITY_ID)?, - CapabilityId::new(JSON_CAPABILITY_ID)?, - CapabilityId::new(HTTP_CAPABILITY_ID)?, - CapabilityId::new(HTTP_SAVE_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_SEARCH_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_WRITE_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_READ_CAPABILITY_ID)?, - CapabilityId::new(MEMORY_TREE_CAPABILITY_ID)?, - CapabilityId::new(PROFILE_SET_CAPABILITY_ID)?, - CapabilityId::new(READ_FILE_CAPABILITY_ID)?, - CapabilityId::new(APPLY_PATCH_CAPABILITY_ID)?, - // slice 5: `builtin.shell` on the surface so scripted shell calls - // route through the process port (recording by default, live via - // `.with_live_shell()`). - CapabilityId::new(SHELL_CAPABILITY_ID)?, - ], - runtime_kind: RuntimeKind::FirstParty, - effect_kinds: vec![ - EffectKind::DispatchCapability, - EffectKind::ReadFilesystem, - EffectKind::WriteFilesystem, - EffectKind::Network, - EffectKind::SpawnProcess, - // slice 5: `builtin.shell` declares ExecuteCode; the grant's - // allowed_effects must include it or the authorizer denies the - // capability before it reaches the process port. - EffectKind::ExecuteCode, - ], - network_policy, - secrets: Vec::new(), - provider_id: ExtensionId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - additional_provider_trust: Vec::new(), - user_id, - invocations: Arc::new(Mutex::new(Vec::new())), - results: Arc::new(Mutex::new(Vec::new())), - http_egress: None, - network_egress: None, - process_port: None, - profile_filesystem: None, - project_service: None, - skill_activation_source: None, - attachment_test_support: None, - outbound_target_tools: None, - scope_capability_by_run_owner: false, - product_auth: None, - tool_permission_overrides: None, - }) - } - - /// Wires the GitHub first-party WASM capabilities behind `GithubHarnessAuthorizer`. - /// See `github_issue_tools_with_credential_result` for the credential-injection - /// coupling this relies on (T0-SECRET-INJECT). - pub(crate) async fn github_issue_tools() -> HarnessResult { - // Credential account resolves to a real handle → capability dispatches. - Self::github_issue_tools_with_credential_result(Ok(SecretHandle::new( - "github_manual_access", - )?)) - } - - /// E-AUTHGATE: the GitHub extension wired so its credential account resolver - /// returns `AuthRequired`, raising a `TurnStatus::BlockedAuth` gate when a - /// `github.*` capability is dispatched. Used by `RebornIntegrationGroup::live_auth_gate`. - pub(crate) async fn github_issue_tools_auth_required() -> HarnessResult { - Self::github_issue_tools_with_credential_result(Err(CredentialStageError::AuthRequired)) - } - - /// Shared GitHub-extension constructor (E-AUTHGATE): the only difference - /// between the happy-path and auth-blocked variants is the credential account - /// resolver result, so the full `Self {..}` literal lives here once. - /// - /// **Credential injection runs through two mechanisms here, not one — worth - /// knowing before you change either.** The authorizer's - /// `InjectCredentialAccountOnce` obligation is one path. The - /// `local_dev_host_runtime_with_registry_and_egress` helper this calls into - /// separately auto-wires `SharedHostWasmRuntimeCredentials` with product-auth - /// restaging via `try_with_wasm_runtime` (since both `.with_secret_store` and - /// `.with_runtime_credential_account_resolver` are always set on that path), - /// which independently resolves the GitHub manifest's declared - /// `runtime_credentials` and stages the same secret. That staging path runs - /// unconditionally on every WASM HTTP call (`WasmRuntimeHttpAdapter::request`) - /// — it is not gated on the authorizer's `Decision`. So a test asserting on the - /// injected header proves the *end-to-end* wire outcome, not that the - /// authorizer's obligation specifically is the sole producer of the header. - /// - /// As currently wired (manually verified once, not re-checked by CI — treat as - /// current-harness observation, not a guaranteed contract): removing the - /// obligation does not make the call fall back to an unauthenticated request; - /// the run instead hangs and never reaches `Completed`. That's why the - /// mutation-verify in `reborn_integration_secret_injection.rs` proves the - /// obligation's secret reaches the wire by flipping the secret *value* (a fast, - /// specific assertion failure) rather than by removing the obligation (which - /// would only yield a slow, ambiguous timeout — a poor mutation-test signal). - fn github_issue_tools_with_credential_result( - credential_account_result: Result, - ) -> HarnessResult { - let root = Arc::new(tempfile::tempdir()?); - let storage_root = root.path().join("local-dev"); - let workspace_root = storage_root.join("workspace"); - std::fs::create_dir_all(&workspace_root)?; - let github_fixture_response = - br#"{"object":{"sha":"abc123def4567890abc123def4567890abc123de"},"ok":true}"#.to_vec(); - let runtime_http_egress = Arc::new(RecordingRuntimeHttpEgress::with_body( - github_fixture_response.clone(), - )); - let network_egress = Arc::new(RecordingNetworkHttpEgress::with_body( - github_fixture_response, - )); - let runtime = local_dev_host_runtime_with_registry_and_egress( - storage_root.clone(), - github_support::extension_registry()?, - runtime_http_egress.clone(), - network_egress.clone(), - credential_account_result, - )?; - let mounts = workspace_mounts(MountPermissions::read_write_list_delete())?; - Ok(Self { - runtime, - approval_parts: None, - auto_approve_settings: None, - pending_approval_scopes: Arc::new(Mutex::new(HashMap::new())), - io: Arc::new(ProductLiveCapabilityIo::default()), - root, - workspace_root, - mounts, - capability_mount_overrides: Vec::new(), - capability_ids: github_support::capability_ids()?, - runtime_kind: RuntimeKind::Wasm, - effect_kinds: github_support::effect_kinds(), - network_policy: github_support::api_policy(), - secrets: github_support::secret_handles()?, - provider_id: github_support::provider_id()?, - additional_provider_trust: Vec::new(), - user_id: UserId::new("reborn-e2e-github-user")?, - invocations: Arc::new(Mutex::new(Vec::new())), - results: Arc::new(Mutex::new(Vec::new())), - http_egress: Some(runtime_http_egress), - network_egress: Some(network_egress), - process_port: None, - profile_filesystem: None, - project_service: None, - skill_activation_source: None, - attachment_test_support: None, - outbound_target_tools: None, - scope_capability_by_run_owner: false, - product_auth: None, - tool_permission_overrides: None, - }) - } - - /// Slice 6: wire a single MCP capability backed by the loopback mock server. - /// - /// `mcp_url` — the mock server's MCP endpoint (e.g. `"http://127.0.0.1:PORT/mcp"`). - /// `provider_id` — extension id used in the registry (e.g. `"mock-mcp"`). - /// `capability_id` — capability id surfaced to the model (e.g. `"mock-mcp.search"`). - /// - /// The harness (via the `harness_mcp` scaffolding) builds a loopback MCP - /// egress that makes REAL HTTP connections to the mock server, injecting a - /// fake Bearer token to satisfy the mock's auth gate. Production egress - /// policy, network policy, and credential stores are bypassed — this path is - /// test-only. - pub(crate) async fn mock_mcp_tools( - mcp_url: &str, - provider_id: &str, - capability_id: &str, - ) -> HarnessResult { - let (root, storage_root, workspace_root) = host_runtime_storage_roots()?; - // Recording egress for any first-party tool paths (unused in MCP tests, - // but HostRuntimeServices requires it when first_party_capabilities are wired). - let first_party_egress = Arc::new(RecordingRuntimeHttpEgress::with_body( - br#"{"accepted":true}"#.to_vec(), - )); - // Real loopback egress + MCP runtime for the mock MCP server; the - // scaffolding (egress, adapter chain, runtime) lives in `harness_mcp`. - let mcp_runtime = build_loopback_mcp_runtime(mcp_url)?; - let mut registry = ExtensionRegistry::new(); - registry.insert(mock_mcp_extension_package( - provider_id, - mcp_url, - capability_id, - )?)?; - let runtime = local_dev_host_runtime_with_registry_egress_and_mcp( - storage_root, - registry, - Arc::clone(&first_party_egress), - mcp_runtime, - provider_id, - )?; - let mounts = workspace_mounts(MountPermissions::read_write_list_delete())?; - Ok(Self { - runtime, - approval_parts: None, - auto_approve_settings: None, - pending_approval_scopes: Arc::new(Mutex::new(HashMap::new())), - io: Arc::new(ProductLiveCapabilityIo::default()), - root, - workspace_root, - mounts, - capability_mount_overrides: Vec::new(), - capability_ids: vec![CapabilityId::new(capability_id)?], - runtime_kind: RuntimeKind::Mcp, - effect_kinds: vec![EffectKind::DispatchCapability, EffectKind::Network], - // The MCP capability declares `EffectKind::Network`, so authorization - // attaches an `ApplyNetworkPolicy` obligation that the host runtime - // rejects when `allowed_targets` is empty (a default `NetworkPolicy`). - // The mock server lives at `http://127.0.0.1:/mcp`, so permit the - // loopback host (and disable the private-IP denial that would otherwise - // block 127.0.0.1) so the MCP egress reaches the loopback server. - network_policy: mcp_loopback_network_policy(), - secrets: Vec::new(), - provider_id: ExtensionId::new(provider_id)?, - additional_provider_trust: Vec::new(), - user_id: UserId::new("reborn-itest-mcp-user")?, - invocations: Arc::new(Mutex::new(Vec::new())), - results: Arc::new(Mutex::new(Vec::new())), - http_egress: None, - network_egress: None, - process_port: None, - profile_filesystem: None, - project_service: None, - skill_activation_source: None, - attachment_test_support: None, - outbound_target_tools: None, - scope_capability_by_run_owner: false, - product_auth: None, - tool_permission_overrides: None, - }) - } - - /// C-WEBACCESS: wires the real first-party `web-access.search` / - /// `web-access.get_content` capabilities via the production - /// `register_bundled_web_access_first_party_handlers` registration - /// (`harness_web_access.rs`), which dispatches through the same - /// `WebAccessExecutor` production composition uses. Unlike - /// `github_issue_tools`, no credential-injecting authorizer is needed — - /// web-access declares zero `runtime_credentials` — so this wires the - /// plain default `GrantAuthorizer`. - /// - /// The three-leg Exa MCP handshake (`initialize` → `notifications/initialized` - /// → `tools/call`) all target the same URL, so script it via - /// `RecordingRuntimeHttpEgress::push_response_body` (FIFO), not the keyed - /// matcher — see [`install_web_access_responses`](Self::install_web_access_responses), - /// called from `RebornIntegrationHarnessBuilder::build` before the harness - /// is returned. - pub(crate) async fn web_access_tools() -> HarnessResult { - let (root, storage_root, workspace_root) = host_runtime_storage_roots()?; - let http_egress = Arc::new(RecordingRuntimeHttpEgress::with_body( - br#"{"accepted":true}"#.to_vec(), - )); - let mut registry = ExtensionRegistry::new(); - registry.insert(harness_web_access::web_access_extension_package()?)?; - let runtime = harness_web_access::local_dev_host_runtime_with_web_access( - storage_root, - registry, - Arc::clone(&http_egress), - )?; - let mounts = workspace_mounts(MountPermissions::read_write_list_delete())?; - Ok(Self { - runtime, - approval_parts: None, - auto_approve_settings: None, - pending_approval_scopes: Arc::new(Mutex::new(HashMap::new())), - io: Arc::new(ProductLiveCapabilityIo::default()), - root, - workspace_root, - mounts, - capability_mount_overrides: Vec::new(), - capability_ids: vec![ - CapabilityId::new(WEB_SEARCH_CAPABILITY_ID)?, - CapabilityId::new(WEB_GET_CONTENT_CAPABILITY_ID)?, - ], - runtime_kind: RuntimeKind::FirstParty, - effect_kinds: vec![EffectKind::DispatchCapability, EffectKind::Network], - network_policy: harness_web_access::exa_mcp_test_network_policy(), - secrets: Vec::new(), - provider_id: ExtensionId::new(harness_web_access::WEB_ACCESS_PROVIDER_ID)?, - additional_provider_trust: Vec::new(), - user_id: UserId::new("reborn-itest-web-access-user")?, - invocations: Arc::new(Mutex::new(Vec::new())), - results: Arc::new(Mutex::new(Vec::new())), - http_egress: Some(http_egress), - network_egress: None, - process_port: None, - profile_filesystem: None, - project_service: None, - skill_activation_source: None, - attachment_test_support: None, - outbound_target_tools: None, - scope_capability_by_run_owner: false, - product_auth: None, - tool_permission_overrides: None, - }) - } - - fn capability_factory( - self: &Arc, - milestone_sink: Arc, - ) -> Arc { - Arc::new(HostRuntimeHarnessCapabilityPortFactory { - harness: Arc::clone(self), - milestone_sink, - }) - } - - fn capability_result_writer(self: &Arc) -> Arc { - Arc::new(RecordingCapabilityResultWriter { - inner: self.io.clone(), - results: Arc::clone(&self.results), - }) - } - - fn invocations(&self) -> Vec { - self.invocations.lock().unwrap().clone() - } - - fn capability_results(&self) -> Vec { - self.results.lock().unwrap().clone() - } - - fn runtime_http_requests(&self) -> Vec { - self.http_egress - .as_ref() - .map(|egress| egress.requests()) - .unwrap_or_default() - } - - /// Install FIFO response bodies (C-WEBACCESS) onto the recording runtime - /// HTTP egress, consumed in call order ahead of the default body. Mirrors - /// [`install_http_responses`](Self::install_http_responses)'s shape but for - /// the web-access backend's `push_response_body` FIFO queue rather than the - /// keyed matcher — the three-leg Exa MCP handshake (`initialize` → - /// `notifications/initialized` → `tools/call`) all target the same - /// URL/method/capability, so only the FIFO queue can script them - /// independently. Errors if this harness wired no recording egress. Called - /// from `RebornIntegrationHarnessBuilder::build` (build-time only — no - /// post-build mutation). - pub(crate) fn install_web_access_responses( - &self, - bodies: impl IntoIterator>, - ) -> HarnessResult<()> { - let egress = self - .http_egress - .as_ref() - .ok_or("web-access host runtime has no recording egress wired")?; - for body in bodies { - egress.push_response_body(body); - } - Ok(()) - } - - /// Snapshot of every command string recorded by the inert process port - /// (slice 5). Empty when the harness uses the live `LocalHostProcessPort` - /// (`.with_live_shell()` path) or predates slice 5. - fn process_commands(&self) -> Vec { - self.process_port - .as_ref() - .map(|port| port.commands()) - .unwrap_or_default() - } - - /// Install URL/method/capability-keyed scripted responses into the recording - /// HTTP egress (§3.6 P1 ergonomics). Errors if this harness wired no - /// recording egress (e.g. the live-HTTP variant). - pub(crate) fn install_http_responses( - &self, - responses: impl IntoIterator, - ) -> HarnessResult<()> { - self.http_egress - .as_ref() - .ok_or("host runtime harness has no recording http egress to script")? - .install_scripted(responses); - Ok(()) - } - - /// W4-AUTHGATE-WIRE: enqueue a FIFO scripted status on the recording - /// **network** HTTP egress (see `RecordingNetworkHttpEgress::push_status`). - /// For `GithubIssueTools`-backed harnesses, the real WASM HTTP call flows - /// through this lane (not `install_http_responses`'s runtime-egress - /// matcher) — see `reborn_integration_secret_injection.rs`'s module doc. - /// Errors if this harness wired no recording network egress. - pub(crate) fn install_network_status_script(&self, status: u16) -> HarnessResult<()> { - self.network_egress - .as_ref() - .ok_or("host runtime harness has no recording network egress to script")? - .push_status(status); - Ok(()) - } - - /// Install a sticky scripted `builtin.shell` process result on the inert - /// recording process port (mirrors `install_http_responses`). Errors if the - /// harness has no recording port (e.g. the `.with_live_shell()` path). - pub(crate) fn install_process_script( - &self, - result: super::process::ScriptedProcessResult, - ) -> HarnessResult<()> { - self.process_port - .as_ref() - .ok_or("host runtime harness has no recording process port to script")? - .set_scripted(result); - Ok(()) - } - - fn network_http_requests(&self) -> Vec { - self.network_egress - .as_ref() - .map(|egress| egress.requests()) - .unwrap_or_default() - } - - fn workspace_file_path(&self, relative: &str) -> PathBuf { - self.workspace_root.join(relative.trim_start_matches('/')) - } - - async fn approve_local_dev_gate(&self, gate_ref: &GateRef) -> HarnessResult<()> { - let approval_parts = self - .approval_parts - .as_ref() - .ok_or("host runtime harness has no local-dev approval stores")?; - let request_id = approval_request_id_from_gate_ref(gate_ref)?; - let scope = self - .pending_approval_scopes - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) - .get(&request_id) - .cloned() - .ok_or("approval gate was not recorded by the host runtime harness")?; - let record = approval_parts - .approval_requests - .get(&scope, request_id) - .await? - .ok_or("approval request was not persisted")?; - let capability = match record.request.action.as_ref() { - Action::Dispatch { capability, .. } | Action::SpawnCapability { capability, .. } => { - capability.clone() - } - other => return Err(format!("unsupported approval action: {other:?}").into()), - }; - let approval = self.lease_approval_for(&capability); - let resolver = ApprovalResolver::new( - approval_parts.approval_requests.as_ref(), - approval_parts.capability_leases.as_ref(), - ); - match record.request.action.as_ref() { - Action::Dispatch { .. } => { - resolver - .approve_dispatch(&scope, request_id, approval) - .await?; - } - Action::SpawnCapability { .. } => { - resolver.approve_spawn(&scope, request_id, approval).await?; - } - other => return Err(format!("unsupported approval action: {other:?}").into()), - } - Ok(()) - } - - /// Deny a pending local-dev approval gate (the model-declined path). Mirrors - /// [`approve_local_dev_gate`](Self::approve_local_dev_gate) but resolves the - /// persisted request to `Denied` (no lease issued) via `ApprovalResolver::deny`. - /// The caller then resumes the run with `GateResumeDisposition::Denied` so the - /// executor surfaces a non-retryable authorization failure to the model. - async fn deny_local_dev_gate(&self, gate_ref: &GateRef) -> HarnessResult<()> { - let approval_parts = self - .approval_parts - .as_ref() - .ok_or("host runtime harness has no local-dev approval stores")?; - let request_id = approval_request_id_from_gate_ref(gate_ref)?; - let scope = self - .pending_approval_scopes - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) - .get(&request_id) - .cloned() - .ok_or("approval gate was not recorded by the host runtime harness")?; - let resolver = ApprovalResolver::new( - approval_parts.approval_requests.as_ref(), - approval_parts.capability_leases.as_ref(), - ); - resolver - .deny( - &scope, - request_id, - DenyApproval { - denied_by: Principal::User(scope.user_id.clone()), - }, - ) - .await?; - Ok(()) - } - - /// The persisted approval-request store, when this harness wires the real - /// local-dev approval stores (`file_tools_requiring_approval`). The - /// integration runtime builds an [`ApprovalGateEvidenceStore`] over it so a - /// `BlockedApproval` run is verified at loop exit (mirrors production - /// `runtime.rs:2799`) and genuinely pauses instead of failing. - pub(crate) fn approval_requests_store( - &self, - ) -> Option> { - self.approval_parts - .as_ref() - .map(|parts| Arc::clone(&parts.approval_requests)) - } - - /// The user id this capability harness's first-party tools execute under. - /// The dispatch-time auto-approve check is keyed `(tenant, user)` on THIS - /// user (not the run's binding owner), so the group derives the auto-approve - /// scope from it — see `GroupSharedStorage::auto_approve_scope`. - pub(crate) fn user_id(&self) -> &UserId { - &self.user_id - } - - /// E-PROFILE: the raw local-dev memory filesystem backing the user-profile - /// source, for write→read-back assertions on `context/profile.json`. `Some` - /// only for `new_with_options`-built harnesses. Consumed by the E-PROFILE - /// `profile_tools()` constructor and the `reborn_integration_profile` test. - pub(crate) fn profile_filesystem_for_test(&self) -> Option> { - self.profile_filesystem.clone() - } - - /// E-PROJ: the project service backing the synthetic `project_create` - /// capability, for write→read-back assertions — mirrors - /// `profile_filesystem_for_test`'s role for E-PROFILE. `Some` only for - /// `project_tools()`-built harnesses. Lets a test read a created project - /// back through the SAME `Arc` instance - /// `apply_synthetic_capability_wrappers` dispatches writes through, rather - /// than reconstructing an equivalent (and possibly unwritten) one. - pub(crate) fn project_service_for_test(&self) -> Option> { - self.project_service.clone() - } - - /// C-SYNTH outbound: the injected facade double, for read-back that a - /// `target_set` actually reached the facade seam - /// (`recorded_set_target_ids`). `Some` only for `outbound_target_tools()`. - pub(crate) fn outbound_preferences_facade_for_test( - &self, - ) -> Option> { - self.outbound_target_tools - .as_ref() - .map(|parts| Arc::clone(&parts.facade)) - } - - /// C-SYNTH outbound: persist a `Disabled` per-tool permission override for - /// `outbound_delivery_target_set` under `(tenant, user)`, driving the - /// handler's settings decision to `Deny` → `Failed{policy_denied}`. The - /// scope must be the run's EFFECTIVE dispatch user (the thread binding actor, - /// `harness.binding.actor_user_id`) — the same `(tenant, user)` - /// `StoreApprovalSettingsProvider::tool_override` reads it back under - /// (`PersistentApprovalScope` = tenant+user, invocation-independent). `Some` - /// only for `outbound_target_tools()`. - pub(crate) async fn disable_outbound_target_set_tool( - &self, - tenant_id: TenantId, - user_id: UserId, - ) -> HarnessResult<()> { - let parts = self - .outbound_target_tools - .as_ref() - .ok_or("harness has no outbound_target_tools backing store")?; - let scope = ResourceScope { - tenant_id, - user_id: user_id.clone(), - agent_id: None, - project_id: None, - mission_id: None, - thread_id: None, - invocation_id: InvocationId::new(), - }; - parts - .tool_permission_overrides - .set(ironclaw_approvals::CapabilityPermissionOverrideInput { - scope, - capability_id: CapabilityId::new( - ironclaw_reborn_composition::test_support::OUTBOUND_DELIVERY_TARGET_SET_CAPABILITY_ID, - )?, - state: ironclaw_approvals::CapabilityPermissionOverride::Disabled, - updated_by: Principal::User(user_id), - }) - .await?; - Ok(()) - } - - /// W4-ASK-EACH-ONCE: install a `ToolPermissionOverride::AskEachTime` - /// override for `capability_id` under `(tenant_id, user_id)` via the real - /// local-dev per-tool permission override store -- generalizes - /// `disable_outbound_target_set_tool`'s shape (same scope-key - /// convention: `agent_id`/`project_id` unset, matching how - /// `StoreApprovalSettingsProvider::tool_override` looks the override up) - /// to any host-runtime-backed harness/group, not just - /// `outbound_target_tools()`. Drives the SAME - /// `require_approval_for_profile_policy` `tool_override` consultation - /// the #5306 fix reordered relative to the one-shot approval-lease check. - /// Errors if this harness wired no local-dev tool-permission-override - /// store (i.e. not built via `new_with_options`). - pub(crate) async fn set_ask_each_time_override_for_test( - &self, - capability_id: &CapabilityId, - tenant_id: TenantId, - user_id: UserId, - ) -> HarnessResult<()> { - let store = self - .tool_permission_overrides - .as_ref() - .ok_or("harness has no local-dev tool-permission-override store")?; - let scope = ResourceScope { - tenant_id, - user_id: user_id.clone(), - agent_id: None, - project_id: None, - mission_id: None, - thread_id: None, - invocation_id: InvocationId::new(), - }; - store - .set(ironclaw_approvals::CapabilityPermissionOverrideInput { - scope, - capability_id: capability_id.clone(), - state: ironclaw_approvals::CapabilityPermissionOverride::AskEachTime, - updated_by: Principal::User(user_id), - }) - .await?; - Ok(()) - } - - /// E-SKILL: the `HostSkillContextSource` to wire as the runtime's - /// `skill_context_source` in `into_group`, so activated-skill instructions - /// inject into the model request. `Some` only for `skill_activation_tools()`. - pub(crate) fn skill_context_source_for_test( - &self, - ) -> Option> { - self.skill_activation_source - .as_ref() - .map(|source| source.context_source()) - } - - /// E-DURABLE: the on-disk local-dev storage root this harness's capability - /// stores persist under (`/local-dev`). Mirrors the `storage_root` - /// computed inline in `new_with_options`. A durability test reopens a fresh, - /// independent store at this path (see - /// `open_local_dev_extension_installation_store_for_test`) to prove capability - /// state survives a reopen, paralleling `assert_reply_persists_after_reopen`. - /// Tests only. - pub(crate) fn storage_root_for_test(&self) -> PathBuf { - self.root.path().join("local-dev") - } - - /// C-DURABLE: resolve `gate_ref` (a `"gate:approval-"` local-dev - /// approval gate) to the `(ApprovalRequestId, ResourceScope)` pair a fresh, - /// independently-reopened `ApprovalRequestStore::get`/`read_versioned` call - /// needs. Reuses the SAME private lookup `approve_local_dev_gate`/ - /// `deny_local_dev_gate` already use (`approval_request_id_from_gate_ref` + - /// `pending_approval_scopes`) so a durability test's scope construction can - /// never drift from the live approve/deny path. Tests only. - pub(crate) fn approval_request_scope_for_test( - &self, - gate_ref: &GateRef, - ) -> HarnessResult<(ApprovalRequestId, ResourceScope)> { - let request_id = approval_request_id_from_gate_ref(gate_ref)?; - let scope = self - .pending_approval_scopes - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) - .get(&request_id) - .cloned() - .ok_or("approval gate was not recorded by the host runtime harness")?; - Ok((request_id, scope)) - } - - /// E-SKILL: seed a system-scoped skill on this harness's on-disk skill - /// filesystem so the model can activate it (`skill_activate`/`$name`). Writes - /// `/system/skills//SKILL.md` — the system bundle root is - /// always present in the skills extension's roots regardless of the run's - /// tenant/user (`FirstPartySkillsExtensionHandles::bundle_roots`), so both - /// `activate_skills_for_run` (the `skill_activate` capability) and the - /// runtime's `skill_context_source` resolve it deterministically without - /// depending on the harness's run-scope owner resolution. User-scoped skill - /// filesystem resolution is already covered by the runtime.rs suite; this - /// seam only needs the skill to exist so the capability + context wiring can - /// be driven. Mirrors the runtime-test system-skill layout - /// (runtime.rs `system/skills//SKILL.md`). Tests only. - pub(crate) fn seed_system_skill_for_test( - &self, - name: &str, - description: &str, - prompt: &str, - ) -> HarnessResult<()> { - let dir = self - .storage_root_for_test() - .join("system") - .join("skills") - .join(name); - std::fs::create_dir_all(&dir)?; - let body = format!( - "---\nname: {name}\ndescription: {description}\nactivation:\n keywords: [\"{name}\"]\n---\n\n{prompt}" - ); - std::fs::write(dir.join("SKILL.md"), body)?; - Ok(()) - } - - /// C-SYNTH `skill_activate` `AmbiguousSkill` seeding arm: seed a - /// USER-scoped skill (writes - /// `/tenants//users//skills//SKILL.md`, - /// mirroring the runtime.rs composition-test layout) so a name shared with - /// a system-scoped skill (`seed_system_skill_for_test`) resolves to TWO - /// Trusted candidates under different `SkillSourceKind`s — `System` root - /// and `User` root both default to `SkillTrust::Trusted` - /// (`FilesystemSkillBundleRoot::system`/`user`, - /// `crates/ironclaw_loop_support/src/filesystem_skill_bundle_source.rs`), - /// so `select_named_skill_activations`'s `active_candidates` filter - /// (`trust == Trusted`) admits both, and - /// `validate_explicit_mentions_are_unambiguous` then rejects the shared - /// name as `SkillActivationSelectionError::AmbiguousSkill`. `tenant`/`user` - /// must be the SAME `(tenant, actor_user_id)` the driving thread's run - /// resolves under (`harness.binding`), or the user root never matches the - /// run's own scoped `/skills` mount. Tests only. - pub(crate) fn seed_user_skill_for_test( - &self, - tenant: &TenantId, - user: &UserId, - name: &str, - description: &str, - prompt: &str, - ) -> HarnessResult<()> { - let dir = self - .storage_root_for_test() - .join("tenants") - .join(tenant.as_str()) - .join("users") - .join(user.as_str()) - .join("skills") - .join(name); - std::fs::create_dir_all(&dir)?; - let body = format!( - "---\nname: {name}\ndescription: {description}\nactivation:\n keywords: [\"{name}\"]\n---\n\n{prompt}" - ); - std::fs::write(dir.join("SKILL.md"), body)?; - Ok(()) - } - - /// C-ATTACH: the attachment read port + inbound lander over this harness's - /// local-dev workspace filesystem, for wiring `DefaultPlannedRuntimeParts.attachment_read_port` - /// and `DefaultInboundTurnService::with_inbound_attachments` — mirrors - /// `profile_filesystem_for_test`'s role for E-PROFILE. - pub(crate) fn attachment_test_support_for_test( - &self, - ) -> Option { - self.attachment_test_support.clone() - } - - /// E-PROJ: wrap `port` with the local-dev synthetic capabilities this harness - /// surfaces, in one linear step (keeps the capability-specific knowledge out - /// of `create_capability_port`'s main assembly chain). - /// - /// Partial synthetic wrap: `project_create` (E-PROJ), `skill_activate` - /// (E-SKILL), and the two `outbound_delivery_*` capabilities (C-SYNTH - /// outbound), each layered independently when this harness holds the backing - /// handle, so other local-dev groups are unaffected. See - /// `LocalDevCapabilityPortFactory::build_inner()` for the full production set. - fn apply_synthetic_capability_wrappers( - &self, - port: Arc, - run_context: &LoopRunContext, - input_resolver: Arc, - result_writer: Arc, - ) -> Result, AgentLoopHostError> { - let mut port = port; - // project_create (E-PROJ): wrapped only for `project_tools`. - if let Some(project_service) = &self.project_service - && self.capability_ids.iter().any(|id| { - id.as_str() - == ironclaw_reborn_composition::test_support::PROJECT_CREATE_CAPABILITY_ID - }) - { - port = - ironclaw_reborn_composition::test_support::wrap_project_create_capability_for_test( - port, - Arc::clone(project_service), - self.user_id.clone(), - run_context.clone(), - input_resolver.clone(), - result_writer.clone(), - )?; - } - // outbound_delivery_* (C-SYNTH outbound): wrapped only for - // `outbound_target_tools`. The facade double is injected at the - // production-wired trait seam; the settings/approval stores are the same - // ones `outbound_delivery_capabilities` consumes in production (the - // auto-approve + approval-request/lease stores are reused from the - // sibling `auto_approve_settings` / `approval_parts` harness fields). - if let Some(parts) = &self.outbound_target_tools { - let approval = self.approval_parts.as_ref().ok_or_else(|| { - AgentLoopHostError::new( - AgentLoopHostErrorKind::Internal, - "outbound_target_tools requires local-dev approval stores", - ) - })?; - let auto_approve = self.auto_approve_settings.clone().ok_or_else(|| { - AgentLoopHostError::new( - AgentLoopHostErrorKind::Internal, - "outbound_target_tools requires the local-dev auto-approve store", - ) - })?; - // Record the synthetic capability's approval scope into - // `pending_approval_scopes` (the host-runtime recorder can't see a - // port-level synthetic gate) so `approve_local_dev_gate` / - // `deny_local_dev_gate` resolve it; delegates to the inner store the - // evidence/approve/deny paths read. - let recording_approval_requests: Arc = - Arc::new(RecordingApprovalRequestStore { - inner: Arc::clone(&approval.approval_requests), - pending_approval_scopes: Arc::clone(&self.pending_approval_scopes), - }); - port = - ironclaw_reborn_composition::test_support::wrap_outbound_delivery_capabilities_for_test( - port, - ironclaw_reborn_composition::test_support::OutboundDeliveryCapabilityTestParts { - facade: Arc::clone(&parts.facade) - as Arc, - fallback_user_id: self.user_id.clone(), - approval_requests: recording_approval_requests, - capability_leases: Arc::clone(&approval.capability_leases), - tool_permission_overrides: Arc::clone(&parts.tool_permission_overrides), - auto_approve, - persistent_policies: Arc::clone(&parts.persistent_approval_policies), - target_set_requires_approval: parts.requires_approval, - run_context: run_context.clone(), - input_resolver: input_resolver.clone(), - result_writer: result_writer.clone(), - }, - )?; - } - // skill_activate (E-SKILL): wrapped only for `skill_activation_tools`. - if let Some(skill_source) = &self.skill_activation_source - && self.capability_ids.iter().any(|id| { - id.as_str() - == ironclaw_reborn_composition::test_support::SKILL_ACTIVATE_CAPABILITY_ID - }) - { - port = - ironclaw_reborn_composition::test_support::wrap_skill_activation_capability_for_test( - port, - skill_source, - run_context.clone(), - input_resolver, - result_writer, - )?; - } - Ok(port) - } - - /// Override the user this capability harness executes first-party tools under. - /// The dispatch ResourceScope, approval-request persistence, auto-approve - /// keying, and the approval-gate-evidence lookup are ALL keyed on this user - /// (`HostRuntimeHarnessCapabilityPortFactory` builds the authority from - /// `self.user_id`). The integration harness sets it to the run's binding owner - /// so capability dispatch and the turn run under the SAME `(tenant, user)` — - /// matching production (where the run owner *is* the capability user) instead - /// of the constructor's fixed test user. Without this, a `BlockedApproval` - /// run's request persists under the capability user but the gate-evidence - /// lookup uses the turn owner, so the gate is never verified and the run goes - /// terminal `Failed`. - pub(crate) fn with_user_id(mut self, user_id: UserId) -> Self { - self.user_id = user_id; - self - } - - /// C-MULTIUSER: opt in to per-actor capability scoping. With this set, - /// [`create_capability_port`] resolves the execution `(tenant, user)` from - /// each run's OWN owner/actor rather than this harness's single fixed - /// `user_id` — so N actors sharing one capability backend dispatch under N - /// distinct scopes. See [`scope_capability_by_run_owner`] and - /// [`dispatch_user_for_run`]. Enabled only by the multiuser group - /// constructors; every other harness keeps the legacy fixed-user behavior. - pub(crate) fn with_run_owner_scoped_capability_dispatch(mut self) -> Self { - self.scope_capability_by_run_owner = true; - self - } - - /// The capability-execution `UserId` for one run. Mirrors production - /// `local_dev_visible_capability_request`'s owner→actor→fallback resolution - /// (`runtime/local_dev.rs`): when [`scope_capability_by_run_owner`] is set, - /// prefer the run scope's explicit owner, then the run actor, then fall back - /// to the fixed harness `user_id`. Without the flag, always the fixed - /// `user_id` (legacy behavior — every existing test unaffected). - fn dispatch_user_for_run(&self, run_context: &LoopRunContext) -> UserId { - if self.scope_capability_by_run_owner { - run_context - .scope - .explicit_owner_user_id() - .cloned() - .or_else(|| run_context.actor().map(|actor| actor.user_id.clone())) - .unwrap_or_else(|| self.user_id.clone()) - } else { - self.user_id.clone() - } - } - - fn lease_approval_for(&self, capability_id: &CapabilityId) -> LeaseApproval { - let mounts = self - .capability_mount_overrides - .iter() - .find(|(override_capability, _)| override_capability == capability_id) - .map(|(_, mounts)| mounts.clone()) - .unwrap_or_else(|| self.mounts.clone()); - LeaseApproval { - issued_by: Principal::HostRuntime, - constraints: GrantConstraints { - allowed_effects: self.effect_kinds.clone(), - mounts, - network: self.network_policy.clone(), - secrets: self.secrets.clone(), - resource_ceiling: None, - expires_at: None, - max_invocations: Some(1), - }, - } - } -} - -fn approval_request_id_from_gate_ref(gate_ref: &GateRef) -> HarnessResult { - const APPROVAL_GATE_PREFIX: &str = "gate:approval-"; - let value = gate_ref - .as_str() - .strip_prefix(APPROVAL_GATE_PREFIX) - .ok_or("gate ref is not a local-dev approval gate")?; - Ok(ApprovalRequestId::parse(value)?) -} - -async fn seed_extension_lifecycle_credentials( - services: &ironclaw_reborn_composition::RebornServices, - user_id: &UserId, -) -> HarnessResult<()> { - let product_auth = services - .product_auth - .as_ref() - .ok_or("extension lifecycle harness missing product auth")?; - let scope = AuthProductScope::credential_owner( - &ResourceScope { - tenant_id: TenantId::new("tenant-e2e")?, - user_id: user_id.clone(), - agent_id: Some(AgentId::new("agent-e2e")?), - project_id: Some(ProjectId::new("project-e2e")?), - mission_id: None, - thread_id: None, - invocation_id: InvocationId::new(), - }, - AuthSurface::Api, - ); - let accounts = product_auth.credential_account_service(); - for seed in extension_lifecycle_credential_seeds() { - accounts - .create_account(NewCredentialAccount { - scope: scope.clone(), - provider: AuthProviderId::new(seed.provider)?, - label: CredentialAccountLabel::new(seed.label)?, - status: CredentialAccountStatus::Configured, - ownership: CredentialOwnership::UserReusable, - owner_extension: None, - granted_extensions: Vec::new(), - access_secret: Some(SecretHandle::new(seed.secret_handle)?), - refresh_secret: None, - scopes: seed - .scopes - .iter() - .map(|scope| ProviderScope::new(*scope)) - .collect::, _>>()?, - }) - .await?; - } - Ok(()) -} - -struct ExtensionLifecycleCredentialSeed { - provider: &'static str, - label: &'static str, - secret_handle: &'static str, - scopes: &'static [&'static str], -} - -fn extension_lifecycle_credential_seeds() -> &'static [ExtensionLifecycleCredentialSeed] { - &[ - ExtensionLifecycleCredentialSeed { - provider: "github", - label: "qa github", - secret_handle: "qa_github_access", - scopes: &[], - }, - ExtensionLifecycleCredentialSeed { - provider: "google", - label: "qa google", - secret_handle: "qa_google_access", - scopes: &[ - "https://www.googleapis.com/auth/calendar.events", - "https://www.googleapis.com/auth/calendar.readonly", - "https://www.googleapis.com/auth/documents", - "https://www.googleapis.com/auth/documents.readonly", - "https://www.googleapis.com/auth/drive", - "https://www.googleapis.com/auth/drive.readonly", - "https://www.googleapis.com/auth/gmail.modify", - "https://www.googleapis.com/auth/gmail.readonly", - "https://www.googleapis.com/auth/gmail.send", - "https://www.googleapis.com/auth/presentations", - "https://www.googleapis.com/auth/presentations.readonly", - "https://www.googleapis.com/auth/spreadsheets", - "https://www.googleapis.com/auth/spreadsheets.readonly", - ], - }, - ExtensionLifecycleCredentialSeed { - provider: "nearai", - label: "qa nearai", - secret_handle: "qa_nearai_access", - scopes: &[], - }, - ExtensionLifecycleCredentialSeed { - provider: "notion", - label: "qa notion", - secret_handle: "qa_notion_access", - scopes: &[], - }, - ] -} - -struct RecordingHostRuntime { - inner: Arc, - pending_approval_scopes: Arc>>, -} - -impl RecordingHostRuntime { - fn new( - inner: Arc, - pending_approval_scopes: Arc>>, - ) -> Self { - Self { - inner, - pending_approval_scopes, - } - } -} - -#[async_trait] -impl HostRuntime for RecordingHostRuntime { - async fn invoke_capability( - &self, - request: RuntimeCapabilityRequest, - ) -> Result { - let scope = request.context.resource_scope.clone(); - let outcome = self.inner.invoke_capability(request).await?; - if let RuntimeCapabilityOutcome::ApprovalRequired(gate) = &outcome { - self.pending_approval_scopes - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) - .insert(gate.approval_request_id, scope); - } - Ok(outcome) - } - - async fn spawn_capability( - &self, - request: RuntimeCapabilityRequest, - ) -> Result { - let scope = request.context.resource_scope.clone(); - let outcome = self.inner.spawn_capability(request).await?; - if let RuntimeCapabilityOutcome::ApprovalRequired(gate) = &outcome { - self.pending_approval_scopes - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) - .insert(gate.approval_request_id, scope); - } - Ok(outcome) - } - - async fn resume_capability( - &self, - request: RuntimeCapabilityResumeRequest, - ) -> Result { - self.inner.resume_capability(request).await - } - - /// C-JOURNEY: forward auth-resume to the real runtime. `auth_resume_capability` - /// is a DEFAULTED trait method whose default fails loudly ("capability - /// auth-resume is unsupported by this host runtime", `ironclaw_host_runtime` - /// lib.rs) precisely so wrappers that forget to forward it fail visibly — - /// which is exactly what happened here: this wrapper predates any test - /// exercising a happy-path auth resume, so the missing forward was latent - /// until the first auth→resolve→re-dispatch journey drove it. Mirrors - /// `invoke_capability`'s ApprovalRequired scope recording because - /// `auth_resume_json` can itself raise an approval gate (the NoPriorLease - /// path re-runs authorization). - async fn auth_resume_capability( - &self, - request: ironclaw_host_runtime::RuntimeCapabilityAuthResumeRequest, - ) -> Result { - let scope = request.context.resource_scope.clone(); - let outcome = self.inner.auth_resume_capability(request).await?; - if let RuntimeCapabilityOutcome::ApprovalRequired(gate) = &outcome { - self.pending_approval_scopes - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) - .insert(gate.approval_request_id, scope); - } - Ok(outcome) - } - - async fn resume_spawn_capability( - &self, - request: RuntimeCapabilityResumeRequest, - ) -> Result { - self.inner.resume_spawn_capability(request).await - } - - async fn visible_capabilities( - &self, - request: RuntimeVisibleCapabilityRequest, - ) -> Result { - self.inner.visible_capabilities(request).await - } - - async fn cancel_work( - &self, - request: CancelRuntimeWorkRequest, - ) -> Result { - self.inner.cancel_work(request).await - } - - async fn runtime_status( - &self, - request: RuntimeStatusRequest, - ) -> Result { - self.inner.runtime_status(request).await - } - - async fn health(&self) -> Result { - self.inner.health().await - } -} - -/// Records `(ApprovalRequestId, ResourceScope)` on `save_pending`, then delegates -/// every method to the inner store. Synthetic local-dev capabilities (e.g. -/// `outbound_delivery_target_set`) persist their approval requests directly to -/// the approval store rather than through the host runtime, so -/// [`RecordingHostRuntime`] (which only observes host-runtime-level gates) never -/// captures their scope. Wrapping the store the synthetic capability writes -/// through restores the same `pending_approval_scopes` bookkeeping the -/// `approve_local_dev_gate` / `deny_local_dev_gate` lookups depend on. Delegation -/// is total, so the inner store the evidence/approve/deny paths read stays the -/// single source of truth. -struct RecordingApprovalRequestStore { - inner: Arc, - pending_approval_scopes: Arc>>, -} - -#[async_trait] -impl ironclaw_run_state::ApprovalRequestStore for RecordingApprovalRequestStore { - async fn save_pending( - &self, - scope: ResourceScope, - request: ironclaw_host_api::approval::ApprovalRequest, - ) -> Result { - self.pending_approval_scopes - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) - .insert(request.id, scope.clone()); - self.inner.save_pending(scope, request).await - } - - async fn get( - &self, - scope: &ResourceScope, - request_id: ApprovalRequestId, - ) -> Result, ironclaw_run_state::RunStateError> { - self.inner.get(scope, request_id).await - } - - async fn approve( - &self, - scope: &ResourceScope, - request_id: ApprovalRequestId, - ) -> Result { - self.inner.approve(scope, request_id).await - } - - async fn deny( - &self, - scope: &ResourceScope, - request_id: ApprovalRequestId, - ) -> Result { - self.inner.deny(scope, request_id).await - } - - async fn discard_pending( - &self, - scope: &ResourceScope, - request_id: ApprovalRequestId, - ) -> Result { - self.inner.discard_pending(scope, request_id).await - } - - async fn records_for_scope( - &self, - scope: &ResourceScope, - ) -> Result, ironclaw_run_state::RunStateError> { - self.inner.records_for_scope(scope).await - } -} - -struct HostRuntimeHarnessCapabilityPortFactory { - harness: Arc, - milestone_sink: Arc, -} - -#[async_trait] -impl LoopCapabilityPortFactory for HostRuntimeHarnessCapabilityPortFactory { - async fn create_capability_port( - &self, - run_context: &LoopRunContext, - ) -> Result, AgentLoopHostError> { - // C-MULTIUSER: resolve the execution user per run (owner/actor) when the - // harness opts in, else the fixed harness user. Both the authority scope - // and the grant grantee MUST use the SAME user so the lease is - // self-consistent (grantee == execution user) — matching production. - let dispatch_user = self.harness.dispatch_user_for_run(run_context); - let mut authority = ProductLiveVisibleCapabilityRequestConfig::new( - dispatch_user.clone(), - self.harness.runtime_kind, - TrustClass::FirstParty, - SurfaceKind::new("agent_loop").map_err(host_runtime_harness_error)?, - CapabilitySurfacePolicy::allow_all(), - ) - .with_mounts(self.harness.mounts.clone()) - .with_grants(capability_grants( - Principal::User(dispatch_user.clone()), - &self.harness.capability_ids, - self.harness.effect_kinds.clone(), - self.harness.mounts.clone(), - &self.harness.capability_mount_overrides, - self.harness.network_policy.clone(), - self.harness.secrets.clone(), - )) - .with_provider_trust_for_effects( - self.harness.provider_id.clone(), - EffectiveTrustClass::user_trusted(), - self.harness.effect_kinds.clone(), - ); - for (provider, effects) in &self.harness.additional_provider_trust { - authority = authority.with_provider_trust_for_effects( - provider.clone(), - EffectiveTrustClass::user_trusted(), - effects.clone(), - ); - } - let execution_mounts = self.harness.mounts.clone(); - let visible_request = visible_capability_request_for_run(run_context, authority) - .map_err(host_runtime_harness_error)?; - let milestone_sink: Arc = self.milestone_sink.clone(); - let result_writer = Arc::new(RecordingCapabilityResultWriter { - inner: self.harness.io.clone(), - results: Arc::clone(&self.harness.results), - }); - let mut factory = HostRuntimeLoopCapabilityPortFactory::new( - Arc::clone(&self.harness.runtime), - visible_request, - self.harness.io.clone(), - result_writer.clone(), - milestone_sink, - ) - .with_execution_mounts(execution_mounts); - for (capability_id, mounts) in &self.harness.capability_mount_overrides { - factory = - factory.with_capability_execution_mount(capability_id.clone(), mounts.clone()); - } - let port = factory.for_run_context(run_context.clone()); - // E-PROJ: see `apply_synthetic_capability_wrappers`'s doc comment. - let port = self.harness.apply_synthetic_capability_wrappers( - port, - run_context, - self.harness.io.clone(), - result_writer, - )?; - Ok(Arc::new(RecordingDelegatingCapabilityPort { - inner: port, - invocations: Arc::clone(&self.harness.invocations), - })) - } -} - -struct RecordingDelegatingCapabilityPort { - inner: Arc, - invocations: Arc>>, -} - -#[async_trait] -impl LoopCapabilityPort for RecordingDelegatingCapabilityPort { - fn tool_definitions(&self) -> Result, AgentLoopHostError> { - self.inner.tool_definitions() - } - - fn validate_provider_tool_call( - &self, - tool_call: &ProviderToolCall, - ) -> Result<(), AgentLoopHostError> { - self.inner.validate_provider_tool_call(tool_call) - } - - async fn register_provider_tool_call( - &self, - request: ironclaw_turns::run_profile::RegisterProviderToolCallRequest, - ) -> Result { - self.inner.register_provider_tool_call(request).await - } - - async fn visible_capabilities( - &self, - request: VisibleCapabilityRequest, - ) -> Result { - self.inner.visible_capabilities(request).await - } - - async fn invoke_capability( - &self, - request: CapabilityInvocation, - ) -> Result { - self.invocations.lock().unwrap().push(request.clone()); - self.inner.invoke_capability(request).await - } - - async fn invoke_capability_batch( - &self, - request: CapabilityBatchInvocation, - ) -> Result { - self.invocations - .lock() - .unwrap() - .extend(request.invocations.iter().cloned()); - self.inner.invoke_capability_batch(request).await - } -} - -fn local_dev_host_runtime_with_http_egress( - storage_root: PathBuf, - egress: Arc, - process_port: Option>, -) -> HarnessResult> { - let mut registry = ExtensionRegistry::new(); - registry.insert(builtin_first_party_package()?)?; - local_dev_host_runtime_with_registry_and_runtime_http_egress( - storage_root, - registry, - egress, - process_port, - ) -} - -fn host_runtime_storage_roots() -> HarnessResult<(Arc, PathBuf, PathBuf)> { - let root = Arc::new(tempfile::tempdir()?); - let storage_root = root.path().join("local-dev"); - let workspace_root = storage_root.join("workspace"); - std::fs::create_dir_all(&workspace_root)?; - Ok((root, storage_root, workspace_root)) -} - -fn local_dev_host_runtime_with_registry_and_runtime_http_egress( - storage_root: PathBuf, - registry: ExtensionRegistry, - egress: Arc, - process_port: Option>, -) -> HarnessResult> { - let mut services = HostRuntimeServices::new( - Arc::new(registry), - local_dev_root_filesystem(storage_root, LocalDevRootMounts::core_builtins())?, - Arc::new(InMemoryResourceGovernor::new()), - Arc::new(GrantAuthorizer::new()), - ironclaw_processes::ProcessServices::in_memory(), - HostRuntimeCapabilitySurfaceVersion::new("reborn-app-v1")?, - ) - .with_secret_store(Arc::new(StaticSecretStore::new( - SecretHandle::new("github_manual_access")?, - SecretMaterial::from("ghp_fake_fixture_token"), - ))) - .with_runtime_credential_account_resolver(Arc::new(FixedRuntimeCredentialAccountResolver { - result: Ok(SecretHandle::new("github_manual_access")?), - })) - .with_first_party_capabilities(Arc::new(builtin_first_party_handlers(Arc::new( - ironclaw_triggers::InMemoryTriggerRepository::default(), - ))?)) - .with_first_party_http_egress(egress) - .with_trust_policy(Arc::new(first_party_trust_policy()?)); - // Inject the recording process port when provided (slice 5). When `None`, - // `HostRuntimeServices` defaults to `LocalHostProcessPort` (real execution). - if let Some(port) = process_port { - services = services.with_runtime_process_port_dyn(port); - } - - Ok(Arc::new(services.host_runtime_for_local_testing())) -} - -fn local_dev_host_runtime_with_registry_and_egress( - storage_root: PathBuf, - registry: ExtensionRegistry, - runtime_http_egress: Arc, - network_egress: Arc, - // E-AUTHGATE: `Ok(handle)` resolves the credential account (capability - // dispatches); `Err(AuthRequired)` raises a `BlockedAuth` gate at dispatch. - credential_account_result: Result, -) -> HarnessResult> { - let services = HostRuntimeServices::new( - Arc::new(registry), - local_dev_root_filesystem(storage_root, LocalDevRootMounts::github_assets())?, - Arc::new(InMemoryResourceGovernor::new()), - Arc::new(GithubHarnessAuthorizer::new()?), - ironclaw_processes::ProcessServices::in_memory(), - HostRuntimeCapabilitySurfaceVersion::new("reborn-app-v1")?, - ) - .with_secret_store(Arc::new(StaticSecretStore::new( - SecretHandle::new("github_manual_access")?, - SecretMaterial::from("ghp_fake_fixture_token"), - ))) - .with_runtime_credential_account_resolver(Arc::new(FixedRuntimeCredentialAccountResolver { - result: credential_account_result, - })) - .with_first_party_capabilities(Arc::new(builtin_first_party_handlers(Arc::new( - ironclaw_triggers::InMemoryTriggerRepository::default(), - ))?)) - .with_runtime_http_egress(runtime_http_egress) - .with_trust_policy(Arc::new(github_first_party_trust_policy()?)) - .try_with_host_http_egress((*network_egress).clone()) - .map_err(|report| std::io::Error::other(format!("host HTTP egress failed: {report:?}")))? - .try_with_wasm_runtime(WitToolRuntimeConfig::default(), WitToolHost::deny_all()) - .map_err(|report| std::io::Error::other(format!("WASM runtime failed: {report:?}")))?; - - Ok(Arc::new(services.host_runtime_for_local_testing())) -} - -fn local_dev_host_runtime_with_live_http_egress( - storage_root: PathBuf, -) -> HarnessResult> { - let mut registry = ExtensionRegistry::new(); - registry.insert(builtin_first_party_package()?)?; - - let services = HostRuntimeServices::new( - Arc::new(registry), - local_dev_root_filesystem(storage_root, LocalDevRootMounts::core_builtins())?, - Arc::new(InMemoryResourceGovernor::new()), - Arc::new(GrantAuthorizer::new()), - ironclaw_processes::ProcessServices::in_memory(), - HostRuntimeCapabilitySurfaceVersion::new("reborn-app-v1")?, - ) - .with_secret_store(Arc::new(InMemorySecretStore::new())) - .with_first_party_capabilities(Arc::new(builtin_first_party_handlers(Arc::new( - ironclaw_triggers::InMemoryTriggerRepository::default(), - ))?)) - .try_with_host_http_egress(PolicyNetworkHttpEgress::new(ReqwestNetworkTransport::new( - Duration::from_secs(2), - ))) - .map_err(|report| { - std::io::Error::other(format!( - "live HTTP egress production wiring failed: {report:?}" - )) - })? - .with_trust_policy(Arc::new(first_party_trust_policy()?)); - - Ok(Arc::new(services.host_runtime_for_local_testing())) -} - -pub(crate) fn local_dev_root_filesystem( - storage_root: PathBuf, - mounts: LocalDevRootMounts, -) -> HarnessResult> { - let mut local = LocalFilesystem::new(); - local.mount_local( - VirtualPath::new("/projects")?, - HostPath::from_path_buf(storage_root), - )?; - if mounts.github_assets { - local.mount_local( - VirtualPath::new("/system/extensions/github")?, - HostPath::from_path_buf(github_support::asset_root()), - )?; - } - if mounts.web_access_assets { - local.mount_local( - VirtualPath::new("/system/extensions/web-access")?, - HostPath::from_path_buf(harness_web_access::asset_root()), - )?; - } - - let local = Arc::new(local); - let mut root = CompositeRootFilesystem::new(); - root.mount( - local_dev_mount_descriptor( - "/projects", - "local-dev-projects", - BackendKind::LocalFilesystem, - StorageClass::FileContent, - ContentKind::ProjectFile, - IndexPolicy::NotIndexed, - BackendCapabilities::bytes_only(), - )?, - Arc::clone(&local), - )?; - if mounts.github_assets { - root.mount( - local_dev_mount_descriptor( - "/system/extensions/github", - "local-dev-github-assets", - BackendKind::LocalFilesystem, - StorageClass::FileContent, - ContentKind::ExtensionPackage, - IndexPolicy::NotIndexed, - BackendCapabilities::bytes_only(), - )?, - Arc::clone(&local), - )?; - } - if mounts.web_access_assets { - root.mount( - local_dev_mount_descriptor( - "/system/extensions/web-access", - "local-dev-web-access-assets", - BackendKind::LocalFilesystem, - StorageClass::FileContent, - ContentKind::ExtensionPackage, - IndexPolicy::NotIndexed, - BackendCapabilities::bytes_only(), - )?, - Arc::clone(&local), - )?; - } - if mounts.memory { - let memory = Arc::new(InMemoryBackend::new()); - root.mount( - local_dev_mount_descriptor( - "/memory", - "local-dev-memory", - BackendKind::MemoryDocuments, - StorageClass::StructuredRecords, - ContentKind::MemoryDocument, - IndexPolicy::FullTextAndVector, - memory.capabilities(), - )?, - memory, - )?; - } - Ok(Arc::new(root)) -} - -#[derive(Clone, Copy)] -pub(crate) struct LocalDevRootMounts { - github_assets: bool, - web_access_assets: bool, - memory: bool, -} - -impl LocalDevRootMounts { - pub(crate) fn core_builtins() -> Self { - Self { - github_assets: false, - web_access_assets: false, - memory: true, - } - } - - fn github_assets() -> Self { - Self { - github_assets: true, - web_access_assets: false, - memory: false, - } - } - - pub(crate) fn web_access_assets() -> Self { - Self { - github_assets: false, - web_access_assets: true, - memory: false, - } - } -} - -fn local_dev_mount_descriptor( - virtual_root: &str, - backend_id: &str, - backend_kind: BackendKind, - storage_class: StorageClass, - content_kind: ContentKind, - index_policy: IndexPolicy, - capabilities: BackendCapabilities, -) -> HarnessResult { - Ok(MountDescriptor { - virtual_root: VirtualPath::new(virtual_root)?, - backend_id: BackendId::new(backend_id)?, - backend_kind, - storage_class, - content_kind, - index_policy, - capabilities, - }) -} - -fn first_party_trust_policy() -> HarnessResult { - Ok(HostTrustPolicy::new(vec![Box::new( - AdminConfig::with_entries(vec![AdminEntry::for_local_manifest( - PackageId::new(BUILTIN_FIRST_PARTY_PROVIDER)?, - "/system/extensions/builtin/manifest.toml".to_string(), - None, - HostTrustAssignment::first_party(), - vec![ - EffectKind::DispatchCapability, - EffectKind::ReadFilesystem, - EffectKind::WriteFilesystem, - EffectKind::DeleteFilesystem, - EffectKind::Network, - EffectKind::SpawnProcess, - EffectKind::ExecuteCode, - EffectKind::ExternalWrite, - ], - None, - )]), - )])?) -} - -fn github_first_party_trust_policy() -> HarnessResult { - Ok(HostTrustPolicy::new(vec![Box::new( - AdminConfig::with_entries(vec![AdminEntry::for_local_manifest( - PackageId::new("github")?, - "/system/extensions/github/manifest.toml".to_string(), - None, - HostTrustAssignment::first_party(), - vec![ - EffectKind::DispatchCapability, - EffectKind::Network, - EffectKind::UseSecret, - EffectKind::ExternalWrite, - ], - None, - )]), - )])?) -} - -fn http_test_policy() -> NetworkPolicy { - NetworkPolicy { - allowed_targets: vec![NetworkTargetPattern { - scheme: Some(NetworkScheme::Https), - host_pattern: "api.example.test".to_string(), - port: None, - }], - deny_private_ip_ranges: true, - max_egress_bytes: Some(10_000), - } -} - -fn wildcard_test_policy() -> NetworkPolicy { - NetworkPolicy { - allowed_targets: vec![NetworkTargetPattern { - scheme: None, - host_pattern: "*".to_string(), - port: None, - }], - deny_private_ip_ranges: true, - max_egress_bytes: Some(1_000_000), - } -} - -/// C-JOURNEY: recursively copy `src` into `dst` (creating `dst` and any -/// intermediate directories). Used to populate a harness's per-test -/// `/system/extensions/` mount with the real bundled-extension asset -/// directory (manifest + wasm module + schemas) so a WASM capability -/// published via `publish_bundled_extension_for_test` is genuinely loadable, -/// not just registered as metadata. -fn copy_dir_recursive(src: &Path, dst: &Path) -> HarnessResult<()> { - std::fs::create_dir_all(dst)?; - for entry in std::fs::read_dir(src)? { - let entry = entry?; - let file_type = entry.file_type()?; - let dst_path = dst.join(entry.file_name()); - if file_type.is_dir() { - copy_dir_recursive(&entry.path(), &dst_path)?; - } else { - std::fs::copy(entry.path(), &dst_path)?; - } - } - Ok(()) -} - -fn capability_ids_from_strs(ids: &[&str]) -> HarnessResult> { - ids.iter() - .map(|id| CapabilityId::new(*id).map_err(Into::into)) - .collect() -} - -fn bundled_extension_provider_trust() -> HarnessResult)>> { - BUNDLED_EXTENSION_IDS - .iter() - .map(|id| Ok((ExtensionId::new(*id)?, local_dev_all_effects()))) - .collect() -} - -fn local_dev_all_effects() -> Vec { - vec![ - EffectKind::DispatchCapability, - EffectKind::ReadFilesystem, - EffectKind::WriteFilesystem, - EffectKind::DeleteFilesystem, - EffectKind::Network, - EffectKind::UseSecret, - EffectKind::SpawnProcess, - EffectKind::ExecuteCode, - EffectKind::ExternalWrite, - ] -} - -#[derive(Debug)] -struct FixedRuntimeCredentialAccountResolver { - result: Result, -} - -#[async_trait] -impl RuntimeCredentialAccountResolver for FixedRuntimeCredentialAccountResolver { - async fn resolve_access_secret( - &self, - request: RuntimeCredentialAccountRequest<'_>, - ) -> Result { - assert_eq!(request.provider.as_str(), "github"); - assert_eq!(request.requester_extension.as_str(), "github"); - self.result - .clone() - .map(|handle| RuntimeCredentialAccessSecret { - scope: request.scope.clone(), - handle, - }) - } -} - -struct StaticSecretStore { - handle: SecretHandle, - material: SecretMaterial, -} - -impl StaticSecretStore { - fn new(handle: SecretHandle, material: SecretMaterial) -> Self { - Self { handle, material } - } -} - -#[async_trait] -impl SecretStore for StaticSecretStore { - async fn put( - &self, - scope: ResourceScope, - handle: SecretHandle, - _material: SecretMaterial, - _expires_at: Option, - ) -> Result { - Ok(SecretMetadata { - scope, - handle, - expires_at: None, - }) - } - - async fn metadata( - &self, - scope: &ResourceScope, - handle: &SecretHandle, - ) -> Result, SecretStoreError> { - Ok((handle == &self.handle).then(|| SecretMetadata { - scope: scope.clone(), - handle: handle.clone(), - expires_at: None, - })) - } - - async fn metadata_for_scope( - &self, - scope: &ResourceScope, - ) -> Result, SecretStoreError> { - Ok(vec![SecretMetadata { - scope: scope.clone(), - handle: self.handle.clone(), - expires_at: None, - }]) - } - - async fn delete( - &self, - _scope: &ResourceScope, - _handle: &SecretHandle, - ) -> Result { - Ok(false) - } - - async fn lease_once( - &self, - scope: &ResourceScope, - handle: &SecretHandle, - ) -> Result { - if handle != &self.handle { - return Err(SecretStoreError::UnknownSecret { - scope: Box::new(scope.clone()), - handle: handle.clone(), - }); - } - Ok(SecretLease { - id: SecretLeaseId::new(), - scope: scope.clone(), - handle: handle.clone(), - status: SecretLeaseStatus::Active, - }) - } - - async fn consume( - &self, - _scope: &ResourceScope, - _lease_id: SecretLeaseId, - ) -> Result { - Ok(self.material.clone()) - } - - async fn revoke( - &self, - scope: &ResourceScope, - lease_id: SecretLeaseId, - ) -> Result { - Ok(SecretLease { - id: lease_id, - scope: scope.clone(), - handle: self.handle.clone(), - status: SecretLeaseStatus::Revoked, - }) - } - - async fn leases_for_scope( - &self, - _scope: &ResourceScope, - ) -> Result, SecretStoreError> { - Ok(Vec::new()) - } -} - -struct GithubHarnessAuthorizer { - obligations: Obligations, -} - -impl GithubHarnessAuthorizer { - fn new() -> HarnessResult { - Ok(Self { - obligations: Obligations::new(vec![ - Obligation::ApplyNetworkPolicy { - policy: github_support::api_policy(), - }, - Obligation::InjectCredentialAccountOnce { - handle: SecretHandle::new("github_runtime_token")?, - provider: RuntimeCredentialAccountProviderId::new("github")?, - setup: ironclaw_host_api::RuntimeCredentialAccountSetup::ManualToken, - provider_scopes: Vec::new(), - requester_extension: ExtensionId::new("github")?, - }, - ])?, - }) - } -} - -#[async_trait] -impl TrustAwareCapabilityDispatchAuthorizer for GithubHarnessAuthorizer { - async fn authorize_dispatch_with_trust( - &self, - _context: &ExecutionContext, - _descriptor: &CapabilityDescriptor, - _estimate: &ResourceEstimate, - _trust_decision: &TrustDecision, - ) -> Decision { - Decision::Allow { - obligations: self.obligations.clone(), - } - } - - async fn authorize_spawn_with_trust( - &self, - _context: &ExecutionContext, - _descriptor: &CapabilityDescriptor, - _estimate: &ResourceEstimate, - _trust_decision: &TrustDecision, - ) -> Decision { - Decision::Allow { - obligations: self.obligations.clone(), - } - } -} - -#[derive(Debug, Clone)] -pub(crate) struct RecordingRuntimeHttpEgress { - default_body: Vec, - /// URL/method/capability-keyed scripted responses (§3.6 P1 ergonomics). - /// Consulted before the FIFO queue; first match wins. - scripted: Arc>>, - response_bodies: Arc>>>, - requests: Arc>>, -} - -impl RecordingRuntimeHttpEgress { - fn with_body(body: Vec) -> Self { - Self { - default_body: body, - scripted: Arc::new(Mutex::new(Vec::new())), - response_bodies: Arc::new(Mutex::new(VecDeque::new())), - requests: Arc::new(Mutex::new(Vec::new())), - } - } - - fn requests(&self) -> Vec { - self.requests.lock().unwrap().clone() - } - - /// Append keyed scripted responses (the canonical keyed-matcher install). - fn install_scripted( - &self, - responses: impl IntoIterator, - ) { - self.scripted.lock().unwrap().extend(responses); - } - - /// Enqueue one FIFO response body (C-WEBACCESS), consumed in call order - /// ahead of `default_body`. Mirrors `install_scripted`'s shape but for the - /// plain FIFO queue rather than keyed matchers — used to script the - /// three-leg Exa MCP handshake (`initialize` → `notifications/initialized` - /// → `tools/call`), which all target the same URL/method/capability and so - /// cannot be told apart by the keyed matcher. - pub(crate) fn push_response_body(&self, body: Vec) { - self.response_bodies.lock().unwrap().push_back(body); - } -} - -#[async_trait::async_trait] -impl RuntimeHttpEgress for RecordingRuntimeHttpEgress { - async fn execute( - &self, - request: RuntimeHttpEgressRequest, - ) -> Result { - let request_bytes = request.body.len() as u64; - // Resolve the keyed outcome BEFORE recording the request: `push(request)` - // moves `request` by value into the log, so any code reading its fields - // (the `.matches()` lookup) must run first. (`RuntimeHttpEgressRequest` - // does implement `Drop`/`ZeroizeOnDrop` to scrub its URL/headers, but that - // fires later when the logged entry is actually dropped, not on push.) - let keyed_outcome = { - let scripted = self.scripted.lock().unwrap(); - scripted - .iter() - .find(|response| response.matches(&request)) - .map(|response| response.outcome()) - }; - self.requests.lock().unwrap().push(request); - // A scripted egress error short-circuits with `Err`, driving the tool's - // error mapping. A body outcome (or the FIFO/default fallback) returns - // `Ok` with the scripted status/body. - let (status, body) = match keyed_outcome { - Some(super::http_matcher::ScriptedHttpOutcome::Error(error)) => return Err(error), - Some(super::http_matcher::ScriptedHttpOutcome::Body { status, bytes }) => { - (status, bytes) - } - None => ( - 200, - self.response_bodies - .lock() - .unwrap() - .pop_front() - .unwrap_or_else(|| self.default_body.clone()), - ), - }; - Ok(RuntimeHttpEgressResponse { - status, - headers: vec![("content-type".to_string(), "application/json".to_string())], - body: body.clone(), - saved_body: None, - request_bytes, - response_bytes: body.len() as u64, - redaction_applied: false, - }) - } -} - -#[async_trait] -impl ironclaw_host_runtime::ToolCallHttpEgress for RecordingRuntimeHttpEgress { - async fn execute_for_model_visible_output( - &self, - request: RuntimeHttpEgressRequest, - ) -> Result { - RuntimeHttpEgress::execute(self, request).await - } -} - -#[derive(Debug, Clone)] -struct RecordingNetworkHttpEgress { - default_body: Vec, - response_bodies: Arc>>>, - /// W4-AUTHGATE-WIRE: FIFO of scripted non-default statuses, consumed ahead - /// of the hardcoded `200` default. Lets a test drive the runtime-401 - /// (credential-injected-but-rejected) path for capabilities whose real - /// HTTP call flows through this **network** lane rather than the runtime - /// egress `ScriptedHttpResponse` matcher (`GithubIssueTools` — see - /// `reborn_integration_secret_injection.rs`'s module doc: `try_with_host_http_egress` - /// overwrites the runtime port with the host pipeline over THIS recorder). - /// Empty by default — every pre-existing caller keeps the old hardcoded-200 - /// behavior byte-identical. - status_queue: Arc>>, - requests: Arc>>, -} - -impl RecordingNetworkHttpEgress { - fn with_body(body: Vec) -> Self { - Self { - default_body: body, - response_bodies: Arc::new(Mutex::new(VecDeque::new())), - status_queue: Arc::new(Mutex::new(VecDeque::new())), - requests: Arc::new(Mutex::new(Vec::new())), - } - } - - fn requests(&self) -> Vec { - self.requests.lock().unwrap().clone() - } - - /// Enqueue one FIFO scripted status, consumed by the next `execute` call - /// ahead of the hardcoded `200` default. - fn push_status(&self, status: u16) { - self.status_queue.lock().unwrap().push_back(status); - } -} - -#[async_trait::async_trait] -impl NetworkHttpEgress for RecordingNetworkHttpEgress { - async fn execute( - &self, - request: NetworkHttpRequest, - ) -> Result { - let request_bytes = request.body.len() as u64; - self.requests.lock().unwrap().push(request); - let body = self - .response_bodies - .lock() - .unwrap() - .pop_front() - .unwrap_or_else(|| self.default_body.clone()); - let status = self.status_queue.lock().unwrap().pop_front().unwrap_or(200); - Ok(NetworkHttpResponse { - status, - headers: vec![("content-type".to_string(), "application/json".to_string())], - body: body.clone(), - usage: NetworkUsage { - request_bytes, - response_bytes: body.len() as u64, - resolved_ip: None, - }, - }) - } -} - -struct RecordingCapabilityResultWriter { - inner: Arc, - results: Arc>>, -} - -#[async_trait] -impl LoopCapabilityResultWriter for RecordingCapabilityResultWriter { - async fn write_capability_result( - &self, - write: CapabilityResultWrite<'_>, - ) -> Result { - let capability_id = write.capability_id.clone(); - let output = write.output.clone(); - let write_result = self.inner.write_capability_result(write).await?; - self.results.lock().unwrap().push(RecordedCapabilityResult { - capability_id, - output, - }); - Ok(write_result) - } - - async fn update_capability_result( - &self, - run_context: &LoopRunContext, - result_ref: &LoopResultRef, - output: serde_json::Value, - ) -> Result { - let byte_len = self - .inner - .update_capability_result(run_context, result_ref, output.clone()) - .await?; - self.results.lock().unwrap().push(RecordedCapabilityResult { - capability_id: CapabilityId::new( - ironclaw_loop_support::DEFAULT_SPAWN_SUBAGENT_CAPABILITY_ID, - ) - .map_err(|error| { - AgentLoopHostError::new(AgentLoopHostErrorKind::Internal, error.to_string()) - })?, - output, - }); - Ok(byte_len) - } -} - -fn workspace_mounts(permissions: MountPermissions) -> HarnessResult { - Ok(MountView::new(vec![MountGrant::new( - MountAlias::new("/workspace")?, - VirtualPath::new("/projects/workspace")?, - permissions, - )])?) -} - -fn memory_mounts(permissions: MountPermissions) -> HarnessResult { - Ok(MountView::new(vec![MountGrant::new( - MountAlias::new("/memory")?, - VirtualPath::new("/memory")?, - permissions, - )])?) -} - -fn skill_mounts() -> HarnessResult { - Ok(MountView::new(vec![ - MountGrant::new( - MountAlias::new("/skills")?, - VirtualPath::new("/projects/skills")?, - MountPermissions::read_write_list_delete(), - ), - MountGrant::new( - MountAlias::new("/system/skills")?, - VirtualPath::new("/projects/system/skills")?, - MountPermissions::read_only(), - ), - ])?) -} - -fn qa_smoke_mounts() -> HarnessResult { - Ok(MountView::new(vec![ - MountGrant::new( - MountAlias::new("/workspace")?, - VirtualPath::new("/projects/workspace")?, - MountPermissions::read_write_list_delete(), - ), - MountGrant::new( - MountAlias::new("/skills")?, - VirtualPath::new("/projects/skills")?, - MountPermissions::read_write_list_delete(), - ), - MountGrant::new( - MountAlias::new("/system/skills")?, - VirtualPath::new("/projects/system/skills")?, - MountPermissions::read_only(), - ), - ])?) -} - -fn capability_grants( - grantee: Principal, - capabilities: &[CapabilityId], - allowed_effects: Vec, - mounts: MountView, - mount_overrides: &[(CapabilityId, MountView)], - network: NetworkPolicy, - secrets: Vec, -) -> CapabilitySet { - CapabilitySet { - grants: capabilities - .iter() - .map(|capability| { - let mounts = mount_overrides - .iter() - .find(|(override_capability, _mounts)| override_capability == capability) - .map(|(_capability, mounts)| mounts.clone()) - .unwrap_or_else(|| mounts.clone()); - CapabilityGrant { - id: CapabilityGrantId::new(), - capability: capability.clone(), - grantee: grantee.clone(), - issued_by: Principal::HostRuntime, - constraints: GrantConstraints { - allowed_effects: allowed_effects.clone(), - mounts, - network: network.clone(), - secrets: secrets.clone(), - resource_ceiling: None, - expires_at: None, - max_invocations: None, - }, - } - }) - .collect(), - } -} - -fn host_runtime_harness_error(error: impl std::fmt::Display) -> AgentLoopHostError { - AgentLoopHostError::new(AgentLoopHostErrorKind::InvalidInvocation, error.to_string()) -} - -#[derive(Clone)] -pub struct RecordingTestCapabilityPort { - mode: CapabilityMode, - expose_spawn_subagent: bool, - use_subagent_allowed_tool: bool, - invocations: Arc>>, - next_result: Arc, - approval_calls: Arc, -} - -#[derive(Debug, Clone, Copy)] -enum CapabilityMode { - Echo, - ApprovalThenEcho, - SpawnAuthThenApprovalThenEcho, -} - -impl RecordingTestCapabilityPort { - pub fn echo() -> Self { - Self::new(CapabilityMode::Echo, false, false) - } - - pub fn echo_with_spawn_subagent() -> Self { - Self::new(CapabilityMode::Echo, true, false) - } - - pub fn approval_then_echo() -> Self { - Self::new(CapabilityMode::ApprovalThenEcho, false, false) - } - - pub fn approval_then_echo_with_spawn_subagent() -> Self { - Self::new(CapabilityMode::ApprovalThenEcho, true, false) - } - - pub fn approval_then_allowed_tool_with_spawn_subagent() -> Self { - Self::new(CapabilityMode::ApprovalThenEcho, true, true) - } - - pub fn spawn_auth_then_approval_then_echo_with_spawn_subagent() -> Self { - Self::new(CapabilityMode::SpawnAuthThenApprovalThenEcho, true, false) - } - - pub fn spawn_auth_then_approval_then_allowed_tool_with_spawn_subagent() -> Self { - Self::new(CapabilityMode::SpawnAuthThenApprovalThenEcho, true, true) - } - - fn new( - mode: CapabilityMode, - expose_spawn_subagent: bool, - use_subagent_allowed_tool: bool, - ) -> Self { - Self { - mode, - expose_spawn_subagent, - use_subagent_allowed_tool, - invocations: Arc::new(Mutex::new(Vec::new())), - next_result: Arc::new(AtomicUsize::new(1)), - approval_calls: Arc::new(AtomicUsize::new(0)), - } - } - - fn primary_capability_id(&self) -> CapabilityId { - let id = if self.use_subagent_allowed_tool { - READ_FILE_CAPABILITY_ID - } else { - TEST_CAPABILITY_ID - }; - CapabilityId::new(id).expect("valid capability id") - } - - fn primary_tool_name(&self) -> &'static str { - if self.use_subagent_allowed_tool { - SUBAGENT_ALLOWED_TEST_TOOL_NAME - } else { - "test_echo" - } - } - - fn invocations(&self) -> Vec { - self.invocations.lock().unwrap().clone() - } - - pub fn invocation_count(&self) -> usize { - self.invocations.lock().unwrap().len() - } - - fn capability_allowlist(&self) -> Vec { - let mut allowlist = vec![self.primary_capability_id()]; - if self.expose_spawn_subagent { - allowlist.push( - CapabilityId::new(DEFAULT_SPAWN_SUBAGENT_CAPABILITY_ID) - .expect("valid capability id"), - ); - } - allowlist - } - - fn completed_result(&self) -> CapabilityOutcome { - let ordinal = self.next_result.fetch_add(1, Ordering::SeqCst); - CapabilityOutcome::Completed(CapabilityResultMessage { - result_ref: ironclaw_turns::LoopResultRef::new(format!("result:test-echo-{ordinal}")) - .expect("valid result ref"), - safe_summary: "echo: hi".to_string(), - progress: ironclaw_turns::run_profile::CapabilityProgress::MadeProgress, - terminate_hint: false, - byte_len: 0, - output_digest: None, - }) - } -} - -#[async_trait] -impl LoopCapabilityPort for RecordingTestCapabilityPort { - fn tool_definitions(&self) -> Result, AgentLoopHostError> { - let definitions = vec![ProviderToolDefinition { - capability_id: self.primary_capability_id(), - name: ProviderToolName::new(self.primary_tool_name()).expect("provider tool name"), - description: "Echo a test payload".to_string(), - parameters: json!({ - "type": "object", - "properties": { - "message": {"type": "string"} - } - }), - }]; - Ok(definitions) - } - - async fn register_provider_tool_call( - &self, - request: ironclaw_turns::run_profile::RegisterProviderToolCallRequest, - ) -> Result { - let call = request.tool_call; - let capability_id = self.primary_capability_id(); - Ok(CapabilityCallCandidate { - activity_id: ironclaw_turns::CapabilityActivityId::new(), - surface_version: CapabilitySurfaceVersion::new(TEST_CAPABILITY_SURFACE_VERSION) - .expect("valid surface version"), - capability_id: capability_id.clone(), - effective_capability_ids: vec![capability_id], - input_ref: CapabilityInputRef::new(format!("input:{}", call.id)) - .expect("valid input ref"), - provider_replay: Some(ProviderToolCallReplay { - provider_id: call.provider_id, - provider_model_id: call.provider_model_id, - provider_turn_id: call.turn_id.unwrap_or_else(|| "trace-turn".to_string()), - provider_call_id: call.id, - provider_tool_name: call.name, - arguments: call.arguments, - response_reasoning: call.response_reasoning, - reasoning: call.reasoning, - signature: call.signature, - }), - }) - } - - async fn visible_capabilities( - &self, - _request: VisibleCapabilityRequest, - ) -> Result { - let descriptors = vec![CapabilityDescriptorView { - capability_id: self.primary_capability_id(), - provider: Some(ExtensionId::new("test").expect("valid provider")), - runtime: RuntimeKind::FirstParty, - safe_name: self.primary_tool_name().to_string(), - safe_description: "Echo a test payload".to_string(), - concurrency_hint: ConcurrencyHint::SafeForParallel, - parameters_schema: json!({"type": "object"}), - }]; - Ok(VisibleCapabilitySurface { - version: CapabilitySurfaceVersion::new(TEST_CAPABILITY_SURFACE_VERSION) - .expect("valid surface version"), - descriptors, - callable_capability_ids: None, - }) - } - - async fn invoke_capability( - &self, - request: CapabilityInvocation, - ) -> Result { - self.invocations.lock().unwrap().push(request); - if matches!(self.mode, CapabilityMode::ApprovalThenEcho) - && self.approval_calls.fetch_add(1, Ordering::SeqCst) == 0 - { - return Ok(CapabilityOutcome::ApprovalRequired { - gate_ref: LoopGateRef::new("gate:test-approval").expect("valid gate ref"), - safe_summary: "test approval required".to_string(), - approval_resume: None, - }); - } - if matches!(self.mode, CapabilityMode::SpawnAuthThenApprovalThenEcho) { - match self.approval_calls.fetch_add(1, Ordering::SeqCst) { - 0 => return Ok(self.completed_result()), - 1 => { - return Ok(CapabilityOutcome::ApprovalRequired { - gate_ref: LoopGateRef::new("gate:test-approval").expect("valid gate ref"), - safe_summary: "test approval required".to_string(), - approval_resume: None, - }); - } - _ => {} - } - } - Ok(self.completed_result()) - } - - async fn invoke_capability_batch( - &self, - request: CapabilityBatchInvocation, - ) -> Result { - let stop_on_first_suspension = request.stop_on_first_suspension; - let mut outcomes = Vec::new(); - let mut stopped_on_suspension = false; - for invocation in request.invocations { - let outcome = self.invoke_capability(invocation).await?; - let is_suspension = outcome.is_suspension(); - outcomes.push(outcome); - if is_suspension && stop_on_first_suspension { - stopped_on_suspension = true; - break; - } - } - Ok(CapabilityBatchOutcome { - outcomes, - stopped_on_suspension, - }) - } -} - -pub(crate) struct HarnessCapabilityPortFactory { - pub(crate) port: Arc, -} - -#[async_trait] -impl LoopCapabilityPortFactory for HarnessCapabilityPortFactory { - async fn create_capability_port( - &self, - _run_context: &LoopRunContext, - ) -> Result, AgentLoopHostError> { - Ok(self.port.clone()) - } -} - -pub(crate) struct StaticCapabilitySurfaceProfileResolver { - pub(crate) allow_set: CapabilityAllowSet, -} - -#[async_trait] -impl CapabilitySurfaceProfileResolver for StaticCapabilitySurfaceProfileResolver { - async fn resolve( - &self, - _run_context: &LoopRunContext, - ) -> Result { - Ok(self.allow_set.clone()) - } -} - -pub(crate) struct EmptyIdentityContextSource; - -#[async_trait] -impl HostIdentityContextSource for EmptyIdentityContextSource { - async fn load_identity_candidates( - &self, - _run_context: &LoopRunContext, - _mode: PromptMode, - ) -> Result, HostIdentityContextBuildError> { - Ok(Vec::new()) - } -} - -fn product_scope() -> ResourceScope { - test_product_scope("tenant-e2e", "host-user", "agent-e2e", Some("project-e2e")) -} - -pub fn test_product_scope( - tenant_id: &str, - host_user_id: &str, - agent_id: &str, - project_id: Option<&str>, -) -> ResourceScope { - resource_scope( - TenantId::new(tenant_id).expect("valid tenant"), - UserId::new(host_user_id).expect("valid user"), - AgentId::new(agent_id).expect("valid agent"), - project_id.map(|id| ProjectId::new(id).expect("valid project")), - ) -} - -fn binding_request( - ingress: &RebornTestIngress, - conversation_id: &str, -) -> HarnessResult { - binding_request_with_trigger(ingress, conversation_id, ProductTriggerReason::DirectChat) -} - -fn binding_request_with_trigger( - ingress: &RebornTestIngress, - conversation_id: &str, - trigger: ProductTriggerReason, -) -> HarnessResult { - binding_request_with_trigger_and_actor(ingress, conversation_id, "alice", trigger) -} - -fn binding_request_with_trigger_and_actor( - ingress: &RebornTestIngress, - conversation_id: &str, - actor_id: &str, - trigger: ProductTriggerReason, -) -> HarnessResult { - let envelope = ingress.verified_text_envelope_with_trigger( - "binding-probe", - actor_id, - conversation_id, - "hi", - trigger, - )?; - Ok(binding_request_from_envelope(&envelope)) -} - -fn binding_request_from_envelope(envelope: &ProductInboundEnvelope) -> ResolveBindingRequest { - ResolveBindingRequest { - adapter_id: envelope.adapter_id().clone(), - installation_id: envelope.installation_id().clone(), - external_actor_ref: envelope.external_actor_ref().clone(), - external_conversation_ref: envelope.external_conversation_ref().clone(), - external_event_id: envelope.external_event_id().clone(), - route_kind: route_kind_for_envelope(envelope), - auth_claim: envelope.auth_claim().clone(), - } -} - -fn thread_scope_from_binding(binding: &ResolvedBinding) -> HarnessResult { - thread_scope_from_binding_with_route_kind(binding, ProductConversationRouteKind::Direct) -} - -fn thread_scope_from_binding_with_route_kind( - binding: &ResolvedBinding, - _route_kind: ProductConversationRouteKind, -) -> HarnessResult { - Ok(ThreadScope { - tenant_id: binding.tenant_id.clone(), - agent_id: binding - .agent_id - .clone() - .ok_or("resolved binding missing agent id")?, - project_id: binding.project_id.clone(), - owner_user_id: binding.subject_user_id.clone(), - mission_id: None, - }) -} - -fn route_kind_for_envelope(envelope: &ProductInboundEnvelope) -> ProductConversationRouteKind { - match envelope.payload() { - ProductInboundPayload::UserMessage(message) => route_kind_for_trigger(message.trigger), - ProductInboundPayload::Command(command) => route_kind_for_trigger(command.trigger), - _ => ProductConversationRouteKind::Direct, - } -} - -fn route_kind_for_trigger(trigger: ProductTriggerReason) -> ProductConversationRouteKind { - match trigger { - ProductTriggerReason::DirectChat => ProductConversationRouteKind::Direct, - ProductTriggerReason::BotMention - | ProductTriggerReason::ReplyToBot - | ProductTriggerReason::BotCommand - | ProductTriggerReason::LinkedThreadAction => ProductConversationRouteKind::Shared, - } -} - -pub(crate) fn scoped_turns_fs( - backend: Arc, - binding: &ResolvedBinding, -) -> HarnessResult>> { - // Include agent_id and project_id in the path when present so that - // distinct agents or projects stored under the same tenant/user - // (e.g. shared-storage multi-harness tests) get isolated turn state - // files and cannot cross-claim each other's queued runs. - // The 4-arm match lives in `super::filesystem::turns_scope_path`; the - // integration tier reuses it with a different prefix via - // `scoped_turns_fs_composite` in builder.rs. - let target = super::filesystem::turns_scope_path("/engine", binding); - let mounts = MountView::new(vec![MountGrant::new( - MountAlias::new("/turns").expect("valid turns alias"), - VirtualPath::new(target).expect("valid turns target"), - MountPermissions::read_write_list_delete(), - )])?; - Ok(Arc::new(ScopedFilesystem::with_fixed_view( - turn_state_root_filesystem(backend)?, - mounts, - ))) -} - -fn turn_state_root_filesystem( - backend: Arc, -) -> HarnessResult> { - let mut root = CompositeRootFilesystem::new(); - root.mount( - local_dev_mount_descriptor( - "/engine", - "reborn-harness-turn-state", - BackendKind::MemoryDocuments, - StorageClass::StructuredRecords, - ContentKind::StructuredRecord, - IndexPolicy::NotIndexed, - backend.capabilities(), - )?, - backend, - )?; - Ok(Arc::new(root)) -} - -pub fn trace_tool_call_response() -> ironclaw_loop_support::HostManagedModelResponse { - ironclaw_loop_support::HostManagedModelResponse { - safe_text_deltas: Vec::new(), - safe_reasoning_deltas: Vec::new(), - usage: None, - output: ParentLoopOutput::CapabilityCalls(vec![CapabilityCallCandidate { - activity_id: ironclaw_turns::CapabilityActivityId::new(), - surface_version: CapabilitySurfaceVersion::new(TEST_CAPABILITY_SURFACE_VERSION) - .expect("valid surface version"), - capability_id: CapabilityId::new(TEST_CAPABILITY_ID).expect("valid capability id"), - effective_capability_ids: vec![ - CapabilityId::new(TEST_CAPABILITY_ID).expect("valid capability id"), - ], - input_ref: CapabilityInputRef::new("input:trace-call-1").expect("valid input ref"), - provider_replay: Some(ProviderToolCallReplay { - provider_id: "trace_replay".to_string(), - provider_model_id: "trace_replay".to_string(), - provider_turn_id: "trace-turn".to_string(), - provider_call_id: "call-1".to_string(), - provider_tool_name: ProviderToolName::new("test_echo").expect("provider tool name"), - arguments: json!({"message": "hi"}), - response_reasoning: None, - reasoning: None, - signature: None, - }), - }]), - } -} - -pub fn assert_milestone_order( - milestones: &[LoopHostMilestone], - before: impl Fn(&LoopHostMilestoneKind) -> bool, - after: impl Fn(&LoopHostMilestoneKind) -> bool, -) { - let before_index = milestones - .iter() - .position(|milestone| before(&milestone.kind)) - .expect("before milestone should be present"); - let after_index = milestones - .iter() - .position(|milestone| after(&milestone.kind)) - .expect("after milestone should be present"); - assert!( - before_index < after_index, - "expected milestone order, got {:?}", - milestones - .iter() - .map(|milestone| milestone.kind.kind_name()) - .collect::>() - ); -} diff --git a/tests/support/reborn/reply.rs b/tests/support/reborn/reply.rs deleted file mode 100644 index c31cf90e3fa..00000000000 --- a/tests/support/reborn/reply.rs +++ /dev/null @@ -1,119 +0,0 @@ -//! `RebornScriptedReply` — the terse façade for scripting one model turn in a -//! Reborn integration test. Each reply maps 1:1 to a `TraceStep`, auto-filling -//! id/tokens/request_hint/expected_tool_results so a test body needs exactly one -//! line per model turn. Raw `TraceStep`/`LlmTrace`/`TraceResponse` construction -//! is forbidden in new Reborn integration tests (design §4.2) — use this. - -// Shared integration-test support: `support_unit_tests.rs` mounts the -// `reborn_support` tree without consuming this module, so its symbols read as -// dead there under `-D warnings`. Module-level allow matches the sibling -// support modules (`assertions.rs`, `test_channel.rs`). -#![allow(dead_code)] - -use crate::support::trace_llm::{TraceResponse, TraceStep, TraceToolCall}; -use std::sync::atomic::{AtomicU64, Ordering}; - -static NEXT_TOOL_CALL_ID: AtomicU64 = AtomicU64::new(1); - -/// One scripted model turn. -pub struct RebornScriptedReply { - step: TraceStep, -} - -impl RebornScriptedReply { - /// A plain assistant text reply. - pub fn text(content: impl Into) -> Self { - Self { - step: TraceStep { - request_hint: None, - response: TraceResponse::Text { - content: content.into(), - input_tokens: 0, - output_tokens: 0, - }, - expected_tool_results: Vec::new(), - }, - } - } - - /// Scripts one model tool-call turn. Accepts a CapabilityId (e.g. `"builtin.http"`). - /// - /// **Why the encoding lives here:** `TraceToolCall.name` flows through `TraceLlm` into - /// `LlmProviderModelGateway::provider_tool_call_from_llm`, which calls - /// `ProviderToolName::new(tool_call.name)` with no intermediate conversion. - /// `ProviderToolName` rejects dots, so the `'.' → "__"` encoding must be applied before - /// storing into `TraceToolCall`. This is distinct from the `RebornTraceReplayModelGateway` - /// JSON-fixture-replay path, which has its own identical encoding in `trace_provider_tool_name` - /// (`model_replay.rs`) at that seam. The two encoders serve different paths and are not - /// redundant; if the mapping ever needs to change (e.g. collision-safety or truncation), - /// update both sites together. - /// - /// **Collision caveat:** this mapping is NOT collision-safe — two distinct capability IDs - /// that differ only by `.` vs `__` would produce the same `ProviderToolName`, and long - /// names are not truncated to `ProviderToolName::MAX_BYTES`. It is valid for the - /// single-capability tests in the current slice. Any future slice that scripts colliding - /// or long capability IDs must instead resolve the name against the advertised - /// `ProviderToolName` from the tool list rather than applying this heuristic mapping. - /// - /// The provisional tool-call `id` is auto-filled from a process-scoped - /// counter. `scripted_provider::scripted_trace_llm` canonicalizes those - /// ids per trace before the model sees them, so assertions can rely on the - /// materialized `call-1`, `call-2`, … order. - pub fn tool_call(capability_id: &str, arguments: serde_json::Value) -> Self { - let name = capability_id.replace('.', "__"); - let id = format!("call-{}", NEXT_TOOL_CALL_ID.fetch_add(1, Ordering::Relaxed)); - Self { - step: TraceStep { - request_hint: None, - response: TraceResponse::ToolCalls { - tool_calls: vec![TraceToolCall { - id, - name, - arguments, - }], - input_tokens: 0, - output_tokens: 0, - }, - expected_tool_results: Vec::new(), - }, - } - } - - /// Scripts one model turn carrying MULTIPLE tool calls in a single - /// assistant response (a "parallel" tool-calls turn — multiple - /// `tool_calls[]` entries from ONE model call, as opposed to separate - /// sequential turns). Each `(capability_id, arguments)` pair gets its own - /// `'.' → "__"` provider-seam encoding and its own provisional id (same - /// counter `tool_call` uses, canonicalized per trace by - /// `scripted_trace_llm`). Still counts as exactly ONE script - /// entry per the harness's "one entry per model call" discipline — the - /// caller must follow it with exactly one more entry (the post-execution - /// model call reacting to however many tool results come back). - pub fn tool_calls<'a>(calls: impl IntoIterator) -> Self { - let tool_calls = calls - .into_iter() - .map(|(capability_id, arguments)| TraceToolCall { - id: format!("call-{}", NEXT_TOOL_CALL_ID.fetch_add(1, Ordering::Relaxed)), - name: capability_id.replace('.', "__"), - arguments, - }) - .collect(); - Self { - step: TraceStep { - request_hint: None, - response: TraceResponse::ToolCalls { - tool_calls, - input_tokens: 0, - output_tokens: 0, - }, - expected_tool_results: Vec::new(), - }, - } - } - - /// Consume into the underlying replay step (crate-internal seam used by - /// `scripted_provider::scripted_trace_llm`). - pub(crate) fn into_step(self) -> TraceStep { - self.step - } -} diff --git a/tests/support/reborn_parity_qa/CLAUDE.md b/tests/support/reborn_parity_qa/CLAUDE.md new file mode 100644 index 00000000000..4597a67ea29 --- /dev/null +++ b/tests/support/reborn_parity_qa/CLAUDE.md @@ -0,0 +1,56 @@ +# Reborn Parity/QA Test Support + +Support tree for the **parity and QA suites** (`tests/reborn_*_parity.rs`, +`tests/reborn_qa_*.rs`, `tests/reborn_*_e2e.rs`). These suites are current and +maintained, but they are **not coverage-bearing** — the coverage program runs +exclusively over `tests/integration/` (see `tests/integration/CLAUDE.md`). + +## Tier + +`RebornBinaryE2EHarness` (in `binary_e2e.rs`) swaps the whole +`HostManagedModelGateway` with `RebornTraceReplayModelGateway` +(`model_replay.rs`) at the *gateway* seam, skipping `ironclaw_llm`. The +integration tier in `tests/integration/` mocks one layer lower (the vendor-SDK +seam) so the real decorator chain runs; prefer that tier for new +coverage-bearing scenarios. + +## Files + +- `binary_e2e.rs` — `RebornBinaryE2EHarness` + `SubmittedTurn` + + `RebornHarnessSharedStorage` + `assert_milestone_order` / + `trace_tool_call_response`. Drives the product caller path (inbound bytes → + ProductAdapter → workflow → coordinator → scheduler → loop) with trace-replay + model + recording capability port. +- `model_replay.rs` — `RebornTraceReplayModelGateway` and trace-replay step + types. +- `qa_trace.rs` — recorded-behavior QA trace tooling (sole consumer + `reborn_qa_recorded_behavior`). +- `qa_scenarios.rs` — QA scenario coverage ledger (sole consumer + `reborn_qa_smoke_scenarios_e2e`). +- `delivery.rs` — `RecordingOutboundDeliverySink` (channel-delivery QA + + outbound reply-target parity). +- `network.rs` — `RecordingNetworkHttpTransport` (doc-grounding / web-fetch QA). + +## Module paths + +Each consuming bin mounts BOTH trees: + +```rust +#[allow(dead_code)] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] +mod reborn_support; +mod support; +``` + +Import the binary-E2E family from `parity_qa_support::…`; the shared adopted +core (`config`, `test_adapter`, `session_thread`, `harness` doubles, +`filesystem`, `product_workflow`, …) stays under `reborn_support::…`. + +## Direction invariant (CI-enforced) + +`parity_qa_support` imports FROM `tests/integration/support/` — never the +reverse. Nothing under `tests/integration/` may reference +`reborn_parity_qa` or `parity_qa_support`. diff --git a/tests/support/reborn_parity_qa/binary_e2e.rs b/tests/support/reborn_parity_qa/binary_e2e.rs new file mode 100644 index 00000000000..9229cd84b8a --- /dev/null +++ b/tests/support/reborn_parity_qa/binary_e2e.rs @@ -0,0 +1,1465 @@ +//! Reborn binary-E2E harness. +//! +//! This harness drives the product caller path used by the #3702 validation +//! ports: +//! +//! inbound bytes -> ProductAdapter -> DefaultProductWorkflow -> +//! DefaultInboundTurnService -> DefaultTurnCoordinator -> TurnRunScheduler -> +//! Reborn planned agent loop -> model/capability/transcript evidence. +//! +//! Documented test-support substitutions: +//! - the model gateway is scripted trace replay; +//! - the capability port is a local recording echo/approval port; +//! - external internet, delivery, and OAuth are not exercised by this harness. + +#![allow(dead_code)] // Shared by staged Reborn binary-E2E validation ports. + +use std::{path::PathBuf, sync::Arc, time::Duration}; + +use async_trait::async_trait; +use ironclaw_filesystem::{InMemoryBackend, LocalFilesystem}; +use ironclaw_host_api::{ + CapabilityId, NetworkPolicy, ProviderToolName, ResourceScope, RuntimeHttpEgressRequest, + ThreadId, +}; +use ironclaw_host_runtime::{SchedulerTurnRunWakeNotifier, TurnRunSchedulerHandle}; +use ironclaw_loop_support::{ + EmptyUserProfileSource, HostIdentityContextSource, HostManagedModelRequest, + JsonSpawnSubagentInputCodec, +}; +use ironclaw_network::NetworkHttpRequest; +use ironclaw_product_adapters::{ + ProductInboundAck, ProductInboundEnvelope, ProductInboundPayload, ProductTriggerReason, + ProductWorkflow, +}; +use ironclaw_product_workflow::{ + ConversationBindingService, DefaultInboundTurnService, DefaultProductWorkflow, + IdempotencyLedger, InboundTurnService, ProductConversationRouteKind, ResolveBindingRequest, + ResolvedBinding, +}; +use ironclaw_reborn::subagent::{ + flavors::StaticSubagentDefinitionResolver, gate_resolution::BoundedSubagentGateResolutionStore, + goal_store::InMemoryBoundedSubagentGoalStore, +}; +use ironclaw_reborn::{ + loop_exit_applier::{ + BlockedEvidenceRequest, CompletionEvidenceRequest, FailureEvidenceRequest, + FinalCheckpointEvidenceRequest, LoopExitEvidencePort, ThreadCheckpointLoopExitEvidencePort, + }, + runtime::{ + DefaultPlannedRuntimeConfig, DefaultPlannedRuntimeParts, RebornRuntimeLoopComposition, + RuntimeTurnStateStore, build_default_planned_runtime, + }, +}; +use ironclaw_threads::{ + FilesystemSessionThreadService, SessionThreadService, ThreadHistoryRequest, + ThreadMessageRecord, ThreadScope, +}; +use ironclaw_turns::{ + CancelRunRequest, FilesystemTurnStateStore, GateRef, GetLoopCheckpointRequest, + GetRunStateRequest, IdempotencyKey, InMemoryCheckpointStateStore, LoopBlockedKind, + LoopCheckpointKind, LoopCheckpointStore, ReplyTargetBindingRef, ResumeTurnRequest, + SanitizedCancelReason, SourceBindingRef, TurnActor, TurnCoordinator, TurnError, TurnRunId, + TurnRunRecord, TurnRunState, TurnScope, TurnSpawnTreeStateStore, TurnStateStore, TurnStatus, + run_profile::{ + CapabilityCallCandidate, CapabilityInputRef, CapabilityInvocation, + CapabilitySurfaceVersion, LoopHostMilestone, LoopHostMilestoneKind, ParentLoopOutput, + ProviderToolCallReplay, + }, +}; +use serde_json::json; + +use super::model_replay::RebornTraceReplayModelGateway; +use crate::reborn_support::config::WaitConfig; +use crate::reborn_support::doubles::{ + EmptyIdentityContextSource, RecordingTestCapabilityPort, TEST_CAPABILITY_ID, + TEST_CAPABILITY_SURFACE_VERSION, +}; +use crate::reborn_support::filesystem::{BlockingTurnStatePutFilesystem, local_filesystem}; +use crate::reborn_support::harness::profiles::core_builtin::{self, CoreBuiltinOptions}; +use crate::reborn_support::harness::{ + HarnessCapabilityMode, HarnessCapabilityRecorder, HarnessResult, HarnessTurnBackend, + HarnessTurnStorageBackend, RecordedCapabilityResult, product_scope, scoped_turns_fs, +}; +use crate::reborn_support::product_workflow::RebornProductWorkflowHarness; +use crate::reborn_support::session_thread::RebornThreadHarness; +use crate::reborn_support::test_adapter::{RebornTestIngress, RebornTestProductAdapter}; + +pub type HarnessWaitConfig = WaitConfig; + +pub struct RebornBinaryE2EHarness { + ingress: RebornTestIngress, + workflow: DefaultProductWorkflow, + external_conversation_id: String, + binding: ResolvedBinding, + thread_scope: ThreadScope, + turn_scope: TurnScope, + turn_store: Arc>, + coordinator: Arc, + _product_harness: RebornProductWorkflowHarness, + thread_harness: RebornThreadHarness, + model_gateway: RebornTraceReplayModelGateway, + capability_recorder: HarnessCapabilityRecorder, + milestone_sink: Arc, + scheduler_handle: Option, + scheduler_notifier: Arc, + _turn_root: Arc, +} + +pub struct SubmittedTurn { + pub ack: ProductInboundAck, + pub run_id: TurnRunId, + pub thread_id: ThreadId, + pub thread_scope: ThreadScope, + pub scope: TurnScope, + pub actor: TurnActor, +} + +#[derive(Clone)] +pub struct RebornHarnessSharedStorage { + product_backend: Arc, + product_root: Arc, + thread_backend: Arc, + turn_backend: Arc, + turn_root: Arc, +} + +impl RebornHarnessSharedStorage { + pub fn new() -> HarnessResult { + let product_root = Arc::new(tempfile::tempdir()?); + let turn_root = Arc::new(tempfile::tempdir()?); + Ok(Self { + product_backend: Arc::new(local_filesystem(product_root.path())?), + product_root, + thread_backend: Arc::new(InMemoryBackend::new()), + turn_backend: Arc::new(BlockingTurnStatePutFilesystem::new(InMemoryBackend::new())), + turn_root, + }) + } + + pub fn block_next_turn_state_put(&self) { + self.turn_backend.block_next_put(); + } + + pub async fn wait_for_blocked_turn_state_put(&self) { + self.turn_backend.wait_for_blocked_put().await; + } + + pub fn release_blocked_turn_state_put(&self) { + self.turn_backend.release_blocked_put(); + } +} + +impl RebornBinaryE2EHarness { + pub async fn reply_only( + conversation_id: &str, + reply: impl Into, + ) -> HarnessResult { + Self::with_model_gateway( + conversation_id, + RebornTraceReplayModelGateway::with_responses([ + ironclaw_loop_support::HostManagedModelResponse::assistant_reply(reply), + ]), + RecordingTestCapabilityPort::echo(), + ) + .await + } + + pub async fn with_model_gateway( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_port: RecordingTestCapabilityPort, + ) -> HarnessResult { + Self::with_model_gateway_options(conversation_id, model_gateway, capability_port, false) + .await + } + + pub async fn with_model_gateway_scope_shared_storage( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_port: RecordingTestCapabilityPort, + scope: ResourceScope, + shared_storage: RebornHarnessSharedStorage, + ) -> HarnessResult { + Self::with_model_gateway_scope_identity_source_trigger_installation_shared_storage( + conversation_id, + model_gateway, + capability_port, + scope, + Arc::new(EmptyIdentityContextSource), + ProductTriggerReason::DirectChat, + "reborn-test", + "install-1", + "alice", + shared_storage, + ) + .await + } + + pub async fn with_model_gateway_scope_installation_shared_storage( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_port: RecordingTestCapabilityPort, + scope: ResourceScope, + adapter_id: &str, + installation_id: &str, + shared_storage: RebornHarnessSharedStorage, + ) -> HarnessResult { + Self::with_model_gateway_scope_initial_actor_installation_shared_storage( + conversation_id, + "alice", + model_gateway, + capability_port, + scope, + adapter_id, + installation_id, + shared_storage, + ) + .await + } + + #[allow(clippy::too_many_arguments)] + pub async fn with_model_gateway_scope_initial_actor_installation_shared_storage( + conversation_id: &str, + initial_actor_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_port: RecordingTestCapabilityPort, + scope: ResourceScope, + adapter_id: &str, + installation_id: &str, + shared_storage: RebornHarnessSharedStorage, + ) -> HarnessResult { + Self::with_model_gateway_scope_identity_source_trigger_installation_shared_storage( + conversation_id, + model_gateway, + capability_port, + scope, + Arc::new(EmptyIdentityContextSource), + ProductTriggerReason::DirectChat, + adapter_id, + installation_id, + initial_actor_id, + shared_storage, + ) + .await + } + + #[allow(clippy::too_many_arguments)] + pub async fn with_model_gateway_scope_identity_source_trigger_installation_shared_storage( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_port: RecordingTestCapabilityPort, + scope: ResourceScope, + identity_context_source: Arc, + initial_trigger: ProductTriggerReason, + adapter_id: &str, + installation_id: &str, + initial_actor_id: &str, + shared_storage: RebornHarnessSharedStorage, + ) -> HarnessResult { + Self::with_model_gateway_capability_mode_identity_source_trigger_storage_and_adapter( + conversation_id, + model_gateway, + HarnessCapabilityMode::Recording(capability_port), + false, + initial_trigger, + identity_context_source, + scope, + Some(shared_storage), + adapter_id, + installation_id, + initial_actor_id, + ) + .await + } + + pub async fn with_model_gateway_identity_source_shared( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_port: RecordingTestCapabilityPort, + identity_context_source: Arc, + ) -> HarnessResult { + Self::with_model_gateway_options_identity_source_trigger( + conversation_id, + model_gateway, + capability_port, + false, + ProductTriggerReason::BotMention, + identity_context_source, + ) + .await + } + + pub async fn with_host_runtime_file_capabilities( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + ) -> HarnessResult { + let host_runtime = + Arc::new(crate::reborn_support::harness::profiles::file::file_tools().await?); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + true, + ) + .await + } + + pub async fn with_host_runtime_file_capabilities_requiring_approval( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + ) -> HarnessResult { + let host_runtime = Arc::new( + crate::reborn_support::harness::profiles::file::file_tools_requiring_approval().await?, + ); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + true, + ) + .await + } + + pub async fn with_host_runtime_write_only( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + ) -> HarnessResult { + let host_runtime = + Arc::new(crate::reborn_support::harness::profiles::file::write_only().await?); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + false, + ) + .await + } + + pub async fn with_host_runtime_coding_read_capabilities( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + ) -> HarnessResult { + let host_runtime = Arc::new( + crate::reborn_support::harness::profiles::coding_read::coding_read_tools().await?, + ); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + false, + ) + .await + } + + pub async fn with_host_runtime_core_builtin_capabilities( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + ) -> HarnessResult { + let host_runtime = Arc::new(core_builtin::core_builtin_tools_default().await?); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + false, + ) + .await + } + + pub async fn with_host_runtime_process_capabilities( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + ) -> HarnessResult { + let host_runtime = + Arc::new(crate::reborn_support::harness::profiles::process::process_tools().await?); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + false, + ) + .await + } + + pub async fn with_host_runtime_qa_smoke_capabilities( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + ) -> HarnessResult { + let host_runtime = + Arc::new(crate::reborn_support::harness::profiles::qa_smoke::qa_smoke_tools().await?); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + false, + ) + .await + } + + pub async fn with_host_runtime_extension_lifecycle_capabilities( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + ) -> HarnessResult { + let host_runtime = Arc::new( + crate::reborn_support::harness::profiles::extension::extension_lifecycle_tools() + .await?, + ); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + false, + ) + .await + } + + pub async fn with_host_runtime_skill_management_capabilities( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + ) -> HarnessResult { + let host_runtime = Arc::new( + crate::reborn_support::harness::profiles::skill::skill_management_tools().await?, + ); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + false, + ) + .await + } + + pub async fn with_host_runtime_trigger_management_capabilities( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + ) -> HarnessResult { + let host_runtime = Arc::new( + crate::reborn_support::harness::profiles::trigger::trigger_management_tools().await?, + ); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + false, + ) + .await + } + + pub async fn with_host_runtime_trace_commons_capabilities( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + ) -> HarnessResult { + let host_runtime = Arc::new( + crate::reborn_support::harness::profiles::trace_commons::trace_commons_tools().await?, + ); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + false, + ) + .await + } + + pub async fn with_host_runtime_core_builtin_capabilities_network_policy( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + network_policy: NetworkPolicy, + ) -> HarnessResult { + let host_runtime = Arc::new( + core_builtin::core_builtin_tools( + CoreBuiltinOptions::default().with_network_policy(network_policy), + ) + .await?, + ); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + false, + ) + .await + } + + pub async fn with_host_runtime_core_builtin_capabilities_live_http_egress( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + network_policy: NetworkPolicy, + ) -> HarnessResult { + let host_runtime = Arc::new( + core_builtin::core_builtin_tools( + CoreBuiltinOptions::default() + .with_live_http_egress() + .with_network_policy(network_policy), + ) + .await?, + ); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + false, + ) + .await + } + + pub async fn with_host_runtime_github_issue_capabilities( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + ) -> HarnessResult { + let host_runtime = + Arc::new(crate::reborn_support::harness::profiles::github::github_issue_tools().await?); + Self::with_model_gateway_capability_mode( + conversation_id, + model_gateway, + HarnessCapabilityMode::HostRuntime(host_runtime), + false, + ) + .await + } + + pub async fn with_harness_blocked_evidence( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_port: RecordingTestCapabilityPort, + ) -> HarnessResult { + Self::with_model_gateway_options(conversation_id, model_gateway, capability_port, true) + .await + } + + async fn with_model_gateway_options( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_port: RecordingTestCapabilityPort, + accept_harness_blocked_evidence: bool, + ) -> HarnessResult { + Self::with_model_gateway_options_identity_source( + conversation_id, + model_gateway, + capability_port, + accept_harness_blocked_evidence, + Arc::new(EmptyIdentityContextSource), + ) + .await + } + + async fn with_model_gateway_options_identity_source( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_port: RecordingTestCapabilityPort, + accept_harness_blocked_evidence: bool, + identity_context_source: Arc, + ) -> HarnessResult { + Self::with_model_gateway_options_identity_source_trigger( + conversation_id, + model_gateway, + capability_port, + accept_harness_blocked_evidence, + ProductTriggerReason::DirectChat, + identity_context_source, + ) + .await + } + + async fn with_model_gateway_options_identity_source_trigger( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_port: RecordingTestCapabilityPort, + accept_harness_blocked_evidence: bool, + initial_trigger: ProductTriggerReason, + identity_context_source: Arc, + ) -> HarnessResult { + Self::with_model_gateway_capability_mode_identity_source_trigger( + conversation_id, + model_gateway, + HarnessCapabilityMode::Recording(capability_port), + accept_harness_blocked_evidence, + initial_trigger, + identity_context_source, + ) + .await + } + + async fn with_model_gateway_capability_mode( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_mode: HarnessCapabilityMode, + accept_harness_blocked_evidence: bool, + ) -> HarnessResult { + Self::with_model_gateway_capability_mode_identity_source( + conversation_id, + model_gateway, + capability_mode, + accept_harness_blocked_evidence, + Arc::new(EmptyIdentityContextSource), + ) + .await + } + + async fn with_model_gateway_capability_mode_identity_source( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_mode: HarnessCapabilityMode, + accept_harness_blocked_evidence: bool, + identity_context_source: Arc, + ) -> HarnessResult { + Self::with_model_gateway_capability_mode_identity_source_trigger( + conversation_id, + model_gateway, + capability_mode, + accept_harness_blocked_evidence, + ProductTriggerReason::DirectChat, + identity_context_source, + ) + .await + } + + async fn with_model_gateway_capability_mode_identity_source_trigger( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_mode: HarnessCapabilityMode, + accept_harness_blocked_evidence: bool, + initial_trigger: ProductTriggerReason, + identity_context_source: Arc, + ) -> HarnessResult { + Self::with_model_gateway_capability_mode_identity_source_trigger_storage_and_adapter( + conversation_id, + model_gateway, + capability_mode, + accept_harness_blocked_evidence, + initial_trigger, + identity_context_source, + product_scope(), + None, + "reborn-test", + "install-1", + "alice", + ) + .await + } + + #[allow(clippy::too_many_arguments)] + async fn with_model_gateway_capability_mode_identity_source_trigger_storage_and_adapter( + conversation_id: &str, + model_gateway: RebornTraceReplayModelGateway, + capability_mode: HarnessCapabilityMode, + accept_harness_blocked_evidence: bool, + initial_trigger: ProductTriggerReason, + identity_context_source: Arc, + product_scope: ResourceScope, + shared_storage: Option, + adapter_id: &str, + installation_id: &str, + initial_actor_id: &str, + ) -> HarnessResult { + let adapter = RebornTestProductAdapter::new(adapter_id, installation_id)?; + let ingress = RebornTestIngress::new(adapter); + let product_harness = if let Some(storage) = shared_storage.as_ref() { + RebornProductWorkflowHarness::filesystem_shared_backend( + product_scope.clone(), + Arc::clone(&storage.product_backend), + Arc::clone(&storage.product_root), + )? + } else { + RebornProductWorkflowHarness::filesystem_temp(product_scope)? + }; + let binding = product_harness + .binding_service()? + .resolve_binding(binding_request_with_trigger_and_actor( + &ingress, + conversation_id, + initial_actor_id, + initial_trigger, + )?) + .await?; + let thread_scope = thread_scope_from_binding_with_route_kind( + &binding, + route_kind_for_trigger(initial_trigger), + )?; + let turn_scope = TurnScope::new_with_owner( + binding.tenant_id.clone(), + binding.agent_id.clone(), + binding.project_id.clone(), + binding.thread_id.clone(), + binding.subject_user_id.clone(), + ); + let thread_harness = if let Some(storage) = shared_storage.as_ref() { + RebornThreadHarness::filesystem_shared_backend( + thread_scope.clone(), + Arc::clone(&storage.thread_backend), + )? + } else { + RebornThreadHarness::filesystem_temp(thread_scope.clone())? + }; + let (turn_backend, turn_root) = if let Some(storage) = shared_storage.as_ref() { + ( + Arc::clone(&storage.turn_backend), + Arc::clone(&storage.turn_root), + ) + } else { + let turn_root = Arc::new(tempfile::tempdir()?); + ( + Arc::new(BlockingTurnStatePutFilesystem::new(InMemoryBackend::new())), + turn_root, + ) + }; + let turn_store = Arc::new(FilesystemTurnStateStore::new(scoped_turns_fs( + turn_backend, + &binding, + )?)); + let checkpoint_state_store = Arc::new(InMemoryCheckpointStateStore::default()); + let loop_checkpoint_store: Arc = turn_store.clone(); + let milestone_sink = + Arc::new(ironclaw_turns::run_profile::InMemoryLoopHostMilestoneSink::default()); + let ( + capability_factory, + capability_surface_resolver, + capability_input_resolver, + capability_result_writer, + capability_recorder, + ) = capability_mode.into_parts(milestone_sink.clone())?; + let turn_state_for_evidence: Arc = turn_store.clone(); + let evidence = Arc::new(HarnessLoopExitEvidencePort { + inner: ThreadCheckpointLoopExitEvidencePort::new_with_thread_scope( + thread_harness.service.clone(), + turn_state_for_evidence, + Arc::clone(&loop_checkpoint_store), + thread_scope.clone(), + ), + loop_checkpoint_store: Arc::clone(&loop_checkpoint_store), + accept_harness_blocked_evidence, + }); + let turn_state_for_runtime: Arc = turn_store.clone(); + let composition = build_default_planned_runtime(DefaultPlannedRuntimeParts { + turn_state: turn_state_for_runtime, + thread_service: thread_harness.service.clone() + as Arc, + thread_scope: thread_scope.clone(), + model_gateway: Arc::new(model_gateway.clone()), + checkpoint_state_store, + loop_checkpoint_store, + milestone_sink: milestone_sink.clone(), + capability_factory, + capability_surface_resolver, + capability_result_writer, + subagent_goal_store: Arc::new(InMemoryBoundedSubagentGoalStore::new()), + subagent_gate_store: Arc::new(BoundedSubagentGateResolutionStore::new()), + subagent_definition_resolver: Arc::new(StaticSubagentDefinitionResolver), + subagent_spawn_input_codec: Arc::new(JsonSpawnSubagentInputCodec::new( + capability_input_resolver, + )), + subagent_spawn_limits: ironclaw_loop_support::SubagentSpawnLimits::default(), + loop_exit_evidence: evidence, + config: DefaultPlannedRuntimeConfig { + // Keep the durable runner heartbeat at its production default; + // test responsiveness comes from fast scheduler polling below. + poll_interval: Duration::from_millis(10), + ..DefaultPlannedRuntimeConfig::default() + }, + model_route_resolver: None, + cancellation_factory: None, + skill_context_source: None, + input_queue: None, + identity_context_source, + user_profile_source: Arc::new(EmptyUserProfileSource), + model_policy_guard: None, + model_budget_accountant: None, + safety_context: None, + hook_dispatcher_builder_factory: None, + communication_context_provider: None, + hook_security_audit_sink: None, + turn_event_sink: None, + attachment_read_port: None, + scheduler_wake_wiring: None, + })?; + let binding_service: Arc = + Arc::new(product_harness.binding_service()?); + let inbound: Arc = Arc::new(DefaultInboundTurnService::new( + Arc::clone(&binding_service), + thread_harness.service_instance()?, + composition.coordinator.clone(), + )); + let ledger: Arc = Arc::new(product_harness.idempotency_ledger()); + let workflow = DefaultProductWorkflow::new(inbound, ledger, binding_service); + + Ok(Self::from_composition( + ingress, + workflow, + conversation_id.to_string(), + binding, + thread_scope, + turn_scope, + turn_store, + product_harness, + thread_harness, + model_gateway, + capability_recorder, + milestone_sink, + composition, + turn_root, + )) + } + + #[allow(clippy::too_many_arguments)] + fn from_composition( + ingress: RebornTestIngress, + workflow: DefaultProductWorkflow, + external_conversation_id: String, + binding: ResolvedBinding, + thread_scope: ThreadScope, + turn_scope: TurnScope, + turn_store: Arc>, + product_harness: RebornProductWorkflowHarness, + thread_harness: RebornThreadHarness, + model_gateway: RebornTraceReplayModelGateway, + capability_recorder: HarnessCapabilityRecorder, + milestone_sink: Arc, + composition: RebornRuntimeLoopComposition< + dyn SessionThreadService, + RebornTraceReplayModelGateway, + >, + turn_root: Arc, + ) -> Self { + let coordinator = Arc::clone(&composition.coordinator); + let scheduler_notifier = composition.scheduler_handle.wake_notifier(); + Self { + ingress, + workflow, + external_conversation_id, + binding, + thread_scope, + turn_scope, + turn_store, + coordinator, + _product_harness: product_harness, + thread_harness, + model_gateway, + capability_recorder, + milestone_sink, + scheduler_handle: Some(composition.scheduler_handle), + scheduler_notifier, + _turn_root: turn_root, + } + } + + pub fn start(&mut self) { + // The scheduler is started automatically inside build_default_planned_runtime. + // This method is kept for API compatibility. + } + + pub fn start_workers(&mut self, _count: usize) { + // The scheduler is started automatically inside build_default_planned_runtime. + // Worker count is configured via DefaultPlannedRuntimeConfig.worker_count. + } + + pub async fn shutdown(&mut self) { + if let Some(scheduler) = self.scheduler_handle.take() { + scheduler.shutdown().await; + } + } + + pub async fn submit_text(&self, event_id: &str, text: &str) -> HarnessResult { + self.submit_text_for(&self.external_conversation_id, "alice", event_id, text) + .await + } + + pub async fn submit_text_for( + &self, + conversation_id: &str, + actor_id: &str, + event_id: &str, + text: &str, + ) -> HarnessResult { + self.submit_text_for_with_trigger( + conversation_id, + actor_id, + event_id, + text, + ProductTriggerReason::DirectChat, + ) + .await + } + + pub async fn submit_text_for_with_trigger( + &self, + conversation_id: &str, + actor_id: &str, + event_id: &str, + text: &str, + trigger: ProductTriggerReason, + ) -> HarnessResult { + let envelope = self.ingress.verified_text_envelope_with_trigger( + event_id, + actor_id, + conversation_id, + text, + trigger, + )?; + let binding_request = binding_request_from_envelope(&envelope); + let route_kind = binding_request.route_kind; + let binding = self + ._product_harness + .binding_service()? + .resolve_binding(binding_request) + .await?; + let thread_scope = thread_scope_from_binding_with_route_kind(&binding, route_kind)?; + let turn_scope = TurnScope::new_with_owner( + binding.tenant_id.clone(), + binding.agent_id.clone(), + binding.project_id.clone(), + binding.thread_id.clone(), + binding.subject_user_id.clone(), + ); + let actor = TurnActor::new(binding.actor_user_id.clone()); + let ack = self.workflow.accept_inbound(envelope).await?; + let run_id = match &ack { + ProductInboundAck::Accepted { + submitted_run_id, .. + } => *submitted_run_id, + other => { + return Err(format!("expected accepted inbound ack, got {other:?}").into()); + } + }; + Ok(SubmittedTurn { + ack, + run_id, + thread_id: binding.thread_id, + thread_scope, + scope: turn_scope, + actor, + }) + } + + pub async fn resume_blocked_turn(&self, run_id: TurnRunId) -> HarnessResult<()> { + let blocked = self + .run_state(run_id) + .await? + .gate_ref + .ok_or("blocked run missing gate ref")?; + self.resume_with_gate(run_id, blocked).await + } + + pub async fn approve_and_resume_local_dev_gate( + &self, + run_id: TurnRunId, + ) -> HarnessResult { + let blocked = self + .run_state(run_id) + .await? + .gate_ref + .ok_or("blocked run missing gate ref")?; + self.capability_recorder + .approve_local_dev_gate(&blocked) + .await?; + self.resume_with_gate(run_id, blocked.clone()).await?; + Ok(blocked) + } + + pub async fn resume_blocked_turn_in_scope( + &self, + scope: TurnScope, + actor: TurnActor, + run_id: TurnRunId, + ) -> HarnessResult<()> { + let blocked = self + .run_state_in_scope(scope.clone(), run_id) + .await? + .gate_ref + .ok_or("blocked run missing gate ref")?; + self.resume_with_gate_as(scope, actor, run_id, blocked, format!("resume-{run_id}")) + .await + } + + pub async fn resume_with_gate( + &self, + run_id: TurnRunId, + gate_ref: GateRef, + ) -> HarnessResult<()> { + self.resume_with_gate_as( + self.turn_scope.clone(), + TurnActor::new(self.binding.actor_user_id.clone()), + run_id, + gate_ref, + format!("resume-{run_id}"), + ) + .await + } + + pub async fn resume_with_gate_as( + &self, + scope: TurnScope, + actor: TurnActor, + run_id: TurnRunId, + gate_ref: GateRef, + idempotency_key: impl Into, + ) -> HarnessResult<()> { + let response = self + .coordinator + .resume_turn(ResumeTurnRequest { + scope, + actor, + run_id, + gate_resolution_ref: gate_ref, + precondition: ironclaw_turns::ResumeTurnPrecondition::AnyBlockedGate, + source_binding_ref: SourceBindingRef::new("src:resume")?, + reply_target_binding_ref: ReplyTargetBindingRef::new("reply:resume")?, + idempotency_key: IdempotencyKey::new(idempotency_key.into())?, + resume_disposition: None, + }) + .await?; + if response.status != TurnStatus::Queued { + return Err(format!("expected resumed run to queue, got {:?}", response.status).into()); + } + Ok(()) + } + + pub async fn cancel_blocked_turn(&self, run_id: TurnRunId) -> HarnessResult<()> { + self.cancel_run_as( + self.turn_scope.clone(), + TurnActor::new(self.binding.actor_user_id.clone()), + run_id, + format!("cancel-{run_id}"), + ) + .await + } + + pub async fn cancel_run_as( + &self, + scope: TurnScope, + actor: TurnActor, + run_id: TurnRunId, + idempotency_key: impl Into, + ) -> HarnessResult<()> { + let response = self + .coordinator + .cancel_run(CancelRunRequest { + scope, + actor, + run_id, + reason: SanitizedCancelReason::UserRequested, + idempotency_key: IdempotencyKey::new(idempotency_key.into())?, + }) + .await?; + if !matches!( + response.status, + TurnStatus::Cancelled | TurnStatus::CancelRequested + ) { + return Err(format!( + "expected run to be cancelled or cancel-requested, got {:?}", + response.status + ) + .into()); + } + Ok(()) + } + + pub async fn wait_for_status( + &self, + run_id: TurnRunId, + expected: TurnStatus, + ) -> HarnessResult { + self.wait_for_status_with_config(run_id, expected, WaitConfig::default()) + .await + } + + pub async fn wait_for_status_with_config( + &self, + run_id: TurnRunId, + expected: TurnStatus, + wait: WaitConfig, + ) -> HarnessResult { + self.wait_for_status_in_scope_with_config(self.turn_scope.clone(), run_id, expected, wait) + .await + } + + pub async fn wait_for_submitted_status( + &self, + submitted: &SubmittedTurn, + expected: TurnStatus, + ) -> HarnessResult { + self.wait_for_status_in_scope(submitted.scope.clone(), submitted.run_id, expected) + .await + } + + pub async fn wait_for_status_in_scope( + &self, + scope: TurnScope, + run_id: TurnRunId, + expected: TurnStatus, + ) -> HarnessResult { + self.wait_for_status_in_scope_with_config(scope, run_id, expected, WaitConfig::default()) + .await + } + + pub async fn wait_for_status_in_scope_with_config( + &self, + scope: TurnScope, + run_id: TurnRunId, + expected: TurnStatus, + wait: WaitConfig, + ) -> HarnessResult { + let deadline = tokio::time::Instant::now() + wait.timeout; + loop { + let state = self.run_state_in_scope(scope.clone(), run_id).await?; + if state.status == expected { + return Ok(state); + } + // A terminal status (Completed/Failed/Cancelled/RecoveryRequired) is + // never left, so once we observe one that is not the target the run + // can never reach `expected`. Fail fast instead of polling to the + // deadline — otherwise a run that fails early (e.g. a spawn capability + // returning a terminal `driver_unavailable`) burns the whole timeout + // and buries the real failure category. + if state.status.is_terminal() { + return Err(format!( + "expected {expected:?} but run reached terminal status {:?}; failure={:?}", + state.status, state.failure + ) + .into()); + } + if tokio::time::Instant::now() >= deadline { + return Err(format!( + "timed out waiting for {expected:?}; last status={:?} failure={:?}", + state.status, state.failure + ) + .into()); + } + tokio::time::sleep(wait.poll_interval).await; + } + } + + pub async fn run_state(&self, run_id: TurnRunId) -> HarnessResult { + self.run_state_in_scope(self.turn_scope.clone(), run_id) + .await + } + + pub async fn run_state_in_scope( + &self, + scope: TurnScope, + run_id: TurnRunId, + ) -> HarnessResult { + Ok(self + .turn_store + .get_run_state(GetRunStateRequest { scope, run_id }) + .await?) + } + + pub async fn assert_final_reply(&self, text: &str) -> HarnessResult<()> { + Ok(self + .thread_harness + .assert_final_reply(self.binding.thread_id.clone(), text) + .await?) + } + + pub async fn history(&self) -> HarnessResult> { + self.history_for_thread(self.binding.thread_id.clone()) + .await + } + + pub async fn history_for_submitted_thread( + &self, + submitted: &SubmittedTurn, + ) -> HarnessResult> { + self.history_for_thread_in_scope( + submitted.thread_scope.clone(), + submitted.thread_id.clone(), + ) + .await + } + + pub async fn history_for_thread( + &self, + thread_id: ThreadId, + ) -> HarnessResult> { + self.history_for_thread_in_scope(self.thread_scope.clone(), thread_id) + .await + } + + pub async fn history_for_thread_in_scope( + &self, + scope: ThreadScope, + thread_id: ThreadId, + ) -> HarnessResult> { + Ok(self + .thread_harness + .service + .list_thread_history(ThreadHistoryRequest { scope, thread_id }) + .await? + .messages) + } + + pub async fn children_of( + &self, + scope: &TurnScope, + run_id: TurnRunId, + ) -> HarnessResult> { + Ok(self.turn_store.children_of(scope, run_id).await?) + } + + pub fn model_requests(&self) -> Vec { + self.model_gateway.requests() + } + + pub fn remaining_model_responses(&self) -> usize { + self.model_gateway.remaining_responses() + } + + pub fn assert_model_exhausted(&self) { + self.model_gateway.assert_exhausted(); + } + + pub fn capability_invocations(&self) -> Vec { + self.capability_recorder.invocations() + } + + pub fn capability_results(&self) -> Vec { + self.capability_recorder.capability_results() + } + + pub fn runtime_http_requests(&self) -> Vec { + self.capability_recorder.runtime_http_requests() + } + + pub fn network_http_requests(&self) -> Vec { + self.capability_recorder.network_http_requests() + } + + pub fn host_workspace_file_path(&self, relative: &str) -> HarnessResult { + self.capability_recorder + .workspace_file_path(relative) + .ok_or_else(|| "harness is not using host-runtime capabilities".into()) + } + + pub fn milestones(&self) -> Vec { + self.milestone_sink.milestones() + } +} + +impl Drop for RebornBinaryE2EHarness { + fn drop(&mut self) { + // Scheduler handle is Option; shutdown is async + // and cannot be called from Drop. The handle is taken in shutdown() and + // here we just let it drop. The scheduler supervisor task exits when the + // command channel closes on drop. + let _ = self.scheduler_handle.take(); + } +} + +struct HarnessLoopExitEvidencePort { + inner: ThreadCheckpointLoopExitEvidencePort>, + loop_checkpoint_store: Arc, + accept_harness_blocked_evidence: bool, +} + +#[async_trait] +impl LoopExitEvidencePort for HarnessLoopExitEvidencePort { + async fn verify_completion_refs( + &self, + request: CompletionEvidenceRequest<'_>, + ) -> Result { + self.inner.verify_completion_refs(request).await + } + + async fn verify_final_checkpoint( + &self, + request: FinalCheckpointEvidenceRequest<'_>, + ) -> Result { + self.inner.verify_final_checkpoint(request).await + } + + async fn verify_blocked_evidence( + &self, + request: BlockedEvidenceRequest<'_>, + ) -> Result { + if self.inner.verify_blocked_evidence(request.clone()).await? { + return Ok(true); + } + if !self.accept_harness_blocked_evidence { + return Ok(false); + } + if !matches!( + request.blocked.kind, + LoopBlockedKind::Approval | LoopBlockedKind::AwaitDependentRun + ) || GateRef::new(request.blocked.gate_ref.as_str()).is_err() + { + return Ok(false); + } + let checkpoint = self + .loop_checkpoint_store + .get_loop_checkpoint(GetLoopCheckpointRequest { + scope: request.scope.clone(), + turn_id: request.turn_id, + run_id: request.run_id, + checkpoint_id: request.blocked.checkpoint_id, + }) + .await?; + Ok(checkpoint + .map(|record| { + record.kind == LoopCheckpointKind::BeforeBlock + && record.state_ref == request.blocked.state_ref + }) + .unwrap_or(false)) + } + + async fn verify_failure_evidence( + &self, + request: FailureEvidenceRequest<'_>, + ) -> Result { + self.inner.verify_failure_evidence(request).await + } + + async fn is_cancellation_observed( + &self, + scope: &TurnScope, + turn_id: ironclaw_turns::TurnId, + run_id: TurnRunId, + ) -> Result { + self.inner + .is_cancellation_observed(scope, turn_id, run_id) + .await + } + + async fn latest_checkpoint_kind( + &self, + scope: &TurnScope, + turn_id: ironclaw_turns::TurnId, + run_id: TurnRunId, + ) -> Result, TurnError> { + self.inner + .latest_checkpoint_kind(scope, turn_id, run_id) + .await + } +} + +fn binding_request( + ingress: &RebornTestIngress, + conversation_id: &str, +) -> HarnessResult { + binding_request_with_trigger(ingress, conversation_id, ProductTriggerReason::DirectChat) +} + +fn binding_request_with_trigger( + ingress: &RebornTestIngress, + conversation_id: &str, + trigger: ProductTriggerReason, +) -> HarnessResult { + binding_request_with_trigger_and_actor(ingress, conversation_id, "alice", trigger) +} + +fn binding_request_with_trigger_and_actor( + ingress: &RebornTestIngress, + conversation_id: &str, + actor_id: &str, + trigger: ProductTriggerReason, +) -> HarnessResult { + let envelope = ingress.verified_text_envelope_with_trigger( + "binding-probe", + actor_id, + conversation_id, + "hi", + trigger, + )?; + Ok(binding_request_from_envelope(&envelope)) +} + +fn binding_request_from_envelope(envelope: &ProductInboundEnvelope) -> ResolveBindingRequest { + ResolveBindingRequest { + adapter_id: envelope.adapter_id().clone(), + installation_id: envelope.installation_id().clone(), + external_actor_ref: envelope.external_actor_ref().clone(), + external_conversation_ref: envelope.external_conversation_ref().clone(), + external_event_id: envelope.external_event_id().clone(), + route_kind: route_kind_for_envelope(envelope), + auth_claim: envelope.auth_claim().clone(), + } +} + +fn thread_scope_from_binding(binding: &ResolvedBinding) -> HarnessResult { + thread_scope_from_binding_with_route_kind(binding, ProductConversationRouteKind::Direct) +} + +fn thread_scope_from_binding_with_route_kind( + binding: &ResolvedBinding, + _route_kind: ProductConversationRouteKind, +) -> HarnessResult { + Ok(ThreadScope { + tenant_id: binding.tenant_id.clone(), + agent_id: binding + .agent_id + .clone() + .ok_or("resolved binding missing agent id")?, + project_id: binding.project_id.clone(), + owner_user_id: binding.subject_user_id.clone(), + mission_id: None, + }) +} + +fn route_kind_for_envelope(envelope: &ProductInboundEnvelope) -> ProductConversationRouteKind { + match envelope.payload() { + ProductInboundPayload::UserMessage(message) => route_kind_for_trigger(message.trigger), + ProductInboundPayload::Command(command) => route_kind_for_trigger(command.trigger), + _ => ProductConversationRouteKind::Direct, + } +} + +fn route_kind_for_trigger(trigger: ProductTriggerReason) -> ProductConversationRouteKind { + match trigger { + ProductTriggerReason::DirectChat => ProductConversationRouteKind::Direct, + ProductTriggerReason::BotMention + | ProductTriggerReason::ReplyToBot + | ProductTriggerReason::BotCommand + | ProductTriggerReason::LinkedThreadAction => ProductConversationRouteKind::Shared, + } +} + +pub fn trace_tool_call_response() -> ironclaw_loop_support::HostManagedModelResponse { + ironclaw_loop_support::HostManagedModelResponse { + safe_text_deltas: Vec::new(), + safe_reasoning_deltas: Vec::new(), + usage: None, + output: ParentLoopOutput::CapabilityCalls(vec![CapabilityCallCandidate { + activity_id: ironclaw_turns::CapabilityActivityId::new(), + surface_version: CapabilitySurfaceVersion::new(TEST_CAPABILITY_SURFACE_VERSION) + .expect("valid surface version"), + capability_id: CapabilityId::new(TEST_CAPABILITY_ID).expect("valid capability id"), + effective_capability_ids: vec![ + CapabilityId::new(TEST_CAPABILITY_ID).expect("valid capability id"), + ], + input_ref: CapabilityInputRef::new("input:trace-call-1").expect("valid input ref"), + provider_replay: Some(ProviderToolCallReplay { + provider_id: "trace_replay".to_string(), + provider_model_id: "trace_replay".to_string(), + provider_turn_id: "trace-turn".to_string(), + provider_call_id: "call-1".to_string(), + provider_tool_name: ProviderToolName::new("test_echo").expect("provider tool name"), + arguments: json!({"message": "hi"}), + response_reasoning: None, + reasoning: None, + signature: None, + }), + }]), + } +} + +pub fn assert_milestone_order( + milestones: &[LoopHostMilestone], + before: impl Fn(&LoopHostMilestoneKind) -> bool, + after: impl Fn(&LoopHostMilestoneKind) -> bool, +) { + let before_index = milestones + .iter() + .position(|milestone| before(&milestone.kind)) + .expect("before milestone should be present"); + let after_index = milestones + .iter() + .position(|milestone| after(&milestone.kind)) + .expect("after milestone should be present"); + assert!( + before_index < after_index, + "expected milestone order, got {:?}", + milestones + .iter() + .map(|milestone| milestone.kind.kind_name()) + .collect::>() + ); +} diff --git a/tests/support/reborn/delivery.rs b/tests/support/reborn_parity_qa/delivery.rs similarity index 100% rename from tests/support/reborn/delivery.rs rename to tests/support/reborn_parity_qa/delivery.rs diff --git a/tests/support/reborn_parity_qa/mod.rs b/tests/support/reborn_parity_qa/mod.rs new file mode 100644 index 00000000000..2807163c933 --- /dev/null +++ b/tests/support/reborn_parity_qa/mod.rs @@ -0,0 +1,14 @@ +//! Reborn binary-E2E harness family + trace-replay model gateway. +//! +//! Extracted from the `tests/integration/support/` tree (which now hosts only +//! the roadmap `RebornIntegrationHarness` family) — this family is the older +//! flat-bin/`reborn_trace_*`/QA-scenario harness, consumed by `tests/reborn_*.rs` +//! parity and QA bins. + +pub mod binary_e2e; +pub mod delivery; +pub mod model_replay; +pub mod network; +#[allow(dead_code)] +pub mod qa_scenarios; +pub mod qa_trace; diff --git a/tests/support/reborn/model_replay.rs b/tests/support/reborn_parity_qa/model_replay.rs similarity index 100% rename from tests/support/reborn/model_replay.rs rename to tests/support/reborn_parity_qa/model_replay.rs diff --git a/tests/support/reborn/network.rs b/tests/support/reborn_parity_qa/network.rs similarity index 100% rename from tests/support/reborn/network.rs rename to tests/support/reborn_parity_qa/network.rs diff --git a/tests/support/reborn/qa_scenarios.rs b/tests/support/reborn_parity_qa/qa_scenarios.rs similarity index 100% rename from tests/support/reborn/qa_scenarios.rs rename to tests/support/reborn_parity_qa/qa_scenarios.rs diff --git a/tests/support/reborn/qa_trace.rs b/tests/support/reborn_parity_qa/qa_trace.rs similarity index 100% rename from tests/support/reborn/qa_trace.rs rename to tests/support/reborn_parity_qa/qa_trace.rs diff --git a/tests/support_unit_tests.rs b/tests/support_unit_tests.rs index a836ffe13f5..11f30ac8eb2 100644 --- a/tests/support_unit_tests.rs +++ b/tests/support_unit_tests.rs @@ -4,7 +4,11 @@ //! and run exactly once, rather than being duplicated across every `e2e_*.rs` //! test binary that declares `mod support;`. -#[path = "support/reborn/mod.rs"] +#[allow(dead_code)] +#[path = "support/reborn_parity_qa/mod.rs"] +mod parity_qa_support; +#[allow(dead_code)] +#[path = "integration/support/mod.rs"] mod reborn_support; mod support; @@ -410,14 +414,14 @@ mod reborn_support_tests { }; use tokio::sync::Barrier; - use crate::reborn_support::delivery::RecordingOutboundDeliverySink; - use crate::reborn_support::filesystem::local_filesystem; - use crate::reborn_support::harness::RecordingTestCapabilityPort; - use crate::reborn_support::model_replay::{ + use crate::parity_qa_support::delivery::RecordingOutboundDeliverySink; + use crate::parity_qa_support::model_replay::{ RebornModelReplayStep, RebornScriptedProviderToolCall, RebornTraceReplayError, RebornTraceReplayModelGateway, capability_call_from_trace_with_surface, }; - use crate::reborn_support::network::RecordingNetworkHttpTransport; + use crate::parity_qa_support::network::RecordingNetworkHttpTransport; + use crate::reborn_support::filesystem::local_filesystem; + use crate::reborn_support::harness::RecordingTestCapabilityPort; use crate::reborn_support::product_workflow::{ FilesystemIdempotencyLedger, RebornProductWorkflowHarness, RebornProductWorkflowHarnessError, resource_scope, From 79b188d5df5d6279ef04c8d8bfd47ca8b436095c Mon Sep 17 00:00:00 2001 From: "firat.sertgoz" Date: Sat, 4 Jul 2026 22:42:51 +0300 Subject: [PATCH 2/5] ci: bucket Reborn crate test jobs (#5635) --- .github/workflows/reborn-tests.yml | 94 ++++++++++-------- scripts/ci/reborn-crate-test-buckets.sh | 123 ++++++++++++++++++++++++ 2 files changed, 175 insertions(+), 42 deletions(-) create mode 100755 scripts/ci/reborn-crate-test-buckets.sh diff --git a/.github/workflows/reborn-tests.yml b/.github/workflows/reborn-tests.yml index 86187162ec0..21a5da4ec0e 100644 --- a/.github/workflows/reborn-tests.yml +++ b/.github/workflows/reborn-tests.yml @@ -94,6 +94,7 @@ jobs: runs-on: ubuntu-latest outputs: packages: ${{ steps.packages.outputs.packages }} + buckets: ${{ steps.packages.outputs.buckets }} steps: - name: Checkout repository uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6 @@ -165,15 +166,20 @@ jobs: fi echo "packages=${packages}" >> "$GITHUB_OUTPUT" + buckets="$(scripts/ci/reborn-crate-test-buckets.sh "${packages}")" + echo "buckets=${buckets}" >> "$GITHUB_OUTPUT" + echo "Testing Reborn package crates:" printf '%s\n' "${packages}" | jq -r '.[] | "- " + .' + echo "Testing Reborn package buckets:" + printf '%s\n' "${buckets}" | jq -r '.[] | "- \(.name): \(.packages | join(", "))"' crate-tests: - name: Test ${{ matrix.package }} + name: Test Reborn crate bucket (${{ matrix.bucket.name }}) needs: [changes, package-matrix] if: needs.changes.outputs.docs_only != 'true' && needs.changes.outputs.has_reborn_tests == 'true' runs-on: ubuntu-latest - timeout-minutes: 30 + timeout-minutes: 60 env: ANTHROPIC_API_KEY: "" LLM_BACKEND: "" @@ -183,7 +189,7 @@ jobs: strategy: fail-fast: false matrix: - package: ${{ fromJSON(needs.package-matrix.outputs.packages) }} + bucket: ${{ fromJSON(needs.package-matrix.outputs.buckets) }} steps: - name: Checkout repository uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6 @@ -201,21 +207,32 @@ jobs: - name: Install Rust uses: dtolnay/rust-toolchain@29eef336d9b2848a0b548edc03f92a220660cdb8 # stable - - name: Resolve crate feature flags - id: crate-feature-flags + - name: Resolve bucket settings + id: bucket-settings env: - PACKAGE: ${{ matrix.package }} + BUCKET_PACKAGES: ${{ toJSON(matrix.bucket.packages) }} run: | - feature_flags="$(scripts/ci/package-feature-flags.sh "$PACKAGE")" - echo "feature_flags=${feature_flags}" >> "$GITHUB_OUTPUT" - if [[ "${feature_flags}" == *"webui-v2-beta"* ]]; then - echo "needs_webui_node=true" >> "$GITHUB_OUTPUT" - else - echo "needs_webui_node=false" >> "$GITHUB_OUTPUT" - fi + needs_webui_node=false + sccache_dist_enabled=true + + while IFS= read -r package; do + feature_flags="$(scripts/ci/package-feature-flags.sh "$package")" + if [[ "${feature_flags}" == *"webui-v2-beta"* ]]; then + needs_webui_node=true + fi + + case "${package}" in + ironclaw_first_party_extension_ports|ironclaw_host_runtime|ironclaw_loop_support|ironclaw_product_workflow|ironclaw_reborn|ironclaw_reborn_cli|ironclaw_reborn_composition|ironclaw_reborn_webui_ingress|ironclaw_wasm|ironclaw_wasm_product_adapters|ironclaw_wasm_sandbox_core) + sccache_dist_enabled=false + ;; + esac + done < <(printf '%s\n' "${BUCKET_PACKAGES}" | jq -r '.[]') + + echo "needs_webui_node=${needs_webui_node}" >> "${GITHUB_OUTPUT}" + echo "sccache_dist_enabled=${sccache_dist_enabled}" >> "${GITHUB_OUTPUT}" - name: Install Node.js for WebUI bundle builds - if: ${{ steps.crate-feature-flags.outputs.needs_webui_node == 'true' }} + if: ${{ steps.bucket-settings.outputs.needs_webui_node == 'true' }} uses: actions/setup-node@60edb5dd545a775178f52524783378180af0d1f8 # v4 with: node-version: "22" @@ -243,26 +260,11 @@ jobs: # repo cache LRU is seeded by shared states, not arbitrary PR branches. save-if: ${{ (github.event_name == 'push' && github.ref == 'refs/heads/main') || github.event_name == 'merge_group' }} - - name: Resolve distributed sccache eligibility - id: sccache-dist-eligibility - env: - PACKAGE: ${{ matrix.package }} - run: | - case "${PACKAGE}" in - ironclaw_first_party_extension_ports|ironclaw_host_runtime|ironclaw_loop_support|ironclaw_product_workflow|ironclaw_reborn|ironclaw_reborn_cli|ironclaw_reborn_composition|ironclaw_reborn_webui_ingress|ironclaw_wasm|ironclaw_wasm_product_adapters|ironclaw_wasm_sandbox_core) - echo "enabled=false" >> "${GITHUB_OUTPUT}" - echo "::notice title=Distributed sccache skipped::${PACKAGE} compiles wasmtime-wasi proc macros that read registry source files unavailable on the remote builder." - ;; - *) - echo "enabled=true" >> "${GITHUB_OUTPUT}" - ;; - esac - - name: Setup OVH sccache uses: ./.github/actions/setup-sccache-dist with: - scheduler-url: ${{ steps.sccache-dist-eligibility.outputs.enabled == 'true' && vars.SCCACHE_DIST_SCHEDULER_URL || '' }} - auth-token: ${{ steps.sccache-dist-eligibility.outputs.enabled == 'true' && secrets.SCCACHE_DIST_AUTH_TOKEN || '' }} + scheduler-url: ${{ steps.bucket-settings.outputs.sccache_dist_enabled == 'true' && vars.SCCACHE_DIST_SCHEDULER_URL || '' }} + auth-token: ${{ steps.bucket-settings.outputs.sccache_dist_enabled == 'true' && secrets.SCCACHE_DIST_AUTH_TOKEN || '' }} cache-ssh-host: ${{ vars.SCCACHE_CACHE_SSH_HOST }} cache-ssh-user: ${{ vars.SCCACHE_CACHE_SSH_USER }} cache-ssh-port: ${{ vars.SCCACHE_CACHE_SSH_PORT }} @@ -272,18 +274,26 @@ jobs: - name: Run crate tests env: - PACKAGE: ${{ matrix.package }} + BUCKET_NAME: ${{ matrix.bucket.name }} + BUCKET_PACKAGES: ${{ toJSON(matrix.bucket.packages) }} run: | - feature_flags="${{ steps.crate-feature-flags.outputs.feature_flags }}" - - if [ "$PACKAGE" = "ironclaw_reborn_composition" ]; then - echo "Disabling incremental compilation for ironclaw_reborn_composition to keep runner disk usage bounded." - export CARGO_INCREMENTAL=0 - fi - - # shellcheck disable=SC2086 # feature_flags intentionally expands to zero or more Cargo args. - timeout --signal=INT --kill-after=30s 28m \ - cargo test -p "$PACKAGE" ${feature_flags} --all-targets -- --nocapture + echo "Running Reborn crate bucket: ${BUCKET_NAME}" + printf '%s\n' "${BUCKET_PACKAGES}" | jq -r '.[] | "- " + .' + + while IFS= read -r package; do + feature_flags="$(scripts/ci/package-feature-flags.sh "$package")" + echo "::group::cargo test -p ${package} ${feature_flags}" + + if [ "$package" = "ironclaw_reborn_composition" ]; then + echo "Disabling incremental compilation for ironclaw_reborn_composition to keep runner disk usage bounded." + export CARGO_INCREMENTAL=0 + fi + + # shellcheck disable=SC2086 # feature_flags intentionally expands to zero or more Cargo args. + timeout --signal=INT --kill-after=30s 28m \ + cargo test -p "$package" ${feature_flags} --all-targets -- --nocapture + echo "::endgroup::" + done < <(printf '%s\n' "${BUCKET_PACKAGES}" | jq -r '.[]') root-reborn-parity-tests: name: Reborn root tests (${{ matrix.partition }}) diff --git a/scripts/ci/reborn-crate-test-buckets.sh b/scripts/ci/reborn-crate-test-buckets.sh new file mode 100755 index 00000000000..8c4595db895 --- /dev/null +++ b/scripts/ci/reborn-crate-test-buckets.sh @@ -0,0 +1,123 @@ +#!/usr/bin/env bash +set -euo pipefail + +if [ "$#" -ne 1 ]; then + echo "usage: $0 " >&2 + exit 2 +fi + +packages_json="$1" + +if ! jq -e 'type == "array" and all(.[]?; type == "string")' >/dev/null 2>&1 <<< "${packages_json}"; then + echo "error: input must be a JSON array of package-name strings" >&2 + exit 1 +fi + +jq -c -n --argjson packages "${packages_json}" ' + def bucket_order: [ + "host-runtime", + "agent-runtime", + "reborn-core", + "composition-core", + "product-workflow", + "webui-ingress", + "wasm-sandbox", + "llm-mcp", + "events-conversations", + "auth-security", + "memory-skills", + "adapters-misc" + ]; + + def bucket_map: + { + ironclaw_host_runtime: "host-runtime", + + ironclaw_agent_loop: "agent-runtime", + ironclaw_approvals: "agent-runtime", + ironclaw_capabilities: "agent-runtime", + ironclaw_dispatcher: "agent-runtime", + ironclaw_host_api: "agent-runtime", + ironclaw_loop_support: "agent-runtime", + + ironclaw_reborn: "reborn-core", + ironclaw_reborn_cli: "reborn-core", + ironclaw_reborn_config: "reborn-core", + ironclaw_reborn_event_store: "reborn-core", + ironclaw_reborn_identity: "reborn-core", + ironclaw_reborn_openai_compat: "reborn-core", + + ironclaw_reborn_composition: "composition-core", + + ironclaw_product_adapter_registry: "product-workflow", + ironclaw_product_adapters: "product-workflow", + ironclaw_product_context: "product-workflow", + ironclaw_product_workflow: "product-workflow", + + ironclaw_attachments: "webui-ingress", + ironclaw_projects: "webui-ingress", + ironclaw_reborn_webui_ingress: "webui-ingress", + ironclaw_resources: "webui-ingress", + ironclaw_webui_v2: "webui-ingress", + + ironclaw_first_party_extension_ports: "wasm-sandbox", + ironclaw_first_party_extensions: "wasm-sandbox", + ironclaw_wasm: "wasm-sandbox", + ironclaw_wasm_limiter: "wasm-sandbox", + ironclaw_wasm_product_adapters: "wasm-sandbox", + ironclaw_wasm_sandbox_core: "wasm-sandbox", + + ironclaw_filesystem: "llm-mcp", + ironclaw_llm: "llm-mcp", + ironclaw_mcp: "llm-mcp", + ironclaw_network: "llm-mcp", + ironclaw_outbound: "llm-mcp", + ironclaw_process_sandbox: "llm-mcp", + ironclaw_processes: "llm-mcp", + + ironclaw_conversations: "events-conversations", + ironclaw_event_projections: "events-conversations", + ironclaw_event_streams: "events-conversations", + ironclaw_events: "events-conversations", + ironclaw_prompt_envelope: "events-conversations", + ironclaw_run_state: "events-conversations", + ironclaw_threads: "events-conversations", + ironclaw_turns: "events-conversations", + + ironclaw_auth: "auth-security", + ironclaw_authorization: "auth-security", + ironclaw_hooks: "auth-security", + ironclaw_runtime_policy: "auth-security", + ironclaw_safety: "auth-security", + ironclaw_secrets: "auth-security", + ironclaw_trust: "auth-security", + + ironclaw_extractors: "memory-skills", + ironclaw_memory: "memory-skills", + ironclaw_memory_native: "memory-skills", + ironclaw_observability: "memory-skills", + ironclaw_scripts: "memory-skills", + ironclaw_skill_learning: "memory-skills", + ironclaw_skills: "memory-skills", + + ironclaw_architecture: "adapters-misc", + ironclaw_common: "adapters-misc", + ironclaw_extensions: "adapters-misc", + ironclaw_reborn_traces: "adapters-misc", + ironclaw_slack_v2_adapter: "adapters-misc", + ironclaw_telegram_v2_adapter: "adapters-misc" + }; + + bucket_map as $bucket_map + | [ + bucket_order[]? as $bucket + | { + name: $bucket, + packages: [ + $packages[]? + | select(($bucket_map[.] // "adapters-misc") == $bucket) + ] + } + | select(.packages | length > 0) + ] +' From 28c3e944840e1ff5f8ba216ecace8f0fd0e1b7b3 Mon Sep 17 00:00:00 2001 From: "firat.sertgoz" Date: Sat, 4 Jul 2026 23:20:29 +0300 Subject: [PATCH 3/5] docs: reborn error recoverability audit + remediation plan (#5383) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Maps every reborn run error to recoverable / run-borking, analyzes PR #4841 coverage, and lays out the path to the two-bucket end state (SecurityStop | Retriable | Explainable). Headline finding: the host_runtime disposition layer intends no capability failure to abort, but the recovery strategy aborts on Dispatcher/InvalidOutput/Unknown — re-bucketing that class makes "model called a nonexistent tool" and malformed-output failures recoverable. Co-authored-by: Claude Opus 4.8 --- ...06-28-reborn-error-recoverability-audit.md | 240 ++++++++++++++++++ 1 file changed, 240 insertions(+) create mode 100644 docs/plans/2026-06-28-reborn-error-recoverability-audit.md diff --git a/docs/plans/2026-06-28-reborn-error-recoverability-audit.md b/docs/plans/2026-06-28-reborn-error-recoverability-audit.md new file mode 100644 index 00000000000..98a6c9699a9 --- /dev/null +++ b/docs/plans/2026-06-28-reborn-error-recoverability-audit.md @@ -0,0 +1,240 @@ +# Reborn Error Recoverability — Audit + Remediation Plan + +**Date:** 2026-06-28 +**Status:** discovery complete (core); two backend sweeps outstanding (see §6.4) +**Builds on:** [`docs/plans/2026-06-12-reborn-no-borking-failures.md`](2026-06-12-reborn-no-borking-failures.md) and PR #4841 (`reborn: no run-borking failures`, OPEN). + +## Goal (decided with user) + +Every reborn run error must end in one of two user-visible outcomes: + +1. **Security-related → stop the run** (deliberate, clean halt). +2. **Everything else → user-explainable OR retriable.** + +The "agent cannot cover / opaque dead-end" category must be **eliminated** by routing every failure into one of three terminal lanes. + +Two decisions locked in: + +- **Bucket 1 (SecurityStop) is minimal** — *only* injection/jailbreak detection and real secret/credential leaks halt a run. Authorization, policy, and egress denials are **not** stops; they stay recoverable or explainable. +- **Hybrid retry** — infra/lease/transient faults auto-retry silently; model/provider faults surface to the user with a retry affordance. + +--- + +## 1. The error spine (how every error is classified today) + +The agent-loop executor (`crates/ironclaw_agent_loop/src/executor.rs:99`) returns `Result`. + +- `LoopExit::Completed` = success +- `LoopExit::Blocked(LoopBlockedKind)` = **parked / resumable** (Approval / Auth / Resource / AwaitDependentRun / ExternalTool) — not a failure +- `LoopExit::Cancelled` = cancel / interrupt +- `LoopExit::Failed(LoopFailureKind)` = **graceful run-bork** (`crates/ironclaw_turns/src/loop_exit.rs:251,432`) +- `AgentLoopExecutorError` = **hard run-bork** — no trustworthy exit (`HostUnavailable`, `HostUnavailableWithDiagnostics`, `PlannerContract`(=driver bug), `CheckpointFailed`, `Cancelled`) + +### Two decision layers govern a capability (tool) failure + +**Layer 1 — host_runtime disposition** (`crates/ironclaw_host_runtime/src/lib.rs:811` `capability_failure_disposition`). Only two outcomes exist: + +- `ModelVisibleToolError` — return a tool error to the model in the same loop +- `RetrySameCall` — retry first; the loop recovery strategy owns the budget and post-exhaustion fallback + +**There is no "abort" disposition. By design, no runtime capability failure should ever end the run.** Default for any kind that is not retryable-infra and not `InvalidInput` is `ModelVisibleToolError`. Retryable-infra set = `Backend / Network / Transient / Unavailable / Internal`. + +**Layer 2 — recovery strategy class** (`crates/ironclaw_agent_loop/src/strategies/recovery.rs` `DefaultRecoveryStrategy`, max 2 retries/class; classification in `crates/ironclaw_agent_loop/src/executor/mapping.rs:131` `capability_error_class`): + +| `CapabilityFailureKind` | class | fate | +|---|---|---| +| Network, Transient | Transient | retry → **recoverable** | +| Backend, Unavailable | Unavailable | retry → **recoverable** | +| InvalidInput | InputInvalid | **recoverable** | +| MissingRuntime, OperationFailed, OutputTooLarge, Process, Resource | OperationFailed | **recoverable** | +| Authorization, GateDeclined, PolicyDenied | PolicyDenied | **recoverable** (denied) | +| Internal | Internal | retry → **recoverable** | +| **Dispatcher, Cancelled, InvalidOutput, Permanent, Unknown(_), + future** | Permanent | **ABORT → `LoopExit::Failed{CapabilityProtocolError}`** | + +### Model-call failures are *always* run-borking + +`model_error_class` (`mapping.rs:94`): transient/unavailable/internal/context-overflow → retry ×2 then Abort (`ModelError`); content-filtered → Abort; budget-approval → parked; cancelled → cancel; **Unauthorized / PolicyDenied / ScopeMismatch / StaleSurface / InvalidInvocation / Invalid / CheckpointRejected / TranscriptWriteFailed → None → hard bork** (`HostUnavailableWithDiagnostics{Model}`), invisible to the model. A model failure can never become a recoverable tool error. + +--- + +## 2. Complete classification map + +**Recoverable (loop continues, agent adapts):** all tool/capability failures by design — bad args, policy/authorization denials, WASM fuel/trap/memory/timeout, MCP timeout/protocol/session-loss, process non-zero-exit/timeout/OOM, filesystem permission/size/path, network DNS/TLS/denied-domain, transient model errors (retried in-loop to `Completed`). The tool backends (`mcp`, `wasm*`, `processes`, `process_sandbox`, `filesystem`, `network`, `outbound`, `secrets`, `resources`) are **clean** — no `panic`/`unwrap`/`expect` on runtime input; every recoverable trigger is a structured `Result::Err` → `Failed` outcome. + +**Run-bork (graceful, `LoopExit::Failed`):** capability `Permanent` class or 8-retry ceiling → `CapabilityProtocolError`; all model errors after retry → `ModelError`; iteration limit (32) → `IterationLimit`; no-progress/stuck → `NoProgressDetected`; ≥3 rejected replies → `InvalidModelOutput`; compaction failure → `CompactionUnavailable`; `SpawnedProcess` unsupported → `CapabilityProtocolError`. + +**Run-bork (hard, `AgentLoopExecutorError`):** any host-port call fails; model class=None; planner/driver contract violations; checkpoint serialize/write; safe-summary validation reject; malformed sandbox-plan; the 4 `unreachable!` index-key constants (regression-only); Postgres `row.get` schema drift. + +**Parked / cancel (not failures):** gate outcomes (resume_turn); host cancel/interrupt. + +`ContextBuildFailed` is **dead** (verified — no construction site); prompt-build failures hard-bork as `HostUnavailable{Prompt}` instead. + +--- + +## 3. PR #4841 coverage (OPEN, "Part 1") + +#4841 delivers the **explainable + retryable backbone** — in the two-bucket model, that *is* bucket 2 for the graceful-failure surface. + +- **Tier-1 model explanation** for model-reachable failures (CapabilityProtocolError, IterationLimit, PolicyDenied, NoProgressDetected, CompactionUnavailable, InvalidModelOutput) — one best-effort no-capability model call before failing. +- **Tier-2 deterministic templates** — `FailureExplanationProvider` covers *every* `LoopFailureKind` + reborn failure category + `host_stage_unavailable:*` (exhaustive-match test). +- **Failed exits carry evidence** (explanation refs, partial assistant refs, diagnostics, `safe_summary` category). +- **Retry-from-failed** — `retry_turn` in both store backends, new run from last resumable checkpoint, idempotent, `webui_v2` retry endpoint, `retryable` flag on the wire. +- **HostUnavailable → categorized retryable failure** (`host_stage_unavailable:`), no longer opaque. + +It does **not** touch `ironclaw_llm`, `ironclaw_processes`, `outbound_delivery`, `process_sandbox`, `ironclaw_safety`, or `turn_scheduler.rs`. + +### Per-case status + +| §4 cannot-cover case | #4841 | +|---|---| +| HostUnavailable infra, lease expiry, checkpoint/transcript, scope/surface, driver-bug, model PolicyDenied | **Covered (explainable; retryable if checkpoint)** | +| model bad key / ModelNotAvailable | **Partial** — surfaced+retryable, but root mis-classification in `ironclaw_llm` remains | +| safe-summary kill | **Partial** — explainable+retryable, not prevented | +| resumed-checkpoint-gone, projection split-brain | **Partial — verify** | +| Codex truncated-stream, detached bg-process | **Not covered** (`ironclaw_llm` / `ironclaw_processes` untouched) | +| evidence stores unwired | **Softened, not fixed** | + +--- + +## 4. The "retriable" gap + +#4841 is **not** fully retriable per the hybrid decision: + +1. **Auto half missing** — all retry is *user-initiated* (the endpoint / `retry_turn`). No silent auto re-drive; `turn_scheduler.rs` is untouched. Everything we marked Auto-retriable (infra `HostUnavailable`, checkpoint/transcript, lease loss, scope/surface) is only manually retryable. +2. **Checkpoint-gated** — retryable *only* when a `BeforeModel`/`BeforeBlock` checkpoint exists (`crates/ironclaw_reborn/src/planned_driver.rs:393`). `BeforeSideEffect` and `Final` are not resumable. Failures before the first `BeforeModel` checkpoint (`crates/ironclaw_agent_loop/src/executor/canonical.rs:111`) — input drain, context/prompt build, surface build, compaction — are explainable but **not retriable at all**. +3. **Codex truncation** — not even detected as a failure, so nothing to retry. + +### No-checkpoint user journey + the from-input gap + +The original input **is durable** — `SubmitTurnRequest.accepted_message_ref` (`crates/ironclaw_turns/src/request.rs:58`). So a from-scratch retry is feasible without asking the user to re-type. Two populations: + +- **(a) Pre-first-checkpoint failures** — nothing durable happened. A from-input re-drive (seed a new run from `accepted_message_ref` when no resumable checkpoint exists) is **safe and cheap** and makes essentially every early failure retryable via the same affordance. **This is a near-free win.** +- **(b) Side-effect-only / final-only failures** — a side effect may have run; blind re-drive double-executes it. This is the genuine open decision: (i) make side-effect dispatch idempotent and add `BeforeSideEffect` to the resumable set; or (ii) explicit user-confirmed "resend (may repeat actions)"; or (iii) explainable-only. + +Today's no-retry journey is a dead end: explanation shown, `retryable: false`, no retry button (copy-mismatch risk: Tier-2 says "Retry the run" while the flag is false), user must manually re-send → new turn, lost continuity. + +--- + +## 5. Residue after #4841 + Part-2 (still neither retriable nor explainable) + +1. **Pre-run failures** — boot/config (operator-explainable via logs only) and API-ingress/submission rejections (HTTP-explainable to the client; the agent never sees them; no run to retry). +2. **SecurityStop** — explainable but deliberately not retriable (correct by the minimal-bucket-1 decision). +3. **Explainable-but-futile** — driver bugs (generic sentence + telemetry; retry re-hits the bug), genuinely-permanent capability failures, and the §6 recoverable→bork defects (retry re-fails). +4. **Out-of-run-loop surfaces** — proactive work (triggers, heartbeat, routines) and projection/SSE/UI-convergence have their own failure handling, not the run-retry/explain machinery. +5. **Persistence floor** — a sustained store outage means even *recording* the explainable/retryable failure can't land. + +--- + +## 6. The recoverable→bork defect hunt (the high-leverage fix) + +**Premise:** turn model-fixable failures into agent-recoverable tool errors so ~90% of failures keep the run alive. There are three shapes. + +### 6.1 Keystone — the table contradicts the disposition layer (verified) + +Layer 1 (host_runtime) dispositions `Dispatcher`, `InvalidOutput`, and `Unknown` as `ModelVisibleToolError` (intended recoverable). Layer 2 (`capability_error_class`, `mapping.rs:155-162`) maps the same three to `Permanent` → **Abort**. `RuntimeFailureKind` has **no `Permanent` variant** (`crates/ironclaw_host_runtime/src/lib.rs:698`), so the abort class is reached *only* through these three (plus `Cancelled`, which is legitimately a cancel). + +Notably, `DispatchFailureKind::UnknownCapability` / `UnknownProvider` → `RuntimeFailureKind::InvalidOutput` (`crates/ironclaw_host_runtime/src/production.rs:2113`), so **"model called a nonexistent tool" → InvalidOutput → Permanent → run dies**, when it should be a model-visible `Failed{InvalidInput}` ("no such tool, choose another"). + +There's also an internal tell: `SameCallRetryConstraint` (`crates/ironclaw_agent_loop/src/executor/capability_helpers.rs:388`) marks `Dispatcher` = `Allowed` and `InvalidOutput` = `RequiresChangedInput` (i.e. retry/adapt makes sense) while the recovery class aborts — the two layers disagree. + +**Fix (single, high-leverage):** in `capability_error_class`, move `Dispatcher`, `InvalidOutput`, and `Unknown(_)` out of `Permanent` into a recoverable class (`OperationFailed` → `ToolErrorResult`), aligning Layer 2 with Layer 1's `ModelVisibleToolError` intent. Keep only genuinely-terminal sources (none from the runtime path) and `Cancelled` (handled as cancel) as abort. This converts the entire "nonexistent tool / malformed output / unknown failure" population from run-ending to recoverable in one change. + +### 6.2 Handler-level Err-instead-of-Ok(Failed) — Invariant 1 (confirmed) + +A handler `invoke` returning `Err(AgentLoopHostError)` is mapped by `capability_host_error` (`mapping.rs:117`) to terminal `HostUnavailable{Capability}` — every non-`Cancelled` kind kills the run. Per `.claude/rules/agent-loop-capabilities.md`, model-fixable conditions must be `Ok(CapabilityOutcome::Failed/Denied)`. + +**Confirmed defects:** + +| Site | Condition | Current | Fix | +|---|---|---|---| +| `crates/ironclaw_reborn_composition/src/runtime/local_dev/outbound_delivery.rs:108,213` (via `outbound_delivery_host_error`, :575) | model picks bad/nonexistent `target_id` (InvalidRequest/NotFound), forbidden, conflict, rate-limited, transient-unavailable | maps **all** `RebornServicesErrorCode` → `Err` → terminal | mirror `project_service_outcome`: InvalidRequest/NotFound→`Failed{InvalidInput}`; Unauthenticated/Forbidden→`Denied`; Conflict→`Failed{OperationFailed}`; RateLimited→`Failed{Resource}`; Unavailable→`Failed{Unavailable}`; only Internal→`Err` | +| `outbound_delivery.rs:223` | model-supplied `target_id` interpolated into `safe_summary` | `format!("set delivery target to {target_id}")` — a delimiter in the id trips validation → terminal | fixed host-authored string; id travels in `output` | +| `outbound_delivery.rs:208,340-351,392-394` (via `approval_lease_error`) | expired/lost approval lease, not-yet-approved gate | → `Unauthorized` `Err` (terminal) | route lease-state arms → `Ok(Denied)` (re-request); keep Persistence/CAS as `Err` | +| `crates/ironclaw_host_runtime/src/production.rs:1990,1995` (`host_runtime_spawn_input_for_capability`) | malformed model-supplied `SandboxProcessPlan` | `HostRuntimeError::invalid_request` → `InvalidInvocation` → terminal | `Ok(CapabilityOutcome::Failed{InvalidInput})` (locked in by a test at `loop_support/.../capability_port.rs:6655` — update it) | + +The exemplars to copy: `skill_activation.rs` (`skill_activation_selection_outcome`) and `project_create.rs` (`project_service_outcome`) — recoverable codes → `Ok(Failed)`, only `Internal` → `Err`. + +### 6.3 Provider fidelity (model path) — accurate explainability + +| Site | Condition | Current | Fix | +|---|---|---|---| +| `crates/ironclaw_llm/src/rig_adapter.rs:1019` (`map_rig_error`) | 401/403 auth on a turn (OpenAI/Anthropic/Ollama/Tinfoil/openai_compatible) | only context-length special-cased; everything else → `RequestFailed` → generic `Unavailable` → retried then bork as "model unavailable" | detect auth → `AuthFailed`/credentials category so the user is told to fix the key | +| `crates/ironclaw_llm/src/bedrock.rs:700` | AccessDenied / Throttling / ValidationException(overflow) | all → `RequestFailed` | map to AuthFailed / RateLimited / ContextLengthExceeded | +| Codex (`openai_codex_provider.rs:830`, `codex_chatgpt.rs:673`) | truncated SSE stream / dropped `error`/`response.failed` events | silent partial success labeled `Stop` | detect incomplete stream → `InvalidResponse`/`Unavailable` (retryable); map SSE error events | +| Codex/Bedrock/Copilot/Anthropic-OAuth | context overflow | no 413 detection → no `ShrinkContext` | detect 413/context-overflow → `ContextLengthExceeded` | + +### 6.4 Outstanding sweeps (not completed — finish these) + +Two parallel read-only sweeps were started and interrupted; finish them with the §6.1 rubric: + +- **host_runtime + dispatcher variant-by-variant**: audit `failure_kind_from` (`production.rs:2068`), `From` (`production.rs:2113`), `RuntimeDispatchErrorKind` / `DispatchFailureKind` / `CapabilityInvocationError` variant lists. For each model-fixable variant, confirm it doesn't land in the abort set or in an infra-retry kind it shouldn't. (`MethodMissing`, `UndeclaredCapability`, `OutputDecode`, `InvalidResult`, `InputEncode` are the suspects.) +- **tool backends + extensions**: `mcp`, `wasm*`, `processes`, `process_sandbox`, `filesystem`, `network`, `outbound`, `first_party_extensions`, `product_adapters`, `skills` — any model-fixable backend failure (unknown method, bad args, malformed output, denied target) that surfaces as a hard `Err` or lands in `InvalidOutput`/`Dispatcher`/`Unknown` (abort set). + +--- + +## 7. Remediation plan — target architecture + +Collapse every terminal failure into three lanes via one classifier: + +| Lane | Retry policy | Membership | +|---|---|---| +| **SecurityStop** | None (clean halt) | *only* injection/jailbreak + real leak (safety layer / leak detector) | +| **Retriable** | `AutoBounded` silent re-drive; exhaustion → Explainable | infra host-port faults, lease loss/runner death, checkpoint/transcript, projection, scope/surface | +| **Explainable** | `UserInitiated` (retry affordance) or `None+restart` | model/provider faults, config/credentials, driver bugs, lost checkpoint | + +### Building blocks (dependency order) + +1. **`RunFailureReason` taxonomy** — wire-stable, user-facing, distinct from internal `LoopFailureKind`; carries `{lane, retry_policy, user_message, correlation_id}`. (#4841's `FailureExplanationProvider` + `safe_summary` category is most of this.) +2. **One exhaustive-match classifier** at the run boundary (`crates/ironclaw_reborn/src/planned_driver.rs` + `turn_run_executor.rs`). Every failure passes through it; a new kind without classification fails to compile. +3. **Re-bucket the `Permanent` class** (§6.1) — the highest-leverage single change. +4. **Fix handler Err sites** (§6.2) + provider fidelity (§6.3). +5. **Scheduler auto re-drive** for the Retriable lane (the missing "Auto" half) — consumes #4841's `retry_turn`, bounded, exhaustion → Explainable. +6. **From-input retry** (§4 population a) — seed a new run from `accepted_message_ref` when no resumable checkpoint exists. +7. **Defense-in-depth degradation** — safe-summary/validation failures fall back to a fixed summary (recoverable), never bork. +8. **Reconcilers** — stuck-`Running` bg-process; lease reclaim → requeue. +9. **Fail-fast composition** — required evidence stores; loud startup failure if unwired. + +### Sequencing + +1. **#6.1 table re-bucket** (small, high-leverage, makes most capability failures recoverable). +2. **#6.2 handler fixes + #6.3 provider fidelity** (accurate recoverable/explainable). +3. **Block 1+2 classifier keystone** (enforce the three-lane invariant). +4. **Retriable lane**: #5 auto re-drive + #6 from-input retry + lease requeue + projection self-heal. +5. **Prevention/security**: #7 degradation, #9 fail-fast, the SecurityStop set, the enforcement test. + +### Enforcement + +A test asserting: every `RunFailureReason` resolves to `SecurityStop` **only if** sourced from the safety/leak layer; everything else is `Retriable` or `Explainable` with a non-empty `user_message`. A new variant without classification fails to compile; a new `SecurityStop` outside the safety layer fails the test. + +--- + +## 8. Open decisions + +1. **Population (b) side-effect retry** (§4) — idempotent dispatch + resumable `BeforeSideEffect`, vs user-confirmed resend, vs explainable-only. +2. **Pre-run / ingress failures** (§5.1) — surface into the same user-facing taxonomy, or accept the HTTP/log boundary for those. + +--- + +## Appendix — key code locations + +| Concern | Location | +|---|---| +| Disposition layer (no-abort intent) | `crates/ironclaw_host_runtime/src/lib.rs:811` (`capability_failure_disposition`), enum `:745`, `RuntimeFailureKind` `:698` | +| Recovery classes | `crates/ironclaw_agent_loop/src/executor/mapping.rs:131` (`capability_error_class`), `:94` (`model_error_class`), `:166` | +| Recovery strategy | `crates/ironclaw_agent_loop/src/strategies/recovery.rs` (`DefaultRecoveryStrategy`) | +| Same-call retry hint | `crates/ironclaw_agent_loop/src/executor/capability_helpers.rs:388` | +| Terminal mapper (Err→HostUnavailable) | `crates/ironclaw_agent_loop/src/executor/mapping.rs:117` (`capability_host_error`) | +| LoopExit / LoopFailureKind | `crates/ironclaw_turns/src/loop_exit.rs:251,432` | +| AgentLoopExecutorError | `crates/ironclaw_agent_loop/src/executor.rs:99` | +| Runtime→loop kind map | `crates/ironclaw_loop_support/src/capability_port.rs:2570` (`runtime_failure_kind_to_loop`) | +| Dispatch kind map | `crates/ironclaw_host_runtime/src/production.rs:2068` (`failure_kind_from`), `:2113` (`From`) | +| outbound_delivery defect | `crates/ironclaw_reborn_composition/src/runtime/local_dev/outbound_delivery.rs:108,213,223,575` | +| sandbox-plan defect | `crates/ironclaw_host_runtime/src/production.rs:1990` | +| Provider fidelity | `crates/ironclaw_llm/src/rig_adapter.rs:1019`, `bedrock.rs:700`, `openai_codex_provider.rs:830`, `codex_chatgpt.rs:673` | +| Model-error path | `crates/ironclaw_agent_loop/src/executor/model.rs:125` | +| Resumable checkpoint kinds | `crates/ironclaw_reborn/src/planned_driver.rs:393`; first checkpoint `canonical.rs:111` | +| Durable turn input | `crates/ironclaw_turns/src/request.rs:58` (`accepted_message_ref`) | +| Lease expiry | `crates/ironclaw_turns/src/memory.rs` (`recover_expired_leases`) | +| Exemplar recoverable handlers | `crates/ironclaw_reborn_composition/src/runtime/local_dev/{skill_activation,project_create}.rs` | + +🤖 Generated with [Claude Code](https://claude.com/claude-code) From 3c54f13fa7b4f8ff205353f635c4bc0ae0906764 Mon Sep 17 00:00:00 2001 From: abbyshekit <153240993+abbyshekit@users.noreply.github.com> Date: Sat, 4 Jul 2026 17:34:55 -0400 Subject: [PATCH 4/5] fix(agent-loop): admit one-line answers that name a __-tool (F3) (#5042) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A genuine single-line final answer that mentions a `__`-bearing tool (e.g. "Tool result from web__fetch: near.ai returned 200 OK") was misclassified as a replayed provider-transcript artifact and rejected, so the real answer vanished. Only treat MULTI-line content as a transcript artifact; a weak model echoing replayed history reproduces multiple lines, which is still caught. Reject tests updated to multi-line; added a single-line accept test. (F1 stalled-turn recovery intentionally NOT included: gating the nudge on any tool call over-fires on pure-failure runs, and the target 3B hang is the runner queue not draining — a separate P0, not a no-reply-at-exit. Doing F1 right needs a successful-tool-call signal that state does not yet track.) cargo test -p ironclaw_agent_loop: 317 passed, 0 failed. Co-authored-by: Abhishek Vaidyanathan Co-authored-by: Claude Opus 4.8 (1M context) --- .../ironclaw_agent_loop/src/executor/tests.rs | 6 ++- .../src/strategies/reply_admission.rs | 40 ++++++++++++++++++- 2 files changed, 43 insertions(+), 3 deletions(-) diff --git a/crates/ironclaw_agent_loop/src/executor/tests.rs b/crates/ironclaw_agent_loop/src/executor/tests.rs index 859ea17d9b8..81b4ab4feff 100644 --- a/crates/ironclaw_agent_loop/src/executor/tests.rs +++ b/crates/ironclaw_agent_loop/src/executor/tests.rs @@ -1367,7 +1367,11 @@ async fn repeated_reply_rejections_stop_as_invalid_model_output() { #[tokio::test] async fn default_reply_admission_rejects_tool_history_echo_and_continues() { let host = MockHost::new(vec![ - reply_response_with_text("Previous tool event: demo__echo was invoked."), + // Multi-line, all provider-transcript-artifact lines: a replayed-history + // echo (still rejected under the multi-line artifact rule). + reply_response_with_text( + "Previous tool event: demo__echo was invoked.\nTool result from demo__echo: hi", + ), reply_response_with_text("done"), ]); let executor = CanonicalAgentLoopExecutor; diff --git a/crates/ironclaw_agent_loop/src/strategies/reply_admission.rs b/crates/ironclaw_agent_loop/src/strategies/reply_admission.rs index 01361c51fb9..e5dd1289868 100644 --- a/crates/ironclaw_agent_loop/src/strategies/reply_admission.rs +++ b/crates/ironclaw_agent_loop/src/strategies/reply_admission.rs @@ -59,7 +59,22 @@ impl ReplyAdmissionStrategy for DefaultReplyAdmissionStrategy { } fn is_non_final_reply_artifact(content: &str) -> bool { - content.trim().is_empty() || is_only_provider_transcript_artifact_lines(content) + let trimmed = content.trim(); + if trimmed.is_empty() { + return true; + } + // Only treat content as a replayed provider-transcript artifact when it is + // MULTI-line. A genuine one-line answer that merely names a `__`-bearing tool + // (e.g. "Tool result from web__fetch: near.ai returned 200 OK") is a real + // final reply, not an echo of provider history — rejecting it on the + // single-line shape was eating valid answers. A weak model echoing replayed + // transcript reproduces multiple history lines, which this still catches. + trimmed + .lines() + .filter(|line| !line.trim().is_empty()) + .count() + > 1 + && is_only_provider_transcript_artifact_lines(content) } pub(crate) fn reply_admission_control_message( @@ -124,10 +139,14 @@ mod tests { #[tokio::test] async fn default_reply_admission_rejects_flattened_tool_history_echo() { + // A weak model echoing replayed provider history reproduces MULTIPLE + // transcript lines — that multi-line, all-artifact shape is still rejected. let context = test_run_context("default-reply-admission-tool-history"); let state = LoopExecutionState::initial_for_run(&context); let reply = AssistantReply { - content: "Previous tool event: demo__echo was invoked.".to_string(), + content: + "Previous tool event: demo__echo was invoked.\nTool result from demo__echo: hi" + .to_string(), }; let outcome = DefaultReplyAdmissionStrategy @@ -137,6 +156,23 @@ mod tests { assert!(matches!(outcome, ReplyAdmissionOutcome::RejectFinal { .. })); } + #[tokio::test] + async fn default_reply_admission_accepts_single_line_answer_naming_a_tool() { + // F3: a one-line final answer that names a `__`-bearing tool is a real + // reply, not a replayed-transcript artifact — it must be admitted. + let context = test_run_context("default-reply-admission-single-line-tool"); + let state = LoopExecutionState::initial_for_run(&context); + let reply = AssistantReply { + content: "Tool result from web__fetch: near.ai returned 200 OK.".to_string(), + }; + + let outcome = DefaultReplyAdmissionStrategy + .admit_reply(&state, &reply) + .await; + + assert_eq!(outcome, ReplyAdmissionOutcome::AcceptFinal); + } + #[tokio::test] async fn default_reply_admission_accepts_reply_that_mentions_tool_history_in_context() { let context = test_run_context("default-reply-admission-tool-history-context"); From 209da7d34c0b0a434f6c9f77e2b8532f93c5ed94 Mon Sep 17 00:00:00 2001 From: Illia Polosukhin Date: Sun, 5 Jul 2026 00:34:46 +0100 Subject: [PATCH 5/5] =?UTF-8?q?feat(migration):=20v1/engine-v2=20=E2=86=92?= =?UTF-8?q?=20Reborn=20state=20migration=20tool=20(#5627)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(migration): add v1/engine-v2 → Reborn state migration tool New crate `ironclaw_reborn_migration` (library + `ironclaw-reborn-migration` binary) that converts legacy IronClaw v1 and engine-v2 persisted state into the Reborn state substrate, recording every unconvertible value in a machine-readable manifest so nothing is silently dropped. Reads a v1 database (PostgreSQL or libSQL) via the root `ironclaw` crate — which is also where engine-v2 mission/project state lives, as JSON blobs in `memory_documents` — and writes through Reborn domain stores built directly over a RootFilesystem / triggers DB. Converts: - conversations + messages → session threads (ids/timestamps preserved) - cron routines + cron missions → TriggerRecords; mission threads under ThreadScope.mission_id - memory documents → ironclaw_memory documents - secrets → decrypt via v1 store, re-encrypt via Reborn SecretStore - user/channel identities → adopt_migrated_identity - installed wasm tools/channels → ExtensionInstallation (placeholder manifest) Gaps recorded in the manifest (no Reborn target): event/webhook/manual trigger sources, non-cron mission cadences, routine guardrails/notify/run-history, mission-only fields, jobs, settings, memory versions, heartbeat, extension manifest fidelity + WASM binary, pairing requests. Adds a `migration-support` feature + `extension_installation_store_for_migration` seam to `ironclaw_reborn_composition` (ships zero bytes without the feature), and read-side channel-store re-exports in `src/channels/wasm`. Acceptance test (`tests/migration_roundtrip.rs`, libSQL, Docker-free) seeds a rich v1+engine-v2 fixture, runs the migration, and asserts converted counts, the exact gap set, triggers read back through the public repo, and on-disk durability of thread/secret/extension documents. A dry-run case asserts full reporting with no writes. Follow-up (documented, deferred per scope): wire run_migration into `ironclaw-reborn` startup. Co-Authored-By: Claude Opus 4.8 * fix(migration): address PR #5627 review — fail-loud on silent drops, TLS, secret redaction Addresses the CodeRabbit/Copilot/Gemini review on the v1→Reborn migration tool. Core theme: the crate's "nothing is silently dropped" contract was not fully upheld, plus a few security/cleanup gaps. Data integrity (silent drops → recorded losses or propagated errors): - source: `distinct_user_ids_in` now tolerates only a missing table; every other connect/query/row-decode failure propagates instead of being swallowed as "0 users" (would have silently dropped every record keyed to an undiscovered user). Adds an explicit user-id column arg — the `users` table keys on `id`, not `user_id` (the old code silently ate the "no such column" error). - automations: unparseable engine-thread blobs and parseable-but-orphaned threads (referenced by no mission) are recorded; dry-run mission threads now record the same per-message losses as the real write path. - extensions: `tool_credential_bindings` propagates capability read errors and records unconvertible secret names; the synthesized ExtensionInstallationId is scoped by owner so per-user same-named installs no longer overwrite each other. - identities: invalid user id in the OAuth path is recorded (matching the channel path); `read_channel_identities` tolerates only a missing table and records malformed rows. - secrets: the expiry re-read failure is recorded, not dropped via `.ok()`. - bad-user-id is now a per-item loss + skip (not a run abort) in threads, secrets, automations (routines + missions), and memory. Security: - target Postgres now enforces the repo's remote-TLS rule (reject sslmode=disable for remote hosts; rustls via ironclaw::db::tls) instead of always using NoTls — migration traffic carries decrypted secrets. - Postgres URLs are held as SecretString (SourceDb/TargetStore) and the `Cli` struct drops its `Debug` derive so creds/keys can't leak via `{:?}`. Cleanup: - mounts map the system-scope sentinel to `__system__` (mirrors production invocation_mount_view) so system-scoped service ops resolve correctly. - v2_model `now()` → `epoch_fallback()` (it returns the epoch, not "now"). - `V1Source` is now pub(crate); secrets `migrate_one` carries the required arch-exempt annotation; CLAUDE.md example drops the absolute /tmp path. - add a `ironclaw_reborn_migration` BoundaryRule so the v1↔Reborn bridge can't grow direct deps on the serving/runtime/engine layers. Tests: the roundtrip acceptance test now asserts the exact per-domain gap set (summing to the whole report so a newly-dropped domain fails the build), seeds tool capabilities to exercise credential-binding + Enabled activation, and asserts extension/identity idempotency on re-run. Declined (replied on the PR): error `domain` String→enum (field carries freeform context, not a strict domain); redacting LossyItem source_id (operator-only tool needs actionable ids, no secret values exposed); gemini `expose_secret` (false positive — DecryptedSecret::expose exists). Co-Authored-By: Claude Opus 4.8 * fix(migration): address CodeRabbit re-review — mission next_run_at, narrower table-missing, shared user-id helper Follow-up to the review round on PR #5627: - automations: a mission with no `next_fire_at` no longer falls back to `created_at` (which can be the `epoch_fallback` synthesized for a drifted blob → a 1970-dated, immediately-due trigger). The trigger's `next_run_at` is now synthesized to the migration time and recorded as a `Degraded` loss (`mission_next_run_at`). - source: `is_missing_table_error` is narrowed to require `relation` alongside `does not exist`, so a PostgreSQL *column*-not-found (`column "…" does not exist`) is no longer downgraded to an empty user set — keeping the exact silent-drop class #21 guards against. - report: add a shared `MigrationReport::valid_user_id` helper and route the six duplicated "validate UserId → record loss → skip" sites (identities x2, secrets, memory, routines, missions) through it so the shape can't drift. - test: assert the serialized `"enabled"` activation token (and absence of `"disabled"`) instead of a loose lowercase substring; bump the exact Mission gap count to 4 (daily-digest now records the synthesized next_fire_at). Co-Authored-By: Claude Opus 4.8 --------- Co-authored-by: Claude Opus 4.8 --- Cargo.lock | 36 + Cargo.toml | 2 +- FEATURE_PARITY.md | 12 + .../tests/reborn_dependency_boundaries.rs | 30 + crates/ironclaw_reborn_composition/Cargo.toml | 4 + .../src/factory.rs | 31 + crates/ironclaw_reborn_composition/src/lib.rs | 2 + crates/ironclaw_reborn_migration/CLAUDE.md | 111 +++ crates/ironclaw_reborn_migration/Cargo.toml | 92 +++ .../src/convert/automations.rs | 560 +++++++++++++++ .../src/convert/extensions.rs | 444 ++++++++++++ .../src/convert/heartbeat.rs | 34 + .../src/convert/identities.rs | 274 ++++++++ .../src/convert/jobs.rs | 41 ++ .../src/convert/memory.rs | 102 +++ .../src/convert/mod.rs | 12 + .../src/convert/secrets.rs | 189 +++++ .../src/convert/settings.rs | 47 ++ .../src/convert/threads.rs | 308 ++++++++ crates/ironclaw_reborn_migration/src/error.rs | 34 + crates/ironclaw_reborn_migration/src/lib.rs | 51 ++ crates/ironclaw_reborn_migration/src/main.rs | 156 +++++ .../ironclaw_reborn_migration/src/mounts.rs | 75 ++ .../ironclaw_reborn_migration/src/options.rs | 49 ++ .../ironclaw_reborn_migration/src/report.rs | 155 ++++ .../ironclaw_reborn_migration/src/source.rs | 150 ++++ .../ironclaw_reborn_migration/src/target.rs | 323 +++++++++ .../ironclaw_reborn_migration/src/v2_model.rs | 191 +++++ .../tests/migration_roundtrip.rs | 660 ++++++++++++++++++ src/channels/wasm/mod.rs | 7 + 30 files changed, 4181 insertions(+), 1 deletion(-) create mode 100644 crates/ironclaw_reborn_migration/CLAUDE.md create mode 100644 crates/ironclaw_reborn_migration/Cargo.toml create mode 100644 crates/ironclaw_reborn_migration/src/convert/automations.rs create mode 100644 crates/ironclaw_reborn_migration/src/convert/extensions.rs create mode 100644 crates/ironclaw_reborn_migration/src/convert/heartbeat.rs create mode 100644 crates/ironclaw_reborn_migration/src/convert/identities.rs create mode 100644 crates/ironclaw_reborn_migration/src/convert/jobs.rs create mode 100644 crates/ironclaw_reborn_migration/src/convert/memory.rs create mode 100644 crates/ironclaw_reborn_migration/src/convert/mod.rs create mode 100644 crates/ironclaw_reborn_migration/src/convert/secrets.rs create mode 100644 crates/ironclaw_reborn_migration/src/convert/settings.rs create mode 100644 crates/ironclaw_reborn_migration/src/convert/threads.rs create mode 100644 crates/ironclaw_reborn_migration/src/error.rs create mode 100644 crates/ironclaw_reborn_migration/src/lib.rs create mode 100644 crates/ironclaw_reborn_migration/src/main.rs create mode 100644 crates/ironclaw_reborn_migration/src/mounts.rs create mode 100644 crates/ironclaw_reborn_migration/src/options.rs create mode 100644 crates/ironclaw_reborn_migration/src/report.rs create mode 100644 crates/ironclaw_reborn_migration/src/source.rs create mode 100644 crates/ironclaw_reborn_migration/src/target.rs create mode 100644 crates/ironclaw_reborn_migration/src/v2_model.rs create mode 100644 crates/ironclaw_reborn_migration/tests/migration_roundtrip.rs diff --git a/Cargo.lock b/Cargo.lock index 7bd005a2aad..3d297a1b2da 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -5047,6 +5047,42 @@ dependencies = [ "uuid", ] +[[package]] +name = "ironclaw_reborn_migration" +version = "0.1.0" +dependencies = [ + "anyhow", + "async-trait", + "chrono", + "clap", + "deadpool-postgres", + "ironclaw", + "ironclaw_common", + "ironclaw_extensions", + "ironclaw_filesystem", + "ironclaw_host_api", + "ironclaw_host_runtime", + "ironclaw_memory", + "ironclaw_memory_native", + "ironclaw_reborn_composition", + "ironclaw_reborn_identity", + "ironclaw_secrets", + "ironclaw_threads", + "ironclaw_triggers", + "libsql", + "secrecy", + "serde", + "serde_json", + "tempfile", + "thiserror 2.0.18", + "tokio", + "tokio-postgres", + "tracing", + "tracing-subscriber", + "ulid", + "uuid", +] + [[package]] name = "ironclaw_reborn_openai_compat" version = "0.1.0" diff --git a/Cargo.toml b/Cargo.toml index 7bc34f65a19..c69cbb262b9 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,5 +1,5 @@ [workspace] -members = [".", "crates/ironclaw_common", "crates/ironclaw_observability", "crates/ironclaw_host_api", "crates/ironclaw_filesystem", "crates/ironclaw_attachments", "crates/ironclaw_extractors", "crates/ironclaw_memory", "crates/ironclaw_memory_native", "crates/ironclaw_events", "crates/ironclaw_event_projections", "crates/ironclaw_event_streams", "crates/ironclaw_reborn_event_store", "crates/ironclaw_extensions", "crates/ironclaw_processes", "crates/ironclaw_dispatcher", "crates/ironclaw_scripts", "crates/ironclaw_process_sandbox", "crates/ironclaw_mcp", "crates/ironclaw_wasm", "crates/ironclaw_wasm_sandbox_core", "crates/ironclaw_wasm_limiter", "crates/ironclaw_capabilities", "crates/ironclaw_secrets", "crates/ironclaw_network", "crates/ironclaw_host_runtime", "crates/ironclaw_runtime_policy", "crates/ironclaw_authorization", "crates/ironclaw_run_state", "crates/ironclaw_approvals", "crates/ironclaw_resources", "crates/ironclaw_auth", "crates/ironclaw_trust", "crates/ironclaw_turns", "crates/ironclaw_agent_loop", "crates/ironclaw_threads", "crates/ironclaw_prompt_envelope", "crates/ironclaw_hooks", "crates/ironclaw_loop_support", "crates/ironclaw_reborn", "crates/ironclaw_reborn_config", "crates/ironclaw_reborn_composition", "crates/ironclaw_reborn_identity", "crates/ironclaw_first_party_extensions", "crates/ironclaw_reborn_cli", "crates/ironclaw_reborn_traces", "crates/ironclaw_reborn_webui_ingress", "crates/ironclaw_reborn_openai_compat", "crates/ironclaw_conversations", "crates/ironclaw_product_adapters", "crates/ironclaw_product_context", "crates/ironclaw_product_workflow", "crates/ironclaw_product_adapter_registry", "crates/ironclaw_wasm_product_adapters", "crates/ironclaw_telegram_v2_adapter", "crates/ironclaw_slack_v2_adapter", "crates/ironclaw_outbound", "crates/ironclaw_triggers", "crates/ironclaw_projects", "crates/ironclaw_architecture", "crates/ironclaw_safety", "crates/ironclaw_skills", "crates/ironclaw_oauth", "crates/ironclaw_llm", "crates/ironclaw_embeddings", "crates/ironclaw_gateway", "crates/ironclaw_tui", "crates/ironclaw_webui_v2", "crates/ironclaw_skill_learning", "tools/ironclaw_stress"] +members = [".", "crates/ironclaw_common", "crates/ironclaw_observability", "crates/ironclaw_host_api", "crates/ironclaw_filesystem", "crates/ironclaw_attachments", "crates/ironclaw_extractors", "crates/ironclaw_memory", "crates/ironclaw_memory_native", "crates/ironclaw_events", "crates/ironclaw_event_projections", "crates/ironclaw_event_streams", "crates/ironclaw_reborn_event_store", "crates/ironclaw_extensions", "crates/ironclaw_processes", "crates/ironclaw_dispatcher", "crates/ironclaw_scripts", "crates/ironclaw_process_sandbox", "crates/ironclaw_mcp", "crates/ironclaw_wasm", "crates/ironclaw_wasm_sandbox_core", "crates/ironclaw_wasm_limiter", "crates/ironclaw_capabilities", "crates/ironclaw_secrets", "crates/ironclaw_network", "crates/ironclaw_host_runtime", "crates/ironclaw_runtime_policy", "crates/ironclaw_authorization", "crates/ironclaw_run_state", "crates/ironclaw_approvals", "crates/ironclaw_resources", "crates/ironclaw_auth", "crates/ironclaw_trust", "crates/ironclaw_turns", "crates/ironclaw_agent_loop", "crates/ironclaw_threads", "crates/ironclaw_prompt_envelope", "crates/ironclaw_hooks", "crates/ironclaw_loop_support", "crates/ironclaw_reborn", "crates/ironclaw_reborn_config", "crates/ironclaw_reborn_composition", "crates/ironclaw_reborn_identity", "crates/ironclaw_first_party_extensions", "crates/ironclaw_reborn_cli", "crates/ironclaw_reborn_traces", "crates/ironclaw_reborn_webui_ingress", "crates/ironclaw_reborn_openai_compat", "crates/ironclaw_conversations", "crates/ironclaw_product_adapters", "crates/ironclaw_product_context", "crates/ironclaw_product_workflow", "crates/ironclaw_product_adapter_registry", "crates/ironclaw_wasm_product_adapters", "crates/ironclaw_telegram_v2_adapter", "crates/ironclaw_slack_v2_adapter", "crates/ironclaw_outbound", "crates/ironclaw_triggers", "crates/ironclaw_projects", "crates/ironclaw_architecture", "crates/ironclaw_safety", "crates/ironclaw_skills", "crates/ironclaw_oauth", "crates/ironclaw_llm", "crates/ironclaw_embeddings", "crates/ironclaw_gateway", "crates/ironclaw_tui", "crates/ironclaw_webui_v2", "crates/ironclaw_skill_learning", "crates/ironclaw_reborn_migration", "tools/ironclaw_stress"] exclude = [ "channels-src/discord", "channels-src/feishu", diff --git a/FEATURE_PARITY.md b/FEATURE_PARITY.md index d1254face3c..8a5c618d6a6 100644 --- a/FEATURE_PARITY.md +++ b/FEATURE_PARITY.md @@ -712,6 +712,18 @@ Trace Commons issuer/TenantCtx note: the server-side `zmanian/tracedao-server` s | Gmail pub/sub | ✅ | ❌ | P3 | | | Inferred follow-up commitments | ✅ | ❌ | P3 | Heartbeat-delivered reminders; opt-in batched extraction | +**State migration (v1/engine-v2 → Reborn):** `crates/ironclaw_reborn_migration` +converts persisted automations. Cron routines and cron missions convert to +Reborn `TriggerRecord`s (mission threads land under `ThreadScope.mission_id`). +Because Reborn's `TriggerSourceKind` is `Schedule`-only, **event / system-event / +webhook / manual routines and non-cron mission cadences have no `TriggerRecord` +target** and are recorded in the migration manifest rather than converted — even +where the runtime supports the *behavior* via hooks/`event_emit`, the durable +automation row does not carry over. Guardrails, notify config, run counters, +`routine_runs` history (no public run-history insert), and mission-only fields +(focus/approach/success-criteria) likewise have no target. See the crate's +CLAUDE.md for the full mapping + gap catalog. + ### Owner: _Unassigned_ --- diff --git a/crates/ironclaw_architecture/tests/reborn_dependency_boundaries.rs b/crates/ironclaw_architecture/tests/reborn_dependency_boundaries.rs index fee85b79a44..8ca42fb7004 100644 --- a/crates/ironclaw_architecture/tests/reborn_dependency_boundaries.rs +++ b/crates/ironclaw_architecture/tests/reborn_dependency_boundaries.rs @@ -1860,6 +1860,36 @@ struct BoundaryRule { fn boundary_rules() -> Vec { vec![ + BoundaryRule { + // The v1→Reborn migration tool is an intentional one-way bridge: + // it reads through the root `ironclaw` crate and writes through the + // Reborn state substrate + composition's `migration-support` seam. + // That bridge must stay a *state* converter — it must not grow + // direct deps on the serving/runtime/engine layers (gateway, engine, + // runtime lanes, dispatcher, webui) or it would quietly become a + // second live entry point into Reborn. + crate_name: "ironclaw_reborn_migration", + forbidden: vec![ + "ironclaw_dispatcher", + "ironclaw_engine", + "ironclaw_gateway", + "ironclaw_llm", + "ironclaw_loop_support", + "ironclaw_mcp", + "ironclaw_network", + "ironclaw_product_adapters", + "ironclaw_product_workflow", + "ironclaw_reborn", + "ironclaw_reborn_cli", + "ironclaw_reborn_event_store", + "ironclaw_run_state", + "ironclaw_runtime_policy", + "ironclaw_scripts", + "ironclaw_tui", + "ironclaw_wasm", + "ironclaw_webui_v2", + ], + }, BoundaryRule { crate_name: "ironclaw_product_workflow", forbidden: vec![ diff --git a/crates/ironclaw_reborn_composition/Cargo.toml b/crates/ironclaw_reborn_composition/Cargo.toml index 152b2f4ef9a..c1773249e80 100644 --- a/crates/ironclaw_reborn_composition/Cargo.toml +++ b/crates/ironclaw_reborn_composition/Cargo.toml @@ -44,6 +44,10 @@ root-llm-provider = [ # test-support` (plus whatever feature surface the test crate also # requires, e.g. `webui-v2-beta`). test-support = ["ironclaw_host_runtime/test-support"] +# Expose the narrow `extension_installation_store_for_migration` accessor for +# the v1→Reborn migration tool (`ironclaw_reborn_migration`). Ships zero bytes in +# a default production binary; mirrors the `test-support` substrate seams. +migration-support = [] # Compose the Reborn WebChat v2 HTTP gateway in this crate. Off by # default; turning this on pulls in the v2 route crate plus the axum / # tower-http middleware stack used by the `webui_serve` module. The diff --git a/crates/ironclaw_reborn_composition/src/factory.rs b/crates/ironclaw_reborn_composition/src/factory.rs index 19a3a86f65e..f7c7c082a6e 100644 --- a/crates/ironclaw_reborn_composition/src/factory.rs +++ b/crates/ironclaw_reborn_composition/src/factory.rs @@ -2707,6 +2707,37 @@ pub(crate) async fn open_local_dev_extension_installation_store_for_test( Ok(Arc::new(store)) } +/// Migration seam: open the extension installation store over a caller-supplied +/// [`RootFilesystem`] at the default installation state path, returning the +/// boxed trait object so the migration tool never touches the concrete +/// `pub(crate)` `FilesystemExtensionInstallationStore`. Mirrors the production +/// binding in [`build_reborn_services`] (the `extension_installation_store` +/// construction via `FilesystemExtensionInstallationStore::load_at`); gated +/// behind `migration-support` so it ships zero bytes in a default production +/// binary, exactly like the `test-support` seams above. +/// +/// The migration tool owns the cross-stack bridge (it depends on the legacy +/// `ironclaw` crate); keeping this narrow accessor here lets composition retain +/// sole ownership of the installation store's construction without composition +/// itself taking any legacy dependency. +#[cfg(feature = "migration-support")] +pub async fn extension_installation_store_for_migration( + filesystem: Arc, +) -> Result, RebornBuildError> { + let state_path = + FilesystemExtensionInstallationStore::default_state_path().map_err(|error| { + RebornBuildError::InvalidConfig { + reason: format!("extension installation state path invalid: {error}"), + } + })?; + let store = FilesystemExtensionInstallationStore::load_at(filesystem, state_path) + .await + .map_err(|error| RebornBuildError::InvalidConfig { + reason: format!("extension installation state could not be loaded: {error}"), + })?; + Ok(Arc::new(store)) +} + /// Test-only (C-DURABLE seam): open a FRESH, independent /// [`ironclaw_run_state::ApprovalRequestStore`] at an existing local-dev /// `storage_root`, paralleling [`open_local_dev_extension_installation_store_for_test`] diff --git a/crates/ironclaw_reborn_composition/src/lib.rs b/crates/ironclaw_reborn_composition/src/lib.rs index 4d3a6355020..aa92b8a087c 100644 --- a/crates/ironclaw_reborn_composition/src/lib.rs +++ b/crates/ironclaw_reborn_composition/src/lib.rs @@ -178,6 +178,8 @@ pub use extension_lifecycle_command::{ pub use factory::AttachmentTestSupport; #[cfg(feature = "test-support")] pub use factory::RebornLocalDevApprovalTestParts; +#[cfg(feature = "migration-support")] +pub use factory::extension_installation_store_for_migration; pub use factory::{RebornServices, build_reborn_services, builtin_first_party_trust_policy}; pub use failure_summary::reborn_failure_summary_for_category; pub use gsuite::{bundled_gsuite_extension_packages, bundled_gsuite_first_party_handlers}; diff --git a/crates/ironclaw_reborn_migration/CLAUDE.md b/crates/ironclaw_reborn_migration/CLAUDE.md new file mode 100644 index 00000000000..45cf4b2bfbd --- /dev/null +++ b/crates/ironclaw_reborn_migration/CLAUDE.md @@ -0,0 +1,111 @@ +# ironclaw_reborn_migration + +Standalone tool + library that converts **IronClaw v1 / engine-v2 persisted +state** into the **Reborn** state substrate. Ships as its own binary +(`ironclaw-reborn-migration`); the conversion engine is a library +(`run_migration`) so it can later be wired into `ironclaw-reborn` startup. + +- **Read side** = the root `ironclaw` crate (`ironclaw::db::connect_with_handles`) + — one v1 database (PostgreSQL **or** libSQL). Engine-v2 state is **not** a + separate DB: missions/projects/threads were persisted by the v2 bridge as JSON + blobs inside the v1 `memory_documents` table under `engine/…` / + `.system/engine/…` paths. Parsed via the serde mirrors in `v2_model.rs` (the + engine-v2 types were deleted; they survive only at git tag `old_engine_v2`). +- **Write side** = Reborn domain stores built directly over a `RootFilesystem` / + triggers DB in `target.rs`, without booting a `RebornRuntime`. Threads / + secrets / identity force a concrete filesystem type, so they are built inside + the backend match arm and stored as `#[async_trait]` trait objects. +- **Philosophy: nothing is silently dropped.** Infrastructure errors abort; a + value with no Reborn representation is recorded as a `LossyItem` on the + `MigrationReport` (the manifest), with the reason and the Reborn gap named. + +``` +cargo run -p ironclaw_reborn_migration -- \ + --source-libsql ~/.ironclaw/ironclaw.db \ + --target-libsql ./reborn-local-dev.db \ + --tenant-id default --agent-id default --dry-run +``` + +## What converts, and where losses go + +| v1 / engine-v2 source | Reborn target | Status | +|---|---|---| +| `conversations` + `conversation_messages` | `SessionThreadRecord` + transcript (orig id preserved via `EnsureThreadRequest.thread_id`; per-message role/ts/id in `metadata_json.legacy_v1`) | **full** | +| routine `Trigger::Cron` | `TriggerRecord` (`TriggerSchedule::Cron`) via `TriggerRepository::upsert_trigger` | **full** | +| engine-v2 mission `Cadence::Cron` | `TriggerRecord`; `thread_history` → threads under `ThreadScope.mission_id` | **full** | +| `memory_documents` (non-engine) | `ironclaw_memory` documents (`MemoryService::write`) | **full** | +| `secrets` | decrypt via v1 `SecretsStore` → re-encrypt via Reborn `SecretStore::put` (needs `--secret-master-key`) | **full** | +| `user_identities` (OAuth) + `channel_identities` | `RebornIdentityResolver::adopt_migrated_identity` (`SurfaceKind::Oauth` / `ChannelActor`) | **full** | +| `wasm_tools` / `wasm_channels` installs | `ExtensionInstallation` (+ synthesized `capability_provider` manifest) via composition's `migration-support` seam; `tool_capabilities.allowed_secrets` → credential bindings | **full (manifest is a placeholder — see below)** | +| routine `Trigger::{Event,SystemEvent,Webhook,Manual}` | — (Reborn `TriggerSourceKind` = `Schedule` only) | **gap → report** | +| mission `Cadence::{OnEvent,OnSystemEvent,Webhook,Manual}` | — | **gap → report** | +| routine guardrails / notify / run counters; mission focus / approach / success-criteria / notify | — (no trigger field / no durable mission entity) | **gap → report** | +| `routine_runs` history | — (`TriggerRepository` has no public run-history insert) | **gap → report** | +| routine/mission `Failed` status | `TriggerState::Paused` | **degraded → report** | +| non-user/assistant transcript messages (system/tool) | retained in thread `metadata_json.legacy_v1`, not a standalone row | **degraded → report** | +| `settings` (key/value) | — (Reborn config is typed `config.toml`/`providers.json`/`LlmKeyStore`, no generic KV store) | **gap → report** | +| `memory_document_versions` | — (no per-doc version history in Reborn) | **gap → report** | +| `agent_jobs` / `job_actions` / `job_events` | — (Reborn has no general job store) | **gap → report** | +| `heartbeat_state` | — (re-establish as a scheduled trigger) | **gap → report** | +| extension manifest fidelity + WASM binary; tool capability config; channel→secret binding; `pairing_requests` | — | **degraded/gap → report** | + +`Domain` + `LossReason` on each `LossyItem` make the manifest greppable; the +acceptance test asserts the **exact** gap set so a regression that silently drops +a domain fails the build. + +## Notes on the "full" deferred converters + +- **Secrets** — needs `--secret-master-key` (used verbatim as the HKDF IKM, as + in v1). The v1 store is built from the raw `DatabaseHandles`; each secret is + listed, decrypted (`get_decrypted`), and re-encrypted through + `RebornTarget::secret_store` (`FilesystemSecretStore`). Expiry is preserved; a + secret that fails to decrypt (expired / wrong key) is a per-secret loss, not a + run abort. Without a key, secrets are skipped with a recorded loss. +- **Identities** — `user_identities` read via the `Database` trait, + `channel_identities` via raw SQL (no trait accessor). Adoption preserves the v1 + `UserId` and seeds the verified-email index. Idempotent (safe to re-run). +- **Extensions** — installed tools/channels become `ExtensionInstallation`s with + activation from the v1 `status` and credential bindings from + `tool_capabilities.allowed_secrets`. The synthesized manifest declares + `ironclaw.capability_provider/v1` + one `ask`-permission placeholder capability + (a non-first-party manifest must declare a host API or capability). **The v1 + capability contract and WASM binary are NOT carried over** — the manifest is a + migration placeholder, recorded as a `manifest_fidelity` loss per installation. + The store is opened through the composition `migration-support` seam + `extension_installation_store_for_migration` (mirrors composition's + `*_for_test` accessors; ships zero bytes without the feature). + +## Remaining follow-up — wire into `ironclaw-reborn` startup + +Call `run_migration` in `crates/ironclaw_reborn_cli/src/runtime/mod.rs` after the +storage root is resolved and before `build_reborn_runtime`, mirroring +`with_run_local_trigger_fire_access_checker`; or add a `Command::Migrate` +subcommand. The `run --dry-run` output already reserves a `v1_state:` line. +Deferred per the original PR scope. + +## Mount layout caveat + +`mounts.rs` reproduces the production alias→path layout (memory `/memory`; +threads/secrets tenant/user-scoped) because the canonical resolver is **private** +in `ironclaw_reborn_composition`. It MUST be reconciled with composition when the +startup wiring lands so the runtime reads back exactly what was migrated. The +acceptance test verifies round-trip through the **same** services the migration +writes with, pinning conversion correctness independently of that reconciliation +(end-to-end runtime-readback is the wiring follow-up). + +## Tests + +`tests/migration_roundtrip.rs` (`required-features = ["libsql"]`, Docker-free): +seeds a rich v1+engine-v2 fixture (conversations, every routine trigger variant, +cron + non-cron missions with a mission thread, memory docs, settings, a secret, +an OAuth + a channel identity, an installed WASM tool), runs the migration, and +asserts converted counts (including secrets/identities/extensions), the exact gap +set, triggers read back through the **public** `LibSqlTriggerRepository`, and +on-disk durability of thread / secret / extension-installation documents via a +fresh connection. A second case asserts `--dry-run` reports fully but writes +nothing. Add a Postgres variant with the `postgres_pool_or_skip()` +skip-if-no-Docker helper (see `crates/ironclaw_reborn_composition/tests/postgres_substrate.rs`). + +``` +cargo test -p ironclaw_reborn_migration --features libsql --test migration_roundtrip +``` diff --git a/crates/ironclaw_reborn_migration/Cargo.toml b/crates/ironclaw_reborn_migration/Cargo.toml new file mode 100644 index 00000000000..6c0edef200e --- /dev/null +++ b/crates/ironclaw_reborn_migration/Cargo.toml @@ -0,0 +1,92 @@ +[package] +name = "ironclaw_reborn_migration" +version = "0.1.0" +edition = "2024" +rust-version.workspace = true +description = "Migrates IronClaw v1 / engine-v2 persisted state into the Reborn state substrate" +authors = ["NEAR AI "] +license = "MIT OR Apache-2.0" +homepage = "https://github.com/nearai/ironclaw" +repository = "https://github.com/nearai/ironclaw" +publish = false + +# Keep out of cargo-dist release packaging; this is an operator/migration tool, +# not a shipped product binary (mirrors ironclaw_reborn_cli). +[package.metadata.dist] +dist = false + +[[bin]] +name = "ironclaw-reborn-migration" +path = "src/main.rs" + +[features] +default = ["libsql", "postgres"] +# Each backend feature forwards into the v1 read stack (root `ironclaw` crate) +# and every Reborn write stack that dispatches on a backend, so a single +# `--features libsql` build wires the whole read→write path over one backend. +libsql = [ + "dep:libsql", + "ironclaw/libsql", + "ironclaw_filesystem/libsql", + "ironclaw_triggers/libsql", + "ironclaw_reborn_composition/libsql", +] +postgres = [ + "dep:deadpool-postgres", + "dep:tokio-postgres", + "ironclaw/postgres", + "ironclaw_filesystem/postgres", + "ironclaw_triggers/postgres", + "ironclaw_reborn_composition/postgres", +] + +[dependencies] +# ── v1 / engine-v2 read side (the legacy monolith) ────────────────────────── +ironclaw = { path = "../..", default-features = false } + +# ── Reborn write side (path deps; versions intentionally unpinned) ─────────── +ironclaw_common = { path = "../ironclaw_common" } +ironclaw_host_api = { path = "../ironclaw_host_api" } +ironclaw_filesystem = { path = "../ironclaw_filesystem" } +ironclaw_threads = { path = "../ironclaw_threads" } +ironclaw_triggers = { path = "../ironclaw_triggers" } +ironclaw_memory = { path = "../ironclaw_memory" } +ironclaw_memory_native = { path = "../ironclaw_memory_native" } +ironclaw_secrets = { path = "../ironclaw_secrets" } +ironclaw_extensions = { path = "../ironclaw_extensions" } +ironclaw_host_runtime = { path = "../ironclaw_host_runtime" } +ironclaw_reborn_composition = { path = "../ironclaw_reborn_composition", features = [ + "migration-support", +] } +ironclaw_reborn_identity = { path = "../ironclaw_reborn_identity" } + +# ── shared ────────────────────────────────────────────────────────────────── +anyhow = "1" +async-trait = "0.1" +chrono = { version = "0.4", features = ["serde"] } +clap = { version = "4", features = ["derive", "env"] } +secrecy = { version = "0.10", features = ["serde"] } +serde = { version = "1", features = ["derive"] } +serde_json = "1" +thiserror = "2" +tokio = { version = "1", features = ["macros", "rt-multi-thread", "fs"] } +tracing = "0.1" +tracing-subscriber = { version = "0.3", features = ["env-filter"] } +ulid = "1" +uuid = { version = "1", features = ["v4", "v5", "serde"] } + +# backend-specific handles (optional, pulled in by feature) +deadpool-postgres = { version = "0.14", optional = true } +libsql = { version = "0.9", optional = true, default-features = false, features = ["core", "replication", "remote", "tls"] } +tokio-postgres = { version = "0.7", optional = true, features = ["with-serde_json-1"] } + +[dev-dependencies] +tempfile = "3" + +# Docker-free acceptance test: seeds a rich v1+engine-v2 libSQL fixture, runs the +# migration, and asserts the round-trip + the exact gap set. Gated on libsql so a +# no-libsql build still compiles. +[[test]] +name = "migration_roundtrip" +path = "tests/migration_roundtrip.rs" +required-features = ["libsql"] diff --git a/crates/ironclaw_reborn_migration/src/convert/automations.rs b/crates/ironclaw_reborn_migration/src/convert/automations.rs new file mode 100644 index 00000000000..2206f28d0b6 --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/convert/automations.rs @@ -0,0 +1,560 @@ +//! Automations converter: v1 routines + engine-v2 missions → Reborn triggers +//! (plus mission threads). +//! +//! **The "no losses" core.** Reborn's trigger substrate only supports scheduled +//! (cron/once) sources, so a cron routine/mission converts to a `TriggerRecord` +//! losslessly, while every other trigger source and every automation-only field +//! Reborn cannot hold is recorded on the report rather than dropped silently: +//! +//! - `Trigger::Cron` / `MissionCadence::Cron` → `TriggerSchedule::Cron` ✅ +//! - `Trigger::{Event,SystemEvent,Webhook,Manual}` / +//! `MissionCadence::{OnEvent,OnSystemEvent,Webhook,Manual}` → **no trigger** +//! (Reborn has only `TriggerSourceKind::Schedule`); recorded as a loss. +//! - routine guardrails / notify / run counters, mission focus / approach / +//! success-criteria / notify → **no target field**; recorded. +//! - `routine_runs` history → **no public run-history insert** on +//! `TriggerRepository`; recorded per routine. +//! - Reborn has no `Failed` trigger state → a failed routine/mission maps to +//! `Paused` and the degrade is recorded. +//! +//! Engine-v2 mission threads (`thread_history`) are migrated as Reborn threads +//! scoped under `ThreadScope.mission_id` — the one place "mission" survives in +//! Reborn (a scope dimension, not a durable entity). + +use std::collections::HashMap; + +use ironclaw::agent::routine::{Routine, RoutineAction, Trigger}; +use ironclaw_host_api::ProjectId; +use ironclaw_triggers::{TriggerRecord, TriggerSchedule, TriggerSourceKind, TriggerState}; +use uuid::Uuid; + +use crate::convert::threads::{ImportMessage, ImportRole, ThreadImport, write_thread}; +use crate::error::MigrationError; +use crate::options::MigrationOptions; +use crate::report::{Domain, LossReason, MigrationReport}; +use crate::source::V1Source; +use crate::target::RebornTarget; +use crate::v2_model::{self, EngineThread, Mission, MissionCadence, MissionStatus}; + +pub(crate) async fn run( + src: &V1Source, + tgt: &mut RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + convert_routines(src, tgt, options, report).await?; + convert_missions(src, tgt, options, report).await?; + Ok(()) +} + +// ── v1 routines ───────────────────────────────────────────────────────────── + +async fn convert_routines( + src: &V1Source, + tgt: &mut RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + let routines = src + .db + .list_all_routines() + .await + .map_err(|e| MigrationError::ReadSource { + domain: "routines".into(), + reason: e.to_string(), + })?; + + for routine in routines { + convert_routine(tgt, options, report, routine).await?; + } + Ok(()) +} + +async fn convert_routine( + tgt: &RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, + routine: Routine, +) -> Result<(), MigrationError> { + let source_id = format!("routine:{}", routine.name); + + // Only cron routines have a Reborn trigger target. + let (schedule, is_cron) = match &routine.trigger { + Trigger::Cron { schedule, timezone } => { + let tz = timezone.clone().unwrap_or_else(|| "UTC".to_string()); + match TriggerSchedule::cron_with_timezone(schedule.clone(), tz) { + Ok(schedule) => (Some(schedule), true), + Err(e) => { + report.record_loss( + Domain::Routine, + &source_id, + "trigger.cron", + LossReason::Unparseable, + format!("invalid cron expression '{schedule}': {e}"), + ); + (None, true) + } + } + } + other => { + report.record_loss( + Domain::Routine, + &source_id, + format!("trigger.{}", trigger_tag(other)), + LossReason::NoTargetConcept, + "Reborn triggers support only scheduled (cron/once) sources; \ + event/system-event/webhook/manual routines have no trigger target \ + (consider a Reborn hook)" + .to_string(), + ); + (None, false) + } + }; + + let Some(schedule) = schedule else { + // Nothing to write; losses already recorded. + record_routine_field_losses(report, &source_id, &routine, is_cron); + return Ok(()); + }; + + let prompt = routine_prompt(&routine.action); + // v1 routines have no terminal-failed status distinct from disabled; + // consecutive_failures is recorded as a field loss below. + let state = if routine.enabled { + TriggerState::Scheduled + } else { + TriggerState::Paused + }; + let now = routine.next_fire_at.unwrap_or(routine.created_at); + + record_routine_field_losses(report, &source_id, &routine, is_cron); + + // A malformed source user id is a per-item loss, not a run abort. + let Some(creator_user_id) = + report.valid_user_id(Domain::Routine, &source_id, "user_id", &routine.user_id) + else { + return Ok(()); + }; + + let record = TriggerRecord { + trigger_id: ironclaw_triggers::TriggerId::new(), + tenant_id: tgt.tenant_id.clone(), + creator_user_id, + agent_id: Some(tgt.agent_id.clone()), + project_id: Option::::None, + name: routine.name.clone(), + source: TriggerSourceKind::Schedule, + schedule, + prompt, + state, + next_run_at: now, + last_run_at: routine.last_run_at, + last_fired_slot: None, + last_status: None, + active_fire_slot: None, + active_run_ref: None, + created_at: routine.created_at, + }; + + if !options.dry_run { + tgt.trigger_repo + .upsert_trigger(record) + .await + .map_err(|e| MigrationError::WriteTarget { + domain: format!("trigger for {source_id}"), + reason: e.to_string(), + })?; + } + report.stats.routines += 1; + Ok(()) +} + +/// Compose the single Reborn trigger prompt from a v1 routine action, recording +/// the action fields Reborn's `prompt`-only trigger cannot hold. +fn routine_prompt(action: &RoutineAction) -> String { + match action { + RoutineAction::Lightweight { prompt, .. } => prompt.clone(), + RoutineAction::FullJob { + title, description, .. + } => { + if title.trim().is_empty() { + description.clone() + } else { + format!("{title}\n\n{description}") + } + } + } +} + +/// Record every routine field that has no Reborn trigger representation. +fn record_routine_field_losses( + report: &mut MigrationReport, + source_id: &str, + routine: &Routine, + is_cron: bool, +) { + // Action extras beyond the prompt. + match &routine.action { + RoutineAction::Lightweight { + context_paths, + max_tokens, + use_tools, + max_tool_rounds, + .. + } => { + if !context_paths.is_empty() || *max_tokens != 0 || *use_tools || *max_tool_rounds != 0 + { + report.record_loss( + Domain::Routine, + source_id, + "action.lightweight_params", + LossReason::NoTargetField, + "Reborn trigger carries only a prompt; context_paths/max_tokens/\ + use_tools/max_tool_rounds are dropped" + .to_string(), + ); + } + } + RoutineAction::FullJob { max_iterations, .. } => { + let _ = max_iterations; + report.record_loss( + Domain::Routine, + source_id, + "action.full_job_params", + LossReason::NoTargetField, + "Reborn trigger carries only a prompt; full-job max_iterations is dropped" + .to_string(), + ); + } + } + + // Guardrails + notify + run counters. + report.record_loss( + Domain::Routine, + source_id, + "guardrails+notify+counters", + LossReason::NoTargetField, + "Reborn triggers have no cooldown/max_concurrent/dedup, notify config, \ + run_count, or consecutive_failures fields" + .to_string(), + ); + + // Run history: no public insert path on TriggerRepository. + if is_cron { + report.record_loss( + Domain::Routine, + source_id, + "routine_runs", + LossReason::NoTargetField, + "TriggerRepository exposes no public run-history insert; historical \ + routine_runs cannot be written to trigger_run_history" + .to_string(), + ); + } +} + +fn trigger_tag(trigger: &Trigger) -> &'static str { + match trigger { + Trigger::Cron { .. } => "cron", + Trigger::Event { .. } => "event", + Trigger::SystemEvent { .. } => "system_event", + Trigger::Webhook { .. } => "webhook", + Trigger::Manual => "manual", + } +} + +// ── engine-v2 missions ────────────────────────────────────────────────────── + +async fn convert_missions( + src: &V1Source, + tgt: &mut RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + let users = src.distinct_users().await?; + for user_id in &users { + let docs = + src.db + .list_documents(user_id, None) + .await + .map_err(|e| MigrationError::ReadSource { + domain: "memory_documents(engine)".into(), + reason: e.to_string(), + })?; + + // Index engine threads by id so mission thread_history can resolve them. + let mut engine_threads: HashMap = HashMap::new(); + let mut missions: Vec = Vec::new(); + for doc in &docs { + if !v2_model::is_engine_path(&doc.path) { + continue; + } + if doc.path.ends_with("mission.json") { + match serde_json::from_str::(&doc.content) { + Ok(mission) => missions.push(mission), + Err(e) => report.record_loss( + Domain::Mission, + doc.path.clone(), + "*", + LossReason::Unparseable, + format!("could not parse mission.json: {e}"), + ), + } + } else if doc.path.contains("/threads/") && doc.path.ends_with(".json") { + match serde_json::from_str::(&doc.content) { + Ok(thread) => { + engine_threads.insert(thread.id, thread); + } + Err(e) => report.record_loss( + Domain::Mission, + doc.path.clone(), + "*", + LossReason::Unparseable, + format!("could not parse engine thread blob: {e}"), + ), + } + } + } + + // Threads referenced by a mission's `thread_history`; anything parsed but + // never referenced has no Reborn owner to migrate it under. + let referenced: std::collections::HashSet = missions + .iter() + .flat_map(|m| m.thread_history.iter().copied()) + .collect(); + + for mission in &missions { + convert_mission(tgt, options, report, user_id, mission, &engine_threads).await?; + } + + for id in engine_threads.keys() { + if !referenced.contains(id) { + report.record_loss( + Domain::Mission, + format!("thread:{id}"), + "*", + LossReason::NoTargetConcept, + "engine thread blob is not referenced by any mission thread_history; \ + there is no Reborn mission owner to migrate it under" + .to_string(), + ); + } + } + } + Ok(()) +} + +async fn convert_mission( + tgt: &RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, + user_id: &str, + mission: &Mission, + engine_threads: &HashMap, +) -> Result<(), MigrationError> { + let source_id = format!("mission:{}", mission.name); + let owner = if mission.user_id.is_empty() { + user_id.to_string() + } else { + mission.user_id.clone() + }; + + // Mission-only fields with no Reborn home. + record_mission_field_losses(report, &source_id, mission); + + // Cadence → trigger (cron only). + match &mission.cadence { + MissionCadence::Cron { + expression, + timezone, + } => { + let tz = timezone.clone().unwrap_or_else(|| "UTC".to_string()); + match TriggerSchedule::cron_with_timezone(expression.clone(), tz) { + Ok(schedule) => { + let state = match mission.status { + MissionStatus::Active => TriggerState::Scheduled, + MissionStatus::Paused => TriggerState::Paused, + MissionStatus::Completed => TriggerState::Completed, + MissionStatus::Failed => { + report.record_loss( + Domain::Mission, + &source_id, + "status.failed", + LossReason::Degraded, + "Reborn has no Failed trigger state; mapped to Paused".to_string(), + ); + TriggerState::Paused + } + }; + // A malformed mission owner id is a per-item loss (the + // trigger is skipped); mission threads below still validate + // their own owner independently. Loss recorded by the helper. + if let Some(creator_user_id) = + report.valid_user_id(Domain::Mission, &source_id, "user_id", &owner) + { + let next_run_at = mission_next_run_at(report, &source_id, mission); + let record = TriggerRecord { + trigger_id: ironclaw_triggers::TriggerId::new(), + tenant_id: tgt.tenant_id.clone(), + creator_user_id, + agent_id: Some(tgt.agent_id.clone()), + project_id: Option::::None, + name: mission.name.clone(), + source: TriggerSourceKind::Schedule, + schedule, + prompt: if mission.goal.trim().is_empty() { + mission.name.clone() + } else { + mission.goal.clone() + }, + state, + next_run_at, + last_run_at: None, + last_fired_slot: None, + last_status: None, + active_fire_slot: None, + active_run_ref: None, + created_at: mission.created_at, + }; + if !options.dry_run { + tgt.trigger_repo.upsert_trigger(record).await.map_err(|e| { + MigrationError::WriteTarget { + domain: format!("trigger for {source_id}"), + reason: e.to_string(), + } + })?; + } + } + } + Err(e) => report.record_loss( + Domain::Mission, + &source_id, + "cadence.cron", + LossReason::Unparseable, + format!("invalid cron expression '{expression}': {e}"), + ), + } + } + other => report.record_loss( + Domain::Mission, + &source_id, + format!("cadence.{}", other.tag()), + LossReason::NoTargetConcept, + "Reborn triggers support only scheduled sources; non-cron mission \ + cadences have no trigger target" + .to_string(), + ), + } + + report.stats.missions += 1; + + // Migrate the mission's threads under ThreadScope.mission_id. + for tid in &mission.thread_history { + let Some(thread) = engine_threads.get(tid) else { + report.record_loss( + Domain::Mission, + &source_id, + format!("thread:{tid}"), + LossReason::Unparseable, + "mission thread_history references a thread blob not found in the \ + engine runtime documents" + .to_string(), + ); + continue; + }; + let import = ThreadImport { + thread_id: thread.id, + owner_user: owner.clone(), + title: thread.title.clone().or_else(|| Some(mission.name.clone())), + mission_id: Some(mission.id), + provenance: serde_json::json!({ + "source": "engine_v2_mission_thread", + "mission": mission.name, + "goal": thread.goal, + "created_at": thread.created_at.to_rfc3339(), + }), + messages: thread + .messages + .iter() + .map(|m| ImportMessage { + role: engine_role(m.role), + raw_role: format!("{:?}", m.role), + content: m.content.clone(), + created_at: m.timestamp, + orig_id: None, + }) + .collect(), + }; + if options.dry_run { + report.stats.threads += 1; + report.stats.messages += import + .messages + .iter() + .filter(|m| m.role != ImportRole::Other) + .count(); + // Match the real write path, which records a loss per non-user/ + // assistant transcript message, so `--dry-run` reports the same gap + // set instead of under-counting. + crate::convert::threads::record_other_role_losses(report, &import); + } else { + write_thread(tgt, options, report, import).await?; + } + } + + Ok(()) +} + +/// The trigger's `next_run_at` for a migrated mission. A mission with an +/// explicit `next_fire_at` uses it; otherwise the fallback is the **migration +/// time**, not `mission.created_at` — the latter can be the `epoch_fallback` +/// synthesized when a drifted blob omits `created_at`, which would produce a +/// `1970`-dated, immediately-due trigger. The synthesized fallback is recorded +/// as a `Degraded` loss so it is never silent. +fn mission_next_run_at( + report: &mut MigrationReport, + source_id: &str, + mission: &Mission, +) -> chrono::DateTime { + match mission.next_fire_at { + Some(next_fire_at) => next_fire_at, + None => { + report.record_loss( + Domain::Mission, + source_id, + "next_fire_at", + LossReason::Degraded, + "mission had no next_fire_at; the trigger's next run was synthesized to the \ + migration time (mission created_at may be an epoch fallback, which would make \ + the trigger immediately due)" + .to_string(), + ); + chrono::Utc::now() + } + } +} + +fn record_mission_field_losses(report: &mut MigrationReport, source_id: &str, mission: &Mission) { + if mission.current_focus.is_some() + || !mission.approach_history.is_empty() + || mission.success_criteria.is_some() + || !mission.notify_channels.is_empty() + { + report.record_loss( + Domain::Mission, + source_id, + "mission_only_fields", + LossReason::NoTargetConcept, + "Reborn has no durable mission entity; current_focus, approach_history, \ + success_criteria, and notify_channels have no target" + .to_string(), + ); + } +} + +fn engine_role(role: v2_model::MessageRole) -> ImportRole { + match role { + v2_model::MessageRole::User => ImportRole::User, + v2_model::MessageRole::Assistant => ImportRole::Assistant, + v2_model::MessageRole::System | v2_model::MessageRole::ActionResult => ImportRole::Other, + } +} diff --git a/crates/ironclaw_reborn_migration/src/convert/extensions.rs b/crates/ironclaw_reborn_migration/src/convert/extensions.rs new file mode 100644 index 00000000000..6328b142c9c --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/convert/extensions.rs @@ -0,0 +1,444 @@ +//! Extensions/channels/tools converter +//! (v1 `wasm_tools` / `wasm_channels` / `tool_capabilities` → Reborn +//! `ExtensionInstallation`). +//! +//! Each installed v1 WASM tool/channel becomes a Reborn `ExtensionInstallation` +//! with a synthesized `InstalledLocal` manifest that declares the +//! `ironclaw.capability_provider/v1` host API plus one placeholder, +//! approval-gated capability (a non-first-party manifest must declare a host API +//! or capability, and rejects top-level `[[capabilities]]`). Activation maps +//! from the v1 `status` column; a tool's `tool_capabilities.allowed_secrets` +//! become `ExtensionCredentialBinding`s pointing at the migrated secrets. The +//! store itself is built by composition's `migration-support` seam +//! (`RebornTarget::extension_store`). +//! +//! Losses recorded (per installation): the manifest is a placeholder — the v1 +//! tool's real capability contract and WASM binary are NOT carried over; tool +//! capability config beyond credential linkage (http_allowlist, rate limits, +//! workspace prefixes); and channel credential linkage (v1 has no explicit +//! channel→secret join; the secret *values* still migrate via the secrets +//! converter). + +use std::sync::Arc; + +use ironclaw::channels::wasm::{StoredWasmChannel, WasmChannelStore}; +use ironclaw::tools::wasm::{StoredWasmTool, ToolStatus, WasmToolStore}; +use ironclaw_extensions::{ + ExtensionActivationState, ExtensionCredentialBinding, ExtensionCredentialHandle, + ExtensionInstallation, ExtensionInstallationId, ExtensionManifestRecord, ExtensionManifestRef, + HostApiContractRegistry, MANIFEST_SCHEMA_VERSION, ManifestSource, +}; +use ironclaw_host_api::{ExtensionId, HostPortCatalog, SecretHandle}; +use ironclaw_host_runtime::{default_host_api_contract_registry, default_host_port_catalog}; + +use crate::error::MigrationError; +use crate::options::MigrationOptions; +use crate::report::{Domain, LossReason, MigrationReport}; +use crate::source::V1Source; +use crate::target::RebornTarget; + +pub(crate) async fn run( + src: &V1Source, + tgt: &mut RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + let catalog = default_host_port_catalog().map_err(|e| MigrationError::WriteTarget { + domain: "extension host-port catalog".into(), + reason: e.to_string(), + })?; + let registry = + default_host_api_contract_registry().map_err(|e| MigrationError::WriteTarget { + domain: "extension host-api contract registry".into(), + reason: e.to_string(), + })?; + + let tool_store = build_tool_store(src); + let channel_store = build_channel_store(src); + + // Installed tools/channels are keyed by user_id; enumerate from both tables. + let mut users: std::collections::BTreeSet = + src.distinct_users().await?.into_iter().collect(); + users.extend(src.distinct_user_ids_in("wasm_tools", "user_id").await?); + users.extend(src.distinct_user_ids_in("wasm_channels", "user_id").await?); + + for user in users { + if let Some(store) = tool_store.as_ref() { + let tools = store + .list(&user) + .await + .map_err(|e| MigrationError::ReadSource { + domain: "wasm_tools".into(), + reason: e.to_string(), + })?; + for tool in tools { + let bindings = tool_credential_bindings(store.as_ref(), &tool, report).await?; + convert_installation( + tgt, + options, + report, + &catalog, + ®istry, + InstallInput { + owner: &user, + raw_name: &tool.name, + version: &tool.version, + description: &tool.description, + active: tool.status == ToolStatus::Active, + updated_at: tool.updated_at, + bindings, + }, + ) + .await?; + } + } + + if let Some(store) = channel_store.as_ref() { + let channels = store + .list(&user) + .await + .map_err(|e| MigrationError::ReadSource { + domain: "wasm_channels".into(), + reason: e.to_string(), + })?; + for channel in channels { + convert_installation( + tgt, + options, + report, + &catalog, + ®istry, + channel_input(&user, &channel), + ) + .await?; + report.record_loss( + Domain::Extension, + format!("channel:{}", channel.name), + "credential_binding", + LossReason::NoTargetField, + "v1 has no explicit channel→secret join; the credential value still \ + migrates via the secrets converter, but the installation binding is \ + not auto-linked" + .to_string(), + ); + } + } + } + Ok(()) +} + +struct InstallInput<'a> { + /// v1 owner user id. Folded into the synthesized `ExtensionInstallationId` + /// so two users with a same-named install do not collide (the store is keyed + /// by installation id, so a bare name would let the second overwrite the + /// first with no loss recorded). + owner: &'a str, + raw_name: &'a str, + version: &'a str, + description: &'a str, + active: bool, + updated_at: chrono::DateTime, + bindings: Vec, +} + +fn channel_input<'a>(owner: &'a str, channel: &'a StoredWasmChannel) -> InstallInput<'a> { + InstallInput { + owner, + raw_name: &channel.name, + version: &channel.version, + description: &channel.description, + active: channel.status == "active", + updated_at: channel.updated_at, + bindings: Vec::new(), + } +} + +#[allow(clippy::too_many_arguments)] // arch-exempt: too_many_args, migration converter threads scope + catalog + registry + input, plan v1-migration +async fn convert_installation( + tgt: &RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, + catalog: &HostPortCatalog, + registry: &HostApiContractRegistry, + input: InstallInput<'_>, +) -> Result<(), MigrationError> { + let source_id = format!("extension:{}", input.raw_name); + let ext_id_str = sanitize_extension_id(input.raw_name); + let extension_id = match ExtensionId::new(&ext_id_str) { + Ok(id) => id, + Err(e) => { + report.record_loss( + Domain::Extension, + &source_id, + "id", + LossReason::Unparseable, + format!("could not derive a valid Reborn extension id: {e}"), + ); + return Ok(()); + } + }; + + // The synthesized manifest is a migration placeholder: v1 tools have no + // Reborn capability contract and the WASM binary is not carried over, so a + // single generic host-mediated capability stands in. Record that gap. + report.record_loss( + Domain::Extension, + &source_id, + "manifest_fidelity", + LossReason::Degraded, + "v1 tool capability contract + WASM binary are not migrated; a placeholder \ + capability_provider manifest is synthesized so the installation record + \ + activation + credential bindings carry over" + .to_string(), + ); + + let manifest_toml = build_manifest_toml( + &ext_id_str, + input.raw_name, + input.version, + input.description, + ); + let manifest = match ExtensionManifestRecord::from_toml_with_contracts( + manifest_toml, + ManifestSource::InstalledLocal, + catalog, + None, + registry, + ) { + Ok(manifest) => manifest, + Err(e) => { + report.record_loss( + Domain::Extension, + &source_id, + "manifest", + LossReason::Unparseable, + format!("synthesized manifest did not validate: {e}"), + ); + return Ok(()); + } + }; + + let activation = if input.active { + ExtensionActivationState::Enabled + } else { + ExtensionActivationState::Disabled + }; + // Installation id is scoped by owner so per-user installs of the same tool + // name each get a distinct record instead of silently overwriting. + let installation_id_str = sanitize_extension_id(&format!("{}-{}", input.owner, input.raw_name)); + let installation_id = match ExtensionInstallationId::new(&installation_id_str) { + Ok(id) => id, + Err(e) => { + report.record_loss( + Domain::Extension, + &source_id, + "installation_id", + LossReason::Unparseable, + format!("invalid installation id: {e}"), + ); + return Ok(()); + } + }; + let manifest_ref = ExtensionManifestRef::new(extension_id.clone(), None); + let installation = match ExtensionInstallation::new( + installation_id, + extension_id, + activation, + manifest_ref, + input.bindings, + input.updated_at, + ) { + Ok(installation) => installation, + Err(e) => { + report.record_loss( + Domain::Extension, + &source_id, + "installation", + LossReason::Unparseable, + format!("could not build installation: {e}"), + ); + return Ok(()); + } + }; + + if !options.dry_run { + tgt.extension_store + .upsert_manifest_and_installation(manifest, installation) + .await + .map_err(|e| MigrationError::WriteTarget { + domain: format!("extension {source_id}"), + reason: e.to_string(), + })?; + } + report.stats.extensions += 1; + Ok(()) +} + +/// Build credential bindings from a tool's `allowed_secrets`, recording the +/// capability config that has no Reborn target. +async fn tool_credential_bindings( + store: &dyn WasmToolStore, + tool: &StoredWasmTool, + report: &mut MigrationReport, +) -> Result, MigrationError> { + // A read *error* is a real infrastructure failure and aborts the run; a + // legitimate "no capabilities row" (`Ok(None)`) just yields no bindings. + let capabilities = match store.get_capabilities(tool.id).await { + Ok(Some(capabilities)) => capabilities, + Ok(None) => return Ok(Vec::new()), + Err(e) => { + return Err(MigrationError::ReadSource { + domain: "tool_capabilities".into(), + reason: e.to_string(), + }); + } + }; + report.record_loss( + Domain::Extension, + format!("tool:{}", tool.name), + "capabilities", + LossReason::NoTargetField, + "tool http_allowlist / rate limits / workspace prefixes have no Reborn \ + installation field" + .to_string(), + ); + let mut bindings = Vec::new(); + for secret_name in capabilities.allowed_secrets { + match ( + ExtensionCredentialHandle::new(secret_name.clone()), + SecretHandle::new(&secret_name), + ) { + (Ok(handle), Ok(secret_handle)) => { + bindings.push(ExtensionCredentialBinding::new(handle, secret_handle)); + } + // An unconvertible secret name is recorded, not dropped silently. + _ => report.record_loss( + Domain::Extension, + format!("tool:{}", tool.name), + "allowed_secret", + LossReason::Unparseable, + format!("secret name '{secret_name}' is not a valid Reborn credential binding"), + ), + } + } + Ok(bindings) +} + +fn build_tool_store(src: &V1Source) -> Option> { + #[cfg(feature = "libsql")] + if let Some(db) = src.handles.libsql_db.as_ref() { + return Some(Arc::new(ironclaw::tools::wasm::LibSqlWasmToolStore::new( + db.clone(), + ))); + } + #[cfg(feature = "postgres")] + if let Some(pool) = src.handles.pg_pool.as_ref() { + return Some(Arc::new(ironclaw::tools::wasm::PostgresWasmToolStore::new( + pool.clone(), + ))); + } + None +} + +fn build_channel_store(src: &V1Source) -> Option> { + #[cfg(feature = "libsql")] + if let Some(db) = src.handles.libsql_db.as_ref() { + return Some(Arc::new( + ironclaw::channels::wasm::LibSqlWasmChannelStore::new(db.clone()), + )); + } + #[cfg(feature = "postgres")] + if let Some(pool) = src.handles.pg_pool.as_ref() { + return Some(Arc::new( + ironclaw::channels::wasm::PostgresWasmChannelStore::new(pool.clone()), + )); + } + None +} + +/// Sanitize a v1 tool/channel name into a valid Reborn `ExtensionId` +/// (`validate_name_segment`: lowercase, starts alnum, `[a-z0-9._-]`, ≤128). +fn sanitize_extension_id(raw: &str) -> String { + let mut out = String::with_capacity(raw.len()); + for ch in raw.chars() { + let lower = ch.to_ascii_lowercase(); + if lower.is_ascii_alphanumeric() || matches!(lower, '_' | '-' | '.') { + out.push(lower); + } else { + out.push('_'); + } + } + // Must start with an alphanumeric. + if !out + .chars() + .next() + .is_some_and(|c| c.is_ascii_alphanumeric()) + { + out.insert(0, 'x'); + } + out.truncate(128); + if out.is_empty() { + out.push_str("ext"); + } + out +} + +fn build_manifest_toml(ext_id: &str, name: &str, version: &str, description: &str) -> String { + // A valid non-first-party manifest declares the capability_provider host API + // and at least one namespaced, host-mediated capability (empty manifests and + // top-level `[[capabilities]]` are both rejected). `ask` permission keeps the + // migrated tool approval-gated. + format!( + r#"schema_version = "{schema}" +id = "{ext_id}" +name = "{name}" +version = "{version}" +description = "{description}" +trust = "third_party" + +[runtime] +kind = "wasm" +module = "wasm/{ext_id}.wasm" + +[[host_api]] +id = "ironclaw.capability_provider/v1" +section = "capability_provider.tools" + +[[capability_provider.tools.capabilities]] +id = "{ext_id}.invoke" +description = "Migrated v1 tool capability (placeholder)." +default_permission = "ask" +visibility = "model" +input_schema_ref = "schemas/{ext_id}/invoke.input.v1.json" +output_schema_ref = "schemas/{ext_id}/invoke.output.v1.json" +prompt_doc_ref = "prompts/{ext_id}/invoke.md" +"#, + schema = MANIFEST_SCHEMA_VERSION, + name = toml_escape(name), + version = toml_escape(normalize_version(version)), + description = toml_escape(description), + ) +} + +fn normalize_version(version: &str) -> &str { + if version.trim().is_empty() { + "0.1.0" + } else { + version + } +} + +/// Escape a value for a TOML basic string: backslash + quote escaped, control +/// characters (incl. newlines) dropped so the synthesized manifest stays valid. +fn toml_escape(value: &str) -> String { + let mut out = String::with_capacity(value.len()); + for ch in value.chars() { + match ch { + '\\' => out.push_str("\\\\"), + '"' => out.push_str("\\\""), + c if c.is_control() => out.push(' '), + c => out.push(c), + } + } + out +} diff --git a/crates/ironclaw_reborn_migration/src/convert/heartbeat.rs b/crates/ironclaw_reborn_migration/src/convert/heartbeat.rs new file mode 100644 index 00000000000..bf4d92c004a --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/convert/heartbeat.rs @@ -0,0 +1,34 @@ +//! Heartbeat converter (v1 `heartbeat_state`). +//! +//! Reborn's periodic execution is expressed through triggers/the poller, not a +//! persisted per-user heartbeat row. There is no durable heartbeat-state target +//! to migrate into, so the presence of v1 heartbeat state is recorded as a loss +//! (its cadence should be re-established as a Reborn scheduled trigger). + +use crate::error::MigrationError; +use crate::options::MigrationOptions; +use crate::report::{Domain, LossReason, MigrationReport}; +use crate::source::V1Source; +use crate::target::RebornTarget; + +pub(crate) async fn run( + src: &V1Source, + _tgt: &mut RebornTarget, + _options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + // heartbeat_state is keyed per (user_id, agent_id); enumerate distinct users + // and record the gap. No typed all-user read API exists for heartbeat state. + for user_id in src.distinct_users().await? { + report.record_loss( + Domain::Heartbeat, + user_id, + "heartbeat_state", + LossReason::NoTargetConcept, + "Reborn has no durable heartbeat-state record; re-establish periodic \ + execution as a scheduled trigger" + .to_string(), + ); + } + Ok(()) +} diff --git a/crates/ironclaw_reborn_migration/src/convert/identities.rs b/crates/ironclaw_reborn_migration/src/convert/identities.rs new file mode 100644 index 00000000000..493c902d95e --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/convert/identities.rs @@ -0,0 +1,274 @@ +//! Identity converter (v1 `user_identities` + `channel_identities` → Reborn). +//! +//! Target is `RebornIdentityResolver::adopt_migrated_identity`, purpose-built to +//! carry a v1 identity over while preserving its `UserId` (idempotent; +//! first-writer-wins verified-email index). v1 OAuth/social identities +//! (`user_identities`) map to `SurfaceKind::Oauth`; channel actor mappings +//! (`channel_identities`, read raw since there is no `Database` accessor) map to +//! `SurfaceKind::ChannelActor`. `pairing_requests` has no Reborn store and is +//! recorded as a gap. + +use ironclaw_host_api::UserId; +use ironclaw_reborn_identity::{ + ExternalSubjectId, ProviderKind, RebornIdentityResolver, ResolveExternalIdentity, SurfaceKind, +}; + +use crate::error::MigrationError; +use crate::options::MigrationOptions; +use crate::report::{Domain, LossReason, MigrationReport}; +use crate::source::V1Source; +use crate::target::RebornTarget; + +pub(crate) async fn run( + src: &V1Source, + tgt: &mut RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + migrate_user_identities(src, tgt, options, report).await?; + migrate_channel_identities(src, tgt, options, report).await?; + + report.record_loss( + Domain::Identity, + "pairing_requests", + "*", + LossReason::NoTargetConcept, + "v1 pairing_requests has no Reborn durable store (pairing is handled live)".to_string(), + ); + Ok(()) +} + +// ── user_identities (OAuth/social) via the Database trait ──────────────────── + +async fn migrate_user_identities( + src: &V1Source, + tgt: &mut RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + // `list_identities_for_user` is per-user and there is no all-users accessor; + // enumerate users from the users table (tolerant) unioned with data-derived + // users so installs without a users table still resolve identities. + let mut users: std::collections::BTreeSet = src + .distinct_user_ids_in("users", "id") + .await? + .into_iter() + .collect(); + users.extend(src.distinct_users().await?); + + for user in users { + let identities = src.db.list_identities_for_user(&user).await.map_err(|e| { + MigrationError::ReadSource { + domain: "user_identities".into(), + reason: e.to_string(), + } + })?; + if identities.is_empty() { + continue; + } + // Consistent with `migrate_channel_identities` below: a malformed source + // user id skips that user's identities but is recorded, not dropped + // silently. + let Some(host_user) = + report.valid_user_id(Domain::Identity, format!("user:{user}"), "user_id", &user) + else { + continue; + }; + let resolver = tgt.identity_store(host_user); + + for rec in identities { + let identity = match build_oauth_identity(tgt, &rec, report) { + Some(identity) => identity, + None => continue, + }; + adopt(&resolver, identity, &rec.user_id, options, report).await?; + } + } + Ok(()) +} + +fn build_oauth_identity( + tgt: &RebornTarget, + rec: &ironclaw::db::UserIdentityRecord, + report: &mut MigrationReport, +) -> Option { + let source_id = format!("identity:{}:{}", rec.provider, rec.provider_user_id); + let provider_kind = match ProviderKind::new(rec.provider.clone()) { + Ok(kind) => kind, + Err(e) => { + report.record_loss( + Domain::Identity, + &source_id, + "provider", + LossReason::Unparseable, + format!("invalid provider kind: {e}"), + ); + return None; + } + }; + let subject = match ExternalSubjectId::new(rec.provider_user_id.clone()) { + Ok(subject) => subject, + Err(e) => { + report.record_loss( + Domain::Identity, + &source_id, + "provider_user_id", + LossReason::Unparseable, + format!("invalid external subject id: {e}"), + ); + return None; + } + }; + Some(ResolveExternalIdentity { + tenant_id: tgt.tenant_id.clone(), + surface_kind: SurfaceKind::Oauth, + provider_kind, + provider_instance_id: None, + external_subject_id: subject, + email: rec.email.clone(), + email_verified: rec.email_verified, + display_name: rec.display_name.clone(), + }) +} + +// ── channel_identities (channel actors) via raw SQL ────────────────────────── + +async fn migrate_channel_identities( + src: &V1Source, + tgt: &mut RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + let rows = read_channel_identities(src, report).await?; + for (owner_id, channel, external_id) in rows { + let source_id = format!("channel_identity:{channel}:{external_id}"); + let Some(host_user) = + report.valid_user_id(Domain::Identity, &source_id, "owner_id", &owner_id) + else { + continue; + }; + let (Ok(provider_kind), Ok(subject)) = ( + ProviderKind::new(channel.clone()), + ExternalSubjectId::new(external_id.clone()), + ) else { + report.record_loss( + Domain::Identity, + &source_id, + "channel/external_id", + LossReason::Unparseable, + "invalid channel or external subject id".to_string(), + ); + continue; + }; + let resolver = tgt.identity_store(host_user); + let identity = ResolveExternalIdentity { + tenant_id: tgt.tenant_id.clone(), + surface_kind: SurfaceKind::ChannelActor, + provider_kind, + provider_instance_id: None, + external_subject_id: subject, + email: None, + email_verified: false, + display_name: None, + }; + adopt(&resolver, identity, &owner_id, options, report).await?; + } + Ok(()) +} + +/// Raw read of `channel_identities` (no `Database` accessor exists). +/// +/// Only an **absent table** is tolerated (returns empty) — v1 installs without +/// the table legitimately have no channel identities. Connect / query failures +/// are real infrastructure errors and propagate. A row that fails to decode is +/// recorded as a per-row loss rather than silently skipped. +async fn read_channel_identities( + src: &V1Source, + report: &mut MigrationReport, +) -> Result, MigrationError> { + let sql = "SELECT owner_id, channel, external_id FROM channel_identities"; + let read_err = |e: &dyn std::fmt::Display| MigrationError::ReadSource { + domain: "channel_identities".to_string(), + reason: e.to_string(), + }; + let record_bad_row = |report: &mut MigrationReport, e: &dyn std::fmt::Display| { + report.record_loss( + Domain::Identity, + "channel_identities", + "row", + LossReason::Unparseable, + format!("channel_identities row could not be decoded (skipped): {e}"), + ); + }; + #[cfg(feature = "libsql")] + if let Some(db) = src.handles.libsql_db.as_ref() { + let conn = db.connect().map_err(|e| read_err(&e))?; + let mut rows = match conn.query(sql, ()).await { + Ok(rows) => rows, + Err(e) if crate::source::is_missing_table_error(&e.to_string()) => { + return Ok(Vec::new()); + } + Err(e) => return Err(read_err(&e)), + }; + let mut out = Vec::new(); + while let Some(row) = rows.next().await.map_err(|e| read_err(&e))? { + match ( + row.get::(0), + row.get::(1), + row.get::(2), + ) { + (Ok(owner), Ok(channel), Ok(external)) => out.push((owner, channel, external)), + (Err(e), ..) | (_, Err(e), _) | (.., Err(e)) => record_bad_row(report, &e), + } + } + return Ok(out); + } + #[cfg(feature = "postgres")] + if let Some(pool) = src.handles.pg_pool.as_ref() { + let client = pool.get().await.map_err(|e| read_err(&e))?; + let rows = match client.query(sql, &[]).await { + Ok(rows) => rows, + Err(e) if crate::source::is_missing_table_error(&e.to_string()) => { + return Ok(Vec::new()); + } + Err(e) => return Err(read_err(&e)), + }; + let mut out = Vec::new(); + for row in &rows { + match ( + row.try_get::<_, String>(0), + row.try_get::<_, String>(1), + row.try_get::<_, String>(2), + ) { + (Ok(owner), Ok(channel), Ok(external)) => out.push((owner, channel, external)), + (Err(e), ..) | (_, Err(e), _) | (.., Err(e)) => record_bad_row(report, &e), + } + } + return Ok(out); + } + Ok(Vec::new()) +} + +async fn adopt( + resolver: &std::sync::Arc, + identity: ResolveExternalIdentity, + migrated_user_id: &str, + options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + let migrated_user = UserId::new(migrated_user_id).map_err(|e| MigrationError::WriteTarget { + domain: format!("identity migrated user_id {migrated_user_id}"), + reason: e.to_string(), + })?; + if !options.dry_run { + resolver + .adopt_migrated_identity(identity, &migrated_user) + .await + .map_err(|e| MigrationError::WriteTarget { + domain: format!("identity for {migrated_user_id}"), + reason: e.to_string(), + })?; + } + report.stats.identities += 1; + Ok(()) +} diff --git a/crates/ironclaw_reborn_migration/src/convert/jobs.rs b/crates/ironclaw_reborn_migration/src/convert/jobs.rs new file mode 100644 index 00000000000..6c4730830f8 --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/convert/jobs.rs @@ -0,0 +1,41 @@ +//! Jobs converter (v1 `agent_jobs` / `job_actions` / `job_events`). +//! +//! Reborn has no general-purpose job store: background execution is either the +//! sandboxed process runtime (`ironclaw_processes`, not a historical job log) or +//! trigger run history (`trigger_run_history`, which has no public insert API). +//! Historical v1 jobs therefore have no Reborn target — each is enumerated and +//! recorded as a loss so the operator sees exactly what did not carry over. + +use crate::error::MigrationError; +use crate::options::MigrationOptions; +use crate::report::{Domain, LossReason, MigrationReport}; +use crate::source::V1Source; +use crate::target::RebornTarget; + +pub(crate) async fn run( + src: &V1Source, + _tgt: &mut RebornTarget, + _options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + let jobs = src + .db + .list_agent_jobs() + .await + .map_err(|e| MigrationError::ReadSource { + domain: "agent_jobs".into(), + reason: e.to_string(), + })?; + for job in jobs { + report.record_loss( + Domain::Job, + job.id.to_string(), + "*", + LossReason::NoTargetConcept, + "Reborn has no general job store; historical agent_jobs (and their \ + job_actions/job_events) have no migration target" + .to_string(), + ); + } + Ok(()) +} diff --git a/crates/ironclaw_reborn_migration/src/convert/memory.rs b/crates/ironclaw_reborn_migration/src/convert/memory.rs new file mode 100644 index 00000000000..9f757ed5405 --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/convert/memory.rs @@ -0,0 +1,102 @@ +//! Memory / workspace document converter (v1 `memory_documents` → Reborn +//! `ironclaw_memory` documents). +//! +//! Each non-engine v1 document is written through the memory service under the +//! migrated (tenant, user, agent) scope; content and path are preserved. +//! Engine-v2 documents (mission/project/runtime blobs) are skipped here — they +//! are consumed by the automations converter. Chunks/embeddings are derived +//! state the memory service recomputes on write, so they are not migrated (not +//! a loss). Version history (`memory_document_versions`) has no Reborn target +//! and is recorded as a loss. + +use ironclaw_host_api::{CorrelationId, InvocationId, ResourceScope}; +use ironclaw_memory::{DocumentMetadata, MemoryInvocation, MemoryServiceWriteRequest}; + +use crate::error::MigrationError; +use crate::options::MigrationOptions; +use crate::report::{Domain, LossReason, MigrationReport}; +use crate::source::V1Source; +use crate::target::RebornTarget; +use crate::v2_model; + +pub(crate) async fn run( + src: &V1Source, + tgt: &mut RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + let users = src.distinct_users().await?; + for user_id in &users { + let docs = + src.db + .list_documents(user_id, None) + .await + .map_err(|e| MigrationError::ReadSource { + domain: "memory_documents".into(), + reason: e.to_string(), + })?; + + for doc in docs { + if v2_model::is_engine_path(&doc.path) { + continue; // engine-v2 state — handled by the automations converter + } + + // A malformed source user id is a per-item loss, not a run abort. + let Some(user) = + report.valid_user_id(Domain::Memory, doc.path.clone(), "user_id", &doc.user_id) + else { + continue; + }; + + if options.dry_run { + report.stats.memory_documents += 1; + continue; + } + + let scope = ResourceScope { + tenant_id: tgt.tenant_id.clone(), + user_id: user, + agent_id: Some(tgt.agent_id.clone()), + project_id: None, + mission_id: None, + thread_id: None, + invocation_id: InvocationId::new(), + }; + let invocation = MemoryInvocation { + scope, + correlation_id: CorrelationId::new(), + }; + let metadata = DocumentMetadata::from_value(&doc.metadata); + let request = MemoryServiceWriteRequest { + target: doc.path.clone(), + content: doc.content.clone(), + append: false, + old_string: None, + new_string: None, + replace_all: false, + metadata: Some(metadata), + timezone: None, + }; + + tgt.memory_service + .write(invocation, request) + .await + .map_err(|e| MigrationError::WriteTarget { + domain: format!("memory document {}", doc.path), + reason: e.to_string(), + })?; + report.stats.memory_documents += 1; + } + } + + report.record_loss( + Domain::Memory, + "memory_document_versions", + "*", + LossReason::NoTargetConcept, + "Reborn memory has no per-document version/undo history; v1 \ + memory_document_versions rows are not migrated" + .to_string(), + ); + Ok(()) +} diff --git a/crates/ironclaw_reborn_migration/src/convert/mod.rs b/crates/ironclaw_reborn_migration/src/convert/mod.rs new file mode 100644 index 00000000000..968efafeadb --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/convert/mod.rs @@ -0,0 +1,12 @@ +//! Per-domain converters. Each `run(src, tgt, options, report)` reads one v1 +//! domain and writes the Reborn equivalent, recording losses on `report`. + +pub(crate) mod automations; +pub(crate) mod extensions; +pub(crate) mod heartbeat; +pub(crate) mod identities; +pub(crate) mod jobs; +pub(crate) mod memory; +pub(crate) mod secrets; +pub(crate) mod settings; +pub(crate) mod threads; diff --git a/crates/ironclaw_reborn_migration/src/convert/secrets.rs b/crates/ironclaw_reborn_migration/src/convert/secrets.rs new file mode 100644 index 00000000000..256d597ae79 --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/convert/secrets.rs @@ -0,0 +1,189 @@ +//! Secrets converter (v1 `secrets` → Reborn `FilesystemSecretStore`). +//! +//! v1 and Reborn both use AES-256-GCM but bind ciphertext to *different* schemes, +//! so migration must **decrypt** each v1 secret and **re-encrypt** through +//! Reborn's `SecretStore::put`. Decryption uses the v1 secrets store constructed +//! with the supplied master key (`--secret-master-key`); the same key builds the +//! Reborn crypto in `RebornTarget::secret_store`. Without a master key, secrets +//! are skipped with a recorded loss. A secret whose decrypt fails (e.g. expired, +//! or wrong key) is recorded per-secret and skipped rather than aborting the run. + +use std::sync::Arc; + +use ironclaw::secrets::{SecretsCrypto, create_secrets_store}; +use ironclaw_host_api::{InvocationId, ResourceScope, SecretHandle}; +use ironclaw_secrets::SecretMaterial; + +use crate::error::MigrationError; +use crate::options::MigrationOptions; +use crate::report::{Domain, LossReason, MigrationReport}; +use crate::source::V1Source; +use crate::target::RebornTarget; + +pub(crate) async fn run( + src: &V1Source, + tgt: &mut RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + let Some(master_key) = options.secret_master_key.as_ref() else { + report.record_loss( + Domain::Secret, + "secrets", + "*", + LossReason::NoTargetField, + "no --secret-master-key supplied; v1 secrets cannot be decrypted and were skipped" + .to_string(), + ); + return Ok(()); + }; + let Some(secret_store) = tgt.secret_store.clone() else { + report.record_loss( + Domain::Secret, + "secrets", + "*", + LossReason::NoTargetField, + "target secret store unavailable".to_string(), + ); + return Ok(()); + }; + + // v1 store, built from the same master key, used only to list + decrypt. + let crypto = Arc::new( + SecretsCrypto::new(master_key.clone()) + .map_err(|e| MigrationError::OpenSource(format!("v1 secrets master key: {e}")))?, + ); + let Some(v1_store) = create_secrets_store(crypto, &src.handles) else { + report.record_loss( + Domain::Secret, + "secrets", + "*", + LossReason::Unparseable, + "could not construct a v1 secrets store for the source backend".to_string(), + ); + return Ok(()); + }; + + // v1 `list`/`get_decrypted` are per-user; enumerate users from the raw table. + let users = src.distinct_user_ids_in("secrets", "user_id").await?; + for user_id in users { + let refs = v1_store + .list(&user_id) + .await + .map_err(|e| MigrationError::ReadSource { + domain: "secrets".into(), + reason: e.to_string(), + })?; + for secret_ref in refs { + migrate_one( + &*v1_store, + secret_store.as_ref(), + tgt, + options, + report, + &user_id, + &secret_ref.name, + ) + .await?; + } + } + Ok(()) +} + +// arch-exempt: too_many_args, secret converter threads both v1 + Reborn store +// handles plus scope/options and the user/name key; the aggregation would be a +// per-secret context struct, plan v1-migration +#[allow(clippy::too_many_arguments)] +async fn migrate_one( + v1_store: &(dyn ironclaw::secrets::SecretsStore + Send + Sync), + secret_store: &dyn ironclaw_secrets::SecretStore, + tgt: &RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, + user_id: &str, + name: &str, +) -> Result<(), MigrationError> { + // Decrypt the plaintext. A failure here (expired row, key mismatch) is a + // per-secret loss, not a run-abort. + let decrypted = match v1_store.get_decrypted(user_id, name).await { + Ok(value) => value, + Err(e) => { + report.record_loss( + Domain::Secret, + format!("{user_id}:{name}"), + "decrypt", + LossReason::Unparseable, + format!("could not decrypt v1 secret (skipped): {e}"), + ); + return Ok(()); + } + }; + // Preserve expiry when the record carries one. `get_decrypted` above does + // not surface the record metadata, so this second read fetches `expires_at`; + // a read failure here is not silently dropped — the secret still migrates, + // but the lost expiry is recorded. + let expires_at = match v1_store.get(user_id, name).await { + Ok(secret) => secret.expires_at, + Err(e) => { + report.record_loss( + Domain::Secret, + format!("{user_id}:{name}"), + "expires_at", + LossReason::Degraded, + format!( + "could not re-read v1 secret metadata for expiry (migrated without it): {e}" + ), + ); + None + } + }; + + let handle = match SecretHandle::new(name) { + Ok(handle) => handle, + Err(e) => { + report.record_loss( + Domain::Secret, + format!("{user_id}:{name}"), + "handle", + LossReason::Unparseable, + format!("v1 secret name is not a valid Reborn secret handle: {e}"), + ); + return Ok(()); + } + }; + // A malformed source user id is a per-item loss, not a run abort: skip this + // secret and keep migrating the rest. + let Some(user) = report.valid_user_id( + Domain::Secret, + format!("{user_id}:{name}"), + "user_id", + user_id, + ) else { + return Ok(()); + }; + + if options.dry_run { + report.stats.secrets += 1; + return Ok(()); + } + + let scope = ResourceScope { + tenant_id: tgt.tenant_id.clone(), + user_id: user, + agent_id: Some(tgt.agent_id.clone()), + project_id: None, + mission_id: None, + thread_id: None, + invocation_id: InvocationId::new(), + }; + let material = SecretMaterial::from(decrypted.expose().to_string()); + secret_store + .put(scope, handle, material, expires_at) + .await + .map_err(|e| MigrationError::WriteTarget { + domain: format!("secret {user_id}:{name}"), + reason: e.to_string(), + })?; + report.stats.secrets += 1; + Ok(()) +} diff --git a/crates/ironclaw_reborn_migration/src/convert/settings.rs b/crates/ironclaw_reborn_migration/src/convert/settings.rs new file mode 100644 index 00000000000..58b277380e7 --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/convert/settings.rs @@ -0,0 +1,47 @@ +//! Settings converter (v1 `settings` key/value → Reborn config). +//! +//! Reborn configuration is a *typed* `config.toml` schema plus `providers.json` +//! and the `LlmKeyStore` — there is **no generic key/value settings store**. +//! A standalone migration cannot safely fold arbitrary v1 keys into that typed +//! schema (and the config file lives at the Reborn home, not in the state +//! store), so every v1 setting is enumerated and recorded as a loss naming the +//! key, so an operator can re-apply the ones that matter via `ironclaw-reborn` +//! config. Nothing is silently dropped. + +use crate::error::MigrationError; +use crate::options::MigrationOptions; +use crate::report::{Domain, LossReason, MigrationReport}; +use crate::source::V1Source; +use crate::target::RebornTarget; + +pub(crate) async fn run( + src: &V1Source, + _tgt: &mut RebornTarget, + _options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + let users = src.distinct_users().await?; + for user_id in &users { + let settings = + src.db + .get_all_settings(user_id) + .await + .map_err(|e| MigrationError::ReadSource { + domain: "settings".into(), + reason: e.to_string(), + })?; + for key in settings.keys() { + report.record_loss( + Domain::Setting, + format!("{user_id}:{key}"), + key.clone(), + LossReason::NoTargetConcept, + "Reborn config is a typed config.toml / providers.json / LlmKeyStore; \ + there is no generic key/value settings store to migrate into. \ + Re-apply via `ironclaw-reborn` config if needed." + .to_string(), + ); + } + } + Ok(()) +} diff --git a/crates/ironclaw_reborn_migration/src/convert/threads.rs b/crates/ironclaw_reborn_migration/src/convert/threads.rs new file mode 100644 index 00000000000..16ef869f151 --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/convert/threads.rs @@ -0,0 +1,308 @@ +//! Conversation-history → Reborn session-thread converter. +//! +//! Each v1 `conversations` row becomes a Reborn `SessionThreadRecord` (original +//! id preserved via `EnsureThreadRequest.thread_id`); its `conversation_messages` +//! become transcript messages in order through `SessionThreadService`. Because +//! the append APIs assign their own timestamps and carry no per-message +//! metadata, the original per-message `(role, created_at, id)` provenance is +//! preserved losslessly in the thread's `metadata_json` under a `legacy_v1` +//! key — content, ordering, and role all survive. +//! +//! The reusable [`write_thread`] helper is also used by the automations +//! converter to migrate engine-v2 mission threads. + +use chrono::{DateTime, Utc}; +use ironclaw_host_api::{MissionId, ThreadId, UserId}; +use ironclaw_threads::{ + AcceptInboundMessageRequest, AppendFinalizedAssistantMessageRequest, EnsureThreadRequest, + MessageContent, ThreadScope, +}; +use serde_json::json; +use uuid::Uuid; + +use crate::error::MigrationError; +use crate::options::MigrationOptions; +use crate::report::{Domain, LossReason, MigrationReport}; +use crate::source::V1Source; +use crate::target::RebornTarget; + +/// Normalized role of a source transcript message. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum ImportRole { + User, + Assistant, + /// System / tool / anything without a first-class Reborn append path. + Other, +} + +impl ImportRole { + pub(crate) fn from_v1(role: &str) -> Self { + match role.to_ascii_lowercase().as_str() { + "user" => ImportRole::User, + "assistant" => ImportRole::Assistant, + _ => ImportRole::Other, + } + } +} + +/// One source transcript message to migrate. +pub(crate) struct ImportMessage { + pub(crate) role: ImportRole, + pub(crate) raw_role: String, + pub(crate) content: String, + pub(crate) created_at: DateTime, + pub(crate) orig_id: Option, +} + +/// A source conversation/thread to migrate into one Reborn thread. +pub(crate) struct ThreadImport { + /// Original id, preserved as the Reborn `ThreadId`. + pub(crate) thread_id: Uuid, + pub(crate) owner_user: String, + pub(crate) title: Option, + pub(crate) mission_id: Option, + /// Provenance stored on the thread (channel, timestamps, source kind…). + pub(crate) provenance: serde_json::Value, + pub(crate) messages: Vec, +} + +pub(crate) async fn run( + src: &V1Source, + tgt: &mut RebornTarget, + options: &MigrationOptions, + report: &mut MigrationReport, +) -> Result<(), MigrationError> { + let users = src.distinct_users().await?; + for user_id in &users { + let conversations = src + .db + .list_conversations_all_channels(user_id, i64::MAX) + .await + .map_err(|e| MigrationError::ReadSource { + domain: "conversations".into(), + reason: e.to_string(), + })?; + + for conv in conversations { + let messages = src + .db + .list_conversation_messages(conv.id) + .await + .map_err(|e| MigrationError::ReadSource { + domain: "conversation_messages".into(), + reason: e.to_string(), + })?; + + let import = ThreadImport { + thread_id: conv.id, + owner_user: user_id.clone(), + title: conv.title.clone(), + mission_id: None, + provenance: json!({ + "channel": conv.channel, + "thread_type": conv.thread_type, + "started_at": conv.started_at.to_rfc3339(), + "last_activity": conv.last_activity.to_rfc3339(), + "source": "v1_conversation", + }), + messages: messages + .into_iter() + .map(|m| ImportMessage { + role: ImportRole::from_v1(&m.role), + raw_role: m.role, + content: m.content, + created_at: m.created_at, + orig_id: Some(m.id.to_string()), + }) + .collect(), + }; + + if options.dry_run { + report.stats.threads += 1; + report.stats.messages += import + .messages + .iter() + .filter(|m| m.role != ImportRole::Other) + .count(); + record_other_role_losses(report, &import); + } else { + write_thread(tgt, options, report, import).await?; + } + } + } + Ok(()) +} + +/// Write one source thread into Reborn, preserving id, ordering, role, content, +/// and original per-message timestamps (the latter in `metadata_json`). Shared +/// by conversation and mission-thread migration. +pub(crate) async fn write_thread( + tgt: &RebornTarget, + _options: &MigrationOptions, + report: &mut MigrationReport, + import: ThreadImport, +) -> Result<(), MigrationError> { + // Malformed identity on one source thread is a per-item loss, not a run + // abort: record it and skip this thread so the rest of the migration + // continues (`run_migration` must not be short-circuited by one bad row). + let owner_user = match UserId::new(import.owner_user.clone()) { + Ok(owner_user) => owner_user, + Err(e) => { + record_thread_id_loss(report, &import.thread_id, "owner_user_id", e.to_string()); + return Ok(()); + } + }; + let thread_id = match ThreadId::new(import.thread_id.to_string()) { + Ok(thread_id) => thread_id, + Err(e) => { + record_thread_id_loss(report, &import.thread_id, "thread_id", e.to_string()); + return Ok(()); + } + }; + let mission_id = match import.mission_id { + Some(id) => match MissionId::new(id.to_string()) { + Ok(mission_id) => Some(mission_id), + Err(e) => { + record_thread_id_loss(report, &import.thread_id, "mission_id", e.to_string()); + return Ok(()); + } + }, + None => None, + }; + + let scope = ThreadScope { + tenant_id: tgt.tenant_id.clone(), + agent_id: tgt.agent_id.clone(), + project_id: None, + owner_user_id: Some(owner_user), + mission_id, + }; + let actor_id = scope + .owner_user_id + .as_ref() + .map(|u| u.as_str().to_string()) + .unwrap_or_default(); + + let metadata_json = build_metadata_json(&import)?; + + tgt.thread_service + .ensure_thread(EnsureThreadRequest { + scope: scope.clone(), + thread_id: Some(thread_id.clone()), + created_by_actor_id: actor_id.clone(), + title: import.title.clone(), + metadata_json: Some(metadata_json), + }) + .await + .map_err(|e| write_err("thread", &import.thread_id, e.to_string()))?; + + for message in import.messages { + match message.role { + ImportRole::User => { + tgt.thread_service + .accept_inbound_message(AcceptInboundMessageRequest { + scope: scope.clone(), + thread_id: thread_id.clone(), + actor_id: actor_id.clone(), + source_binding_id: None, + reply_target_binding_id: None, + external_event_id: message.orig_id.clone(), + content: MessageContent::text(message.content), + }) + .await + .map_err(|e| write_err("message", &import.thread_id, e.to_string()))?; + report.stats.messages += 1; + } + ImportRole::Assistant => { + tgt.thread_service + .append_finalized_assistant_message(AppendFinalizedAssistantMessageRequest { + scope: scope.clone(), + thread_id: thread_id.clone(), + turn_run_id: message + .orig_id + .clone() + .unwrap_or_else(|| import.thread_id.to_string()), + content: MessageContent::text(message.content), + }) + .await + .map_err(|e| write_err("message", &import.thread_id, e.to_string()))?; + report.stats.messages += 1; + } + ImportRole::Other => { + // No first-class Reborn append path for system/tool transcript + // messages. Content is retained in `legacy_v1.messages` (built + // above), but it does not become a standalone transcript row. + record_other_role_loss(report, import.thread_id, &message.raw_role); + } + } + } + + report.stats.threads += 1; + Ok(()) +} + +fn build_metadata_json(import: &ThreadImport) -> Result { + let legacy_messages: Vec = import + .messages + .iter() + .map(|m| { + json!({ + "role": m.raw_role, + "created_at": m.created_at.to_rfc3339(), + "orig_id": m.orig_id, + }) + }) + .collect(); + serde_json::to_string(&json!({ + "legacy_v1": { + "orig_thread_id": import.thread_id.to_string(), + "provenance": import.provenance, + "messages": legacy_messages, + } + })) + .map_err(MigrationError::Serde) +} + +pub(crate) fn record_other_role_losses(report: &mut MigrationReport, import: &ThreadImport) { + for message in &import.messages { + if message.role == ImportRole::Other { + record_other_role_loss(report, import.thread_id, &message.raw_role); + } + } +} + +/// Record a thread skipped because one of its identity fields +/// (`owner_user_id` / `thread_id` / `mission_id`) is not a valid Reborn id. +fn record_thread_id_loss( + report: &mut MigrationReport, + thread_id: &Uuid, + field: &str, + reason: String, +) { + report.record_loss( + Domain::Thread, + thread_id.to_string(), + field, + LossReason::Unparseable, + format!("v1 thread {field} is not a valid Reborn id (thread skipped): {reason}"), + ); +} + +fn record_other_role_loss(report: &mut MigrationReport, thread_id: Uuid, raw_role: &str) { + report.record_loss( + Domain::Message, + thread_id.to_string(), + format!("role={raw_role}"), + LossReason::Degraded, + "no Reborn append path for non-user/assistant transcript roles; content \ + retained in thread metadata legacy_v1.messages" + .to_string(), + ); +} + +fn write_err(what: &str, id: &Uuid, reason: String) -> MigrationError { + MigrationError::WriteTarget { + domain: format!("{what} for thread {id}"), + reason, + } +} diff --git a/crates/ironclaw_reborn_migration/src/error.rs b/crates/ironclaw_reborn_migration/src/error.rs new file mode 100644 index 00000000000..a42e783b687 --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/error.rs @@ -0,0 +1,34 @@ +//! Migration error type. +//! +//! Migration is fail-loud on infrastructure errors (can't open a DB, can't +//! write a record) — those abort the run. Per-item conversion problems are NOT +//! errors: they become [`crate::report::LossyItem`]s and the run continues. + +use thiserror::Error; + +#[derive(Debug, Error)] +pub enum MigrationError { + #[error("failed to open v1 source database: {0}")] + OpenSource(String), + + #[error("failed to open Reborn target store: {0}")] + OpenTarget(String), + + #[error("failed to read v1 state ({domain}): {reason}")] + ReadSource { domain: String, reason: String }, + + #[error("failed to write Reborn state ({domain}): {reason}")] + WriteTarget { domain: String, reason: String }, + + #[error("secrets master key required to migrate secrets but none was provided")] + MissingSecretKey, + + #[error("invalid migration input: {0}")] + InvalidInput(String), + + #[error("serialization error: {0}")] + Serde(#[from] serde_json::Error), + + #[error("i/o error: {0}")] + Io(#[from] std::io::Error), +} diff --git a/crates/ironclaw_reborn_migration/src/lib.rs b/crates/ironclaw_reborn_migration/src/lib.rs new file mode 100644 index 00000000000..d50a77d629e --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/lib.rs @@ -0,0 +1,51 @@ +//! v1 / engine-v2 → Reborn state migration. +//! +//! Reads a legacy IronClaw v1 database (PostgreSQL or libSQL) — which is also +//! where engine-v2 state lives, as JSON blobs in `memory_documents` — and +//! writes the equivalent Reborn state into the `RootFilesystem` KV substrate +//! plus the triggers database. Threads and automations (routines + engine-v2 +//! missions) convert without loss; anything that has no Reborn representation +//! today is recorded in a [`MigrationReport`] rather than silently dropped. +//! +//! The crate is a library (this module) plus a thin binary (`src/main.rs`) so +//! the conversion engine can later be reused inside `ironclaw-reborn` startup. + +pub mod error; +pub mod options; +pub mod report; + +mod mounts; +mod source; +mod target; +mod v2_model; + +mod convert; + +pub use error::MigrationError; +pub use options::{MigrationOptions, SourceDb, TargetStore}; +pub use report::{Domain, LossReason, LossyItem, MigrationReport, MigrationStats}; + +/// Run a full migration: open the v1 source and Reborn target, convert every +/// in-scope domain, and return the outcome report. +/// +/// Infrastructure failures (cannot open a store, cannot write a record) abort +/// with a [`MigrationError`]. Per-item representation gaps do not abort — they +/// are accumulated as [`LossyItem`]s on the returned report. +pub async fn run_migration(options: MigrationOptions) -> Result { + let mut report = MigrationReport::new(options.dry_run); + + let src = source::V1Source::open(&options.source).await?; + let mut tgt = target::RebornTarget::open(&options).await?; + + convert::threads::run(&src, &mut tgt, &options, &mut report).await?; + convert::automations::run(&src, &mut tgt, &options, &mut report).await?; + convert::memory::run(&src, &mut tgt, &options, &mut report).await?; + convert::jobs::run(&src, &mut tgt, &options, &mut report).await?; + convert::secrets::run(&src, &mut tgt, &options, &mut report).await?; + convert::extensions::run(&src, &mut tgt, &options, &mut report).await?; + convert::identities::run(&src, &mut tgt, &options, &mut report).await?; + convert::heartbeat::run(&src, &mut tgt, &options, &mut report).await?; + convert::settings::run(&src, &mut tgt, &options, &mut report).await?; + + Ok(report) +} diff --git a/crates/ironclaw_reborn_migration/src/main.rs b/crates/ironclaw_reborn_migration/src/main.rs new file mode 100644 index 00000000000..cce15bafd4e --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/main.rs @@ -0,0 +1,156 @@ +//! `ironclaw-reborn-migration` — convert IronClaw v1 / engine-v2 state into the +//! Reborn state substrate. +//! +//! Thin CLI wrapper over [`ironclaw_reborn_migration::run_migration`]. The +//! conversion engine lives in the library so it can later be reused inside +//! `ironclaw-reborn` startup (documented follow-up). + +use std::path::PathBuf; +use std::process::ExitCode; + +use clap::Parser; +use ironclaw_host_api::{AgentId, TenantId}; +use ironclaw_reborn_migration::{ + MigrationOptions, MigrationReport, SourceDb, TargetStore, run_migration, +}; +use secrecy::SecretString; + +/// Migrate IronClaw v1 / engine-v2 persisted state into Reborn state. +/// +/// Deliberately no `Debug` derive: this struct holds the secrets master key and +/// PostgreSQL connection URLs (which embed `user:password@host`) as plain +/// strings, so a stray `{cli:?}` must not be able to leak them. `clap::Parser` +/// does not require `Debug`. +#[derive(Parser)] +#[command(name = "ironclaw-reborn-migration", version, about)] +struct Cli { + /// v1 source: path to a libSQL/SQLite database file. + #[arg( + long, + conflicts_with = "source_postgres", + env = "MIGRATION_SOURCE_LIBSQL" + )] + source_libsql: Option, + + /// v1 source: PostgreSQL connection URL. + #[arg( + long, + conflicts_with = "source_libsql", + env = "MIGRATION_SOURCE_POSTGRES" + )] + source_postgres: Option, + + /// Reborn target: path to the libSQL store to write (created if absent). + #[arg( + long, + conflicts_with = "target_postgres", + env = "MIGRATION_TARGET_LIBSQL" + )] + target_libsql: Option, + + /// Reborn target: PostgreSQL connection URL. + #[arg( + long, + conflicts_with = "target_libsql", + env = "MIGRATION_TARGET_POSTGRES" + )] + target_postgres: Option, + + /// Reborn tenant all migrated state is written under. + #[arg(long, default_value = "default")] + tenant_id: String, + + /// Reborn agent migrated threads/triggers/memory are scoped to. + #[arg(long, default_value = "default")] + agent_id: String, + + /// Secrets master key (needed only to migrate secrets). Prefer the env var. + #[arg(long, env = "MIGRATION_SECRET_MASTER_KEY")] + secret_master_key: Option, + + /// Report what would be migrated without writing to the Reborn store. + #[arg(long)] + dry_run: bool, + + /// Write the JSON report to this path (otherwise printed to stdout). + #[arg(long)] + report: Option, +} + +impl Cli { + fn into_options(self) -> anyhow::Result<(MigrationOptions, Option)> { + let source = match (self.source_libsql, self.source_postgres) { + (Some(path), None) => SourceDb::LibSql { path }, + (None, Some(url)) => SourceDb::Postgres { + url: SecretString::from(url), + }, + _ => anyhow::bail!("exactly one of --source-libsql / --source-postgres is required"), + }; + let target = match (self.target_libsql, self.target_postgres) { + (Some(path), None) => TargetStore::LibSql { path }, + (None, Some(url)) => TargetStore::Postgres { + url: SecretString::from(url), + }, + _ => anyhow::bail!("exactly one of --target-libsql / --target-postgres is required"), + }; + let tenant_id = TenantId::new(self.tenant_id)?; + let agent_id = AgentId::new(self.agent_id)?; + let options = MigrationOptions { + source, + target, + tenant_id, + agent_id, + secret_master_key: self.secret_master_key.map(SecretString::from), + dry_run: self.dry_run, + }; + Ok((options, self.report)) + } +} + +#[tokio::main] +async fn main() -> ExitCode { + tracing_subscriber::fmt() + .with_env_filter( + tracing_subscriber::EnvFilter::try_from_default_env() + .unwrap_or_else(|_| tracing_subscriber::EnvFilter::new("info")), + ) + .with_writer(std::io::stderr) + .init(); + + match run().await { + Ok(()) => ExitCode::SUCCESS, + Err(error) => { + eprintln!("migration failed: {error:#}"); + ExitCode::FAILURE + } + } +} + +async fn run() -> anyhow::Result<()> { + let cli = Cli::parse(); + let (options, report_path) = cli.into_options()?; + let dry_run = options.dry_run; + + let report = run_migration(options).await?; + emit_report(&report, report_path.as_deref()).await?; + + let losses = report.lossy.len(); + if dry_run { + eprintln!("dry run complete: {} lossy item(s) reported", losses); + } else { + eprintln!("migration complete: {} lossy item(s) reported", losses); + } + Ok(()) +} + +async fn emit_report( + report: &MigrationReport, + path: Option<&std::path::Path>, +) -> anyhow::Result<()> { + let json = report.to_json()?; + match path { + Some(path) => tokio::fs::write(path, json).await?, + None => println!("{json}"), + } + Ok(()) +} diff --git a/crates/ironclaw_reborn_migration/src/mounts.rs b/crates/ironclaw_reborn_migration/src/mounts.rs new file mode 100644 index 00000000000..d2622e0ea65 --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/mounts.rs @@ -0,0 +1,75 @@ +//! Canonical per-domain mount views. +//! +//! Each Reborn domain service resolves records through a `ScopedFilesystem` +//! whose resolver maps a [`ResourceScope`] to a [`MountView`] — alias → a +//! tenant/user-scoped virtual path. These builders mirror the production +//! layout (`ironclaw_reborn_composition::local_dev_mounts` for memory; the +//! tenant/user `/threads` + `/secrets` shape the runtime resolves through). +//! +//! FOLLOW-UP: the production mount resolver lives (private) in +//! `ironclaw_reborn_composition`. Until a shared `pub` accessor exists, this +//! module reproduces that layout; it MUST be reconciled with composition when +//! the migration step is wired into `ironclaw-reborn` startup so the runtime +//! reads back exactly what was migrated. The acceptance test verifies +//! round-trip through these same services, which pins conversion correctness +//! independently of that reconciliation. + +use ironclaw_host_api::{ + HostApiError, MountAlias, MountGrant, MountPermissions, MountView, ResourceScope, + SYSTEM_RESERVED_ID, VirtualPath, +}; + +fn grant(alias: &str, target: String) -> Result { + Ok(MountGrant::new( + MountAlias::new(alias)?, + VirtualPath::new(target)?, + MountPermissions::read_write_list_delete(), + )) +} + +/// Map a scope segment to its on-disk path form, mirroring production's +/// `ironclaw_reborn_composition::invocation_mount_view`: the system sentinel +/// ([`SYSTEM_RESERVED_ID`]) carries control bytes and must render as +/// `__system__` (a valid path segment) so system-scoped service operations +/// (e.g. `FilesystemSessionThreadService` idempotency lookups under +/// `ResourceScope::system()`) resolve to the same paths the runtime reads back. +fn scope_segment(value: &str) -> &str { + if value == SYSTEM_RESERVED_ID { + "__system__" + } else { + value + } +} + +/// `/threads` → `/tenants//users//threads`. Sub-scope (agent, project, +/// mission) is path-encoded by `FilesystemSessionThreadService` inside the alias. +pub(crate) fn threads_mount_view(scope: &ResourceScope) -> Result { + MountView::new(vec![grant( + "/threads", + format!( + "/tenants/{}/users/{}/threads", + scope_segment(scope.tenant_id.as_str()), + scope_segment(scope.user_id.as_str()) + ), + )?]) +} + +/// `/secrets` → `/tenants//users//secrets`. `FilesystemSecretStore` +/// path-encodes agent/project inside the alias. +pub(crate) fn secrets_mount_view(scope: &ResourceScope) -> Result { + MountView::new(vec![grant( + "/secrets", + format!( + "/tenants/{}/users/{}/secrets", + scope_segment(scope.tenant_id.as_str()), + scope_segment(scope.user_id.as_str()) + ), + )?]) +} + +/// Identity records live under the store's fixed `/tenant-shared/reborn-identity` +/// root (partitioned by tenant inside the record path), so the mount exposes the +/// `/tenant-shared` alias — matching the identity crate's own store wiring. +pub(crate) fn identity_mount_view(_scope: &ResourceScope) -> Result { + MountView::new(vec![grant("/tenant-shared", "/tenant-shared".to_string())?]) +} diff --git a/crates/ironclaw_reborn_migration/src/options.rs b/crates/ironclaw_reborn_migration/src/options.rs new file mode 100644 index 00000000000..929d7ab4515 --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/options.rs @@ -0,0 +1,49 @@ +//! Typed inputs to a migration run. + +use std::path::PathBuf; + +use ironclaw_host_api::{AgentId, TenantId}; +use secrecy::SecretString; + +/// Everything a migration run needs: where the v1 state is, where Reborn state +/// should be written, and the Reborn scope dimensions that v1 never had (tenant, +/// agent) so single-user v1 rows land in the right Reborn cell. +#[derive(Clone)] +pub struct MigrationOptions { + /// Backend + connection details for the v1 source database. + pub source: SourceDb, + /// Where to write Reborn state. + pub target: TargetStore, + /// Reborn tenant that all migrated state belongs to. + pub tenant_id: TenantId, + /// Reborn agent that migrated threads/triggers/memory are scoped to. + pub agent_id: AgentId, + /// Secrets master key (v1 ciphertext re-encrypts under Reborn's ported + /// AES-256-GCM scheme). Required only when migrating secrets. + pub secret_master_key: Option, + /// Report only; write nothing to the Reborn store. + pub dry_run: bool, +} + +/// v1 source database selector. Mirrors `ironclaw::config::DatabaseConfig` +/// enough to open a read connection via `ironclaw::db::connect_with_handles`. +#[derive(Debug, Clone)] +pub enum SourceDb { + /// libSQL/SQLite file on disk. + LibSql { path: PathBuf }, + /// PostgreSQL connection URL. Held as a `SecretString` because the URL + /// typically embeds `user:password@host`; `secrecy` redacts it under the + /// derived `Debug`. + Postgres { url: SecretString }, +} + +/// Where Reborn state is written. The `RootFilesystem` KV substrate (threads, +/// memory, secrets, extensions, identity) and the triggers DB share the same +/// underlying backend handle. +#[derive(Debug, Clone)] +pub enum TargetStore { + /// Local libSQL file (the `reborn-local-dev.db` shape). + LibSql { path: PathBuf }, + /// PostgreSQL connection URL. Held as a `SecretString` (see [`SourceDb`]). + Postgres { url: SecretString }, +} diff --git a/crates/ironclaw_reborn_migration/src/report.rs b/crates/ironclaw_reborn_migration/src/report.rs new file mode 100644 index 00000000000..7845f981fa2 --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/report.rs @@ -0,0 +1,155 @@ +//! Migration outcome accounting. +//! +//! Two shapes: [`MigrationStats`] counts what was converted per domain, and +//! [`LossyItem`] records every source field/entity that could **not** be +//! represented in Reborn. Together they form the [`MigrationReport`], which is +//! JSON-serializable so an operator (or a follow-up in-process migration step) +//! can inspect exactly what carried over and what was dropped. Nothing is ever +//! silently lost: a value that has no Reborn home lands here as a `LossyItem`. + +use ironclaw_host_api::UserId; +use serde::{Deserialize, Serialize}; + +/// The domain a converted item or a loss belongs to. Keeps report entries +/// grouped and greppable rather than free-text. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum Domain { + Thread, + Message, + Routine, + Mission, + Job, + Memory, + Secret, + Extension, + Identity, + Heartbeat, + Setting, +} + +/// Why a source value did not fully carry over into Reborn. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum LossReason { + /// The target concept does not exist in Reborn at all (e.g. a durable + /// mission entity, event/webhook trigger sources). + NoTargetConcept, + /// The target type exists but has no field for this value (e.g. routine + /// guardrails, notify config, run counters). + NoTargetField, + /// The value was degraded onto a coarser target (e.g. a `Failed` routine + /// mapped onto `Paused` because Reborn has no failed trigger state). + Degraded, + /// The source value was malformed / unparseable and skipped. + Unparseable, +} + +/// One thing that could not be losslessly represented in Reborn. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct LossyItem { + pub domain: Domain, + /// Source identifier (v1 row id, routine name, mission slug, settings key…). + pub source_id: String, + /// The specific field/aspect that was lost, or `"*"` for the whole entity. + pub field: String, + pub reason: LossReason, + /// Human-readable explanation, ideally naming the Reborn gap. + pub detail: String, +} + +/// Per-domain counts of successfully converted source entities. +#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] +pub struct MigrationStats { + pub threads: usize, + pub messages: usize, + pub routines: usize, + pub missions: usize, + pub trigger_runs: usize, + pub jobs: usize, + pub memory_documents: usize, + pub secrets: usize, + pub extensions: usize, + pub identities: usize, + pub heartbeats: usize, + pub settings: usize, +} + +/// The full outcome of a migration run: what converted, and everything that +/// could not be represented in Reborn. +#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] +pub struct MigrationReport { + /// True when the run was a dry run (nothing written to the Reborn store). + pub dry_run: bool, + pub stats: MigrationStats, + pub lossy: Vec, +} + +impl MigrationReport { + pub fn new(dry_run: bool) -> Self { + Self { + dry_run, + ..Default::default() + } + } + + /// Record a value that could not be losslessly represented in Reborn. + pub fn record_loss( + &mut self, + domain: Domain, + source_id: impl Into, + field: impl Into, + reason: LossReason, + detail: impl Into, + ) { + self.lossy.push(LossyItem { + domain, + source_id: source_id.into(), + field: field.into(), + reason, + detail: detail.into(), + }); + } + + /// Validate a source user id, returning the Reborn [`UserId`] or recording an + /// [`LossReason::Unparseable`] loss and returning `None` when it is invalid. + /// + /// Shared by every converter that scopes a record to a per-user `UserId` + /// (threads owner, secrets, memory docs, identities, routines, missions) so + /// the "validate → skip + record" shape has a single definition and cannot + /// drift between copies. + pub(crate) fn valid_user_id( + &mut self, + domain: Domain, + source_id: impl Into, + field: &'static str, + raw_user_id: &str, + ) -> Option { + match UserId::new(raw_user_id) { + Ok(user) => Some(user), + Err(e) => { + self.record_loss( + domain, + source_id, + field, + LossReason::Unparseable, + format!("source {field} is not a valid Reborn UserId (record skipped): {e}"), + ); + None + } + } + } + + /// Count of losses for a given domain — used by tests and summaries. + pub fn losses_in(&self, domain: Domain) -> usize { + self.lossy + .iter() + .filter(|item| item.domain == domain) + .count() + } + + /// Pretty JSON for `--report ` / stdout. + pub fn to_json(&self) -> serde_json::Result { + serde_json::to_string_pretty(self) + } +} diff --git a/crates/ironclaw_reborn_migration/src/source.rs b/crates/ironclaw_reborn_migration/src/source.rs new file mode 100644 index 00000000000..13bb21f6049 --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/source.rs @@ -0,0 +1,150 @@ +//! v1 / engine-v2 read side. +//! +//! Opens the legacy database through the root `ironclaw` crate and exposes it +//! as an `Arc` plus the backend-specific handles that satellite +//! v1 stores (secrets, wasm tools, identities) need. Engine-v2 mission/project +//! state is not a separate connection — it lives as JSON blobs inside the +//! `memory_documents` table and is read through the same `Database` handle +//! (see [`crate::convert::automations`] and [`crate::v2_model`]). + +use std::sync::Arc; + +use ironclaw::config::{DatabaseBackend, DatabaseConfig, SslMode}; +use ironclaw::db::{Database, DatabaseHandles, connect_with_handles}; +use secrecy::SecretString; + +use crate::error::MigrationError; +use crate::options::SourceDb; + +/// A live, migrations-applied handle to the v1 source database. +/// +/// Crate-internal: the only public entry point is [`crate::run_migration`], and +/// this handle is consumed exclusively by the in-crate converters (mirrors the +/// symmetric `RebornTarget` visibility). +pub(crate) struct V1Source { + pub(crate) db: Arc, + /// Backend-specific handles for satellite v1 stores (secrets) and raw + /// distinct-user / channel-identity discovery. + pub(crate) handles: DatabaseHandles, +} + +/// Tables a v1 user_id can appear in. Queried independently so a DB missing one +/// (e.g. a minimal libSQL install without `settings`) still discovers users +/// from the others. +const USER_ID_TABLES: [&str; 4] = ["conversations", "routines", "memory_documents", "settings"]; + +impl V1Source { + pub(crate) async fn open(source: &SourceDb) -> Result { + let config = source_to_config(source); + let (db, handles) = connect_with_handles(&config) + .await + .map_err(|e| MigrationError::OpenSource(e.to_string()))?; + Ok(Self { db, handles }) + } + + /// Discover every distinct v1 `user_id` present in the source. v1 single-user + /// installs (especially libSQL) may have no `users` table, so users are + /// discovered from the data rows themselves, tolerating any table that does + /// not exist. + pub(crate) async fn distinct_users(&self) -> Result, MigrationError> { + let mut users = std::collections::BTreeSet::new(); + for table in USER_ID_TABLES { + for uid in self.distinct_user_ids_in(table, "user_id").await? { + if !uid.is_empty() { + users.insert(uid); + } + } + } + Ok(users.into_iter().collect()) + } + + /// `SELECT DISTINCT FROM ` against the raw handle. `column` + /// is the user-id column, which is `user_id` on data tables but `id` on the + /// `users` table. + /// + /// A **missing table** is tolerated (returns an empty vec) — minimal v1 + /// installs legitimately lack some tables (e.g. libSQL without `settings`), + /// and "table absent" means "no users here". Every *other* failure — + /// connect, query, or row decode — is a real infrastructure error and + /// propagates, so a transient pool/permission/connection fault can never be + /// silently mistaken for "0 users" and drop everything keyed to them. + /// + /// `table`/`column` are always internal constants, never user input. + pub(crate) async fn distinct_user_ids_in( + &self, + table: &str, + column: &str, + ) -> Result, MigrationError> { + let read_err = |e: &dyn std::fmt::Display| MigrationError::ReadSource { + domain: table.to_string(), + reason: e.to_string(), + }; + let sql = format!("SELECT DISTINCT {column} FROM {table}"); + #[cfg(feature = "libsql")] + if let Some(db) = self.handles.libsql_db.as_ref() { + let conn = db.connect().map_err(|e| read_err(&e))?; + let mut rows = match conn.query(&sql, ()).await { + Ok(rows) => rows, + Err(e) if is_missing_table_error(&e.to_string()) => return Ok(Vec::new()), + Err(e) => return Err(read_err(&e)), + }; + let mut out = Vec::new(); + while let Some(row) = rows.next().await.map_err(|e| read_err(&e))? { + out.push(row.get::(0).map_err(|e| read_err(&e))?); + } + return Ok(out); + } + #[cfg(feature = "postgres")] + if let Some(pool) = self.handles.pg_pool.as_ref() { + let client = pool.get().await.map_err(|e| read_err(&e))?; + let stmt_rows = match client.query(sql.as_str(), &[]).await { + Ok(rows) => rows, + Err(e) if is_missing_table_error(&e.to_string()) => return Ok(Vec::new()), + Err(e) => return Err(read_err(&e)), + }; + return stmt_rows + .iter() + .map(|row| row.try_get::<_, String>(0).map_err(|e| read_err(&e))) + .collect(); + } + Ok(Vec::new()) + } +} + +/// True when a DB error string denotes an absent table/relation, the one case +/// [`V1Source::distinct_user_ids_in`] tolerates. Covers SQLite/libSQL +/// (`no such table`) and PostgreSQL (`relation "…" does not exist`). +/// +/// Deliberately narrow: a bare `does not exist` also matches PostgreSQL's +/// *column*-not-found message (`column "…" does not exist`), so requiring +/// `relation` keeps a schema drift on a real table from being downgraded to an +/// empty user set — exactly the silent-drop class this converter guards against. +pub(crate) fn is_missing_table_error(message: &str) -> bool { + let lower = message.to_ascii_lowercase(); + lower.contains("no such table") + || (lower.contains("relation") && lower.contains("does not exist")) +} + +fn source_to_config(source: &SourceDb) -> DatabaseConfig { + match source { + SourceDb::LibSql { path } => DatabaseConfig { + backend: DatabaseBackend::LibSql, + // libSQL backend ignores `url`; the resolver uses this sentinel too. + url: SecretString::from("unused://libsql"), + pool_size: 4, + ssl_mode: SslMode::default(), + libsql_path: Some(path.clone()), + libsql_url: None, + libsql_auth_token: None, + }, + SourceDb::Postgres { url } => DatabaseConfig { + backend: DatabaseBackend::Postgres, + url: url.clone(), + pool_size: 4, + ssl_mode: SslMode::default(), + libsql_path: None, + libsql_url: None, + libsql_auth_token: None, + }, + } +} diff --git a/crates/ironclaw_reborn_migration/src/target.rs b/crates/ironclaw_reborn_migration/src/target.rs new file mode 100644 index 00000000000..d04ec8dd6e2 --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/target.rs @@ -0,0 +1,323 @@ +//! Reborn write side. +//! +//! Opens the Reborn `RootFilesystem` substrate (and, for triggers, the raw +//! backend DB handle), builds every per-domain write service, and hands them to +//! the converters. Threads / secrets / identity force a concrete filesystem +//! type, so they are constructed inside the backend match arm where `F` is +//! known, then stored as `#[async_trait]` trait objects so the converters stay +//! backend-agnostic. All state is written under one (tenant, agent) scope from +//! [`MigrationOptions`]; each v1 `user_id` becomes the per-record Reborn `UserId`. + +use std::sync::Arc; + +use ironclaw_extensions::ExtensionInstallationStore; +use ironclaw_filesystem::{RootFilesystem, ScopedFilesystem}; +use ironclaw_host_api::{AgentId, ProjectId, TenantId, UserId}; +use ironclaw_memory::MemoryService; +use ironclaw_memory_native::NativeMemoryService; +use ironclaw_reborn_identity::{FilesystemRebornIdentityStore, RebornIdentityResolver}; +use ironclaw_secrets::{FilesystemSecretStore, SecretStore, SecretsCrypto}; +use ironclaw_threads::{FilesystemSessionThreadService, SessionThreadService}; +use ironclaw_triggers::TriggerRepository; +use secrecy::SecretString; + +use crate::error::MigrationError; +use crate::mounts; +use crate::options::{MigrationOptions, TargetStore}; + +/// The concrete Reborn backend the migration writes into. Both the KV substrate +/// and the triggers DB share this one handle. +pub(crate) enum Backend { + #[cfg(feature = "libsql")] + LibSql { + root: Arc, + /// Shared handle for the triggers repo, which uses the raw DB (not the + /// KV substrate). `LibSqlRootFilesystem` does not re-expose it. + db: Arc, + }, + #[cfg(feature = "postgres")] + Postgres { + root: Arc, + pool: deadpool_postgres::Pool, + }, +} + +/// Live Reborn write target: opened backend plus every constructed write +/// service and the scope migrated records are written under. +pub(crate) struct RebornTarget { + /// Held for `identity_store` (the identity row-by-row follow-up); the other + /// services already retain their own root/db Arcs. + #[allow(dead_code)] + pub(crate) backend: Backend, + pub(crate) tenant_id: TenantId, + pub(crate) agent_id: AgentId, + pub(crate) thread_service: Arc, + pub(crate) memory_service: Arc, + pub(crate) trigger_repo: Arc, + pub(crate) extension_store: Arc, + /// Present only when a secrets master key was supplied. + pub(crate) secret_store: Option>, +} + +impl RebornTarget { + pub(crate) async fn open(options: &MigrationOptions) -> Result { + let crypto = match &options.secret_master_key { + Some(key) => Some(Arc::new(build_crypto(key)?)), + None => None, + }; + + let backend = open_backend(&options.target).await?; + let (thread_service, memory_service, secret_store) = match &backend { + #[cfg(feature = "libsql")] + Backend::LibSql { root, .. } => build_kv_services(root.clone(), crypto.clone()), + #[cfg(feature = "postgres")] + Backend::Postgres { root, .. } => build_kv_services(root.clone(), crypto.clone()), + }; + let trigger_repo = build_trigger_repo(&backend).await?; + + // Extension installation store is owned by composition; the migration + // seam builds it over our root filesystem at the default state path. + let root_dyn: Arc = match &backend { + #[cfg(feature = "libsql")] + Backend::LibSql { root, .. } => root.clone(), + #[cfg(feature = "postgres")] + Backend::Postgres { root, .. } => root.clone(), + }; + let extension_store = + ironclaw_reborn_composition::extension_installation_store_for_migration(root_dyn) + .await + .map_err(|e| { + MigrationError::OpenTarget(format!("extension installation store: {e}")) + })?; + + Ok(Self { + backend, + tenant_id: options.tenant_id.clone(), + agent_id: options.agent_id.clone(), + thread_service, + memory_service, + trigger_repo, + extension_store, + secret_store, + }) + } + + /// Build a per-user identity store. Identity records are scoped to a fixed + /// (tenant, user, agent); the store type is generic over the concrete + /// backend, so it is constructed here where `F` is known and returned as a + /// trait object. + pub(crate) fn identity_store(&self, user_id: UserId) -> Arc { + let tenant = self.tenant_id.clone(); + let agent = self.agent_id.clone(); + match &self.backend { + #[cfg(feature = "libsql")] + Backend::LibSql { root, .. } => { + build_identity_store(root.clone(), tenant, user_id, agent) + } + #[cfg(feature = "postgres")] + Backend::Postgres { root, .. } => { + build_identity_store(root.clone(), tenant, user_id, agent) + } + } + } +} + +fn build_crypto(key: &SecretString) -> Result { + SecretsCrypto::new(key.clone()) + .map_err(|e| MigrationError::OpenTarget(format!("secrets master key: {e}"))) +} + +/// The KV-substrate write services built over one backend. +type KvServices = ( + Arc, + Arc, + Option>, +); + +/// Build the filesystem-backed KV services over one concrete backend, returning +/// them as trait objects. +fn build_kv_services(root: Arc, crypto: Option>) -> KvServices +where + F: RootFilesystem + 'static, +{ + let threads_scoped = Arc::new(ScopedFilesystem::new( + root.clone(), + mounts::threads_mount_view, + )); + let thread_service: Arc = + Arc::new(FilesystemSessionThreadService::new(threads_scoped)); + + let root_dyn: Arc = root.clone(); + let memory_service: Arc = + Arc::new(NativeMemoryService::from_filesystem(root_dyn, None)); + + let secret_store: Option> = crypto.map(|crypto| { + let secrets_scoped = Arc::new(ScopedFilesystem::new( + root.clone(), + mounts::secrets_mount_view, + )); + let store: Arc = + Arc::new(FilesystemSecretStore::new(secrets_scoped, crypto)); + store + }); + + (thread_service, memory_service, secret_store) +} + +#[allow(dead_code)] // wired for the identity row-by-row follow-up +fn build_identity_store( + root: Arc, + tenant_id: TenantId, + user_id: UserId, + agent_id: AgentId, +) -> Arc +where + F: RootFilesystem + 'static, +{ + let scoped = Arc::new(ScopedFilesystem::new(root, mounts::identity_mount_view)); + let project_id: Option = None; + let store: Arc = Arc::new(FilesystemRebornIdentityStore::new( + scoped, tenant_id, user_id, agent_id, project_id, + )); + store +} + +async fn build_trigger_repo( + backend: &Backend, +) -> Result, MigrationError> { + match backend { + #[cfg(feature = "libsql")] + Backend::LibSql { db, .. } => { + let repo = ironclaw_triggers::LibSqlTriggerRepository::new(db.clone()); + repo.run_migrations() + .await + .map_err(|e| MigrationError::OpenTarget(format!("trigger migrations: {e}")))?; + let repo: Arc = Arc::new(repo); + Ok(repo) + } + #[cfg(feature = "postgres")] + Backend::Postgres { pool, .. } => { + let repo = ironclaw_triggers::PostgresTriggerRepository::new(pool.clone()); + repo.run_migrations() + .await + .map_err(|e| MigrationError::OpenTarget(format!("trigger migrations: {e}")))?; + let repo: Arc = Arc::new(repo); + Ok(repo) + } + } +} + +async fn open_backend(target: &TargetStore) -> Result { + match target { + #[cfg(feature = "libsql")] + TargetStore::LibSql { path } => { + if let Some(parent) = path.parent() { + tokio::fs::create_dir_all(parent).await?; + } + let db = Arc::new( + libsql::Builder::new_local(path) + .build() + .await + .map_err(|e| MigrationError::OpenTarget(e.to_string()))?, + ); + let root = Arc::new(ironclaw_filesystem::LibSqlRootFilesystem::new(db.clone())); + root.run_migrations() + .await + .map_err(|e| MigrationError::OpenTarget(e.to_string()))?; + Ok(Backend::LibSql { root, db }) + } + #[cfg(not(feature = "libsql"))] + TargetStore::LibSql { .. } => Err(MigrationError::OpenTarget( + "binary built without the libsql feature".into(), + )), + #[cfg(feature = "postgres")] + TargetStore::Postgres { url } => { + let pool = open_postgres_pool(url)?; + let root = Arc::new(ironclaw_filesystem::PostgresRootFilesystem::new( + pool.clone(), + )); + root.run_migrations() + .await + .map_err(|e| MigrationError::OpenTarget(e.to_string()))?; + Ok(Backend::Postgres { root, pool }) + } + #[cfg(not(feature = "postgres"))] + TargetStore::Postgres { .. } => Err(MigrationError::OpenTarget( + "binary built without the postgres feature".into(), + )), + } +} + +/// Build the Reborn target Postgres pool with the repo's remote-TLS rule: +/// remote hosts must use TLS (mirrors `ironclaw_reborn_event_store` and +/// `src/db/tls.rs`). A remote `sslmode=disable` is rejected rather than sending +/// migration traffic — including decrypted secrets — in cleartext; local +/// connections keep plain TCP. TLS wiring is reused from `ironclaw::db::tls`. +#[cfg(feature = "postgres")] +fn open_postgres_pool( + url: &secrecy::SecretString, +) -> Result { + use secrecy::ExposeSecret; + + let raw = url.expose_secret(); + let pg_config = raw + .parse::() + .map_err(|e| MigrationError::OpenTarget(format!("parse Postgres URL: {e}")))?; + let remote = !is_local_postgres_config(&pg_config); + let ssl_mode = match pg_config.get_ssl_mode() { + tokio_postgres::config::SslMode::Disable => { + if remote { + return Err(MigrationError::OpenTarget( + "remote Postgres target requires TLS; sslmode=disable is rejected for \ + migration traffic (it carries decrypted secrets)" + .into(), + )); + } + ironclaw::config::SslMode::Disable + } + // `Prefer`/`Require`/future variants: force TLS on remote, allow the + // parsed intent on local. + _ if remote => ironclaw::config::SslMode::Require, + _ => ironclaw::config::SslMode::Prefer, + }; + + let mut dp_config = deadpool_postgres::Config::new(); + dp_config.url = Some(raw.to_string()); + ironclaw::db::tls::create_pool(&dp_config, ssl_mode) + .map_err(|e| MigrationError::OpenTarget(e.to_string())) +} + +/// True when the parsed Postgres `Config` targets only loopback hosts / Unix +/// sockets. Anything else is treated as remote and must use TLS. Mirrors the +/// event-store's `is_local_postgres_config`. +#[cfg(feature = "postgres")] +fn is_local_postgres_config(config: &tokio_postgres::Config) -> bool { + use tokio_postgres::config::Host; + + let hosts = config.get_hosts(); + let hostaddrs = config.get_hostaddrs(); + if hosts.is_empty() && hostaddrs.is_empty() { + // Empty host list means libpq's compiled-in default socket directory. + return true; + } + for host in hosts { + match host { + #[cfg(unix)] + Host::Unix(_) => continue, + Host::Tcp(name) => { + if !matches!( + name.as_str(), + "localhost" | "127.0.0.1" | "::1" | "[::1]" | "0.0.0.0" + ) { + return false; + } + } + } + } + for addr in hostaddrs { + if !addr.is_loopback() && !addr.is_unspecified() { + return false; + } + } + true +} diff --git a/crates/ironclaw_reborn_migration/src/v2_model.rs b/crates/ironclaw_reborn_migration/src/v2_model.rs new file mode 100644 index 00000000000..3ec28a1a54e --- /dev/null +++ b/crates/ironclaw_reborn_migration/src/v2_model.rs @@ -0,0 +1,191 @@ +//! Deserialization mirror of engine-v2 persisted types. +//! +//! Engine v2 was deleted from the tree (it survives only at git tag +//! `old_engine_v2`), but its state persists as JSON documents inside the v1 +//! `memory_documents` table under `engine/…` / `.system/engine/…` paths. These +//! structs mirror the exact serde representation of +//! `ironclaw_engine::types::{mission, project, thread, message}` at that tag so +//! the migration can parse those blobs. They are read-only parse targets — only +//! the fields the migration consumes are declared; everything else is ignored +//! via `#[serde(default)]` tolerance, so a schema-drifted blob still parses what +//! it can rather than failing the whole run. +//! +//! Serde note: engine-v2 enums used *default* (externally-tagged, PascalCase) +//! derives, and id newtypes are transparent single-field tuple structs — these +//! mirrors reproduce that exactly. +//! +//! `dead_code` is allowed module-wide: several fields exist only to match the +//! persisted wire shape (read by serde during deserialization, then ignored by +//! the converters). Keeping them documents the on-disk contract even when a +//! given migration path does not consume every field. +#![allow(dead_code)] + +use chrono::{DateTime, Utc}; +use serde::Deserialize; +use uuid::Uuid; + +/// Path prefixes under which engine-v2 state was persisted in `memory_documents`. +/// The `.system/engine/…` form is post-#2049; the bare `engine/…` form is the +/// legacy layout. A document whose path starts with either is engine-v2 state. +pub(crate) const ENGINE_PREFIXES: [&str; 2] = ["engine/", ".system/engine/"]; + +/// Returns true if a `memory_documents.path` holds engine-v2 state. +pub(crate) fn is_engine_path(path: &str) -> bool { + ENGINE_PREFIXES.iter().any(|p| path.starts_with(p)) +} + +/// `ironclaw_engine::types::mission::Mission` (subset). +#[derive(Debug, Clone, Deserialize)] +pub(crate) struct Mission { + pub id: Uuid, + #[serde(default)] + pub project_id: Option, + #[serde(default)] + pub user_id: String, + pub name: String, + #[serde(default)] + pub description: Option, + #[serde(default)] + pub goal: String, + #[serde(default)] + pub status: MissionStatus, + #[serde(default)] + pub cadence: MissionCadence, + #[serde(default)] + pub current_focus: Option, + #[serde(default)] + pub approach_history: Vec, + #[serde(default)] + pub thread_history: Vec, + #[serde(default)] + pub success_criteria: Option, + #[serde(default)] + pub notify_channels: Vec, + #[serde(default)] + pub next_fire_at: Option>, + #[serde(default = "epoch_fallback")] + pub created_at: DateTime, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Deserialize, Default)] +pub(crate) enum MissionStatus { + #[default] + Active, + Paused, + Completed, + Failed, +} + +/// `ironclaw_engine::types::mission::MissionCadence`. Externally tagged; unit +/// variant `Manual` is a bare string, struct variants are `{"Cron": {...}}`. +#[derive(Debug, Clone, Deserialize, Default)] +pub(crate) enum MissionCadence { + Cron { + expression: String, + #[serde(default)] + timezone: Option, + }, + OnEvent { + event_pattern: String, + #[serde(default)] + channel: Option, + }, + OnSystemEvent { + source: String, + event_type: String, + }, + Webhook { + path: String, + #[serde(default)] + secret: Option, + }, + #[default] + Manual, +} + +impl MissionCadence { + /// Short tag used in loss reports for non-cron cadences. + pub(crate) fn tag(&self) -> &'static str { + match self { + Self::Cron { .. } => "cron", + Self::OnEvent { .. } => "on_event", + Self::OnSystemEvent { .. } => "on_system_event", + Self::Webhook { .. } => "webhook", + Self::Manual => "manual", + } + } +} + +/// `ironclaw_engine::types::project::Project` (subset). +#[derive(Debug, Clone, Deserialize)] +pub(crate) struct Project { + pub id: Uuid, + #[serde(default)] + pub user_id: String, + pub name: String, + #[serde(default)] + pub description: String, + #[serde(default)] + pub goals: Vec, + #[serde(default = "epoch_fallback")] + pub created_at: DateTime, +} + +/// `ironclaw_engine::types::thread::Thread` (subset) — a mission's execution +/// thread, whose `messages` are the user-visible transcript we migrate. +#[derive(Debug, Clone, Deserialize)] +pub(crate) struct EngineThread { + pub id: Uuid, + #[serde(default)] + pub goal: String, + #[serde(default)] + pub title: Option, + #[serde(default)] + pub project_id: Option, + #[serde(default)] + pub user_id: String, + #[serde(default)] + pub state: EngineThreadState, + #[serde(default)] + pub messages: Vec, + #[serde(default = "epoch_fallback")] + pub created_at: DateTime, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Deserialize, Default)] +pub(crate) enum EngineThreadState { + Created, + #[default] + Running, + Waiting, + Suspended, + Completed, + Done, + Failed, +} + +/// `ironclaw_engine::types::message::ThreadMessage` (subset). +#[derive(Debug, Clone, Deserialize)] +pub(crate) struct ThreadMessage { + pub role: MessageRole, + pub content: String, + #[serde(default = "epoch_fallback")] + pub timestamp: DateTime, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Deserialize)] +pub(crate) enum MessageRole { + System, + User, + Assistant, + ActionResult, +} + +/// Epoch fallback timestamp (`1970-01-01T00:00:00Z`) for optional timestamps in +/// drifted blobs. Migration converters prefer the real persisted timestamp; this +/// only fires when a blob omits one entirely, and the epoch value makes such +/// synthesized timestamps obvious in logs/reports rather than masquerading as a +/// real "now". +fn epoch_fallback() -> DateTime { + DateTime::::from_timestamp(0, 0).unwrap_or_default() +} diff --git a/crates/ironclaw_reborn_migration/tests/migration_roundtrip.rs b/crates/ironclaw_reborn_migration/tests/migration_roundtrip.rs new file mode 100644 index 00000000000..5d42ad5acdd --- /dev/null +++ b/crates/ironclaw_reborn_migration/tests/migration_roundtrip.rs @@ -0,0 +1,660 @@ +//! Acceptance test: seed a rich v1 + engine-v2 libSQL fixture, run the +//! migration into a fresh Reborn libSQL store, and assert the full round-trip — +//! threads/messages/routines/missions convert, and the manifest lists exactly +//! the expected lossy items so nothing is silently dropped. A separate dry-run +//! case asserts the report is produced with nothing written. +//! +//! Docker-free (libSQL on tempdirs). Gated `required-features = ["libsql"]`. + +use std::path::{Path, PathBuf}; +use std::sync::Arc; + +use chrono::{TimeZone, Utc}; +use ironclaw::agent::routine::{NotifyConfig, Routine, RoutineAction, RoutineGuardrails, Trigger}; +use ironclaw::config::{DatabaseBackend, DatabaseConfig, SslMode}; +use ironclaw::db::{Database, DatabaseHandles, UserIdentityRecord, connect_with_handles}; +use ironclaw::secrets::{CreateSecretParams, SecretsCrypto, create_secrets_store}; +use ironclaw::tools::wasm::{LibSqlWasmToolStore, StoreToolParams, TrustLevel, WasmToolStore}; +use ironclaw_host_api::TenantId; +use ironclaw_reborn_migration::{Domain, MigrationOptions, SourceDb, TargetStore, run_migration}; +use ironclaw_triggers::{LibSqlTriggerRepository, TriggerRepository, TriggerSchedule}; +use secrecy::SecretString; +use uuid::Uuid; + +const TENANT: &str = "acme"; +const AGENT: &str = "assistant"; +const USER: &str = "alice"; +/// 64-char string ≥ 32 bytes (used verbatim as HKDF IKM by v1 + Reborn crypto). +const MASTER_KEY: &str = "0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"; + +fn libsql_config(path: &Path) -> DatabaseConfig { + DatabaseConfig { + backend: DatabaseBackend::LibSql, + url: SecretString::from("unused://libsql"), + pool_size: 4, + ssl_mode: SslMode::default(), + libsql_path: Some(path.to_path_buf()), + libsql_url: None, + libsql_auth_token: None, + } +} + +/// Build a v1 routine with a given trigger + action. All the guardrail/notify +/// fields are populated so the field-loss assertions have something to find. +fn routine(name: &str, trigger: Trigger, action: RoutineAction, enabled: bool) -> Routine { + let now = Utc.with_ymd_and_hms(2024, 1, 2, 3, 4, 5).unwrap(); + Routine { + id: Uuid::new_v4(), + name: name.to_string(), + description: format!("desc for {name}"), + user_id: USER.to_string(), + enabled, + trigger, + action, + guardrails: RoutineGuardrails { + cooldown: std::time::Duration::from_secs(300), + max_concurrent: 2, + dedup_window: Some(std::time::Duration::from_secs(60)), + }, + notify: NotifyConfig { + channel: Some("telegram".into()), + user: Some(USER.into()), + on_attention: true, + on_failure: true, + on_success: false, + }, + last_run_at: Some(now), + next_fire_at: Some(now), + run_count: 7, + consecutive_failures: 1, + state: serde_json::json!({}), + created_at: now, + updated_at: now, + } +} + +fn lightweight() -> RoutineAction { + RoutineAction::Lightweight { + prompt: "summarize my day".into(), + context_paths: vec!["context/priorities.md".into()], + max_tokens: 2048, + use_tools: true, + max_tool_rounds: 3, + } +} + +fn full_job() -> RoutineAction { + RoutineAction::FullJob { + title: "Nightly report".into(), + description: "produce the nightly report".into(), + max_iterations: 10, + } +} + +/// Seed a rich v1 + engine-v2 fixture and return its libSQL path (kept alive by +/// the returned `TempDir`). +async fn seed_v1_fixture(dir: &std::path::Path) -> PathBuf { + let path = dir.join("v1.db"); + let (db, handles) = connect_with_handles(&libsql_config(&path)) + .await + .expect("open v1 fixture"); + + // ── conversations + messages ── + let c1 = db + .create_conversation("gateway", USER, None) + .await + .expect("conv1"); + db.add_conversation_message(c1, "user", "hello there") + .await + .expect("m1"); + db.add_conversation_message(c1, "assistant", "hi! how can I help?") + .await + .expect("m2"); + db.add_conversation_message(c1, "system", "session started") + .await + .expect("m3 (system → recorded loss)"); + + let c2 = db + .create_conversation("telegram", USER, None) + .await + .expect("conv2"); + db.add_conversation_message(c2, "user", "what's on my calendar?") + .await + .expect("m4"); + db.add_conversation_message(c2, "assistant", "you have 2 meetings") + .await + .expect("m5"); + + // ── routines: every trigger variant × both actions ── + for r in [ + routine("cron-light", cron("0 9 * * *"), lightweight(), true), + routine("cron-fulljob", cron("0 18 * * MON-FRI"), full_job(), false), + routine( + "event-r", + Trigger::Event { + channel: Some("telegram".into()), + pattern: "deploy".into(), + }, + lightweight(), + true, + ), + routine( + "sysevent-r", + Trigger::SystemEvent { + source: "github".into(), + event_type: "issue.opened".into(), + filters: Default::default(), + }, + lightweight(), + true, + ), + routine( + "webhook-r", + Trigger::Webhook { + path: Some("gh".into()), + secret: None, + }, + lightweight(), + true, + ), + routine("manual-r", Trigger::Manual, lightweight(), true), + ] { + db.create_routine(&r).await.expect("create routine"); + } + + // ── engine-v2 state: mission + thread blobs in memory_documents ── + let mission_thread_id = Uuid::new_v4(); + let cron_mission = serde_json::json!({ + "id": Uuid::new_v4(), + "project_id": Uuid::new_v4(), + "user_id": USER, + "name": "daily-digest", + "goal": "compile a daily digest of important updates", + // Failed status on a cron mission exercises the degrade-to-Paused path. + "status": "Failed", + "cadence": { "Cron": { "expression": "0 7 * * *", "timezone": "UTC" } }, + "current_focus": "news", + "approach_history": ["v1", "v2"], + "thread_history": [mission_thread_id.to_string()], + "success_criteria": "digest delivered by 8am", + "notify_channels": ["telegram"], + "created_at": "2024-01-01T00:00:00Z", + }); + write_engine_doc( + db.as_ref(), + ".system/engine/projects/p1/missions/daily-digest/mission.json", + &cron_mission, + ) + .await; + + let event_mission = serde_json::json!({ + "id": Uuid::new_v4(), + "user_id": USER, + "name": "on-deploy", + "goal": "react to deploys", + "status": "Failed", + "cadence": { "OnEvent": { "event_pattern": "deploy", "channel": null } }, + "thread_history": [], + "created_at": "2024-01-01T00:00:00Z", + }); + write_engine_doc( + db.as_ref(), + ".system/engine/projects/p1/missions/on-deploy/mission.json", + &event_mission, + ) + .await; + + let thread_blob = serde_json::json!({ + "id": mission_thread_id.to_string(), + "goal": "compile digest", + "title": "Digest run", + "user_id": USER, + "state": "Completed", + "messages": [ + { "role": "User", "content": "run the digest", "timestamp": "2024-01-01T07:00:00Z" }, + { "role": "Assistant", "content": "digest ready", "timestamp": "2024-01-01T07:00:05Z" }, + ], + "created_at": "2024-01-01T07:00:00Z", + }); + write_engine_doc( + db.as_ref(), + &format!(".system/engine/runtime/threads/active/{mission_thread_id}.json"), + &thread_blob, + ) + .await; + + // ── a non-engine memory document ── + write_doc(db.as_ref(), "context/vision.md", "# Vision\nbe helpful").await; + + // ── settings ── + let mut settings = std::collections::HashMap::new(); + settings.insert("model".to_string(), serde_json::json!("gpt-4")); + settings.insert("timezone".to_string(), serde_json::json!("UTC")); + db.set_all_settings(USER, &settings) + .await + .expect("seed settings"); + + seed_identities(db.as_ref(), &handles).await; + seed_secret(&handles).await; + seed_wasm_tool(&handles).await; + + path +} + +/// A v1 user + OAuth identity (via the Database trait) and a channel identity +/// (raw insert — no trait writer exists). +async fn seed_identities(db: &dyn Database, handles: &DatabaseHandles) { + let now = Utc.with_ymd_and_hms(2024, 1, 2, 3, 4, 5).unwrap(); + let conn = handles + .libsql_db + .as_ref() + .expect("libsql handle") + .connect() + .expect("connect"); + // Raw users insert — `get_or_create_user` would additionally seed an + // assistant thread (a real v1 behavior, but it would perturb the thread + // counts this test pins), so insert the row directly. + conn.execute( + "INSERT INTO users (id, email, display_name, status, role, created_at, updated_at, metadata) \ + VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?6, ?7)", + ( + USER.to_string(), + "alice@example.com".to_string(), + "Alice".to_string(), + "active".to_string(), + "member".to_string(), + now.to_rfc3339(), + "{}".to_string(), + ), + ) + .await + .expect("seed user row"); + + db.create_identity(&UserIdentityRecord { + id: Uuid::new_v4(), + user_id: USER.to_string(), + provider: "google".into(), + provider_user_id: "google-sub-123".into(), + email: Some("alice@example.com".into()), + email_verified: true, + display_name: Some("Alice".into()), + avatar_url: None, + raw_profile: serde_json::json!({}), + created_at: now, + updated_at: now, + }) + .await + .expect("seed user identity"); + + // channel_identities has no Database writer; insert raw (reuse `conn`). + conn.execute( + "INSERT INTO channel_identities (id, owner_id, channel, external_id, created_at) \ + VALUES (?1, ?2, ?3, ?4, ?5)", + ( + Uuid::new_v4().to_string(), + USER.to_string(), + "telegram".to_string(), + "tg-999".to_string(), + now.to_rfc3339(), + ), + ) + .await + .expect("seed channel identity"); +} + +async fn seed_secret(handles: &DatabaseHandles) { + let crypto = Arc::new(SecretsCrypto::new(SecretString::from(MASTER_KEY)).expect("crypto")); + let store = create_secrets_store(crypto, handles).expect("v1 secrets store"); + let mut params = CreateSecretParams::new("openai_api_key", "sk-secret-value"); + params.provider = Some("openai".into()); + store.create(USER, params).await.expect("seed secret"); +} + +async fn seed_wasm_tool(handles: &DatabaseHandles) { + let db = handles.libsql_db.clone().expect("libsql handle"); + let store = LibSqlWasmToolStore::new(db.clone()); + // A default (Active) install so the migration derives Enabled activation. + let tool = store + .store(StoreToolParams { + user_id: USER.to_string(), + name: "weather".to_string(), + version: "1.0.0".to_string(), + wit_version: "0.1.0".to_string(), + description: "Weather lookup".to_string(), + wasm_binary: b"\0asm-fake-binary".to_vec(), + parameters_schema: serde_json::json!({"type": "object"}), + source_url: None, + trust_level: TrustLevel::User, + }) + .await + .expect("seed wasm tool"); + + // Seed tool_capabilities with an allowed secret (no trait writer exists) so + // the migration derives a credential binding to the migrated secret and + // records the capability-config gap. `openai_api_key` matches the seeded + // secret above. + let conn = db.connect().expect("connect"); + conn.execute( + "INSERT INTO tool_capabilities (id, wasm_tool_id, allowed_secrets) VALUES (?1, ?2, ?3)", + ( + Uuid::new_v4().to_string(), + tool.id.to_string(), + serde_json::json!(["openai_api_key"]).to_string(), + ), + ) + .await + .expect("seed tool capabilities"); +} + +fn cron(expr: &str) -> Trigger { + Trigger::Cron { + schedule: expr.to_string(), + timezone: Some("UTC".to_string()), + } +} + +async fn write_engine_doc(db: &dyn Database, path: &str, value: &serde_json::Value) { + write_doc(db, path, &serde_json::to_string(value).unwrap()).await; +} + +async fn write_doc(db: &dyn Database, path: &str, content: &str) { + let doc = db + .get_or_create_document_by_path(USER, None, path) + .await + .expect("create doc"); + db.update_document(doc.id, content) + .await + .expect("write doc content"); +} + +fn options(src: PathBuf, dst: PathBuf, dry_run: bool) -> MigrationOptions { + MigrationOptions { + source: SourceDb::LibSql { path: src }, + target: TargetStore::LibSql { path: dst }, + tenant_id: TenantId::new(TENANT).unwrap(), + agent_id: ironclaw_host_api::AgentId::new(AGENT).unwrap(), + secret_master_key: Some(SecretString::from(MASTER_KEY)), + dry_run, + } +} + +/// Count rows in the Reborn store whose resolved path matches a LIKE pattern, +/// via a fresh connection — proves on-disk durability of a domain's documents. +async fn reborn_entry_count(path: &Path, like: &str) -> i64 { + let db = libsql::Builder::new_local(path) + .build() + .await + .expect("open reborn db"); + let conn = db.connect().expect("connect"); + let mut rows = conn + .query( + "SELECT count(*) FROM root_filesystem_entries WHERE path LIKE ?1", + [like], + ) + .await + .expect("query entries"); + let row = rows.next().await.expect("row").expect("some row"); + row.get::(0).expect("count") +} + +/// Read the `contents` blob of the first Reborn entry matching a LIKE pattern, +/// as UTF-8 — used to assert the shape of a written installation/thread doc. +async fn reborn_entry_content(path: &Path, like: &str) -> String { + let db = libsql::Builder::new_local(path) + .build() + .await + .expect("open reborn db"); + let conn = db.connect().expect("connect"); + let mut rows = conn + .query( + "SELECT contents FROM root_filesystem_entries WHERE path LIKE ?1 LIMIT 1", + [like], + ) + .await + .expect("query entry contents"); + let row = rows.next().await.expect("row").expect("some row"); + let blob = row.get::>(0).expect("contents blob"); + String::from_utf8(blob).expect("utf-8 contents") +} + +async fn reborn_triggers(path: &Path) -> Vec { + let db = Arc::new( + libsql::Builder::new_local(path) + .build() + .await + .expect("open reborn db"), + ); + let repo = LibSqlTriggerRepository::new(db); + repo.run_migrations().await.expect("trigger migrations"); + repo.list_triggers(TenantId::new(TENANT).unwrap()) + .await + .expect("list triggers") +} + +/// Count thread.json documents in the Reborn store via a fresh connection +/// (proves on-disk durability independent of the migration's live handles). +async fn reborn_thread_doc_count(path: &Path) -> i64 { + let db = libsql::Builder::new_local(path) + .build() + .await + .expect("open reborn db"); + let conn = db.connect().expect("connect"); + let mut rows = conn + .query( + "SELECT count(*) FROM root_filesystem_entries WHERE path LIKE '%/thread.json'", + (), + ) + .await + .expect("query thread docs"); + let row = rows.next().await.expect("row").expect("some row"); + row.get::(0).expect("count") +} + +#[tokio::test] +async fn migrates_v1_and_engine_v2_state_without_loss() { + let dir = tempfile::tempdir().unwrap(); + let src = seed_v1_fixture(dir.path()).await; + let dst = dir.path().join("reborn.db"); + + let report = run_migration(options(src.clone(), dst.clone(), false)) + .await + .expect("migration runs"); + + // ── converted counts ── + // 2 conversations + 1 mission thread. + assert_eq!(report.stats.threads, 3, "threads: {:?}", report.stats); + // 2 cron routines converted (event/sysevent/webhook/manual do not). + assert_eq!(report.stats.routines, 2, "routines: {:?}", report.stats); + // both missions are counted (even the non-cron one). + assert_eq!(report.stats.missions, 2, "missions: {:?}", report.stats); + // user+assistant messages: conv1 (2) + conv2 (2) + mission thread (2) = 6. + assert_eq!(report.stats.messages, 6, "messages: {:?}", report.stats); + assert_eq!( + report.stats.memory_documents, 1, + "memory: {:?}", + report.stats + ); + + // ── the gap set: the EXACT expected lossy count per domain, so any newly + // dropped (or newly-recovered) value fails the build. Counts are pinned to + // the fixture above; see the inline breakdown per domain. ── + let expected_losses = [ + // owner/thread/mission ids are all valid → no thread-identity losses. + (Domain::Thread, 0), + // conv1's single "system" transcript message (no first-class append path). + (Domain::Message, 1), + // 6 routines: each cron routine records 3 field losses + // (action + guardrails/notify/counters + routine_runs); each non-cron + // routine records 1 trigger-source loss + 2 field losses. 6 × 3 = 18. + (Domain::Routine, 18), + // daily-digest: mission_only_fields + status.failed + next_fire_at + // (the fixture mission has no next_fire_at → synthesized) = 3; + // on-deploy: cadence.on_event (1). No orphan threads (blob referenced). + (Domain::Mission, 4), + // fixture seeds no jobs → the job converter records nothing. + (Domain::Job, 0), + // single unconditional memory_document_versions gap. + (Domain::Memory, 1), + // the seeded secret decrypts, re-encrypts, and carries no expiry → 0. + (Domain::Secret, 0), + // the migrated wasm tool: manifest_fidelity + capabilities (2). + (Domain::Extension, 2), + // unconditional pairing_requests gap (both identities adopt cleanly). + (Domain::Identity, 1), + // unconditional heartbeat_state gap. + (Domain::Heartbeat, 1), + // one gap per seeded setting key (model, timezone). + (Domain::Setting, 2), + ]; + for (domain, expected) in expected_losses { + assert_eq!( + report.losses_in(domain), + expected, + "expected exactly {expected} recorded gap(s) for {domain:?}; \ + all losses: {:#?}", + report.lossy + ); + } + // The per-domain buckets must sum to the whole report — a newly-dropped + // value in an unasserted domain would break this even if it slipped past the + // per-domain checks above. + let expected_total: usize = expected_losses.iter().map(|(_, n)| n).sum(); + assert_eq!( + report.lossy.len(), + expected_total, + "total lossy count must equal the sum of every asserted domain bucket" + ); + // Semantic spot-checks on the gap set (field names, not just counts). + let routine_trigger_gaps = report + .lossy + .iter() + .filter(|l| l.domain == Domain::Routine && l.field.starts_with("trigger.")) + .count(); + assert_eq!( + routine_trigger_gaps, 4, + "event/sysevent/webhook/manual routines" + ); + for (domain, field) in [ + (Domain::Mission, "cadence.on_event"), + (Domain::Mission, "status.failed"), + (Domain::Extension, "manifest_fidelity"), + (Domain::Extension, "capabilities"), + ] { + assert!( + report + .lossy + .iter() + .any(|l| l.domain == domain && l.field == field), + "expected a recorded {domain:?} gap for field `{field}`" + ); + } + + // ── deferred domains now convert ── + // 1 v1 secret decrypted + re-encrypted. + assert_eq!(report.stats.secrets, 1, "secrets: {:?}", report.stats); + // 1 OAuth identity + 1 channel identity adopted. + assert_eq!(report.stats.identities, 2, "identities: {:?}", report.stats); + // 1 installed wasm tool → ExtensionInstallation. + assert_eq!(report.stats.extensions, 1, "extensions: {:?}", report.stats); + + // Extension installation invariants: the on-disk installation record must + // carry Enabled activation (v1 tool status was Active) and the credential + // binding derived from tool_capabilities.allowed_secrets (openai_api_key). + let installation_doc = + reborn_entry_content(&dst, "%/system/extensions/.installations/state.json").await; + assert!( + installation_doc.contains("openai_api_key"), + "installation record must carry the allowed_secrets credential binding; got: {installation_doc}" + ); + // Match the exact serialized `ExtensionActivationState` token (snake_case + // JSON value), not a loose substring, so a future `*_enabled`/`disabled` + // state can't pass by accident. + assert!( + installation_doc.contains("\"enabled\""), + "an Active v1 tool must migrate to Enabled activation; got: {installation_doc}" + ); + assert!( + !installation_doc.contains("\"disabled\""), + "the migrated installation must not be Disabled; got: {installation_doc}" + ); + + // On-disk durability of the deferred domains (fresh connection). + assert!( + reborn_entry_count(&dst, "%/secrets/%openai_api_key.json").await >= 1, + "expected the migrated secret document on disk" + ); + assert!( + reborn_entry_count(&dst, "%/system/extensions/.installations/state.json").await >= 1, + "expected the extension installation state document on disk" + ); + + // ── round-trip through the Reborn triggers repo ── + let triggers = reborn_triggers(&dst).await; + // 2 cron routines + 1 cron mission. + assert_eq!(triggers.len(), 3, "triggers: {triggers:#?}"); + let names: Vec<&str> = triggers.iter().map(|t| t.name.as_str()).collect(); + assert!(names.contains(&"cron-light")); + assert!(names.contains(&"cron-fulljob")); + assert!(names.contains(&"daily-digest")); + let digest = triggers.iter().find(|t| t.name == "daily-digest").unwrap(); + match &digest.schedule { + TriggerSchedule::Cron { expression, .. } => assert_eq!(expression, "0 7 * * *"), + other => panic!("expected cron schedule, got {other:?}"), + } + + // ── on-disk durability of threads (fresh connection) ── + assert!( + reborn_thread_doc_count(&dst).await >= 3, + "expected >=3 persisted thread.json docs" + ); + + // ── idempotency: re-running the migration into the same target re-adopts + // identities (first-writer-wins) and upserts the extension installation by + // its deterministic id, so no duplicate installation doc is written. (Trigger + // ids are freshly minted per run, so triggers are intentionally not + // deduplicated — the tool is a one-shot converter.) ── + let report2 = run_migration(options(src, dst.clone(), false)) + .await + .expect("second migration run"); + assert_eq!( + report2.stats.identities, 2, + "re-run must re-adopt the same 2 identities" + ); + assert_eq!( + report2.stats.extensions, 1, + "re-run must upsert the same installation, not duplicate" + ); + assert_eq!( + reborn_entry_count(&dst, "%/system/extensions/.installations/state.json").await, + 1, + "re-run must not write a second installation state document" + ); +} + +#[tokio::test] +async fn dry_run_reports_without_writing() { + let dir = tempfile::tempdir().unwrap(); + let src = seed_v1_fixture(dir.path()).await; + let dst = dir.path().join("reborn-dry.db"); + + let report = run_migration(options(src, dst.clone(), true)) + .await + .expect("dry run"); + + // Same counts as a real run … + assert_eq!(report.stats.threads, 3); + assert_eq!(report.stats.routines, 2); + assert_eq!(report.stats.missions, 2); + assert!(report.dry_run); + + // … but nothing was written to the Reborn store. + assert!( + reborn_triggers(&dst).await.is_empty(), + "dry run wrote triggers" + ); + assert_eq!( + reborn_thread_doc_count(&dst).await, + 0, + "dry run wrote thread docs" + ); +} diff --git a/src/channels/wasm/mod.rs b/src/channels/wasm/mod.rs index ced28b837ff..80fa374c070 100644 --- a/src/channels/wasm/mod.rs +++ b/src/channels/wasm/mod.rs @@ -116,5 +116,12 @@ pub use schema::{ }; pub use setup::{WasmChannelSetup, inject_channel_credentials, setup_wasm_channels}; pub(crate) use setup::{is_reserved_wasm_channel_name, owner_id_from_capabilities}; +// Read-side store types, exposed for the v1→Reborn migration tool +// (`ironclaw_reborn_migration`) which enumerates installed channels. +#[cfg(feature = "libsql")] +pub use storage::LibSqlWasmChannelStore; +#[cfg(feature = "postgres")] +pub use storage::PostgresWasmChannelStore; +pub use storage::{StoredWasmChannel, WasmChannelStore}; pub(crate) use telegram_host_config::{TELEGRAM_CHANNEL_NAME, bot_username_setting_key}; pub use wrapper::{HttpResponse, SharedWasmChannel, WasmChannel};