Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions .github/workflows/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -110,6 +110,13 @@ external services.

`nightly-deep-ci.yml` (04:00 UTC) reuses `platform-and-compat.yml`,
`reborn-tests.yml`, and `reborn-e2e.yml` via `workflow_call` at full scope.
`reborn-e2e.yml` owns the deterministic Reborn surface coverage used by pull
requests, the merge queue candidate check, and main. The standalone
`reborn-playwright.yml` schedule owns the broader six-shard browser matrix; it
is post-merge nightly coverage, not a required merge check. Failed nightly
shards upload server logs, Playwright traces, screenshots, and videos, and the
nightly watchdog owns alerting for that workflow.

The legacy v1 suite (`test.yml`) is deliberately not invoked — see the
freeze note in `nightly-deep-ci.yml`. Two hard-won gotchas are encoded in
the configuration:
Expand Down
17 changes: 11 additions & 6 deletions .github/workflows/reborn-playwright.yml
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,7 @@ jobs:
[
{
"group": "webui-smoke-files",
"files": "tests/e2e/scenarios/test_reborn_webui_v2_smoke.py tests/e2e/scenarios/test_reborn_v2_file_download.py"
"files": "tests/e2e/scenarios/test_reborn_webui_harness_artifacts.py tests/e2e/scenarios/test_reborn_webui_v2_smoke.py tests/e2e/scenarios/test_reborn_v2_file_download.py"
},
{
"group": "served-api-routes",
Expand All @@ -48,8 +48,7 @@ jobs:
},
{
"group": "legacy-settings-extensions",
"files": "tests/e2e/scenarios/test_reborn_webui_v2_legacy_extensions.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_settings_search.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_skills.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_tool_permissions.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_projects.py",
"pytest_args": "-k 'not test_reborn_legacy_always_approve_survives_reborn_restart'"
"files": "tests/e2e/scenarios/test_reborn_webui_v2_legacy_extensions.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_settings_search.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_skills.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_tool_permissions.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_projects.py"
},
{
"group": "legacy-runtime",
Expand Down Expand Up @@ -163,15 +162,21 @@ jobs:
playwright install --with-deps chromium

- name: Run Reborn Playwright shard
env:
IRONCLAW_E2E_ARTIFACT_DIR: tests/e2e/artifacts/${{ matrix.group }}
IRONCLAW_E2E_ARTIFACT_MAX_BYTES: "268435456" # 256 MiB per shard
run: pytest ${{ matrix.files }} ${{ matrix.pytest_args }} -v --timeout=120 --durations=25

- name: Upload screenshots on failure
- name: Upload diagnostics on failure
if: failure()
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: reborn-playwright-screenshots-${{ matrix.group }}
path: tests/e2e/screenshots/
name: reborn-playwright-diagnostics-${{ matrix.group }}
path: |
tests/e2e/artifacts/${{ matrix.group }}/
tests/e2e/screenshots/
if-no-files-found: ignore
retention-days: 7

reborn-playwright:
name: Reborn Playwright
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -165,11 +165,14 @@ impl ProductCommandHandler {
let _: EmptyProductCommandInput = product_command_input(input)?;
command_output(services.trace_account_login_link(caller).await?)
}
Self::TraceHoldAuthorize => command_output(
services
.authorize_trace_hold(caller, product_command_input(input)?)
.await?,
),
Self::TraceHoldAuthorize => {
let request: RebornTraceHoldAuthorizeProductRequest = product_command_input(input)?;
command_output(
services
.authorize_trace_hold(caller, request.submission_id)
.await?,
)
}
Self::OperatorConfigSetKey => {
let request: RebornOperatorConfigSetProductRequest = product_command_input(input)?;
command_output(
Expand Down
48 changes: 41 additions & 7 deletions crates/ironclaw_product/tests/reborn_services_contract.rs
Original file line number Diff line number Diff line change
Expand Up @@ -117,13 +117,14 @@ use ironclaw_product::{
RebornSkillListResponse, RebornSkillSearchResponse, RebornSkillSourceKind,
RebornSkillTrustLevel, RebornStreamEventsRequest, RebornSubmitTurnResponse,
RebornTimelineRequest, RebornTimelineResponse, RebornTraceCreditsResponse,
RebornUpdateMemberRoleRequest, RebornUpdateProjectRequest, RebornViewPage, RebornViewQuery,
ResolveApprovalInteractionRequest, ResolveApprovalInteractionResponse,
ResolveAuthInteractionRequest, ResolveAuthInteractionResponse, SKILL_CONTENT_VIEW,
SKILL_SEARCH_VIEW, SKILLS_VIEW, SetActiveLlmRequest, SkillsProductService,
StaticOperatorStatusService, THREAD_DELETE_CAPABILITY_ID, THREADS_VIEW, TIMELINE_VIEW,
TRACE_ACCOUNT_TRACES_VIEW, TRACE_CREDITS_VIEW, TriggerRunThreadScope, UpsertLlmProviderRequest,
approval_gate_ref, automation_trigger_thread_metadata_json,
RebornTraceHoldAuthorizeProductRequest, RebornUpdateMemberRoleRequest,
RebornUpdateProjectRequest, RebornViewPage, RebornViewQuery, ResolveApprovalInteractionRequest,
ResolveApprovalInteractionResponse, ResolveAuthInteractionRequest,
ResolveAuthInteractionResponse, SKILL_CONTENT_VIEW, SKILL_SEARCH_VIEW, SKILLS_VIEW,
SetActiveLlmRequest, SkillsProductService, StaticOperatorStatusService,
THREAD_DELETE_CAPABILITY_ID, THREADS_VIEW, TIMELINE_VIEW, TRACE_ACCOUNT_TRACES_VIEW,
TRACE_CREDITS_VIEW, TRACE_HOLD_AUTHORIZE_COMMAND, TriggerRunThreadScope,
UpsertLlmProviderRequest, approval_gate_ref, automation_trigger_thread_metadata_json,
};
use ironclaw_product::{
AdminCreateUserFields, AdminCreatedUser, AdminUserError, AdminUserRecord, AdminUserRole,
Expand Down Expand Up @@ -2325,6 +2326,39 @@ async fn default_invoke_uses_canonical_host_types_and_fails_closed() {
assert!(!error.retryable);
}

#[tokio::test]
async fn trace_hold_authorize_capability_decodes_typed_product_input() {
let services = RebornServices::new(
Arc::new(InMemorySessionThreadService::default()),
Arc::new(FakeTurnCoordinator::default()),
);

let error = ProductSurface::invoke(
&services,
caller(),
ironclaw_host_api::ProductSurfaceInvokeRequest {
operation_id: TRACE_HOLD_AUTHORIZE_COMMAND
.capability_id()
.expect("trace hold capability id"),
input: serde_json::to_value(RebornTraceHoldAuthorizeProductRequest {
submission_id: "not-a-submission-id".to_string(),
})
.expect("trace hold input"),
activity_id: ActivityId::new(),
},
)
.await
.expect_err("invalid submission id must fail validation");

assert_eq!(error.code, ProductSurfaceErrorCode::InvalidRequest);
assert_eq!(error.kind, ProductSurfaceErrorKind::Validation);
assert_eq!(error.field.as_deref(), Some("submission_id"));
assert_eq!(
error.validation_code,
Some(ProductSurfaceValidationCode::InvalidId)
);
}

#[tokio::test]
async fn duplicate_create_thread_replays_generated_thread_for_same_client_action() {
let services = RebornServices::new(
Expand Down
31 changes: 31 additions & 0 deletions tests/e2e/CLAUDE.md
Original file line number Diff line number Diff line change
Expand Up @@ -64,6 +64,37 @@ pytest scenarios/ --timeout=60
HEADED=1 pytest scenarios/
```

## Reborn Playwright diagnostics

Set `IRONCLAW_E2E_ARTIFACT_DIR` to preserve diagnostics from the shared Reborn
browser and server fixtures. Leave it unset for ordinary local runs, where the
harness keeps logs under pytest's temporary directory and does not enable
Playwright trace, screenshot, or video capture.

```bash
IRONCLAW_E2E_ARTIFACT_DIR=tests/e2e/artifacts/local-debug \
pytest scenarios/test_reborn_webui_v2_smoke.py
```

The harness writes server stdout/stderr beneath `server-logs/`. Each browser
context gets a directory beneath `browser/` whose name is derived from the
sanitized pytest node ID plus a short unique suffix. A context bundle contains
final page screenshots, a screenshot-only Playwright trace (DOM snapshots and
source capture are disabled), and 960×540 video. The nightly workflow uploads
these diagnostics only when its shard fails.

Browser artifacts have a 256 MiB per-shard default soft budget. Override it
with a positive byte count in `IRONCLAW_E2E_ARTIFACT_MAX_BYTES`. Context
bundles remain protected while their test outcome is pending. Once pytest
reports the final outcome, the harness removes the oldest successful bundles
first; if the newest successful bundle alone exceeds the remaining budget, its
largest files are removed until it fits. Failed-test bundles remain protected
until the workflow uploads them, so a shard with failures can temporarily
exceed the soft budget rather than delete the traces needed to diagnose those
failures. Server logs remain outside this browser budget so startup and process
failures retain textual evidence even when successful browser bundles are
pruned; nightly servers run at warn-level logging to limit that volume.

## Test Scenarios

The suite has grown to ~65+ scenario files. The table below is a **representative
Expand Down
16 changes: 16 additions & 0 deletions tests/e2e/conftest.py
Original file line number Diff line number Diff line change
Expand Up @@ -82,6 +82,22 @@
TEST_TOOL_NAMES = ("ascii-renderer", "hacker-news", "market-data")


@pytest.hookimpl(hookwrapper=True)
def pytest_runtest_makereport(item, call):
"""Forward final pytest outcomes to the optional Reborn artifact recorder."""
outcome = yield
report = outcome.get_result()
from reborn_webui_harness import (
_finalize_registered_artifact_bundles,
_mark_registered_artifact_bundles_failed,
)

if report.failed:
_mark_registered_artifact_bundles_failed(item.nodeid)
if report.when == "teardown":
_finalize_registered_artifact_bundles(item.nodeid)


def _latest_mtime(path: Path) -> float:
"""Return the newest mtime under a file or directory."""
if not path.exists():
Expand Down
Loading
Loading