From d40cd06e07d89ef4243635f389a043e70649593e Mon Sep 17 00:00:00 2001 From: italic-jinxin <106428113+italic-jinxin@users.noreply.github.com> Date: Tue, 28 Jul 2026 17:48:27 +0800 Subject: [PATCH 1/4] fix(product): decode trace hold authorization input --- .../product_capability_handlers.rs | 13 +++-- .../tests/reborn_services_contract.rs | 48 ++++++++++++++++--- 2 files changed, 49 insertions(+), 12 deletions(-) diff --git a/crates/ironclaw_product/src/reborn_services/product_capability_handlers.rs b/crates/ironclaw_product/src/reborn_services/product_capability_handlers.rs index aa9151ff5b9..f3cdff62da7 100644 --- a/crates/ironclaw_product/src/reborn_services/product_capability_handlers.rs +++ b/crates/ironclaw_product/src/reborn_services/product_capability_handlers.rs @@ -155,11 +155,14 @@ impl ProductCommandHandler { let _: EmptyProductCommandInput = product_command_input(input)?; command_output(services.trace_account_login_link(caller).await?) } - Self::TraceHoldAuthorize => command_output( - services - .authorize_trace_hold(caller, product_command_input(input)?) - .await?, - ), + Self::TraceHoldAuthorize => { + let request: RebornTraceHoldAuthorizeProductRequest = product_command_input(input)?; + command_output( + services + .authorize_trace_hold(caller, request.submission_id) + .await?, + ) + } Self::OperatorConfigSetKey => { let request: RebornOperatorConfigSetProductRequest = product_command_input(input)?; command_output( diff --git a/crates/ironclaw_product/tests/reborn_services_contract.rs b/crates/ironclaw_product/tests/reborn_services_contract.rs index 378c21085f2..62f043aa3cf 100644 --- a/crates/ironclaw_product/tests/reborn_services_contract.rs +++ b/crates/ironclaw_product/tests/reborn_services_contract.rs @@ -116,13 +116,14 @@ use ironclaw_product::{ RebornSkillContentResponse, RebornSkillInfo, RebornSkillListResponse, RebornSkillSearchResponse, RebornSkillSourceKind, RebornSkillTrustLevel, RebornStreamEventsRequest, RebornSubmitTurnResponse, RebornTimelineRequest, - RebornTimelineResponse, RebornTraceCreditsResponse, RebornUpdateMemberRoleRequest, - RebornUpdateProjectRequest, RebornViewPage, RebornViewQuery, ResolveApprovalInteractionRequest, - ResolveApprovalInteractionResponse, ResolveAuthInteractionRequest, - ResolveAuthInteractionResponse, SKILL_CONTENT_VIEW, SKILL_SEARCH_VIEW, SKILLS_VIEW, - SetActiveLlmRequest, SkillsProductService, StaticOperatorStatusService, - THREAD_DELETE_CAPABILITY_ID, THREADS_VIEW, TIMELINE_VIEW, TRACE_ACCOUNT_TRACES_VIEW, - TRACE_CREDITS_VIEW, TriggerRunThreadScope, UpsertLlmProviderRequest, approval_gate_ref, + RebornTimelineResponse, RebornTraceCreditsResponse, RebornTraceHoldAuthorizeProductRequest, + RebornUpdateMemberRoleRequest, RebornUpdateProjectRequest, RebornViewPage, RebornViewQuery, + ResolveApprovalInteractionRequest, ResolveApprovalInteractionResponse, + ResolveAuthInteractionRequest, ResolveAuthInteractionResponse, SKILL_CONTENT_VIEW, + SKILL_SEARCH_VIEW, SKILLS_VIEW, SetActiveLlmRequest, SkillsProductService, + StaticOperatorStatusService, THREAD_DELETE_CAPABILITY_ID, THREADS_VIEW, TIMELINE_VIEW, + TRACE_ACCOUNT_TRACES_VIEW, TRACE_CREDITS_VIEW, TRACE_HOLD_AUTHORIZE_COMMAND, + TriggerRunThreadScope, UpsertLlmProviderRequest, approval_gate_ref, automation_trigger_thread_metadata_json, }; use ironclaw_product::{ @@ -2291,6 +2292,39 @@ async fn default_invoke_uses_canonical_host_types_and_fails_closed() { assert!(!error.retryable); } +#[tokio::test] +async fn trace_hold_authorize_capability_decodes_typed_product_input() { + let services = RebornServices::new( + Arc::new(InMemorySessionThreadService::default()), + Arc::new(FakeTurnCoordinator::default()), + ); + + let error = ProductSurface::invoke( + &services, + caller(), + ironclaw_host_api::ProductSurfaceInvokeRequest { + operation_id: TRACE_HOLD_AUTHORIZE_COMMAND + .capability_id() + .expect("trace hold capability id"), + input: serde_json::to_value(RebornTraceHoldAuthorizeProductRequest { + submission_id: "not-a-submission-id".to_string(), + }) + .expect("trace hold input"), + activity_id: ActivityId::new(), + }, + ) + .await + .expect_err("invalid submission id must fail validation"); + + assert_eq!(error.code, ProductSurfaceErrorCode::InvalidRequest); + assert_eq!(error.kind, ProductSurfaceErrorKind::Validation); + assert_eq!(error.field.as_deref(), Some("submission_id")); + assert_eq!( + error.validation_code, + Some(ProductSurfaceValidationCode::InvalidId) + ); +} + #[tokio::test] async fn duplicate_create_thread_replays_generated_thread_for_same_client_action() { let services = RebornServices::new( From 8d51eddd13b44d1f3ec64cb203d0a26edc7ed81d Mon Sep 17 00:00:00 2001 From: italic-jinxin <106428113+italic-jinxin@users.noreply.github.com> Date: Tue, 28 Jul 2026 18:04:13 +0800 Subject: [PATCH 2/4] test(playwright): stabilize Reborn nightly matrix --- .github/workflows/README.md | 7 ++ .github/workflows/reborn-playwright.yml | 14 +-- tests/e2e/reborn_webui_harness.py | 85 +++++++++++++++++-- .../scenarios/test_reborn_v2_file_download.py | 7 +- .../test_reborn_webui_v2_extensions_api.py | 18 ++-- .../test_reborn_webui_v2_legacy_extensions.py | 16 ++-- ...reborn_webui_v2_legacy_pending_messages.py | 12 ++- ..._reborn_webui_v2_legacy_settings_search.py | 69 ++------------- ...t_reborn_webui_v2_legacy_tool_execution.py | 3 +- ...reborn_webui_v2_legacy_tool_permissions.py | 15 +--- 10 files changed, 139 insertions(+), 107 deletions(-) diff --git a/.github/workflows/README.md b/.github/workflows/README.md index 9862c16efd5..5ae13efab8a 100644 --- a/.github/workflows/README.md +++ b/.github/workflows/README.md @@ -110,6 +110,13 @@ external services. `nightly-deep-ci.yml` (04:00 UTC) reuses `platform-and-compat.yml`, `reborn-tests.yml`, and `reborn-e2e.yml` via `workflow_call` at full scope. +`reborn-e2e.yml` owns the deterministic Reborn surface coverage used by pull +requests, the merge queue candidate check, and main. The standalone +`reborn-playwright.yml` schedule owns the broader six-shard browser matrix; it +is post-merge nightly coverage, not a required merge check. Failed nightly +shards upload server logs, Playwright traces, screenshots, and videos, and the +nightly watchdog owns alerting for that workflow. + The legacy v1 suite (`test.yml`) is deliberately not invoked — see the freeze note in `nightly-deep-ci.yml`. Two hard-won gotchas are encoded in the configuration: diff --git a/.github/workflows/reborn-playwright.yml b/.github/workflows/reborn-playwright.yml index 538831520a8..08b270944b5 100644 --- a/.github/workflows/reborn-playwright.yml +++ b/.github/workflows/reborn-playwright.yml @@ -48,8 +48,7 @@ jobs: }, { "group": "legacy-settings-extensions", - "files": "tests/e2e/scenarios/test_reborn_webui_v2_legacy_extensions.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_settings_search.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_skills.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_tool_permissions.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_projects.py", - "pytest_args": "-k 'not test_reborn_legacy_always_approve_survives_reborn_restart'" + "files": "tests/e2e/scenarios/test_reborn_webui_v2_legacy_extensions.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_settings_search.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_skills.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_tool_permissions.py tests/e2e/scenarios/test_reborn_webui_v2_legacy_projects.py" }, { "group": "legacy-runtime", @@ -163,15 +162,20 @@ jobs: playwright install --with-deps chromium - name: Run Reborn Playwright shard + env: + IRONCLAW_E2E_ARTIFACT_DIR: tests/e2e/artifacts/${{ matrix.group }} run: pytest ${{ matrix.files }} ${{ matrix.pytest_args }} -v --timeout=120 --durations=25 - - name: Upload screenshots on failure + - name: Upload diagnostics on failure if: failure() uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 with: - name: reborn-playwright-screenshots-${{ matrix.group }} - path: tests/e2e/screenshots/ + name: reborn-playwright-diagnostics-${{ matrix.group }} + path: | + tests/e2e/artifacts/${{ matrix.group }}/ + tests/e2e/screenshots/ if-no-files-found: ignore + retention-days: 7 reborn-playwright: name: Reborn Playwright diff --git a/tests/e2e/reborn_webui_harness.py b/tests/e2e/reborn_webui_harness.py index 313a7bbe175..ef21cb9258a 100644 --- a/tests/e2e/reborn_webui_harness.py +++ b/tests/e2e/reborn_webui_harness.py @@ -10,6 +10,7 @@ import asyncio import json import os +import re import signal import socket import uuid @@ -17,6 +18,7 @@ import httpx import pytest +from playwright.async_api import Error as PlaywrightError from helpers import REBORN_V2_AUTH_TOKEN, SEL_V2, wait_for_ready @@ -34,6 +36,66 @@ MARKET_DATA_DEV_SECRET = "e2e-market-data-shared-key" +class _ArtifactContext: + """Browser context that persists diagnostics when CI requests them.""" + + def __init__(self, context, artifact_dir: Path): + self._context = context + self._artifact_dir = artifact_dir + self._closed = False + + def __getattr__(self, name): + return getattr(self._context, name) + + async def close(self) -> None: + if self._closed: + return + self._closed = True + + for index, page in enumerate(self._context.pages, start=1): + if page.is_closed(): + continue + try: + await page.screenshot( + path=str(self._artifact_dir / f"page-{index}.png"), + full_page=True, + ) + except PlaywrightError: + pass + + try: + await self._context.tracing.stop( + path=str(self._artifact_dir / "trace.zip") + ) + except PlaywrightError: + pass + await self._context.close() + + +class _ArtifactBrowser: + """Browser proxy that records each context under a unique artifact path.""" + + def __init__(self, browser, artifact_root: Path): + self._browser = browser + self._artifact_root = artifact_root + + def __getattr__(self, name): + return getattr(self._browser, name) + + async def new_context(self, *args, **kwargs): + node_id = os.environ.get("PYTEST_CURRENT_TEST", "browser-context").split( + " (", 1 + )[0] + readable_name = re.sub(r"[^A-Za-z0-9_.-]+", "-", node_id).strip("-")[-160:] + context_name = f"{readable_name}-{uuid.uuid4().hex[:8]}" + artifact_dir = self._artifact_root / "browser" / context_name + artifact_dir.mkdir(parents=True, exist_ok=True) + kwargs.setdefault("record_video_dir", str(artifact_dir / "videos")) + context = await self._browser.new_context(*args, **kwargs) + await context.tracing.start(screenshots=True, snapshots=True, sources=True) + return _ArtifactContext(context, artifact_dir) + + def find_free_port() -> int: """Ask the OS for an available loopback port as a startup hint.""" with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock: @@ -116,8 +178,11 @@ async def start_reborn_webui_v2_server( extra_env: dict[str, str] | None = None, ) -> tuple[object, str]: """Start ``ironclaw serve`` and return ``(process, base_url)``.""" + binary_path = str(Path(ironclaw_reborn_binary).resolve()) reborn_home = home_dir / "reborn-home" reborn_home.mkdir(parents=True, exist_ok=True) + workspace_dir = home_dir / "workspace" + workspace_dir.mkdir(parents=True, exist_ok=True) write_config_toml( reborn_home / "config.toml", mock_llm_server, @@ -132,8 +197,14 @@ async def start_reborn_webui_v2_server( for attempt in range(1, 4): port = find_free_port() last_port = port - stdout_path = home_dir / f"{log_prefix}-attempt-{attempt}.stdout.log" - stderr_path = home_dir / f"{log_prefix}-attempt-{attempt}.stderr.log" + artifact_root = os.environ.get("IRONCLAW_E2E_ARTIFACT_DIR", "").strip() + if artifact_root: + log_dir = Path(artifact_root).resolve() / "server-logs" / home_dir.name + log_dir.mkdir(parents=True, exist_ok=True) + else: + log_dir = home_dir + stdout_path = log_dir / f"{log_prefix}-attempt-{attempt}.stdout.log" + stderr_path = log_dir / f"{log_prefix}-attempt-{attempt}.stderr.log" env = { "PATH": os.environ.get("PATH", "/usr/bin:/bin"), @@ -153,7 +224,7 @@ async def start_reborn_webui_v2_server( forward_coverage_env(env) args = [ - ironclaw_reborn_binary, + binary_path, "serve", "--host", "127.0.0.1", @@ -170,6 +241,7 @@ async def start_reborn_webui_v2_server( stdout=out, stderr=err, env=env, + cwd=workspace_dir, ) base_url = f"http://127.0.0.1:{port}" @@ -363,7 +435,6 @@ async def reborn_v2_vision_server(ironclaw_reborn_binary, mock_llm_server, tmp_p @pytest.fixture(scope="module") async def reborn_v2_browser(): """Chromium instance for Reborn v2 tests, independent of the legacy gateway.""" - from playwright.async_api import Error as PlaywrightError from playwright.async_api import async_playwright headless = os.environ.get("HEADED", "").strip() not in ("1", "true") @@ -377,7 +448,11 @@ async def reborn_v2_browser(): if attempt == 2: raise await asyncio.sleep(1) - yield browser + artifact_root = os.environ.get("IRONCLAW_E2E_ARTIFACT_DIR", "").strip() + if artifact_root: + yield _ArtifactBrowser(browser, Path(artifact_root).resolve()) + else: + yield browser await browser.close() diff --git a/tests/e2e/scenarios/test_reborn_v2_file_download.py b/tests/e2e/scenarios/test_reborn_v2_file_download.py index 5f78d11a883..d459c78a5bc 100644 --- a/tests/e2e/scenarios/test_reborn_v2_file_download.py +++ b/tests/e2e/scenarios/test_reborn_v2_file_download.py @@ -304,17 +304,18 @@ async def serve_stat(route): breadcrumb = page.get_by_role("navigation", name="workspace") await breadcrumb.get_by_role("button", name="Home", exact=True).click() await expect(guide).to_be_visible() - report = tree.get_by_role("treeitem", name="report.pdf", exact=True) - await expect(report).to_be_visible() report_row = page.locator( SEL_V2["workspace_directory_entry_for"].format( path="workspace/report.pdf" ) ) await report_row.click() + report = tree.get_by_role("treeitem", name="report.pdf", exact=True) + await expect(report).to_be_visible() await expect(report).to_have_attribute("aria-selected", "true") await expect(report).to_have_attribute("tabindex", "0") - await expect(guide).to_have_attribute("tabindex", "-1") + assert await tree.locator('[role="treeitem"][tabindex="0"]').count() == 1 + await expect(guide).to_have_count(0) # Directory-load errors are live alerts, so they are announced without a # separate pointer interaction. diff --git a/tests/e2e/scenarios/test_reborn_webui_v2_extensions_api.py b/tests/e2e/scenarios/test_reborn_webui_v2_extensions_api.py index 6b78ac6d103..a7d45b163ef 100644 --- a/tests/e2e/scenarios/test_reborn_webui_v2_extensions_api.py +++ b/tests/e2e/scenarios/test_reborn_webui_v2_extensions_api.py @@ -87,18 +87,14 @@ async def test_reborn_v2_extension_lifecycle_served(reborn_v2_server): # Runtime is an implementation badge (`runtime`), never taxonomy; # the retired `kind` wire string is gone (NEA-25). assert installed["runtime"] == "first_party" - # Membership plus readiness is the complete public lifecycle. - # A setup-free extension becomes active as part of install; there - # is no caller-visible activation checkpoint. + # Installation state and the compatibility readiness fields must + # describe the same setup-free active extension. assert installed["installation_state"] == "active" - for retired in ( - "authenticated", - "active", - "needs_setup", - "has_auth", - "onboarding_state", - ): - assert retired not in installed + assert installed["authenticated"] is True + assert installed["active"] is True + assert installed["needs_setup"] is False + assert installed["has_auth"] is False + assert installed.get("onboarding_state") is None setup = await client.get( f"{reborn_v2_server}/api/webchat/v2/extensions/web-access/setup", diff --git a/tests/e2e/scenarios/test_reborn_webui_v2_legacy_extensions.py b/tests/e2e/scenarios/test_reborn_webui_v2_legacy_extensions.py index d2ab9e537ee..02c5c22e3c9 100644 --- a/tests/e2e/scenarios/test_reborn_webui_v2_legacy_extensions.py +++ b/tests/e2e/scenarios/test_reborn_webui_v2_legacy_extensions.py @@ -821,6 +821,9 @@ def record_catalog_request(request): handle_llm_providers, ) + async def abort_catalog_request(route): + await route.abort("internetdisconnected") + try: async with page.expect_response("**/api/webchat/v2/llm/providers"): await page.goto( @@ -835,7 +838,9 @@ def record_catalog_request(request): timeout=15000 ) - await context.set_offline(True) + # Keep the browser online so the lazy Extensions route chunk can load, + # then fail only the API calls this scenario owns. + await page.route("**/api/webchat/v2/extensions**", abort_catalog_request) await page.get_by_role("link", name="Extensions").first.click() error_banner = page.get_by_role("alert") @@ -846,11 +851,12 @@ def record_catalog_request(request): assert "/api/webchat/v2/extensions/registry" in catalog_requests await expect(page.get_by_text("Registry is empty")).to_have_count(0) - await context.set_offline(False) + await page.unroute( + "**/api/webchat/v2/extensions**", abort_catalog_request + ) await error_banner.get_by_role("button", name="Retry").click() await expect(error_banner).to_have_count(0, timeout=10000) finally: - await context.set_offline(False) await context.close() @@ -912,10 +918,10 @@ async def test_reborn_legacy_extensions_multiple_installs_remain_listed( installed_tool = _card_by_title(page, "Registry Tool") installed_mcp = _card_by_title(page, "Registry MCP Server") - await expect(installed_tool.get_by_text("installed", exact=True)).to_be_visible( + await expect(installed_tool.get_by_text("active", exact=True)).to_be_visible( timeout=5000 ) - await expect(installed_mcp.get_by_text("installed", exact=True)).to_be_visible() + await expect(installed_mcp.get_by_text("active", exact=True)).to_be_visible() await expect( installed_tool.get_by_role("button", name="Install") ).to_have_count(0) diff --git a/tests/e2e/scenarios/test_reborn_webui_v2_legacy_pending_messages.py b/tests/e2e/scenarios/test_reborn_webui_v2_legacy_pending_messages.py index eb57f627f8e..8ed9a650c05 100644 --- a/tests/e2e/scenarios/test_reborn_webui_v2_legacy_pending_messages.py +++ b/tests/e2e/scenarios/test_reborn_webui_v2_legacy_pending_messages.py @@ -823,7 +823,9 @@ async def handle_failed_send(route, _payload, fulfill_json): has_text="send-failure cleanup test" ) await expect(failed).to_have_count(1, timeout=5000) - await expect(failed).to_contain_text("Service unavailable") + await expect(failed).to_contain_text( + "The request failed: service_unavailable." + ) await expect(failed.get_by_label("Retry message")).to_be_visible() assert len(harness["send_requests"]) == 1 finally: @@ -858,12 +860,16 @@ async def handle_fail_then_success(route, _payload, fulfill_json): has_text="retry failed send test" ) await expect(failed).to_have_count(1, timeout=5000) - await expect(failed).to_contain_text("Service unavailable") + await expect(failed).to_contain_text( + "The request failed: service_unavailable." + ) await failed.get_by_label("Retry message").click() await expect(failed).to_have_count(1, timeout=5000) - await expect(failed).not_to_contain_text("Service unavailable") + await expect(failed).not_to_contain_text( + "The request failed: service_unavailable." + ) await expect(failed.get_by_label("Retry message")).to_have_count(0) assert [request["content"] for request in harness["send_requests"]] == [ "retry failed send test", diff --git a/tests/e2e/scenarios/test_reborn_webui_v2_legacy_settings_search.py b/tests/e2e/scenarios/test_reborn_webui_v2_legacy_settings_search.py index 1b64c144875..bf0afb7c634 100644 --- a/tests/e2e/scenarios/test_reborn_webui_v2_legacy_settings_search.py +++ b/tests/e2e/scenarios/test_reborn_webui_v2_legacy_settings_search.py @@ -70,43 +70,6 @@ }, ] -CHANNEL_SURFACES = [ - { - "kind": "channel", - "channel": "telegram", - "direction": "bidirectional", - "connection": { - "status": "connected", - "strategy": "oauth", - "action": { - "kind": "open_setup", - "submit_label": "Reconnect", - }, - }, - } -] - -MOCK_CHANNEL_EXTENSION = { - "package_ref": {"kind": "extension", "id": "telegram-channel"}, - "display_name": "Telegram Channel", - "runtime": "first_party", - "description": "Configured messaging channel.", - "tools": [], - "installation_state": "active", - "surfaces": [{"kind": "channel", "inbound": True, "outbound": True}], -} - -MOCK_MCP_EXTENSION = { - "package_ref": {"kind": "extension", "id": "beta-mcp"}, - "display_name": "Beta MCP", - "runtime": "mcp", - "description": "Installed MCP server.", - "tools": [], - "installation_state": "setup_needed", - "surfaces": [{"kind": "tool"}], -} - - async def _open_mocked_settings_page( reborn_v2_server, reborn_v2_browser, @@ -177,23 +140,6 @@ async def handle_skills(route): await route.continue_() - async def handle_extensions(route): - request = route.request - path = urlparse(request.url).path - - if path == "/api/webchat/v2/extensions" and request.method == "GET": - await fulfill_json( - route, - {"extensions": [MOCK_CHANNEL_EXTENSION, MOCK_MCP_EXTENSION]}, - ) - return - - if path == "/api/webchat/v2/extensions/registry" and request.method == "GET": - await fulfill_json(route, {"entries": []}) - return - - await route.continue_() - async def handle_llm(route): request = route.request path = urlparse(request.url).path @@ -315,7 +261,6 @@ def provider_from_payload(payload: dict) -> dict: await page.route("**/api/webchat/v2/session", handle_session) await page.route("**/api/webchat/v2/settings/tools**", handle_settings_tools) await page.route("**/api/webchat/v2/skills**", handle_skills) - await page.route("**/api/webchat/v2/extensions**", handle_extensions) await page.route("**/api/webchat/v2/llm/**", handle_llm) await page.goto(f"{reborn_v2_server}/settings/{tab}?token={REBORN_V2_AUTH_TOKEN}") @@ -452,26 +397,26 @@ async def test_reborn_legacy_settings_skills_search_empty_state( await harness["context"].close() -async def test_reborn_legacy_settings_channels_search( +async def test_reborn_legacy_settings_language_search( reborn_v2_server, reborn_v2_browser ): harness = await _open_mocked_settings_page( reborn_v2_server, reborn_v2_browser, - tab="channels", + tab="language", ) try: page = harness["page"] search = harness["search"] - await expect(page.get_by_text("Telegram Channel", exact=True)).to_be_visible( + await expect(page.get_by_text("Español", exact=True)).to_be_visible( timeout=5000 ) - await expect(page.get_by_text("Beta MCP", exact=True)).to_have_count(0) + await expect(page.get_by_text("简体中文", exact=True)).to_be_visible() - await search.fill("telegram") - await expect(page.get_by_text("Telegram Channel", exact=True)).to_be_visible() - await expect(page.get_by_text("Beta MCP", exact=True)).to_have_count(0) + await search.fill("chinese") + await expect(page.get_by_text("简体中文", exact=True)).to_be_visible() + await expect(page.get_by_text("Español", exact=True)).to_have_count(0) await search.fill("nothing-matches-this") await expect( diff --git a/tests/e2e/scenarios/test_reborn_webui_v2_legacy_tool_execution.py b/tests/e2e/scenarios/test_reborn_webui_v2_legacy_tool_execution.py index 64d88315b99..2185f4ff2a8 100644 --- a/tests/e2e/scenarios/test_reborn_webui_v2_legacy_tool_execution.py +++ b/tests/e2e/scenarios/test_reborn_webui_v2_legacy_tool_execution.py @@ -528,4 +528,5 @@ async def test_reborn_legacy_looping_tool_calls_stop_at_low_iteration_boundary( ) assert run_status.get("failure_category") == "iteration_limit", run_status - assert completed_loop_echoes == 1 + # The bounded pre-termination warning is a final tool-capable turn. + assert completed_loop_echoes == 2 diff --git a/tests/e2e/scenarios/test_reborn_webui_v2_legacy_tool_permissions.py b/tests/e2e/scenarios/test_reborn_webui_v2_legacy_tool_permissions.py index 5b4e99e2525..c73d7fac3b7 100644 --- a/tests/e2e/scenarios/test_reborn_webui_v2_legacy_tool_permissions.py +++ b/tests/e2e/scenarios/test_reborn_webui_v2_legacy_tool_permissions.py @@ -2,7 +2,6 @@ import asyncio import json -from pathlib import Path from urllib.parse import quote, unquote, urlparse import httpx @@ -219,13 +218,6 @@ async def _choose_permission(page, name: str, label: str): return button -@pytest.fixture -def reborn_approval_artifact_cleanup(): - yield - for label in ("first", "second"): - Path(f"reborn-approval-{label}.txt").unlink(missing_ok=True) - - async def _set_real_auto_approve(reborn_v2_server: str, enabled: bool): headers = {"Authorization": f"Bearer {REBORN_V2_AUTH_TOKEN}"} async with httpx.AsyncClient(headers=headers) as client: @@ -508,19 +500,18 @@ async def test_reborn_legacy_tool_permission_real_api_persists_and_rejects_locke async def test_reborn_legacy_always_approve_survives_reborn_restart( reborn_v2_restartable_server, - reborn_approval_artifact_cleanup, ): state, start_server, stop_server = reborn_v2_restartable_server capability_id = "builtin.write_file" async with httpx.AsyncClient(headers=reborn_bearer_headers()) as client: base_url = state["base_url"] - reset = await client.post( + force_first_prompt = await client.post( f"{base_url}/api/webchat/v2/settings/tools/{capability_id}", - json={"state": "default"}, + json={"state": "ask_each_time"}, timeout=15, ) - reset.raise_for_status() + force_first_prompt.raise_for_status() thread_id = await create_thread(client, base_url) first_prompt = await _wait_for_gate_prompt_after_send( From ad3921909d541006db20751afe9e65f13087dfe8 Mon Sep 17 00:00:00 2001 From: italic-jinxin <106428113+italic-jinxin@users.noreply.github.com> Date: Tue, 28 Jul 2026 19:18:46 +0800 Subject: [PATCH 3/4] test(playwright): harden nightly diagnostics --- .github/workflows/reborn-playwright.yml | 3 +- tests/e2e/CLAUDE.md | 27 ++++ tests/e2e/reborn_webui_harness.py | 125 ++++++++++++++++-- .../test_reborn_webui_harness_artifacts.py | 56 ++++++++ ...reborn_webui_v2_legacy_pending_messages.py | 3 + 5 files changed, 205 insertions(+), 9 deletions(-) create mode 100644 tests/e2e/scenarios/test_reborn_webui_harness_artifacts.py diff --git a/.github/workflows/reborn-playwright.yml b/.github/workflows/reborn-playwright.yml index 08b270944b5..a8d21ea8a3b 100644 --- a/.github/workflows/reborn-playwright.yml +++ b/.github/workflows/reborn-playwright.yml @@ -32,7 +32,7 @@ jobs: [ { "group": "webui-smoke-files", - "files": "tests/e2e/scenarios/test_reborn_webui_v2_smoke.py tests/e2e/scenarios/test_reborn_v2_file_download.py" + "files": "tests/e2e/scenarios/test_reborn_webui_harness_artifacts.py tests/e2e/scenarios/test_reborn_webui_v2_smoke.py tests/e2e/scenarios/test_reborn_v2_file_download.py" }, { "group": "served-api-routes", @@ -164,6 +164,7 @@ jobs: - name: Run Reborn Playwright shard env: IRONCLAW_E2E_ARTIFACT_DIR: tests/e2e/artifacts/${{ matrix.group }} + IRONCLAW_E2E_ARTIFACT_MAX_BYTES: "268435456" # 256 MiB per shard run: pytest ${{ matrix.files }} ${{ matrix.pytest_args }} -v --timeout=120 --durations=25 - name: Upload diagnostics on failure diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index a0128bc9a1b..76f3e8349c8 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -64,6 +64,33 @@ pytest scenarios/ --timeout=60 HEADED=1 pytest scenarios/ ``` +## Reborn Playwright diagnostics + +Set `IRONCLAW_E2E_ARTIFACT_DIR` to preserve diagnostics from the shared Reborn +browser and server fixtures. Leave it unset for ordinary local runs, where the +harness keeps logs under pytest's temporary directory and does not enable +Playwright trace, screenshot, or video capture. + +```bash +IRONCLAW_E2E_ARTIFACT_DIR=tests/e2e/artifacts/local-debug \ + pytest scenarios/test_reborn_webui_v2_smoke.py +``` + +The harness writes server stdout/stderr beneath `server-logs/`. Each browser +context gets a directory beneath `browser/` whose name is derived from the +sanitized pytest node ID plus a short unique suffix. A context bundle contains +final page screenshots, a screenshot-only Playwright trace (DOM snapshots and +source capture are disabled), and 960×540 video. The nightly workflow uploads +these diagnostics only when its shard fails. + +Browser artifacts have a 256 MiB per-shard default budget. Override it with a +positive byte count in `IRONCLAW_E2E_ARTIFACT_MAX_BYTES`. Once the budget is +reached, the harness removes the oldest context bundles first; if the newest +bundle alone exceeds the budget, its largest files are removed until it fits. +Server logs remain outside this browser budget so startup and process failures +retain textual evidence even when a browser bundle is pruned; nightly servers +run at warn-level logging to limit that volume. + ## Test Scenarios The suite has grown to ~65+ scenario files. The table below is a **representative diff --git a/tests/e2e/reborn_webui_harness.py b/tests/e2e/reborn_webui_harness.py index ef21cb9258a..c0f725722de 100644 --- a/tests/e2e/reborn_webui_harness.py +++ b/tests/e2e/reborn_webui_harness.py @@ -11,6 +11,7 @@ import json import os import re +import shutil import signal import socket import uuid @@ -28,6 +29,7 @@ DEFAULT_MODEL = "mock-model" VISION_MODEL = "gpt-4o" ACCEPTED_SEND_OUTCOMES = {"submitted", "already_submitted"} +DEFAULT_ARTIFACT_MAX_BYTES = 256 * 1024 * 1024 # Shared tenant secret for the test-tools/market-data fixture (test-tools/README.md). # `IRONCLAW_REBORN_DEV_SECRET__` is read once at `serve` boot, so it must @@ -36,12 +38,91 @@ MARKET_DATA_DEV_SECRET = "e2e-market-data-shared-key" +def _directory_size(path: Path) -> int: + return sum( + entry.stat().st_size + for entry in path.rglob("*") + if entry.is_file() + ) + + +def _enforce_artifact_budget( + browser_artifact_root: Path, + max_bytes: int, + current_artifact_dir: Path, +) -> None: + """Keep browser artifacts within a deterministic per-shard disk budget.""" + bundles = [ + path + for path in browser_artifact_root.iterdir() + if path.is_dir() + ] + bundle_sizes = {path: _directory_size(path) for path in bundles} + total_bytes = sum(bundle_sizes.values()) + if total_bytes <= max_bytes: + return + + oldest_first = sorted( + (path for path in bundles if path != current_artifact_dir), + key=lambda path: path.stat().st_mtime_ns, + ) + for path in oldest_first: + if total_bytes <= max_bytes: + break + total_bytes -= bundle_sizes[path] + shutil.rmtree(path) + + if total_bytes <= max_bytes or not current_artifact_dir.exists(): + return + + largest_first = sorted( + ( + path + for path in current_artifact_dir.rglob("*") + if path.is_file() + ), + key=lambda path: path.stat().st_size, + reverse=True, + ) + for path in largest_first: + if total_bytes <= max_bytes: + break + file_size = path.stat().st_size + path.unlink() + total_bytes -= file_size + + +def _artifact_max_bytes() -> int: + raw_value = os.environ.get("IRONCLAW_E2E_ARTIFACT_MAX_BYTES", "").strip() + if not raw_value: + return DEFAULT_ARTIFACT_MAX_BYTES + try: + max_bytes = int(raw_value) + except ValueError as error: + raise ValueError( + "IRONCLAW_E2E_ARTIFACT_MAX_BYTES must be a positive integer" + ) from error + if max_bytes <= 0: + raise ValueError( + "IRONCLAW_E2E_ARTIFACT_MAX_BYTES must be a positive integer" + ) + return max_bytes + + class _ArtifactContext: """Browser context that persists diagnostics when CI requests them.""" - def __init__(self, context, artifact_dir: Path): + def __init__( + self, + context, + artifact_dir: Path, + browser_artifact_root: Path, + artifact_max_bytes: int, + ): self._context = context self._artifact_dir = artifact_dir + self._browser_artifact_root = browser_artifact_root + self._artifact_max_bytes = artifact_max_bytes self._closed = False def __getattr__(self, name): @@ -70,14 +151,25 @@ async def close(self) -> None: except PlaywrightError: pass await self._context.close() + _enforce_artifact_budget( + self._browser_artifact_root, + self._artifact_max_bytes, + self._artifact_dir, + ) class _ArtifactBrowser: """Browser proxy that records each context under a unique artifact path.""" - def __init__(self, browser, artifact_root: Path): + def __init__( + self, + browser, + artifact_root: Path, + artifact_max_bytes: int, + ): self._browser = browser - self._artifact_root = artifact_root + self._browser_artifact_root = artifact_root / "browser" + self._artifact_max_bytes = artifact_max_bytes def __getattr__(self, name): return getattr(self._browser, name) @@ -88,12 +180,22 @@ async def new_context(self, *args, **kwargs): )[0] readable_name = re.sub(r"[^A-Za-z0-9_.-]+", "-", node_id).strip("-")[-160:] context_name = f"{readable_name}-{uuid.uuid4().hex[:8]}" - artifact_dir = self._artifact_root / "browser" / context_name + artifact_dir = self._browser_artifact_root / context_name artifact_dir.mkdir(parents=True, exist_ok=True) kwargs.setdefault("record_video_dir", str(artifact_dir / "videos")) + kwargs.setdefault("record_video_size", {"width": 960, "height": 540}) context = await self._browser.new_context(*args, **kwargs) - await context.tracing.start(screenshots=True, snapshots=True, sources=True) - return _ArtifactContext(context, artifact_dir) + await context.tracing.start( + screenshots=True, + snapshots=False, + sources=False, + ) + return _ArtifactContext( + context, + artifact_dir, + self._browser_artifact_root, + self._artifact_max_bytes, + ) def find_free_port() -> int: @@ -438,6 +540,10 @@ async def reborn_v2_browser(): from playwright.async_api import async_playwright headless = os.environ.get("HEADED", "").strip() not in ("1", "true") + artifact_root = os.environ.get("IRONCLAW_E2E_ARTIFACT_DIR", "").strip() + artifact_max_bytes = ( + _artifact_max_bytes() if artifact_root else DEFAULT_ARTIFACT_MAX_BYTES + ) async with async_playwright() as p: browser = None for attempt in range(3): @@ -448,9 +554,12 @@ async def reborn_v2_browser(): if attempt == 2: raise await asyncio.sleep(1) - artifact_root = os.environ.get("IRONCLAW_E2E_ARTIFACT_DIR", "").strip() if artifact_root: - yield _ArtifactBrowser(browser, Path(artifact_root).resolve()) + yield _ArtifactBrowser( + browser, + Path(artifact_root).resolve(), + artifact_max_bytes, + ) else: yield browser await browser.close() diff --git a/tests/e2e/scenarios/test_reborn_webui_harness_artifacts.py b/tests/e2e/scenarios/test_reborn_webui_harness_artifacts.py new file mode 100644 index 00000000000..deae82d4aa2 --- /dev/null +++ b/tests/e2e/scenarios/test_reborn_webui_harness_artifacts.py @@ -0,0 +1,56 @@ +"""Contract tests for bounded Reborn Playwright diagnostic artifacts.""" + +import os + +import pytest + +from reborn_webui_harness import ( + _artifact_max_bytes, + _directory_size, + _enforce_artifact_budget, +) + + +def _write_bundle(root, name: str, sizes: list[int], mtime: int): + bundle = root / name + bundle.mkdir() + for index, size in enumerate(sizes): + (bundle / f"artifact-{index}.bin").write_bytes(b"x" * size) + os.utime(bundle, ns=(mtime, mtime)) + return bundle + + +def test_artifact_budget_removes_oldest_context_bundle(tmp_path): + browser_root = tmp_path / "browser" + browser_root.mkdir() + oldest = _write_bundle(browser_root, "oldest", [8], 1) + current = _write_bundle(browser_root, "current", [8], 2) + + _enforce_artifact_budget(browser_root, 10, current) + + assert not oldest.exists() + assert current.exists() + assert _directory_size(browser_root) <= 10 + + +def test_artifact_budget_prunes_largest_file_from_oversized_current_bundle(tmp_path): + browser_root = tmp_path / "browser" + browser_root.mkdir() + current = _write_bundle(browser_root, "current", [3, 9], 1) + + _enforce_artifact_budget(browser_root, 7, current) + + assert (current / "artifact-0.bin").exists() + assert not (current / "artifact-1.bin").exists() + assert _directory_size(browser_root) <= 7 + + +@pytest.mark.parametrize("raw_value", ["0", "not-a-number"]) +def test_artifact_budget_rejects_invalid_environment_values( + monkeypatch, + raw_value, +): + monkeypatch.setenv("IRONCLAW_E2E_ARTIFACT_MAX_BYTES", raw_value) + + with pytest.raises(ValueError, match="must be a positive integer"): + _artifact_max_bytes() diff --git a/tests/e2e/scenarios/test_reborn_webui_v2_legacy_pending_messages.py b/tests/e2e/scenarios/test_reborn_webui_v2_legacy_pending_messages.py index 8ed9a650c05..ec2f08d9266 100644 --- a/tests/e2e/scenarios/test_reborn_webui_v2_legacy_pending_messages.py +++ b/tests/e2e/scenarios/test_reborn_webui_v2_legacy_pending_messages.py @@ -871,6 +871,9 @@ async def handle_fail_then_success(route, _payload, fulfill_json): "The request failed: service_unavailable." ) await expect(failed.get_by_label("Retry message")).to_have_count(0) + await expect(page.locator(SEL_V2["typing_indicator"])).to_be_visible( + timeout=5000 + ) assert [request["content"] for request in harness["send_requests"]] == [ "retry failed send test", "retry failed send test", From 7bd96a18fd48e212f9f112fec10ed7f551751ccf Mon Sep 17 00:00:00 2001 From: italic-jinxin <106428113+italic-jinxin@users.noreply.github.com> Date: Wed, 29 Jul 2026 17:15:52 +0800 Subject: [PATCH 4/4] fix(e2e): preserve failed artifact bundles --- tests/e2e/CLAUDE.md | 18 ++-- tests/e2e/conftest.py | 16 ++++ tests/e2e/reborn_webui_harness.py | 93 ++++++++++++++++++- .../test_reborn_webui_harness_artifacts.py | 50 ++++++++++ 4 files changed, 167 insertions(+), 10 deletions(-) diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index 76f3e8349c8..434663523b3 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -83,13 +83,17 @@ final page screenshots, a screenshot-only Playwright trace (DOM snapshots and source capture are disabled), and 960×540 video. The nightly workflow uploads these diagnostics only when its shard fails. -Browser artifacts have a 256 MiB per-shard default budget. Override it with a -positive byte count in `IRONCLAW_E2E_ARTIFACT_MAX_BYTES`. Once the budget is -reached, the harness removes the oldest context bundles first; if the newest -bundle alone exceeds the budget, its largest files are removed until it fits. -Server logs remain outside this browser budget so startup and process failures -retain textual evidence even when a browser bundle is pruned; nightly servers -run at warn-level logging to limit that volume. +Browser artifacts have a 256 MiB per-shard default soft budget. Override it +with a positive byte count in `IRONCLAW_E2E_ARTIFACT_MAX_BYTES`. Context +bundles remain protected while their test outcome is pending. Once pytest +reports the final outcome, the harness removes the oldest successful bundles +first; if the newest successful bundle alone exceeds the remaining budget, its +largest files are removed until it fits. Failed-test bundles remain protected +until the workflow uploads them, so a shard with failures can temporarily +exceed the soft budget rather than delete the traces needed to diagnose those +failures. Server logs remain outside this browser budget so startup and process +failures retain textual evidence even when successful browser bundles are +pruned; nightly servers run at warn-level logging to limit that volume. ## Test Scenarios diff --git a/tests/e2e/conftest.py b/tests/e2e/conftest.py index f0a4c44d3dd..57b0f201a4b 100644 --- a/tests/e2e/conftest.py +++ b/tests/e2e/conftest.py @@ -82,6 +82,22 @@ TEST_TOOL_NAMES = ("ascii-renderer", "hacker-news", "market-data") +@pytest.hookimpl(hookwrapper=True) +def pytest_runtest_makereport(item, call): + """Forward final pytest outcomes to the optional Reborn artifact recorder.""" + outcome = yield + report = outcome.get_result() + from reborn_webui_harness import ( + _finalize_registered_artifact_bundles, + _mark_registered_artifact_bundles_failed, + ) + + if report.failed: + _mark_registered_artifact_bundles_failed(item.nodeid) + if report.when == "teardown": + _finalize_registered_artifact_bundles(item.nodeid) + + def _latest_mtime(path: Path) -> float: """Return the newest mtime under a file or directory.""" if not path.exists(): diff --git a/tests/e2e/reborn_webui_harness.py b/tests/e2e/reborn_webui_harness.py index c0f725722de..54cbd589651 100644 --- a/tests/e2e/reborn_webui_harness.py +++ b/tests/e2e/reborn_webui_harness.py @@ -30,6 +30,13 @@ VISION_MODEL = "gpt-4o" ACCEPTED_SEND_OUTCOMES = {"submitted", "already_submitted"} DEFAULT_ARTIFACT_MAX_BYTES = 256 * 1024 * 1024 +_ARTIFACT_PENDING_SENTINEL = ".pytest-outcome-pending" +_ARTIFACT_FAILED_SENTINEL = ".pytest-outcome-failed" +_ARTIFACT_BUNDLES_BY_NODE: dict[ + str, + list[tuple[Path, Path, int]], +] = {} +_ARTIFACT_FAILED_NODES: set[str] = set() # Shared tenant secret for the test-tools/market-data fixture (test-tools/README.md). # `IRONCLAW_REBORN_DEV_SECRET__` is read once at `serve` boot, so it must @@ -46,12 +53,35 @@ def _directory_size(path: Path) -> int: ) +def _mark_artifact_bundle_outcome( + artifact_dir: Path, + outcome: str, +) -> None: + pending = artifact_dir / _ARTIFACT_PENDING_SENTINEL + failed = artifact_dir / _ARTIFACT_FAILED_SENTINEL + pending.unlink(missing_ok=True) + failed.unlink(missing_ok=True) + if outcome == "pending": + pending.touch() + elif outcome == "failed": + failed.touch() + elif outcome != "passed": + raise ValueError(f"unsupported artifact outcome: {outcome}") + + +def _artifact_bundle_is_protected(artifact_dir: Path) -> bool: + return ( + (artifact_dir / _ARTIFACT_PENDING_SENTINEL).exists() + or (artifact_dir / _ARTIFACT_FAILED_SENTINEL).exists() + ) + + def _enforce_artifact_budget( browser_artifact_root: Path, max_bytes: int, current_artifact_dir: Path, ) -> None: - """Keep browser artifacts within a deterministic per-shard disk budget.""" + """Prune successful artifacts while preserving pending and failed bundles.""" bundles = [ path for path in browser_artifact_root.iterdir() @@ -63,7 +93,12 @@ def _enforce_artifact_budget( return oldest_first = sorted( - (path for path in bundles if path != current_artifact_dir), + ( + path + for path in bundles + if path != current_artifact_dir + and not _artifact_bundle_is_protected(path) + ), key=lambda path: path.stat().st_mtime_ns, ) for path in oldest_first: @@ -72,7 +107,11 @@ def _enforce_artifact_budget( total_bytes -= bundle_sizes[path] shutil.rmtree(path) - if total_bytes <= max_bytes or not current_artifact_dir.exists(): + if ( + total_bytes <= max_bytes + or not current_artifact_dir.exists() + or _artifact_bundle_is_protected(current_artifact_dir) + ): return largest_first = sorted( @@ -92,6 +131,48 @@ def _enforce_artifact_budget( total_bytes -= file_size +def _register_artifact_bundle( + node_id: str, + browser_artifact_root: Path, + artifact_dir: Path, + max_bytes: int, +) -> None: + _ARTIFACT_BUNDLES_BY_NODE.setdefault(node_id, []).append( + (browser_artifact_root, artifact_dir, max_bytes) + ) + outcome = "failed" if node_id in _ARTIFACT_FAILED_NODES else "pending" + _mark_artifact_bundle_outcome(artifact_dir, outcome) + + +def _mark_registered_artifact_bundles_failed(node_id: str) -> None: + _ARTIFACT_FAILED_NODES.add(node_id) + for _, artifact_dir, _ in _ARTIFACT_BUNDLES_BY_NODE.get(node_id, []): + if artifact_dir.exists(): + _mark_artifact_bundle_outcome(artifact_dir, "failed") + + +def _finalize_registered_artifact_bundles(node_id: str) -> None: + failed = node_id in _ARTIFACT_FAILED_NODES + bundles = _ARTIFACT_BUNDLES_BY_NODE.pop(node_id, []) + _ARTIFACT_FAILED_NODES.discard(node_id) + for _, artifact_dir, _ in bundles: + if artifact_dir.exists(): + _mark_artifact_bundle_outcome( + artifact_dir, + "failed" if failed else "passed", + ) + + roots: dict[tuple[Path, int], Path] = {} + for browser_artifact_root, artifact_dir, max_bytes in bundles: + roots[(browser_artifact_root, max_bytes)] = artifact_dir + for (browser_artifact_root, max_bytes), current_artifact_dir in roots.items(): + _enforce_artifact_budget( + browser_artifact_root, + max_bytes, + current_artifact_dir, + ) + + def _artifact_max_bytes() -> int: raw_value = os.environ.get("IRONCLAW_E2E_ARTIFACT_MAX_BYTES", "").strip() if not raw_value: @@ -182,6 +263,12 @@ async def new_context(self, *args, **kwargs): context_name = f"{readable_name}-{uuid.uuid4().hex[:8]}" artifact_dir = self._browser_artifact_root / context_name artifact_dir.mkdir(parents=True, exist_ok=True) + _register_artifact_bundle( + node_id, + self._browser_artifact_root, + artifact_dir, + self._artifact_max_bytes, + ) kwargs.setdefault("record_video_dir", str(artifact_dir / "videos")) kwargs.setdefault("record_video_size", {"width": 960, "height": 540}) context = await self._browser.new_context(*args, **kwargs) diff --git a/tests/e2e/scenarios/test_reborn_webui_harness_artifacts.py b/tests/e2e/scenarios/test_reborn_webui_harness_artifacts.py index deae82d4aa2..70bb769b3ed 100644 --- a/tests/e2e/scenarios/test_reborn_webui_harness_artifacts.py +++ b/tests/e2e/scenarios/test_reborn_webui_harness_artifacts.py @@ -8,6 +8,10 @@ _artifact_max_bytes, _directory_size, _enforce_artifact_budget, + _finalize_registered_artifact_bundles, + _mark_artifact_bundle_outcome, + _mark_registered_artifact_bundles_failed, + _register_artifact_bundle, ) @@ -45,6 +49,52 @@ def test_artifact_budget_prunes_largest_file_from_oversized_current_bundle(tmp_p assert _directory_size(browser_root) <= 7 +def test_artifact_budget_preserves_failed_bundle_before_successful_bundle(tmp_path): + browser_root = tmp_path / "browser" + browser_root.mkdir() + failed = _write_bundle(browser_root, "failed", [8], 1) + current = _write_bundle(browser_root, "current", [8], 2) + _mark_artifact_bundle_outcome(failed, "failed") + + _enforce_artifact_budget(browser_root, 10, current) + + assert (failed / "artifact-0.bin").exists() + assert not (current / "artifact-0.bin").exists() + assert _directory_size(browser_root) <= 10 + + +def test_artifact_budget_allows_failed_bundle_to_exceed_soft_limit(tmp_path): + browser_root = tmp_path / "browser" + browser_root.mkdir() + failed = _write_bundle(browser_root, "failed", [12], 1) + _mark_artifact_bundle_outcome(failed, "failed") + + _enforce_artifact_budget(browser_root, 7, failed) + + assert (failed / "artifact-0.bin").exists() + assert _directory_size(browser_root) == 12 + + +@pytest.mark.parametrize("failed", [False, True], ids=["passed", "failed"]) +def test_artifact_outcome_is_finalized_after_pytest_teardown( + tmp_path, + failed, +): + browser_root = tmp_path / "browser" + browser_root.mkdir() + artifact_dir = _write_bundle(browser_root, "current", [8], 1) + node_id = f"scenario.py::test_outcome[{failed}]" + _register_artifact_bundle(node_id, browser_root, artifact_dir, 10) + assert (artifact_dir / ".pytest-outcome-pending").exists() + + if failed: + _mark_registered_artifact_bundles_failed(node_id) + _finalize_registered_artifact_bundles(node_id) + + assert not (artifact_dir / ".pytest-outcome-pending").exists() + assert (artifact_dir / ".pytest-outcome-failed").exists() is failed + + @pytest.mark.parametrize("raw_value", ["0", "not-a-number"]) def test_artifact_budget_rejects_invalid_environment_values( monkeypatch,