diff --git a/registry/tools/github.json b/registry/tools/github.json index 3c5d9ce1446..65738341e42 100644 --- a/registry/tools/github.json +++ b/registry/tools/github.json @@ -4,7 +4,7 @@ "kind": "tool", "version": "0.2.3", "wit_version": "0.3.0", - "description": "GitHub integration for repositories, issues, pull requests, search, branches, file writes, releases, and workflows", + "description": "GitHub integration for repos, issues, PRs, branches, files, releases, and workflows. Search actions: search_repositories, search_code, search_issues_pull_requests.", "keywords": [ "git", "code", diff --git a/skills/code-review/SKILL.md b/skills/code-review/SKILL.md index c7e287fe332..de0f6d49d8f 100644 --- a/skills/code-review/SKILL.md +++ b/skills/code-review/SKILL.md @@ -1,7 +1,7 @@ --- name: code-review -version: "1.0.0" -description: Review code changes for bugs, style, and security issues +version: "2.0.0" +description: Paranoid architect review of code changes for bugs, security, missing tests, and undocumented assumptions. Works on local git diffs OR a GitHub pull request (e.g. `owner/repo N`). For PRs, can post findings as line-level review comments. activation: keywords: - "review" @@ -10,24 +10,221 @@ activation: patterns: - "(?i)review\\s.*(code|changes|diff|PR|pull request|commit)" - "(?i)(check|look at|inspect)\\s.*(changes|diff|code)" + - "(?i)review\\s+[a-z0-9._-]+/[a-z0-9._-]+\\s+#?\\d+" tags: - "code-review" - "quality" - max_context_tokens: 1200 + - "security" + max_context_tokens: 2500 +requires: + skills: + - github --- -# Code Review Workflow - -When the user asks to review code: - -1. **Get the changes**: Run `shell` with `git diff` (unstaged) or `git diff --cached` (staged) or `git diff HEAD~1` (last commit) depending on context. -2. **Focus on what changed**, not surrounding code. Don't review unchanged code unless it's directly relevant to the change. -3. **Check for these categories:** - - **Bugs**: Logic errors, off-by-one, null/undefined handling, race conditions - - **Security**: Injection vulnerabilities, credential exposure, path traversal, XSS - - **Error handling**: Missing error cases, swallowed errors, unclear error messages - - **Edge cases**: Empty inputs, large inputs, concurrent access, unicode handling - - **Style**: Inconsistency with surrounding code, unclear naming, missing/excessive comments -4. **Provide actionable feedback** with specific file:line references. Don't just say "this could be improved" - say what to change and why. -5. **Be proportional**: A one-line typo fix doesn't need a full security audit. Match review depth to change scope. -6. If the changes look good, say so clearly. Don't invent problems. +# Paranoid Architect Code Review + +You are reviewing this change as a paranoid architect. Your job is to find every bug, vulnerability, race condition, edge case, and undocumented assumption before it ships. Assume adversarial users, concurrent access, and Murphy's law. + +You handle two input shapes: + +- **Local changes** — uncommitted edits or recent commits in the working tree. +- **GitHub pull request** — `owner/repo N`, `owner/repo#N`, or a `github.com/.../pull/N` URL. If the message contains anything shaped like `owner/repo` followed by a number, treat it as a PR request and use the GitHub path, not git. Exception: if the message also contains `locally` or `local`, use the local path instead. + +## Step 1 — Load the changes + +### GitHub PR path + +Wrap the whole flow in `async def` and `return` from it, then `FINAL(await review())`. `FINAL()` only records the answer, it does not stop execution; without a `return`, code after `FINAL(...)` keeps running and crashes when it tries to use a variable that was never set on an error path. + +```repl +async def review(): + pr_url = f"https://api.github.com/repos/{owner}/{repo}/pulls/{number}" + files_url = f"{pr_url}/files?per_page=100" + # Sequential awaits instead of asyncio.gather — the Monty sandbox + # does NOT reliably capture `import asyncio` into the function + # closure, and the LLM has hit `NameError: name 'asyncio' is not + # defined` when calls cross repl-block boundaries. Three serial + # GETs against api.github.com are fast enough, and this avoids the + # whole class of closure-capture bugs. + meta_r = await http(method="GET", url=pr_url) + diff_r = await http( + method="GET", url=pr_url, + headers=[{"name": "Accept", "value": "application/vnd.github.v3.diff"}], + ) + files_r = await http(method="GET", url=files_url) + for r, label in [(meta_r, "metadata"), (diff_r, "diff"), (files_r, "files")]: + if r["status"] != 200: + return (f"GitHub {label} fetch for {owner}/{repo}#{number} " + f"returned HTTP {r['status']}: {r['body']}") + pr = meta_r["body"] # dict: title, state, head, base, user, head.sha, ... + diff = diff_r["body"] # str: unified diff + files = files_r["body"] # list: per-file summaries with patch hunks + head_sha = pr["head"]["sha"] # needed if you post line-level comments later + # ... build the review ... + return body + +FINAL(await review()) +``` + +Do NOT wrap `body` with `.get("body", body)` or `isinstance(..., str)` normalization. On a 2xx, `meta_r["body"]` is the parsed JSON; on the diff request it is a string. If status is not 2xx, return fast — silently falling through produces empty reviews where every field is "unknown". + +### Local path + +Run `shell` with `git diff` (unstaged), `git diff --cached` (staged), or `git diff HEAD~1` (last commit). For local reviews, skip the GitHub posting steps entirely and present findings in the chat. + +## Step 2 — Read every changed file in full + +For each file in the diff, read the **entire current file**, not just the hunks. You need surrounding context to catch: + +- Callers of modified functions that now behave differently +- Trait/interface contracts the change may violate +- Invariants established elsewhere that the diff breaks + +Fetch file contents via GitHub's raw media type so you get the text directly — the default `/contents/` response is base64-encoded and Monty's CodeAct sandbox does **not** ship the `base64` module (so `import base64` raises `ModuleNotFoundError`). Use the `application/vnd.github.raw` Accept header and read `body` as a plain string: + +```repl +r = await http( + method="GET", + url=f"https://api.github.com/repos/{owner}/{repo}/contents/{urllib.parse.quote(path, safe='')}?ref={head_sha}", + headers=[{"name": "Accept", "value": "application/vnd.github.raw"}], +) +if r["status"] != 200: + # Missing-on-head usually means the PR deleted the file; fall back to + # `base=pr['base']['sha']` or skip. + continue +file_text = r["body"] # plain str, no base64 +``` + +If the PR touches more than 20 files, prioritize: service logic > routes/handlers > models/types > tests > docs. + +## Step 3 — Deep review (six lenses) + +Walk the changes through each lens. For every finding, capture: file, line range, severity, category, concrete description, and a suggested fix. + +### 3a. Correctness and bugs + +- Off-by-one errors, wrong comparison operators, inverted conditions +- Unreachable code, dead branches, impossible match arms +- Type confusion (mixing up IDs, wrong enum variant, string vs newtype) +- Incorrect error propagation (swallowed errors, wrong error type or status) +- Broken invariants (uniqueness, ordering, state-machine transitions) +- Concurrency issues (TOCTOU, missing locks, races between check and use) + +### 3b. Edge cases and failure handling + +- Empty input, `None`/`null`, zero-length collections +- External-service failure (DB down, HTTP timeout, malformed response) +- Integer boundaries (overflow, underflow, `i64::MAX`, negative when expecting positive) +- Adversarial input (invalid UTF-8, huge payloads, deeply nested JSON) +- Are all error paths tested? Does every `?` propagation make sense? +- Partial-failure handling (wrote to DB but failed to emit event, or vice versa) + +### 3c. Security (assume a malicious actor) + +- **AuthN/AuthZ bypass**: Can an unauthenticated user reach this? Can workspace A access workspace B's data? IDOR? +- **Injection**: SQL via string interpolation, command, log, header, prompt injection +- **Data leakage**: Secrets, PII, or conversation content in logs, error messages, or API responses +- **Resource exhaustion / DoS**: Unbounded input, expensive operations without rate limits, OOM via large allocations +- **Financial abuse**: Tokens or credits consumed without tracking, usage limits bypassed, billing manipulated +- **Replay / races**: Same request replayed for double-spend, concurrent requests bypassing limits +- **Cryptographic issues**: Timing attacks on comparisons, weak randomness, missing HMAC verification + +### 3d. Test coverage + +- Every new public function/method tested? +- Error paths tested, not just happy paths? +- Edge cases covered (empty, boundary, concurrent)? +- Do existing tests still make sense, or do they assert stale behavior? +- Are there integration/e2e tests for the full flow? +- If a test is missing, name the exact test that should exist. + +### 3e. Documentation and assumptions + +- New assumptions documented in comments? ("this field is always non-empty because X") +- Non-obvious algorithms or business rules explained? +- Module-level docs updated to reflect new capabilities? +- API contracts (request/response shapes, error codes) documented? +- New patterns explained for future contributors? +- TODO/FIXME/HACK that should be tracked as issues? + +### 3f. Architectural concerns + +- Follows existing patterns, or introduces a new one without justification? +- Unnecessary abstractions or premature generalizations? +- Duplicated logic that should be extracted? +- Module dependencies clean, or circular/tight coupling introduced? +- Will this make future work harder? + +## Step 4 — Present findings + +The review **must**: + +- Start with `Review of {owner}/{repo}#{number}: {pr["title"]}` (or `Review of local changes` for the local path) +- Cite at least one specific `path:line` from the diff, never a generic "looks good" +- Use this severity scale: + +| Severity | Meaning | +|----------|---------| +| **Critical** | Security vulnerability, data loss, or financial exploit | +| **High** | Bug that will cause incorrect behavior in production | +| **Medium** | Robustness issue, missing validation, incomplete error handling | +| **Low** | Style, naming, documentation, minor improvement | +| **Nit** | Optional, take-it-or-leave-it | + +Render findings as a table: + +| # | Severity | Category | File:Line | Finding | Suggested fix | +|---|----------|----------|-----------|---------|---------------| + +Then ask the user which findings to post as PR comments. Default: all Critical, High, and Medium. Skip this prompt for the local path. + +## Step 5 — Post comments on GitHub (PR path only) + +Use the same `async def` + `FINAL(await ...)` pattern. Line-level review comments require `commit_id` (the head SHA you captured in step 1) and the line number on the **post-image** side of the diff (`side: "RIGHT"`). For findings spanning multiple files or architectural critiques, post a single PR-level issue comment instead. + +```repl +async def post(): + # Line-level comment on a specific file:line + r = await http( + method="POST", + url=f"https://api.github.com/repos/{owner}/{repo}/pulls/{number}/comments", + body={ + "body": "**High** — `state.store` accessed directly, bypassing dispatch. See `.claude/rules/tools.md`.", + "commit_id": head_sha, + "path": "src/channels/web/handlers/foo.rs", + "start_line": 140, + "start_side": "RIGHT", + "line": 142, + "side": "RIGHT", + }, + ) + if r["status"] not in (200, 201): + return f"Posting line comment failed: HTTP {r['status']}: {r['body']}" + + # Architectural / multi-file finding as a PR-level comment + r2 = await http( + method="POST", + url=f"https://api.github.com/repos/{owner}/{repo}/issues/{number}/comments", + body={"body": "**Architectural note**: ..."}, + ) + if r2["status"] not in (200, 201): + return f"Posting PR comment failed: HTTP {r2['status']}: {r2['body']}" + + return f"Posted {len(line_findings)} line comments and {len(pr_findings)} PR comments." + +FINAL(await post()) +``` + +Format every comment as: bold severity tag, one-line summary, detailed explanation, concrete fix (with code if useful). + +## Rules + +- **Read every changed file in full before writing a single finding.** Context matters more than throughput. +- **Never comment on code you have not actually read.** Verify line numbers against the file you fetched, not against the diff offset. +- **Be specific.** "This might have issues" is useless. "Line 42 returns 404 but should return 400 because X" is useful. +- **Distinguish "this IS a bug" from "this COULD be a bug if X."** Be honest about certainty. +- **Don't nitpick formatting or style** unless it causes actual confusion. Focus on substance. +- **If the code is good, say so.** Don't invent problems to look thorough. An honest "no issues, here is what I checked" beats a padded list. +- **Round severity up when in doubt.** Cheaper to dismiss a false alarm than to miss a real bug. +- **Respect privacy:** never include customer data, secrets, or PII in posted comments. +- **Be proportional.** A one-line typo fix does not need a full security audit. Match depth to change scope. diff --git a/skills/github/SKILL.md b/skills/github/SKILL.md index a3d93460737..9795e6c8ee9 100644 --- a/skills/github/SKILL.md +++ b/skills/github/SKILL.md @@ -103,12 +103,80 @@ http(method="GET", url="https://api.github.com/repos/{owner}/{repo}/branches") http(method="GET", url="https://api.github.com/repos/{owner}/{repo}/commits?per_page=10") ``` +### Authenticated User & Cross-Repo Queries + +When the user says "my PRs", "my issues", or "my repos", they mean the user who owns `github_token`. Don't try to list a single repo, hit the search/user endpoints instead. + +**Get the authenticated user (resolves who `@me` is):** +``` +http(method="GET", url="https://api.github.com/user") +``` + +**My latest PRs across all repos:** +``` +http(method="GET", url="https://api.github.com/search/issues?q=is:pr+author:%40me+sort:updated-desc&per_page=20") +``` + +**My open issues across all repos (assigned to me):** +``` +http(method="GET", url="https://api.github.com/search/issues?q=is:issue+is:open+assignee:%40me&per_page=20") +``` + +**PRs that need my review:** +``` +http(method="GET", url="https://api.github.com/search/issues?q=is:pr+is:open+review-requested:%40me") +``` + +**My repos (list all repos accessible to the token):** +``` +http(method="GET", url="https://api.github.com/user/repos?sort=updated&per_page=30") +``` + +### Search + +GitHub has three search endpoints. Build queries with the [search syntax](https://docs.github.com/en/search-github/searching-on-github). + +**Search issues and PRs (one endpoint, filter with `is:pr` or `is:issue`):** +``` +http(method="GET", url="https://api.github.com/search/issues?q=repo:{owner}/{repo}+is:pr+is:open+label:bug") +``` + +**Search code:** +``` +http(method="GET", url="https://api.github.com/search/code?q=fn+main+language:rust+repo:{owner}/{repo}") +``` + +**Search repositories:** +``` +http(method="GET", url="https://api.github.com/search/repositories?q=tetris+language:rust&sort=stars") +``` + +URL-encode `@` as `%40` and spaces as `+` in `q=` values. + ## Response Handling -- GitHub returns JSON. Parse the response to extract relevant fields. +The `http` tool returns an envelope: + +```python +{"status": 200, "headers": {...}, "body": } +``` + +- **JSON endpoints** — `body` is already a parsed Python dict or list. Do **not** call `json.loads()` on it. Example: + ```python + r = await http(method="GET", url="https://api.github.com/repos/{owner}/{repo}/pulls/123") + if r["status"] != 200: + FINAL(f"GitHub returned HTTP {r['status']}: {r['body']}") + pr = r["body"] # dict, not a string + title = pr["title"] # use direct indexing; these keys always exist on a 2xx + state = pr["state"] + head = pr["head"]["ref"] + base = pr["base"]["ref"] + ``` +- **Diff / plain text endpoints** (`Accept: application/vnd.github.v3.diff` etc.) — `body` is a `str` containing the raw unified diff; use it as-is. +- **Never** write `body = pr_meta.get("body", pr_meta)` as a "safety net" — it hides real errors. If `status` isn't 2xx, fail fast. - For list endpoints, check the `Link` header for pagination. -- Rate limit: 5000 req/hour authenticated. Check `X-RateLimit-Remaining` header if doing bulk operations. -- Errors return `{"message": "..."}` — always check for error responses. +- Rate limit: 5000 req/hour authenticated. Check `X-RateLimit-Remaining` if doing bulk ops. +- Error responses are JSON of the form `{"message": "..."}` with a non-2xx `status` — surface them literally in your FINAL answer. ## Common Mistakes @@ -117,3 +185,4 @@ http(method="GET", url="https://api.github.com/repos/{owner}/{repo}/commits?per_ - For creating PRs, always set `draft: true` unless the user explicitly says "ready for review". - The `state` parameter for issues/PRs is `open`, `closed`, or `all` — not `active`/`inactive`. - Use `per_page` to control result count (max 100). Default is 30. +- For "my PRs / my issues" across all repos, hit `/search/issues?q=...+author:%40me`. Do NOT loop over `/repos/{owner}/{repo}/pulls` for every repo; that's slow and you usually don't have the full repo list. diff --git a/src/bridge/auth_manager.rs b/src/bridge/auth_manager.rs index d12ba5622bf..0909a8c6d23 100644 --- a/src/bridge/auth_manager.rs +++ b/src/bridge/auth_manager.rs @@ -333,10 +333,15 @@ impl AuthManager { if matches!( action_name, "tool_install" | "tool-install" | "tool_activate" | "tool_auth" - ) && let Some(name) = parameters.get("name").and_then(|v| v.as_str()) - && !name.trim().is_empty() - { - return name.to_string(); + ) { + let trimmed = parameters + .get("name") + .and_then(|v| v.as_str()) + .map(str::trim) + .unwrap_or(""); + if !trimmed.is_empty() { + return trimmed.to_string(); + } } if let Some(tools) = self.tools.as_ref() diff --git a/src/channels/wasm/setup.rs b/src/channels/wasm/setup.rs index 9672415fa54..aba19cdd6f1 100644 --- a/src/channels/wasm/setup.rs +++ b/src/channels/wasm/setup.rs @@ -141,6 +141,7 @@ pub async fn setup_wasm_channels( }) } +#[allow(clippy::too_many_arguments)] async fn register_startup_channels( loaded_channels: Vec, config: &Config, diff --git a/src/channels/web/server.rs b/src/channels/web/server.rs index 2dccf141a6b..35e2c0d0ca7 100644 --- a/src/channels/web/server.rs +++ b/src/channels/web/server.rs @@ -2896,12 +2896,9 @@ async fn pending_gate_extension_name( ); } - if let Some(tools) = state.tool_registry.as_ref() - && let Some(name) = tools.provider_extension_for_tool(tool_name).await - { - return Some(name); - } - + // auth_manager is None only when no secrets backend exists (e.g. bare + // test harness). Fall back to the raw credential name rather than + // duplicating AuthManager resolution logic here. Some(credential_name.clone()) } @@ -4613,11 +4610,32 @@ mod tests { test_gateway_state_with_dependencies(ext_mgr, None, None, None) } + /// Build a minimal `AuthManager` backed by an in-memory secrets store. + fn test_auth_manager( + tool_registry: Option>, + ) -> Arc { + let secrets: Arc = + Arc::new(crate::secrets::InMemorySecretsStore::new(Arc::new( + crate::secrets::SecretsCrypto::new(secrecy::SecretString::from( + TEST_GATEWAY_CRYPTO_KEY.to_string(), + )) + .expect("crypto"), + ))); + Arc::new(crate::bridge::auth_manager::AuthManager::new( + secrets, + None, + None, + tool_registry, + )) + } + #[tokio::test] async fn pending_gate_extension_name_uses_install_parameters_for_post_install_auth() { + let registry = Arc::new(ToolRegistry::new()); let mut state = test_gateway_state(None); let state_mut = Arc::get_mut(&mut state).expect("test state must be uniquely owned"); - state_mut.tool_registry = Some(Arc::new(ToolRegistry::new())); + state_mut.tool_registry = Some(Arc::clone(®istry)); + state_mut.auth_manager = Some(test_auth_manager(Some(Arc::clone(®istry)))); let extension_name = pending_gate_extension_name( state_mut, @@ -4672,6 +4690,7 @@ mod tests { let mut state = test_gateway_state(None); let state_mut = Arc::get_mut(&mut state).expect("test state must be uniquely owned"); state_mut.tool_registry = Some(Arc::clone(®istry)); + state_mut.auth_manager = Some(test_auth_manager(Some(Arc::clone(®istry)))); let extension_name = pending_gate_extension_name( state_mut, diff --git a/src/cli/snapshots/ironclaw__cli__tests__help_output.snap b/src/cli/snapshots/ironclaw__cli__tests__help_output.snap index 4d0ff28b7bf..393c30e1811 100644 --- a/src/cli/snapshots/ironclaw__cli__tests__help_output.snap +++ b/src/cli/snapshots/ironclaw__cli__tests__help_output.snap @@ -8,8 +8,8 @@ Usage: ironclaw [OPTIONS] [COMMAND] Commands: run Run the AI agent - onboard Run interactive setup wizard - config Manage app configs + onboard Run interactive setup wizard (start here if new to IronClaw) + config Manage app configuration settings tool Manage WASM tools registry Browse/install extensions channels Manage channels @@ -17,16 +17,17 @@ Commands: mcp Manage MCP servers memory Manage workspace memory pairing Manage DM pairing + profile Manage deployment profiles service Manage OS service skills Manage skills hooks Manage lifecycle hooks models Manage LLM providers and models - doctor Run diagnostics + doctor Run diagnostics (check if everything is configured correctly) logs View and manage gateway logs status Show system status completion Generate completions import Import from other AI systems - login Authenticate with a provider + login Authenticate with a provider (re-login) acp Manage ACP agents help Print this message or the help of the given subcommand(s) diff --git a/src/cli/snapshots/ironclaw__cli__tests__long_help_output.snap b/src/cli/snapshots/ironclaw__cli__tests__long_help_output.snap index 24c1d8ef9cc..b8c0f44edaf 100644 --- a/src/cli/snapshots/ironclaw__cli__tests__long_help_output.snap +++ b/src/cli/snapshots/ironclaw__cli__tests__long_help_output.snap @@ -2,17 +2,28 @@ source: src/cli/mod.rs expression: help --- -IronClaw is a secure AI assistant. Use 'ironclaw --help' for details. -Examples: - ironclaw run # Start the agent - ironclaw config list # List configs +IronClaw is a secure AI assistant. + +Getting started: + ironclaw onboard # Interactive setup wizard (recommended for first run) + ironclaw onboard --quick # Quick setup: just pick a provider and model + ironclaw models set-provider openai # Switch to a specific provider + ironclaw doctor # Check your configuration + +Common commands: + ironclaw run # Start the agent + ironclaw config list # View all settings + ironclaw models status # Show current provider and model + ironclaw models list # List available providers + +Use 'ironclaw --help' for details on any command. Usage: ironclaw [OPTIONS] [COMMAND] Commands: run Run the AI agent - onboard Run interactive setup wizard - config Manage app configs + onboard Run interactive setup wizard (start here if new to IronClaw) + config Manage app configuration settings tool Manage WASM tools registry Browse/install extensions channels Manage channels @@ -20,16 +31,17 @@ Commands: mcp Manage MCP servers memory Manage workspace memory pairing Manage DM pairing + profile Manage deployment profiles service Manage OS service skills Manage skills hooks Manage lifecycle hooks models Manage LLM providers and models - doctor Run diagnostics + doctor Run diagnostics (check if everything is configured correctly) logs View and manage gateway logs status Show system status completion Generate completions import Import from other AI systems - login Authenticate with a provider + login Authenticate with a provider (re-login) acp Manage ACP agents help Print this message or the help of the given subcommand(s) diff --git a/tests/e2e/scenarios/test_extensions.py b/tests/e2e/scenarios/test_extensions.py index 365ef204574..b4091407132 100644 --- a/tests/e2e/scenarios/test_extensions.py +++ b/tests/e2e/scenarios/test_extensions.py @@ -510,7 +510,7 @@ async def test_member_pairing_claim_submission_shows_success(browser, ironclaw_s await wait_for_toast(page, "Pairing approved") assert pairing_hits["approve"] == 1 - assert pairing_hits["approve_body"] == {"code": "PAIR-1234"} + assert pairing_hits["approve_body"]["code"] == "PAIR-1234" assert await input_field.input_value() == "" finally: await context.close() @@ -559,7 +559,7 @@ async def test_admin_pairing_manual_code_submit(browser, ironclaw_server): await wait_for_toast(page, "Pairing approved") assert pairing_hits["approve"] == 1 - assert pairing_hits["approve_body"] == {"code": "PAIR-1234"} + assert pairing_hits["approve_body"]["code"] == "PAIR-1234" assert await input_field.input_value() == "" finally: await context.close() diff --git a/tests/e2e_live_code_review.rs b/tests/e2e_live_code_review.rs new file mode 100644 index 00000000000..58d7c9977f9 --- /dev/null +++ b/tests/e2e_live_code_review.rs @@ -0,0 +1,349 @@ +//! Live/replay test for `/code-review owner/repo N`. +//! +//! Drives the `code-review` skill against a real (or replayed) pull +//! request on `nearai/ironclaw` and verifies: +//! +//! 1. The `code-review` skill actually activated from the `/code-review` +//! slash mention. +//! 2. The agent fetched the *correct* PR via the GitHub API (the URL +//! contains `/repos/nearai/ironclaw/pulls/2483`, not some other PR). +//! 3. The response text references the PR the user asked about, so a +//! silent-substitution regression (agent reviews the wrong PR but +//! confidently answers) is caught. +//! +//! # Running +//! +//! **Replay mode** (default, deterministic, needs committed trace fixture): +//! ```bash +//! cargo test --features libsql --test e2e_live_code_review -- --ignored +//! ``` +//! +//! **Live mode** (real LLM + real GitHub API, records/updates fixture): +//! ```bash +//! IRONCLAW_LIVE_TEST=1 cargo test --features libsql \ +//! --test e2e_live_code_review -- --ignored --test-threads=1 --nocapture +//! ``` +//! +//! Live mode requires a `github_token` secret in the developer's +//! `~/.ironclaw/ironclaw.db` (read scope is enough; the PR is public). +//! Replay mode does not need any credentials — the trace fixture carries +//! the LLM side and the harness stubs HTTP interactions recorded in the +//! fixture. + +#[cfg(feature = "libsql")] +mod support; + +#[cfg(feature = "libsql")] +mod code_review_test { + use std::path::PathBuf; + use std::time::Duration; + + use crate::support::live_harness::{LiveTestHarness, LiveTestHarnessBuilder}; + use ironclaw::channels::StatusUpdate; + + const TEST_NAME: &str = "code_review_pr_2483"; + const REPO_OWNER: &str = "nearai"; + const REPO_NAME: &str = "ironclaw"; + const PR_NUMBER: u64 = 2483; + + fn repo_skills_dir() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("skills") + } + + fn trace_fixture_path(test_name: &str) -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("tests") + .join("fixtures") + .join("llm_traces") + .join("live") + .join(format!("{test_name}.json")) + } + + /// Extract the PR title from the trace fixture's HTTP exchanges. + /// + /// Finds the first exchange whose URL contains `pulls/{PR_NUMBER}` and + /// whose response body parses as JSON with a `"title"` field (i.e. the + /// PR metadata request, not the diff). This keeps the expected title in + /// sync with the recorded fixture so drift is caught by the replay + /// machinery rather than a stale hard-coded constant. + fn pr_title_from_fixture(test_name: &str) -> Option { + let path = trace_fixture_path(test_name); + let data = std::fs::read_to_string(&path).ok()?; + let trace: serde_json::Value = serde_json::from_str(&data).ok()?; + let expected_url_fragment = format!("pulls/{PR_NUMBER}"); + for exchange in trace.get("http_exchanges")?.as_array()? { + let url = exchange + .pointer("/request/url") + .and_then(|v| v.as_str()) + .unwrap_or(""); + if !url.contains(&expected_url_fragment) { + continue; + } + let body_str = exchange + .pointer("/response/body") + .and_then(|v| v.as_str()) + .unwrap_or(""); + if let Ok(body) = serde_json::from_str::(body_str) + && let Some(title) = body.get("title").and_then(|v| v.as_str()) + { + return Some(title.to_string()); + } + } + None + } + + /// Mirror of the pattern in `e2e_github_dev_workflow`: skip in replay + /// mode unless the fixture is already committed. In live mode we + /// always run (and the fixture gets recorded). + fn should_run_test(test_name: &str) -> bool { + if trace_fixture_path(test_name).exists() + || std::env::var("IRONCLAW_LIVE_TEST") + .ok() + .filter(|v| !v.is_empty() && v != "0") + .is_some() + { + true + } else { + eprintln!( + "[{}] replay fixture missing at {}; skipping until recorded in live mode", + test_name, + trace_fixture_path(test_name).display() + ); + false + } + } + + async fn build_harness(test_name: &str) -> LiveTestHarness { + LiveTestHarnessBuilder::new(test_name) + .with_engine_v2(true) + .with_auto_approve_tools(true) + // Fetching the PR, parsing metadata, then fetching the diff + // is at most a handful of tool calls, but the LLM may branch + // on large diffs — give it enough headroom to finish. + .with_max_tool_iterations(30) + .with_skills_dir(repo_skills_dir()) + // The agent hits api.github.com. In live mode we need the + // real token so the request is authenticated (avoids the + // 60/hr unauthenticated rate limit). In replay mode the + // token is unused — the trace fixture carries the response. + .with_secrets(["github_token"]) + .build() + .await + } + + /// Dump activity for a failed run so CI logs show what the agent + /// actually did. + fn dump_activity(harness: &LiveTestHarness, label: &str) { + eprintln!("───── [{label}] activity dump ─────"); + eprintln!("active skills: {:?}", harness.rig().active_skill_names()); + for event in harness.rig().captured_status_events() { + match event { + StatusUpdate::SkillActivated { skill_names, .. } => { + eprintln!(" ◆ skills activated: {}", skill_names.join(", ")); + } + StatusUpdate::ToolStarted { name, detail, .. } => { + eprintln!(" ● {name} {}", detail.unwrap_or_default()); + } + StatusUpdate::ToolCompleted { + name, + success, + error, + .. + } => { + if success { + eprintln!(" ✓ {name}"); + } else { + eprintln!(" ✗ {name}: {}", error.unwrap_or_default()); + } + } + StatusUpdate::ToolResult { name, preview, .. } => { + let short: String = preview.chars().take(200).collect(); + eprintln!(" {name} → {short}"); + } + _ => {} + } + } + eprintln!("───── end activity ─────"); + } + + /// End-to-end: `/code-review nearai/ironclaw 2483` must (a) activate + /// the `code-review` skill, (b) hit + /// `api.github.com/repos/nearai/ironclaw/pulls/2483`, and (c) + /// produce a response naming the PR it reviewed. + #[tokio::test] + #[ignore] // Live tier: requires LLM API keys or a recorded trace fixture + async fn code_review_real_pr() { + if !should_run_test(TEST_NAME) { + return; + } + + let harness = build_harness(TEST_NAME).await; + let rig = harness.rig(); + + let user_input = format!("/code-review {REPO_OWNER}/{REPO_NAME} {PR_NUMBER}"); + rig.send_message(&user_input).await; + + // Reviewing a real PR with diff-fetching + reasoning can take a + // while in live mode — wait up to 5 minutes. + let responses = rig.wait_for_responses(1, Duration::from_secs(300)).await; + let response_text: Vec = responses.iter().map(|r| r.content.clone()).collect(); + let joined = response_text.join("\n"); + + // ── Activity-level diagnostics before assertions ────────────── + dump_activity(&harness, "code_review_real_pr"); + + // Assertion 1: the code-review skill activated. + // + // Without this the test could pass on a generic "here's what I'd + // do" reply that never touched the skill body. `/code-review` is + // an explicit slash mention — selector::extract_skill_mentions + // should force-select it regardless of score. + let active = rig.active_skill_names(); + assert!( + active.iter().any(|s| s == "code-review"), + "Expected `code-review` skill to activate from the `/code-review` \ + mention. Active skills: {active:?}" + ); + + // Assertion 2: the agent actually called the `http` tool. + // + // The skill body tells the agent to reach api.github.com. If it + // falls back to shell/git or hallucinates a review from training + // data, this catches it. `tool_calls_started` decorates the name + // with a short summary (e.g. `"http(https://.../pulls/2483)"`), + // so we match on a prefix rather than bare equality. + let tools = rig.tool_calls_started(); + assert!( + tools.iter().any(|t| t == "http" || t.starts_with("http(")), + "Expected the `http` tool to be invoked for the GitHub PR fetch. \ + Tools used: {tools:?}" + ); + + // Assertion 3: the request targeted *this* PR, not a different one. + // + // We inspect both ToolStarted.detail (which the http tool populates + // with the URL summary) and ToolResult.preview (which echoes the + // PR metadata). The ToolStarted.name also embeds the URL as + // `http()` in this harness, so we check that too. Any of + // the three surfaces containing `pulls/2483` proves the request + // went to the correct endpoint. Checking only the response text + // is not enough — the LLM could repeat the number from the prompt + // without ever fetching the right PR. + let expected_path = format!("pulls/{PR_NUMBER}"); + let pr_endpoint_hit = rig + .captured_status_events() + .iter() + .any(|event| match event { + StatusUpdate::ToolStarted { name, detail, .. } => { + let name_hit = name.contains(&expected_path); + let detail_hit = detail + .as_deref() + .map(|d| d.contains(&expected_path)) + .unwrap_or(false); + (name.starts_with("http") || name == "http") && (name_hit || detail_hit) + } + StatusUpdate::ToolResult { + name: _, preview, .. + } => preview.contains(&expected_path), + _ => false, + }); + assert!( + pr_endpoint_hit, + "Expected at least one http call or result referencing `{expected_path}`. \ + The agent invoked http() but did not appear to target PR #{PR_NUMBER}. \ + Full response preview: {}", + joined.chars().take(400).collect::() + ); + + // Assertion 4: the response names the PR it reviewed. + // + // Catches the "silent substitution" regression — agent fetches + // the right PR but writes about a different one, or answers + // generically without naming the PR at all. + let lower = joined.to_lowercase(); + let names_pr = + lower.contains(&format!("#{PR_NUMBER}")) || lower.contains(&PR_NUMBER.to_string()); + assert!( + names_pr, + "Response should reference PR #{PR_NUMBER}; got: {}", + joined.chars().take(400).collect::() + ); + let names_repo = lower.contains("nearai/ironclaw") || lower.contains("ironclaw"); + assert!( + names_repo, + "Response should name the repo that was reviewed; got: {}", + joined.chars().take(400).collect::() + ); + + // Assertion 5: the review must reference the *real* PR content, + // not a blank shell. + // + // The earlier fixture captured a green-ticket "looks good" + // reply where every PR field was "unknown" because the LLM's + // generated code mishandled the `http` envelope shape. Guard + // against that class of silent-empty review by requiring: + // (a) the exact PR title from GitHub (so the agent actually + // extracted `body["title"]` instead of falling back to + // `"(unknown title)"`), and + // (b) at least one concrete `path:line` or fenced-code + // reference — a review with zero specifics is not a + // review. + // + // Extract the expected PR title from the trace fixture so that + // re-recording the fixture automatically updates the expectation. + // Falls back to a hard-coded value if the fixture is missing or + // doesn't contain the PR metadata exchange (e.g. live mode before + // the fixture is committed). + let pr_title = pr_title_from_fixture(TEST_NAME).unwrap_or_else(|| { + "feat(engine): add code execution failure categorization instrumentation".to_string() + }); + assert!( + joined.contains(&pr_title), + "Response should include the real PR title \"{pr_title}\" — \ + absence usually means the agent never parsed the JSON body. \ + Got: {}", + joined.chars().take(600).collect::() + ); + + // At least one concrete file reference. The pattern is loose + // on purpose: any of these signals a grounded review: + // - `path/to/file.rs` inside backticks + // - `path/to/file.rs:42` line reference + // - a fenced code block with diff content + let has_concrete_reference = joined.contains("```") + || joined.contains(".rs:") + || joined.contains(".py:") + || joined.contains(".ts:") + || joined.contains(".md:") + || joined.contains("crates/") + || joined.contains("src/"); + assert!( + has_concrete_reference, + "Response should cite at least one concrete file or code reference. \ + A review without specifics is not a review. Got: {}", + joined.chars().take(600).collect::() + ); + + // Assertion 6: the sidebar / header must not say every field + // is "unknown". This is the exact failure mode the earlier + // fixture recorded, and it is the strongest indicator that + // the `http` envelope handling is broken again. + let unknown_markers = [ + "(unknown title)", + "state: `unknown`", + "base ← head: `unknown`", + "files changed (reported): `0`", + ]; + for marker in unknown_markers { + assert!( + !lower.contains(&marker.to_lowercase()), + "Response contains the blank-shell marker {marker:?} — the \ + agent's `http` response handling produced an empty PR \ + snapshot. Got: {}", + joined.chars().take(600).collect::() + ); + } + + harness.finish(&user_input, &response_text).await; + } +} diff --git a/tests/fixtures/llm_traces/live/code_review_pr_2483.json b/tests/fixtures/llm_traces/live/code_review_pr_2483.json new file mode 100644 index 00000000000..d374b137435 --- /dev/null +++ b/tests/fixtures/llm_traces/live/code_review_pr_2483.json @@ -0,0 +1,2979 @@ +{ + "model_name": "live-code_review_pr_2483", + "http_exchanges": [ + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/pulls/2483" + }, + "response": { + "status": 200, + "headers": [ + [ + "x-oauth-scopes", + "repo" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:23:08 GMT" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "x-xss-protection", + "0" + ], + [ + "x-ratelimit-remaining", + "4982" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP,Accept-Encoding, Accept, X-Requested-With" + ], + [ + "x-github-media-type", + "github.v3; format=json" + ], + [ + "x-ratelimit-used", + "18" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "content-type", + "application/json; charset=utf-8" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:06 GMT" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "etag", + "W/\"9e16e6116f0660a9e6887ab1801d96ff4f14bb9a0f40491f309b0afb108e8b95\"" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-frame-options", + "deny" + ], + [ + "server", + "github.com" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-github-request-id", + "F500:365517:3FF2E4:4B4B88:69DFAEC1" + ] + ], + "body": "{\"url\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483\",\"id\":3532069864,\"node_id\":\"PR_kwDORHZ7Z87Shxvo\",\"html_url\":\"https://github.com/nearai/ironclaw/pull/2483\",\"diff_url\":\"https://github.com/nearai/ironclaw/pull/2483.diff\",\"patch_url\":\"https://github.com/nearai/ironclaw/pull/2483.patch\",\"issue_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/2483\",\"number\":2483,\"state\":\"open\",\"locked\":false,\"title\":\"feat(engine): add code execution failure categorization instrumentation\",\"user\":{\"login\":\"serrrfirat\",\"id\":5748809,\"node_id\":\"MDQ6VXNlcjU3NDg4MDk=\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/5748809?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/serrrfirat\",\"html_url\":\"https://github.com/serrrfirat\",\"followers_url\":\"https://api.github.com/users/serrrfirat/followers\",\"following_url\":\"https://api.github.com/users/serrrfirat/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/serrrfirat/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/serrrfirat/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/serrrfirat/subscriptions\",\"organizations_url\":\"https://api.github.com/users/serrrfirat/orgs\",\"repos_url\":\"https://api.github.com/users/serrrfirat/repos\",\"events_url\":\"https://api.github.com/users/serrrfirat/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/serrrfirat/received_events\",\"type\":\"User\",\"user_view_type\":\"public\",\"site_admin\":false},\"body\":\"## Summary\\n- Adds `CodeExecutionFailure` enum (7 variants: SyntaxError, RuntimeError, NameLookup, VmPanic, ResourceLimit, ToolError, OsDenied) to classify why code execution fails\\n- Threads `failure` through `CodeExecutionResult` (replacing `had_error: bool`) and emits structured `CodeExecutionFailed` events for aggregate analysis of failure modes (Monty limitation vs LLM logic error vs tool dispatch failure)\\n- Upgrades trace analysis to use structured events with severity escalation (VmPanic/ResourceLimit → Error), with backward-compatible fallback for pre-instrumentation threads\\n\\n### Caller audit (Err → Ok(failure=VmPanic) shift)\\n`execute_code` / `execute_code_with_skills` previously returned `Err(EngineError::Effect)` on VM panic; now returns `Ok(CodeExecutionResult { failure: Some(VmPanic) })`. The only production call site is `orchestrator.rs:handle_execute_code_step` which correctly dispatches on `result.failure`. Test-only callers in `scripting.rs` were also updated. No other callers exist.\\n\\n### Known limitation: mixed-era fallback\\nThe trace analyzer's backward-compatible fallback (message-scraping for pre-instrumentation threads) is all-or-nothing: if a thread has *any* `CodeExecutionFailed` event, message-scraping is skipped entirely. Errors from pre-instrumentation steps in a mixed-era thread will go unreported. This is acceptable during the transition period since all new threads will have full instrumentation.\\n\\n## Test plan\\n- [x] 11 new unit tests for error classifier, code hash, and trace detection\\n- [x] `cargo test -p ironclaw_engine --lib` — 395 pass\\n- [x] `cargo clippy --all --all-features` — zero engine warnings\\n\\n🤖 Generated with [Claude Code](https://claude.com/claude-code)\",\"created_at\":\"2026-04-15T05:50:13Z\",\"updated_at\":\"2026-04-15T14:23:08Z\",\"closed_at\":null,\"merged_at\":null,\"merge_commit_sha\":\"35b0564ebf9bd9aef17e24c821ea98158844030d\",\"assignees\":[],\"requested_reviewers\":[{\"login\":\"ilblackdragon\",\"id\":175486,\"node_id\":\"MDQ6VXNlcjE3NTQ4Ng==\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/175486?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/ilblackdragon\",\"html_url\":\"https://github.com/ilblackdragon\",\"followers_url\":\"https://api.github.com/users/ilblackdragon/followers\",\"following_url\":\"https://api.github.com/users/ilblackdragon/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/ilblackdragon/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/ilblackdragon/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/ilblackdragon/subscriptions\",\"organizations_url\":\"https://api.github.com/users/ilblackdragon/orgs\",\"repos_url\":\"https://api.github.com/users/ilblackdragon/repos\",\"events_url\":\"https://api.github.com/users/ilblackdragon/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/ilblackdragon/received_events\",\"type\":\"User\",\"user_view_type\":\"public\",\"site_admin\":false}],\"requested_teams\":[],\"labels\":[{\"id\":10248775942,\"node_id\":\"LA_kwDORHZ7Z88AAAACYt_pBg\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/size:%20XL\",\"name\":\"size: XL\",\"color\":\"B71C1C\",\"default\":false,\"description\":\"500+ changed lines\"},{\"id\":10248776001,\"node_id\":\"LA_kwDORHZ7Z88AAAACYt_pQQ\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/risk:%20low\",\"name\":\"risk: low\",\"color\":\"4CAF50\",\"default\":false,\"description\":\"Changes to docs, tests, or low-risk modules\"},{\"id\":10248776633,\"node_id\":\"LA_kwDORHZ7Z88AAAACYt_ruQ\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/scope:%20db/postgres\",\"name\":\"scope: db/postgres\",\"color\":\"6A1B9A\",\"default\":false,\"description\":\"PostgreSQL backend\"},{\"id\":10248777536,\"node_id\":\"LA_kwDORHZ7Z88AAAACYt_vQA\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/contributor:%20core\",\"name\":\"contributor: core\",\"color\":\"FF8A65\",\"default\":false,\"description\":\"20+ merged PRs\"},{\"id\":10665130192,\"node_id\":\"LA_kwDORHZ7Z88AAAACe7D40A\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/DB%20MIGRATION\",\"name\":\"DB MIGRATION\",\"color\":\"C62828\",\"default\":false,\"description\":\"PR adds or modifies PostgreSQL or libSQL migration definitions\"}],\"milestone\":null,\"draft\":false,\"commits_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483/commits\",\"review_comments_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483/comments\",\"review_comment_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls/comments{/number}\",\"comments_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/2483/comments\",\"statuses_url\":\"https://api.github.com/repos/nearai/ironclaw/statuses/5496341c49b2abaa9842e2f939ea349c75fd7340\",\"head\":{\"label\":\"nearai:claude/audit-v2-engine-usage-I7dIz\",\"ref\":\"claude/audit-v2-engine-usage-I7dIz\",\"sha\":\"5496341c49b2abaa9842e2f939ea349c75fd7340\",\"user\":{\"login\":\"nearai\",\"id\":29134221,\"node_id\":\"MDEyOk9yZ2FuaXphdGlvbjI5MTM0MjIx\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/29134221?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/nearai\",\"html_url\":\"https://github.com/nearai\",\"followers_url\":\"https://api.github.com/users/nearai/followers\",\"following_url\":\"https://api.github.com/users/nearai/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/nearai/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/nearai/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/nearai/subscriptions\",\"organizations_url\":\"https://api.github.com/users/nearai/orgs\",\"repos_url\":\"https://api.github.com/users/nearai/repos\",\"events_url\":\"https://api.github.com/users/nearai/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/nearai/received_events\",\"type\":\"Organization\",\"user_view_type\":\"public\",\"site_admin\":false},\"repo\":{\"id\":1148615527,\"node_id\":\"R_kgDORHZ7Zw\",\"name\":\"ironclaw\",\"full_name\":\"nearai/ironclaw\",\"private\":false,\"owner\":{\"login\":\"nearai\",\"id\":29134221,\"node_id\":\"MDEyOk9yZ2FuaXphdGlvbjI5MTM0MjIx\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/29134221?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/nearai\",\"html_url\":\"https://github.com/nearai\",\"followers_url\":\"https://api.github.com/users/nearai/followers\",\"following_url\":\"https://api.github.com/users/nearai/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/nearai/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/nearai/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/nearai/subscriptions\",\"organizations_url\":\"https://api.github.com/users/nearai/orgs\",\"repos_url\":\"https://api.github.com/users/nearai/repos\",\"events_url\":\"https://api.github.com/users/nearai/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/nearai/received_events\",\"type\":\"Organization\",\"user_view_type\":\"public\",\"site_admin\":false},\"html_url\":\"https://github.com/nearai/ironclaw\",\"description\":\"IronClaw is OpenClaw inspired implementation in Rust focused on privacy and security\",\"fork\":false,\"url\":\"https://api.github.com/repos/nearai/ironclaw\",\"forks_url\":\"https://api.github.com/repos/nearai/ironclaw/forks\",\"keys_url\":\"https://api.github.com/repos/nearai/ironclaw/keys{/key_id}\",\"collaborators_url\":\"https://api.github.com/repos/nearai/ironclaw/collaborators{/collaborator}\",\"teams_url\":\"https://api.github.com/repos/nearai/ironclaw/teams\",\"hooks_url\":\"https://api.github.com/repos/nearai/ironclaw/hooks\",\"issue_events_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/events{/number}\",\"events_url\":\"https://api.github.com/repos/nearai/ironclaw/events\",\"assignees_url\":\"https://api.github.com/repos/nearai/ironclaw/assignees{/user}\",\"branches_url\":\"https://api.github.com/repos/nearai/ironclaw/branches{/branch}\",\"tags_url\":\"https://api.github.com/repos/nearai/ironclaw/tags\",\"blobs_url\":\"https://api.github.com/repos/nearai/ironclaw/git/blobs{/sha}\",\"git_tags_url\":\"https://api.github.com/repos/nearai/ironclaw/git/tags{/sha}\",\"git_refs_url\":\"https://api.github.com/repos/nearai/ironclaw/git/refs{/sha}\",\"trees_url\":\"https://api.github.com/repos/nearai/ironclaw/git/trees{/sha}\",\"statuses_url\":\"https://api.github.com/repos/nearai/ironclaw/statuses/{sha}\",\"languages_url\":\"https://api.github.com/repos/nearai/ironclaw/languages\",\"stargazers_url\":\"https://api.github.com/repos/nearai/ironclaw/stargazers\",\"contributors_url\":\"https://api.github.com/repos/nearai/ironclaw/contributors\",\"subscribers_url\":\"https://api.github.com/repos/nearai/ironclaw/subscribers\",\"subscription_url\":\"https://api.github.com/repos/nearai/ironclaw/subscription\",\"commits_url\":\"https://api.github.com/repos/nearai/ironclaw/commits{/sha}\",\"git_commits_url\":\"https://api.github.com/repos/nearai/ironclaw/git/commits{/sha}\",\"comments_url\":\"https://api.github.com/repos/nearai/ironclaw/comments{/number}\",\"issue_comment_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/comments{/number}\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/{+path}\",\"compare_url\":\"https://api.github.com/repos/nearai/ironclaw/compare/{base}...{head}\",\"merges_url\":\"https://api.github.com/repos/nearai/ironclaw/merges\",\"archive_url\":\"https://api.github.com/repos/nearai/ironclaw/{archive_format}{/ref}\",\"downloads_url\":\"https://api.github.com/repos/nearai/ironclaw/downloads\",\"issues_url\":\"https://api.github.com/repos/nearai/ironclaw/issues{/number}\",\"pulls_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls{/number}\",\"milestones_url\":\"https://api.github.com/repos/nearai/ironclaw/milestones{/number}\",\"notifications_url\":\"https://api.github.com/repos/nearai/ironclaw/notifications{?since,all,participating}\",\"labels_url\":\"https://api.github.com/repos/nearai/ironclaw/labels{/name}\",\"releases_url\":\"https://api.github.com/repos/nearai/ironclaw/releases{/id}\",\"deployments_url\":\"https://api.github.com/repos/nearai/ironclaw/deployments\",\"created_at\":\"2026-02-03T06:57:10Z\",\"updated_at\":\"2026-04-15T15:21:34Z\",\"pushed_at\":\"2026-04-15T15:27:33Z\",\"git_url\":\"git://github.com/nearai/ironclaw.git\",\"ssh_url\":\"git@github.com:nearai/ironclaw.git\",\"clone_url\":\"https://github.com/nearai/ironclaw.git\",\"svn_url\":\"https://github.com/nearai/ironclaw\",\"homepage\":\"https://www.ironclaw.com\",\"size\":30129,\"stargazers_count\":11789,\"watchers_count\":11789,\"language\":\"Rust\",\"has_issues\":true,\"has_projects\":false,\"has_downloads\":true,\"has_wiki\":false,\"has_pages\":false,\"has_discussions\":false,\"forks_count\":1349,\"mirror_url\":null,\"archived\":false,\"disabled\":false,\"open_issues_count\":640,\"license\":{\"key\":\"apache-2.0\",\"name\":\"Apache License 2.0\",\"spdx_id\":\"Apache-2.0\",\"url\":\"https://api.github.com/licenses/apache-2.0\",\"node_id\":\"MDc6TGljZW5zZTI=\"},\"allow_forking\":true,\"is_template\":false,\"web_commit_signoff_required\":false,\"has_pull_requests\":true,\"pull_request_creation_policy\":\"all\",\"topics\":[],\"visibility\":\"public\",\"forks\":1349,\"open_issues\":640,\"watchers\":11789,\"default_branch\":\"staging\"}},\"base\":{\"label\":\"nearai:staging\",\"ref\":\"staging\",\"sha\":\"16a07316d03430066f420e0d19e98a7506fec032\",\"user\":{\"login\":\"nearai\",\"id\":29134221,\"node_id\":\"MDEyOk9yZ2FuaXphdGlvbjI5MTM0MjIx\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/29134221?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/nearai\",\"html_url\":\"https://github.com/nearai\",\"followers_url\":\"https://api.github.com/users/nearai/followers\",\"following_url\":\"https://api.github.com/users/nearai/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/nearai/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/nearai/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/nearai/subscriptions\",\"organizations_url\":\"https://api.github.com/users/nearai/orgs\",\"repos_url\":\"https://api.github.com/users/nearai/repos\",\"events_url\":\"https://api.github.com/users/nearai/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/nearai/received_events\",\"type\":\"Organization\",\"user_view_type\":\"public\",\"site_admin\":false},\"repo\":{\"id\":1148615527,\"node_id\":\"R_kgDORHZ7Zw\",\"name\":\"ironclaw\",\"full_name\":\"nearai/ironclaw\",\"private\":false,\"owner\":{\"login\":\"nearai\",\"id\":29134221,\"node_id\":\"MDEyOk9yZ2FuaXphdGlvbjI5MTM0MjIx\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/29134221?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/nearai\",\"html_url\":\"https://github.com/nearai\",\"followers_url\":\"https://api.github.com/users/nearai/followers\",\"following_url\":\"https://api.github.com/users/nearai/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/nearai/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/nearai/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/nearai/subscriptions\",\"organizations_url\":\"https://api.github.com/users/nearai/orgs\",\"repos_url\":\"https://api.github.com/users/nearai/repos\",\"events_url\":\"https://api.github.com/users/nearai/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/nearai/received_events\",\"type\":\"Organization\",\"user_view_type\":\"public\",\"site_admin\":false},\"html_url\":\"https://github.com/nearai/ironclaw\",\"description\":\"IronClaw is OpenClaw inspired implementation in Rust focused on privacy and security\",\"fork\":false,\"url\":\"https://api.github.com/repos/nearai/ironclaw\",\"forks_url\":\"https://api.github.com/repos/nearai/ironclaw/forks\",\"keys_url\":\"https://api.github.com/repos/nearai/ironclaw/keys{/key_id}\",\"collaborators_url\":\"https://api.github.com/repos/nearai/ironclaw/collaborators{/collaborator}\",\"teams_url\":\"https://api.github.com/repos/nearai/ironclaw/teams\",\"hooks_url\":\"https://api.github.com/repos/nearai/ironclaw/hooks\",\"issue_events_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/events{/number}\",\"events_url\":\"https://api.github.com/repos/nearai/ironclaw/events\",\"assignees_url\":\"https://api.github.com/repos/nearai/ironclaw/assignees{/user}\",\"branches_url\":\"https://api.github.com/repos/nearai/ironclaw/branches{/branch}\",\"tags_url\":\"https://api.github.com/repos/nearai/ironclaw/tags\",\"blobs_url\":\"https://api.github.com/repos/nearai/ironclaw/git/blobs{/sha}\",\"git_tags_url\":\"https://api.github.com/repos/nearai/ironclaw/git/tags{/sha}\",\"git_refs_url\":\"https://api.github.com/repos/nearai/ironclaw/git/refs{/sha}\",\"trees_url\":\"https://api.github.com/repos/nearai/ironclaw/git/trees{/sha}\",\"statuses_url\":\"https://api.github.com/repos/nearai/ironclaw/statuses/{sha}\",\"languages_url\":\"https://api.github.com/repos/nearai/ironclaw/languages\",\"stargazers_url\":\"https://api.github.com/repos/nearai/ironclaw/stargazers\",\"contributors_url\":\"https://api.github.com/repos/nearai/ironclaw/contributors\",\"subscribers_url\":\"https://api.github.com/repos/nearai/ironclaw/subscribers\",\"subscription_url\":\"https://api.github.com/repos/nearai/ironclaw/subscription\",\"commits_url\":\"https://api.github.com/repos/nearai/ironclaw/commits{/sha}\",\"git_commits_url\":\"https://api.github.com/repos/nearai/ironclaw/git/commits{/sha}\",\"comments_url\":\"https://api.github.com/repos/nearai/ironclaw/comments{/number}\",\"issue_comment_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/comments{/number}\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/{+path}\",\"compare_url\":\"https://api.github.com/repos/nearai/ironclaw/compare/{base}...{head}\",\"merges_url\":\"https://api.github.com/repos/nearai/ironclaw/merges\",\"archive_url\":\"https://api.github.com/repos/nearai/ironclaw/{archive_format}{/ref}\",\"downloads_url\":\"https://api.github.com/repos/nearai/ironclaw/downloads\",\"issues_url\":\"https://api.github.com/repos/nearai/ironclaw/issues{/number}\",\"pulls_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls{/number}\",\"milestones_url\":\"https://api.github.com/repos/nearai/ironclaw/milestones{/number}\",\"notifications_url\":\"https://api.github.com/repos/nearai/ironclaw/notifications{?since,all,participating}\",\"labels_url\":\"https://api.github.com/repos/nearai/ironclaw/labels{/name}\",\"releases_url\":\"https://api.github.com/repos/nearai/ironclaw/releases{/id}\",\"deployments_url\":\"https://api.github.com/repos/nearai/ironclaw/deployments\",\"created_at\":\"2026-02-03T06:57:10Z\",\"updated_at\":\"2026-04-15T15:21:34Z\",\"pushed_at\":\"2026-04-15T15:27:33Z\",\"git_url\":\"git://github.com/nearai/ironclaw.git\",\"ssh_url\":\"git@github.com:nearai/ironclaw.git\",\"clone_url\":\"https://github.com/nearai/ironclaw.git\",\"svn_url\":\"https://github.com/nearai/ironclaw\",\"homepage\":\"https://www.ironclaw.com\",\"size\":30129,\"stargazers_count\":11789,\"watchers_count\":11789,\"language\":\"Rust\",\"has_issues\":true,\"has_projects\":false,\"has_downloads\":true,\"has_wiki\":false,\"has_pages\":false,\"has_discussions\":false,\"forks_count\":1349,\"mirror_url\":null,\"archived\":false,\"disabled\":false,\"open_issues_count\":640,\"license\":{\"key\":\"apache-2.0\",\"name\":\"Apache License 2.0\",\"spdx_id\":\"Apache-2.0\",\"url\":\"https://api.github.com/licenses/apache-2.0\",\"node_id\":\"MDc6TGljZW5zZTI=\"},\"allow_forking\":true,\"is_template\":false,\"web_commit_signoff_required\":false,\"has_pull_requests\":true,\"pull_request_creation_policy\":\"all\",\"topics\":[],\"visibility\":\"public\",\"forks\":1349,\"open_issues\":640,\"watchers\":11789,\"default_branch\":\"staging\"}},\"_links\":{\"self\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483\"},\"html\":{\"href\":\"https://github.com/nearai/ironclaw/pull/2483\"},\"issue\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/issues/2483\"},\"comments\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/issues/2483/comments\"},\"review_comments\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483/comments\"},\"review_comment\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/pulls/comments{/number}\"},\"commits\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483/commits\"},\"statuses\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/statuses/5496341c49b2abaa9842e2f939ea349c75fd7340\"}},\"author_association\":\"MEMBER\",\"auto_merge\":null,\"assignee\":null,\"active_lock_reason\":null,\"merged\":false,\"mergeable\":true,\"rebaseable\":false,\"mergeable_state\":\"blocked\",\"merged_by\":null,\"comments\":5,\"review_comments\":4,\"maintainer_can_modify\":false,\"commits\":5,\"additions\":570,\"deletions\":78,\"changed_files\":6}" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/pulls/2483", + "headers": [ + [ + "Accept", + "application/vnd.github.v3.diff" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "strict-transport-security", + "max-age=31536000; includeSubdomains; preload" + ], + [ + "etag", + "\"4b1d931184c11e44c3e76e79b2ff928f0612e710d77ecace99caaf32c5d7d12b\"" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "x-ratelimit-remaining", + "4981" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "x-github-media-type", + "github.v3; param=diff" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:23:08 GMT" + ], + [ + "x-github-request-id", + "F521:77454:402653:4B7F8B:69DFAEC2" + ], + [ + "x-ratelimit-used", + "19" + ], + [ + "x-frame-options", + "deny" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP,Accept-Encoding, Accept, X-Requested-With" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:07 GMT" + ], + [ + "content-type", + "application/vnd.github.v3.diff; charset=utf-8" + ], + [ + "content-length", + "45867" + ], + [ + "x-xss-protection", + "0" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "server", + "github.com" + ] + ], + "body": "diff --git a/crates/ironclaw_engine/src/executor/orchestrator.rs b/crates/ironclaw_engine/src/executor/orchestrator.rs\nindex bf16598490..76bf321dee 100644\n--- a/crates/ironclaw_engine/src/executor/orchestrator.rs\n+++ b/crates/ironclaw_engine/src/executor/orchestrator.rs\n@@ -720,6 +720,7 @@ async fn handle_execute_code_step(\n };\n \n // Run user code in a nested Monty VM (same pattern as rlm_query)\n+ let code_start = std::time::Instant::now();\n match Box::pin(execute_code(\n &code,\n thread,\n@@ -748,10 +749,12 @@ async fn handle_execute_code_step(\n // error, etc.), surface it as an ActionFailed event so traces and\n // observers see the failure. Without this, parse errors silently\n // fall back to the LLM via the result dict and never warn callers.\n- if result.had_error {\n+ if let Some(ref category) = result.failure {\n let error_msg = if !result.stdout.is_empty() {\n- let snippet: String = result.stdout.chars().take(500).collect();\n- format!(\"CodeAct execution failed: {snippet}\")\n+ format!(\n+ \"CodeAct execution failed: {}\",\n+ tail_chars(&result.stdout, 500)\n+ )\n } else {\n \"CodeAct execution failed (no stdout)\".to_string()\n };\n@@ -774,6 +777,25 @@ async fn handle_execute_code_step(\n let _ = tx.send(failed_event.clone());\n }\n thread.events.push(failed_event);\n+\n+ // Emit structured CodeExecutionFailed event for instrumentation.\n+ // This enables aggregate analysis of WHY code execution fails\n+ // (Monty limitation vs LLM logic error vs tool dispatch failure).\n+ let error_text = tail_chars(&result.stdout, 500);\n+ let instrumentation_event = ThreadEvent::new(\n+ thread.id,\n+ EventKind::CodeExecutionFailed {\n+ step_id: exec_ctx.step_id,\n+ category: category.clone(),\n+ error: error_text,\n+ code_hash: Some(crate::executor::scripting::code_hash(&code)),\n+ duration_ms: code_start.elapsed().as_millis() as u64,\n+ },\n+ );\n+ if let Some(tx) = event_tx {\n+ let _ = tx.send(instrumentation_event.clone());\n+ }\n+ thread.events.push(instrumentation_event);\n }\n thread.updated_at = chrono::Utc::now();\n \n@@ -795,7 +817,7 @@ async fn handle_execute_code_step(\n \"stdout\": result.stdout,\n \"action_results\": action_results,\n \"final_answer\": result.final_answer,\n- \"had_error\": result.had_error,\n+ \"had_error\": result.failure.is_some(),\n \"pending_gate\": result.need_approval.as_ref().map(|na| {\n match na {\n ThreadOutcome::GatePaused { gate_name, action_name, call_id, parameters, resume_kind, resume_output } => serde_json::json!({\n@@ -2196,6 +2218,19 @@ fn action_calls_to_python_json(calls: &[ActionCall]) -> Vec {\n .collect()\n }\n \n+/// Extract the last `n` characters from `s`.\n+///\n+/// Error tracebacks appear at the end of stdout, after any `print()` output.\n+/// Using the head would capture the print statements instead of the error.\n+fn tail_chars(s: &str, n: usize) -> String {\n+ let char_count = s.chars().count();\n+ if char_count > n {\n+ s.chars().skip(char_count - n).collect()\n+ } else {\n+ s.to_owned()\n+ }\n+}\n+\n /// Build a PII-safe summary of an `action_calls` JSON value for log output.\n ///\n /// The action_calls payload contains tool parameters, which can carry user\n@@ -3860,4 +3895,83 @@ FINAL(batch_error_count)\n );\n }\n }\n+\n+ // ── CodeExecutionFailed event emission (caller test) ────────\n+\n+ #[tokio::test]\n+ async fn execute_code_step_emits_code_execution_failed_event() {\n+ let llm: Arc = Arc::new(ModelCapturingLlm {\n+ captured: tokio::sync::Mutex::new(Vec::new()),\n+ });\n+ let effects: Arc = Arc::new(NoopEffects);\n+ let leases = Arc::new(LeaseManager::new());\n+ let policy = Arc::new(PolicyEngine::new());\n+\n+ let mut thread = Thread::new(\n+ \"test code execution failure instrumentation\",\n+ crate::types::thread::ThreadType::Foreground,\n+ ProjectId::new(),\n+ \"test-user\",\n+ crate::types::thread::ThreadConfig::default(),\n+ );\n+ thread.transition_to(ThreadState::Running, None).unwrap();\n+\n+ // Pass intentionally broken Python code (syntax error)\n+ let args = &[\n+ json_to_monty(&serde_json::json!(\"def ==\")),\n+ json_to_monty(&serde_json::json!({})),\n+ ];\n+\n+ let (tx, _rx) = tokio::sync::broadcast::channel(16);\n+ let _result = handle_execute_code_step(\n+ args,\n+ &[],\n+ &mut thread,\n+ &llm,\n+ &effects,\n+ &leases,\n+ &policy,\n+ Some(&tx),\n+ )\n+ .await;\n+\n+ // Verify CodeExecutionFailed event was emitted on thread.events\n+ let code_failed_events: Vec<_> = thread\n+ .events\n+ .iter()\n+ .filter(|e| matches!(&e.kind, EventKind::CodeExecutionFailed { .. }))\n+ .collect();\n+\n+ assert_eq!(\n+ code_failed_events.len(),\n+ 1,\n+ \"expected exactly one CodeExecutionFailed event, got {}\",\n+ code_failed_events.len()\n+ );\n+\n+ if let EventKind::CodeExecutionFailed {\n+ category,\n+ code_hash,\n+ ..\n+ } = &code_failed_events[0].kind\n+ {\n+ assert_eq!(\n+ *category,\n+ crate::types::step::CodeExecutionFailure::SyntaxError\n+ );\n+ assert!(code_hash.is_some());\n+ } else {\n+ panic!(\"expected CodeExecutionFailed event kind\");\n+ }\n+\n+ // Also verify ActionFailed was emitted (existing behavior)\n+ let action_failed = thread\n+ .events\n+ .iter()\n+ .any(|e| matches!(&e.kind, EventKind::ActionFailed { .. }));\n+ assert!(\n+ action_failed,\n+ \"expected ActionFailed event alongside CodeExecutionFailed\"\n+ );\n+ }\n }\ndiff --git a/crates/ironclaw_engine/src/executor/scripting.rs b/crates/ironclaw_engine/src/executor/scripting.rs\nindex 11f3f09c6c..674fca2750 100644\n--- a/crates/ironclaw_engine/src/executor/scripting.rs\n+++ b/crates/ironclaw_engine/src/executor/scripting.rs\n@@ -33,7 +33,7 @@ use crate::traits::llm::{LlmBackend, LlmCallConfig};\n use crate::types::error::EngineError;\n use crate::types::event::EventKind;\n use crate::types::message::{MessageRole, ThreadMessage};\n-use crate::types::step::{ActionResult, LlmResponse, TokenUsage};\n+use crate::types::step::{ActionResult, CodeExecutionFailure, LlmResponse, TokenUsage};\n use crate::types::thread::Thread;\n use ironclaw_common::ValidTimezone;\n \n@@ -72,8 +72,10 @@ pub struct CodeExecutionResult {\n pub recursive_tokens: TokenUsage,\n /// If set, the code called FINAL() or FINAL_VAR() with this answer.\n pub final_answer: Option,\n- /// Whether the code execution hit an error (traceback included in stdout).\n- pub had_error: bool,\n+ /// Classified failure category. `None` when execution succeeded or was\n+ /// paused by a gate. `Some(category)` when code execution failed —\n+ /// `failure.is_some()` replaces the former `had_error: bool` field.\n+ pub failure: Option,\n }\n \n /// Build a compact output summary for inclusion in LLM context between steps.\n@@ -297,7 +299,6 @@ pub async fn execute_code_with_skills(\n let mut events = Vec::new();\n let mut recursive_tokens = TokenUsage::default();\n let mut final_answer: Option = None;\n- let mut had_error = false;\n \n // Build context variables including persisted state from prior steps\n let (input_names, input_values) = build_context_inputs(thread, persisted_state);\n@@ -335,12 +336,19 @@ pub async fn execute_code_with_skills(\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n- had_error: true,\n+ failure: Some(CodeExecutionFailure::SyntaxError),\n });\n }\n Err(_) => {\n- return Err(EngineError::Effect {\n- reason: \"Monty VM panicked during code parsing\".into(),\n+ return Ok(CodeExecutionResult {\n+ return_value: serde_json::Value::Null,\n+ stdout: format!(\"{stdout}\\nVmPanic: Monty VM panicked during code parsing\"),\n+ action_results,\n+ events,\n+ need_approval: None,\n+ recursive_tokens,\n+ final_answer: None,\n+ failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n };\n@@ -356,6 +364,7 @@ pub async fn execute_code_with_skills(\n Ok(Ok(p)) => p,\n Ok(Err(e)) => {\n // Runtime error flows back to LLM\n+ let category = classify_runtime_error(&e.to_string());\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"{stdout}\\nError: {e}\"),\n@@ -364,12 +373,19 @@ pub async fn execute_code_with_skills(\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n- had_error: true,\n+ failure: Some(category),\n });\n }\n Err(_) => {\n- return Err(EngineError::Effect {\n- reason: \"Monty VM panicked during execution start\".into(),\n+ return Ok(CodeExecutionResult {\n+ return_value: serde_json::Value::Null,\n+ stdout: format!(\"{stdout}\\nVmPanic: Monty VM panicked during execution start\"),\n+ action_results,\n+ events,\n+ need_approval: None,\n+ recursive_tokens,\n+ final_answer: None,\n+ failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n };\n@@ -392,7 +408,7 @@ pub async fn execute_code_with_skills(\n need_approval: None,\n recursive_tokens,\n final_answer,\n- had_error,\n+ failure: None,\n });\n }\n \n@@ -479,7 +495,6 @@ pub async fn execute_code_with_skills(\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n- had_error = true;\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n@@ -488,12 +503,21 @@ pub async fn execute_code_with_skills(\n need_approval: None,\n recursive_tokens,\n final_answer,\n- had_error,\n+ failure: Some(classify_runtime_error(&e.to_string())),\n });\n }\n Err(_) => {\n- return Err(EngineError::Effect {\n- reason: \"Monty VM panicked during resume\".into(),\n+ return Ok(CodeExecutionResult {\n+ return_value: serde_json::Value::Null,\n+ stdout: format!(\n+ \"{stdout}\\nVmPanic: Monty VM panicked during resume\"\n+ ),\n+ action_results,\n+ events,\n+ need_approval: None,\n+ recursive_tokens,\n+ final_answer,\n+ failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n@@ -509,7 +533,6 @@ pub async fn execute_code_with_skills(\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n- had_error = true;\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n@@ -518,12 +541,21 @@ pub async fn execute_code_with_skills(\n need_approval: None,\n recursive_tokens,\n final_answer,\n- had_error,\n+ failure: Some(classify_runtime_error(&e.to_string())),\n });\n }\n Err(_) => {\n- return Err(EngineError::Effect {\n- reason: \"Monty VM panicked during resume_pending\".into(),\n+ return Ok(CodeExecutionResult {\n+ return_value: serde_json::Value::Null,\n+ stdout: format!(\n+ \"{stdout}\\nVmPanic: Monty VM panicked during resume_pending\"\n+ ),\n+ action_results,\n+ events,\n+ need_approval: None,\n+ recursive_tokens,\n+ final_answer,\n+ failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n@@ -585,7 +617,6 @@ pub async fn execute_code_with_skills(\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n- had_error = true;\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n@@ -594,12 +625,21 @@ pub async fn execute_code_with_skills(\n need_approval: None,\n recursive_tokens,\n final_answer,\n- had_error,\n+ failure: Some(CodeExecutionFailure::ToolError),\n });\n }\n Err(_) => {\n- return Err(EngineError::Effect {\n- reason: \"Monty VM panicked during resume_pending\".into(),\n+ return Ok(CodeExecutionResult {\n+ return_value: serde_json::Value::Null,\n+ stdout: format!(\n+ \"{stdout}\\nVmPanic: Monty VM panicked during resume_pending\"\n+ ),\n+ action_results,\n+ events,\n+ need_approval: None,\n+ recursive_tokens,\n+ final_answer,\n+ failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n@@ -612,7 +652,6 @@ pub async fn execute_code_with_skills(\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n- had_error = true;\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n@@ -621,12 +660,21 @@ pub async fn execute_code_with_skills(\n need_approval: None,\n recursive_tokens,\n final_answer,\n- had_error,\n+ failure: Some(CodeExecutionFailure::ToolError),\n });\n }\n Err(_) => {\n- return Err(EngineError::Effect {\n- reason: \"Monty VM panicked during resume\".into(),\n+ return Ok(CodeExecutionResult {\n+ return_value: serde_json::Value::Null,\n+ stdout: format!(\n+ \"{stdout}\\nVmPanic: Monty VM panicked during resume\"\n+ ),\n+ action_results,\n+ events,\n+ need_approval: None,\n+ recursive_tokens,\n+ final_answer,\n+ failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n@@ -640,7 +688,7 @@ pub async fn execute_code_with_skills(\n need_approval: Some(outcome),\n recursive_tokens,\n final_answer: None,\n- had_error,\n+ failure: None,\n });\n }\n }\n@@ -702,7 +750,6 @@ pub async fn execute_code_with_skills(\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n- had_error = true;\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n@@ -711,12 +758,21 @@ pub async fn execute_code_with_skills(\n need_approval: None,\n recursive_tokens,\n final_answer,\n- had_error,\n+ failure: Some(classify_runtime_error(&e.to_string())),\n });\n }\n Err(_) => {\n- return Err(EngineError::Effect {\n- reason: \"Monty VM panicked during ResolveFutures resume\".into(),\n+ return Ok(CodeExecutionResult {\n+ return_value: serde_json::Value::Null,\n+ stdout: format!(\n+ \"{stdout}\\nVmPanic: Monty VM panicked during ResolveFutures resume\"\n+ ),\n+ action_results,\n+ events,\n+ need_approval: None,\n+ recursive_tokens,\n+ final_answer,\n+ failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n@@ -747,7 +803,6 @@ pub async fn execute_code_with_skills(\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nNameError: {e}\"));\n- had_error = true;\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n@@ -756,12 +811,21 @@ pub async fn execute_code_with_skills(\n need_approval: None,\n recursive_tokens,\n final_answer,\n- had_error,\n+ failure: Some(CodeExecutionFailure::NameLookup),\n });\n }\n Err(_) => {\n- return Err(EngineError::Effect {\n- reason: \"Monty VM panicked during name lookup\".into(),\n+ return Ok(CodeExecutionResult {\n+ return_value: serde_json::Value::Null,\n+ stdout: format!(\n+ \"{stdout}\\nVmPanic: Monty VM panicked during name lookup\"\n+ ),\n+ action_results,\n+ events,\n+ need_approval: None,\n+ recursive_tokens,\n+ final_answer,\n+ failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n@@ -779,7 +843,6 @@ pub async fn execute_code_with_skills(\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nOSError: {e}\"));\n- had_error = true;\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n@@ -788,12 +851,19 @@ pub async fn execute_code_with_skills(\n need_approval: None,\n recursive_tokens,\n final_answer,\n- had_error,\n+ failure: Some(CodeExecutionFailure::OsDenied),\n });\n }\n Err(_) => {\n- return Err(EngineError::Effect {\n- reason: \"Monty VM panicked during OS call\".into(),\n+ return Ok(CodeExecutionResult {\n+ return_value: serde_json::Value::Null,\n+ stdout: format!(\"{stdout}\\nVmPanic: Monty VM panicked during OS call\"),\n+ action_results,\n+ events,\n+ need_approval: None,\n+ recursive_tokens,\n+ final_answer,\n+ failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n@@ -802,6 +872,52 @@ pub async fn execute_code_with_skills(\n }\n }\n \n+// ── Error classification ────────────────────────────────────\n+\n+/// Classify a runtime error message into a failure category.\n+///\n+/// Parses the error text from Monty to distinguish between LLM logic bugs\n+/// (NameError, TypeError, etc.), resource limit hits, and Monty VM issues.\n+fn classify_runtime_error(error_msg: &str) -> CodeExecutionFailure {\n+ let lower = error_msg.to_ascii_lowercase();\n+\n+ // Most specific checks first to avoid substring false positives.\n+ if lower.contains(\"timed out\")\n+ || lower.contains(\"timeout\")\n+ || lower.contains(\"memory limit\")\n+ || lower.contains(\"allocation limit\")\n+ || lower.contains(\"out of fuel\")\n+ || lower.contains(\"fuel exhausted\")\n+ || lower.contains(\"resource limit\")\n+ {\n+ CodeExecutionFailure::ResourceLimit\n+ } else if lower.contains(\"os operations are not permitted\") || lower.contains(\"oserror\") {\n+ CodeExecutionFailure::OsDenied\n+ } else if lower.contains(\"syntaxerror\") {\n+ CodeExecutionFailure::SyntaxError\n+ } else {\n+ // NameError, TypeError, ValueError, AttributeError, IndexError,\n+ // KeyError, ModuleNotFoundError, NotImplementedError, etc.\n+ CodeExecutionFailure::RuntimeError\n+ }\n+}\n+\n+/// Compute a short hash of Python code for dedup/correlation in events.\n+///\n+/// Uses FNV-1a (64-bit) which is stable across Rust versions, unlike\n+/// `DefaultHasher`. Not cryptographic — collision probability is ~2^-32\n+/// at typical usage levels, sufficient for dedup but not for security.\n+pub fn code_hash(code: &str) -> String {\n+ const FNV_OFFSET: u64 = 0xcbf29ce484222325;\n+ const FNV_PRIME: u64 = 0x00000100000001B3;\n+ let mut hash = FNV_OFFSET;\n+ for byte in code.as_bytes() {\n+ hash ^= *byte as u64;\n+ hash = hash.wrapping_mul(FNV_PRIME);\n+ }\n+ format!(\"{hash:016x}\")\n+}\n+\n // ── Pending future tracking ─────────────────────────────────\n \n /// A deferred computation spawned as a tokio task, pending resolution\n@@ -1803,7 +1919,7 @@ FINAL(str(result))\n result.stdout\n );\n assert!(\n- !result.had_error,\n+ result.failure.is_none(),\n \"should not error, stdout: {}\",\n result.stdout\n );\n@@ -1855,7 +1971,7 @@ FINAL(str(a + b))\n result.stdout\n );\n assert_eq!(result.action_results.len(), 2);\n- assert!(!result.had_error);\n+ assert!(result.failure.is_none());\n }\n \n // ── asyncio.gather three tools ──────────────────────────\n@@ -1905,7 +2021,7 @@ FINAL(str(s) + \"|\" + str(h) + \"|\" + str(m))\n \"#;\n \n let result = run_code(code, effects, &thread).await.unwrap();\n- assert!(!result.had_error, \"stdout: {}\", result.stdout);\n+ assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.action_results.len(), 3);\n let answer = result.final_answer.unwrap();\n assert!(answer.contains(\"search results\"), \"got: {answer}\");\n@@ -1945,7 +2061,7 @@ FINAL(str(b))\n \"#;\n \n let result = run_code(code, effects, &thread).await.unwrap();\n- assert!(!result.had_error, \"stdout: {}\", result.stdout);\n+ assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.action_results.len(), 2);\n assert_eq!(result.final_answer.as_deref(), Some(\"final\"));\n }\n@@ -1980,7 +2096,7 @@ FINAL(\"should not reach\")\n let result = run_code(code, effects, &thread).await.unwrap();\n // Error in gather propagates as exception — code should error\n assert!(\n- result.had_error,\n+ result.failure.is_some(),\n \"should have error, stdout: {}\",\n result.stdout\n );\n@@ -2024,7 +2140,7 @@ FINAL(\"hello from sync\")\n \n let result = run_code(code, effects, &thread).await.unwrap();\n assert_eq!(result.final_answer.as_deref(), Some(\"hello from sync\"));\n- assert!(!result.had_error);\n+ assert!(result.failure.is_none());\n }\n \n // ── globals() still works ───────────────────────────────\n@@ -2045,7 +2161,7 @@ FINAL(str(has_search) + \"|\" + str(has_http))\n \"#;\n \n let result = run_code(code, effects, &thread).await.unwrap();\n- assert!(!result.had_error, \"stdout: {}\", result.stdout);\n+ assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.final_answer.as_deref(), Some(\"True|True\"));\n }\n \n@@ -2063,7 +2179,7 @@ FINAL(str(len(results)))\n \"#;\n \n let result = run_code(code, effects, &thread).await.unwrap();\n- assert!(!result.had_error, \"stdout: {}\", result.stdout);\n+ assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.final_answer.as_deref(), Some(\"0\"));\n }\n \n@@ -2090,7 +2206,7 @@ FINAL(str(results[0]))\n \"#;\n \n let result = run_code(code, effects, &thread).await.unwrap();\n- assert!(!result.had_error, \"stdout: {}\", result.stdout);\n+ assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.final_answer.as_deref(), Some(\"gathered\"));\n assert_eq!(result.action_results.len(), 1);\n }\n@@ -2137,7 +2253,7 @@ while True:\n // the key assertion is that it DOES NOT run forever.\n if let Ok(r) = result {\n assert!(\n- r.had_error || r.stdout.contains(\"Error\") || r.stdout.contains(\"limit\"),\n+ r.failure.is_some() || r.stdout.contains(\"Error\") || r.stdout.contains(\"limit\"),\n \"resource limit should terminate infinite loop, got stdout: {}\",\n truncate_for_assert(&r.stdout, 500),\n );\n@@ -2279,7 +2395,7 @@ while True:\n // Must terminate — either via error or resource limit\n if let Ok(r) = result {\n assert!(\n- r.had_error || r.stdout.contains(\"Error\") || r.stdout.contains(\"limit\"),\n+ r.failure.is_some() || r.stdout.contains(\"Error\") || r.stdout.contains(\"limit\"),\n \"cpu-bound loop should be terminated, stdout: {}\",\n truncate_for_assert(&r.stdout, 500),\n );\n@@ -2313,7 +2429,7 @@ FINAL(str(x))\n \n let code = \"def broken(\\nFINAL('nope')\";\n let result = run_code(code, effects, &thread).await.unwrap();\n- assert!(result.had_error, \"syntax error should set had_error\");\n+ assert!(result.failure.is_some(), \"syntax error should set failure\");\n assert!(\n result.stdout.contains(\"SyntaxError\") || result.stdout.contains(\"Error\"),\n \"should contain SyntaxError, got: {}\",\n@@ -2736,4 +2852,86 @@ FINAL(str(x))\n assert!(matches!(result, ExtFunctionResult::Error(_)));\n assert!(llm.calls.lock().await.is_empty());\n }\n+\n+ // ── Error classification tests ──────────────────────────────\n+\n+ #[test]\n+ fn classify_syntax_error() {\n+ let cat = classify_runtime_error(\"SyntaxError: unexpected token\");\n+ assert_eq!(cat, CodeExecutionFailure::SyntaxError);\n+ }\n+\n+ #[test]\n+ fn classify_timeout() {\n+ let cat = classify_runtime_error(\"execution timed out after 30s\");\n+ assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\n+ }\n+\n+ #[test]\n+ fn classify_memory_limit() {\n+ let cat = classify_runtime_error(\"memory limit exceeded\");\n+ assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\n+ }\n+\n+ #[test]\n+ fn classify_fuel_exhaustion() {\n+ let cat = classify_runtime_error(\"fuel exhausted during execution\");\n+ assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\n+ }\n+\n+ #[test]\n+ fn classify_os_denied() {\n+ let cat = classify_runtime_error(\"OS operations are not permitted in CodeAct scripts\");\n+ assert_eq!(cat, CodeExecutionFailure::OsDenied);\n+ }\n+\n+ #[test]\n+ fn classify_name_error_as_runtime() {\n+ // NameError from Monty (not NameLookup) is classified as RuntimeError\n+ let cat = classify_runtime_error(\"NameError: name 'foo' is not defined\");\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n+ }\n+\n+ #[test]\n+ fn classify_type_error_as_runtime() {\n+ let cat = classify_runtime_error(\"TypeError: unsupported operand\");\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n+ }\n+\n+ #[test]\n+ fn classify_module_not_found_as_runtime() {\n+ let cat = classify_runtime_error(\"ModuleNotFoundError: No module named 'csv'\");\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n+ }\n+\n+ #[test]\n+ fn classify_syntax_word_is_not_syntaxerror() {\n+ // \"syntax\" alone should not trigger SyntaxError — only \"syntaxerror\" should.\n+ let cat = classify_runtime_error(\"unexpected syntax in expression\");\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n+ }\n+\n+ #[test]\n+ fn vm_panic_variant_serializes_as_snake_case() {\n+ // VmPanic is set directly by catch_unwind paths, not by classify_runtime_error.\n+ // Verify it serializes consistently with Display (both snake_case).\n+ let failure = CodeExecutionFailure::VmPanic;\n+ assert_eq!(failure.to_string(), \"vm_panic\");\n+ let json = serde_json::to_value(&failure).unwrap();\n+ assert_eq!(json, serde_json::json!(\"vm_panic\"));\n+ }\n+\n+ #[test]\n+ fn code_hash_deterministic() {\n+ let h1 = code_hash(\"print('hello')\");\n+ let h2 = code_hash(\"print('hello')\");\n+ assert_eq!(h1, h2);\n+ }\n+\n+ #[test]\n+ fn code_hash_differs_for_different_code() {\n+ let h1 = code_hash(\"print('hello')\");\n+ let h2 = code_hash(\"print('world')\");\n+ assert_ne!(h1, h2);\n+ }\n }\ndiff --git a/crates/ironclaw_engine/src/executor/trace.rs b/crates/ironclaw_engine/src/executor/trace.rs\nindex 0252c30c3d..563d5e8dd6 100644\n--- a/crates/ironclaw_engine/src/executor/trace.rs\n+++ b/crates/ironclaw_engine/src/executor/trace.rs\n@@ -195,33 +195,76 @@ fn analyze_trace(thread: &Thread) -> Vec {\n }\n }\n \n- // 4. Check for code execution errors in output messages.\n- // Code output appears as User-role messages (Monty stdout/stderr) with\n- // prefixes like \"[stdout]\" or \"[stderr]\". Skip the System prompt (index 0)\n- // and Assistant messages to avoid false positives from example text.\n- let error_patterns = [\n- \"NameError\",\n- \"SyntaxError\",\n- \"TypeError\",\n- \"NotImplementedError\",\n- ];\n- for (i, msg) in thread.messages.iter().enumerate() {\n- let is_code_output = msg.role == crate::types::message::MessageRole::User\n- && (msg.content.starts_with(\"[stdout]\")\n- || msg.content.starts_with(\"[stderr]\")\n- || msg.content.starts_with(\"[code \")\n- || msg.content.starts_with(\"Traceback\"));\n- if is_code_output && error_patterns.iter().any(|p| msg.content.contains(p)) {\n- let preview: String = msg.content.chars().take(200).collect();\n+ // 4. Check for code execution errors via structured CodeExecutionFailed events.\n+ // These carry a classified failure category that tells us exactly what kind\n+ // of error occurred (syntax, runtime, name lookup, VM panic, resource limit,\n+ // tool error, OS denied, gate pause).\n+ let code_failures: Vec<&ThreadEvent> = thread\n+ .events\n+ .iter()\n+ .filter(|e| {\n+ matches!(\n+ e.kind,\n+ crate::types::event::EventKind::CodeExecutionFailed { .. }\n+ )\n+ })\n+ .collect();\n+ for event in &code_failures {\n+ if let crate::types::event::EventKind::CodeExecutionFailed {\n+ category, error, ..\n+ } = &event.kind\n+ {\n+ let preview: String = error.chars().take(200).collect();\n+ let severity = match category {\n+ crate::types::step::CodeExecutionFailure::VmPanic => IssueSeverity::Error,\n+ crate::types::step::CodeExecutionFailure::ResourceLimit => IssueSeverity::Error,\n+ _ => IssueSeverity::Warning,\n+ };\n issues.push(TraceIssue {\n- severity: IssueSeverity::Warning,\n- category: \"code_error\".into(),\n- description: format!(\"Code execution error in message {i}: {preview}\"),\n+ severity,\n+ category: format!(\"code_{category}\"),\n+ description: format!(\"Code execution failed ({category}): {preview}\"),\n step: None,\n });\n }\n }\n \n+ // Fallback: also check message-level patterns for backward compatibility\n+ // with threads that ran before the CodeExecutionFailed instrumentation\n+ // was added (PR #2483). Note: threads from mixed eras (some steps\n+ // instrumented, some not) will only report structured events when any\n+ // exist, silently skipping message-level errors from uninstrumented steps.\n+ if code_failures.is_empty() {\n+ let error_patterns = [\n+ \"NameError\",\n+ \"SyntaxError\",\n+ \"TypeError\",\n+ \"NotImplementedError\",\n+ \"ValueError\",\n+ \"AttributeError\",\n+ \"IndexError\",\n+ \"KeyError\",\n+ \"ModuleNotFoundError\",\n+ \"RuntimeError\",\n+ ];\n+ for (i, msg) in thread.messages.iter().enumerate() {\n+ let is_code_output = msg.role == crate::types::message::MessageRole::User\n+ && (msg.content.starts_with(\"[stdout]\")\n+ || msg.content.starts_with(\"[stderr]\")\n+ || msg.content.starts_with(\"[code \")\n+ || msg.content.starts_with(\"Traceback\"));\n+ if is_code_output && error_patterns.iter().any(|p| msg.content.contains(p)) {\n+ let preview: String = msg.content.chars().take(200).collect();\n+ issues.push(TraceIssue {\n+ severity: IssueSeverity::Warning,\n+ category: \"code_error\".into(),\n+ description: format!(\"Code execution error in message {i}: {preview}\"),\n+ step: None,\n+ });\n+ }\n+ }\n+ }\n+\n // 5. Check for empty call_id on ActionResult messages (causes LLM API rejection).\n for (i, msg) in thread.messages.iter().enumerate() {\n if msg.role == crate::types::message::MessageRole::ActionResult {\n@@ -536,6 +579,77 @@ mod tests {\n );\n }\n \n+ // ── CodeExecutionFailed event detection ────────────────────\n+\n+ #[test]\n+ fn detects_code_execution_failure_from_event() {\n+ let mut thread = make_thread();\n+ thread.add_message(ThreadMessage::system(\"sys\"));\n+ thread.add_message(ThreadMessage::assistant(\"```repl\\nimport csv\\n```\"));\n+ thread.events.push(ThreadEvent::new(\n+ thread.id,\n+ EventKind::CodeExecutionFailed {\n+ step_id: StepId::new(),\n+ category: crate::types::step::CodeExecutionFailure::RuntimeError,\n+ error: \"ModuleNotFoundError: No module named 'csv'\".into(),\n+ code_hash: Some(\"abc123\".into()),\n+ duration_ms: 42,\n+ },\n+ ));\n+\n+ let issues = analyze_trace(&thread);\n+ let code_issues: Vec<_> = issues\n+ .iter()\n+ .filter(|i| i.category.starts_with(\"code_\"))\n+ .collect();\n+ assert_eq!(code_issues.len(), 1);\n+ assert_eq!(code_issues[0].category, \"code_runtime_error\");\n+ assert_eq!(code_issues[0].severity, IssueSeverity::Warning);\n+ assert!(code_issues[0].description.contains(\"ModuleNotFoundError\"));\n+ }\n+\n+ #[test]\n+ fn vm_panic_is_error_severity() {\n+ let mut thread = make_thread();\n+ thread.add_message(ThreadMessage::system(\"sys\"));\n+ thread.add_message(ThreadMessage::assistant(\"code\"));\n+ thread.events.push(ThreadEvent::new(\n+ thread.id,\n+ EventKind::CodeExecutionFailed {\n+ step_id: StepId::new(),\n+ category: crate::types::step::CodeExecutionFailure::VmPanic,\n+ error: \"Monty panicked: unreachable\".into(),\n+ code_hash: None,\n+ duration_ms: 0,\n+ },\n+ ));\n+\n+ let issues = analyze_trace(&thread);\n+ let panic_issues: Vec<_> = issues\n+ .iter()\n+ .filter(|i| i.category == \"code_vm_panic\")\n+ .collect();\n+ assert_eq!(panic_issues.len(), 1);\n+ assert_eq!(panic_issues[0].severity, IssueSeverity::Error);\n+ }\n+\n+ #[test]\n+ fn fallback_message_detection_when_no_events() {\n+ // Threads from before instrumentation should still be detected\n+ let mut thread = make_thread();\n+ thread.add_message(ThreadMessage::system(\"sys\"));\n+ thread.add_message(ThreadMessage::assistant(\"code\"));\n+ thread.add_message(ThreadMessage::user(\n+ \"[stdout]\\nNameError: name 'foo' is not defined\",\n+ ));\n+\n+ let issues = analyze_trace(&thread);\n+ assert!(\n+ issues.iter().any(|i| i.category == \"code_error\"),\n+ \"should detect code error from message when no CodeExecutionFailed events exist\"\n+ );\n+ }\n+\n #[test]\n fn trace_serializes_approval_request_payload() {\n let mut thread = make_thread();\ndiff --git a/crates/ironclaw_engine/src/lib.rs b/crates/ironclaw_engine/src/lib.rs\nindex 5894d7a7a6..9ecaa6b575 100644\n--- a/crates/ironclaw_engine/src/lib.rs\n+++ b/crates/ironclaw_engine/src/lib.rs\n@@ -46,7 +46,8 @@ pub use types::mission::{Mission, MissionCadence, MissionId, MissionStatus, Vali\n pub use types::project::{Project, ProjectId};\n pub use types::provenance::Provenance;\n pub use types::step::{\n- ActionCall, ActionResult, ExecutionTier, LlmResponse, Step, StepId, StepStatus, TokenUsage,\n+ ActionCall, ActionResult, CodeExecutionFailure, ExecutionTier, LlmResponse, Step, StepId,\n+ StepStatus, TokenUsage,\n };\n pub use types::thread::{\n ActiveSkillProvenance, Thread, ThreadConfig, ThreadId, ThreadState, ThreadType,\ndiff --git a/crates/ironclaw_engine/src/types/event.rs b/crates/ironclaw_engine/src/types/event.rs\nindex 4c707dcdd0..b61185f071 100644\n--- a/crates/ironclaw_engine/src/types/event.rs\n+++ b/crates/ironclaw_engine/src/types/event.rs\n@@ -215,10 +215,35 @@ pub enum EventKind {\n skill_names: Vec,\n },\n \n+ // ── Code execution instrumentation ────────────────────────\n+ /// Emitted when a code (REPL) execution attempt fails. Enables aggregate\n+ /// analysis of code execution failure modes to determine whether the\n+ /// runtime (Monty), the LLM, or tool dispatch is the primary source of\n+ /// failures.\n+ CodeExecutionFailed {\n+ step_id: StepId,\n+ /// Classified failure category.\n+ category: crate::types::step::CodeExecutionFailure,\n+ /// The error message text (truncated to 500 chars).\n+ error: String,\n+ /// Hash of the Python code that was executed, for dedup/correlation.\n+ #[serde(default, skip_serializing_if = \"Option::is_none\")]\n+ code_hash: Option,\n+ /// Duration of the code execution attempt in milliseconds.\n+ #[serde(default)]\n+ duration_ms: u64,\n+ },\n+\n // ── Orchestrator versioning ───────────────────────────────\n OrchestratorRollback {\n from_version: u64,\n to_version: u64,\n reason: String,\n },\n+\n+ /// Unknown event kind — catch-all for forward compatibility during\n+ /// rolling deploys. Older binaries deserializing events written by\n+ /// newer binaries will produce this variant instead of failing.\n+ #[serde(other)]\n+ Unknown,\n }\ndiff --git a/crates/ironclaw_engine/src/types/step.rs b/crates/ironclaw_engine/src/types/step.rs\nindex f09bcb0ed4..161096e9b0 100644\n--- a/crates/ironclaw_engine/src/types/step.rs\n+++ b/crates/ironclaw_engine/src/types/step.rs\n@@ -132,6 +132,46 @@ pub struct ActionResult {\n pub duration: Duration,\n }\n \n+/// Classification of code execution failures.\n+///\n+/// Used by the instrumentation layer to distinguish Monty VM limitations\n+/// from LLM logic errors, tool dispatch failures, and resource exhaustion.\n+/// This data enables informed decisions about runtime alternatives.\n+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]\n+#[serde(rename_all = \"snake_case\")]\n+pub enum CodeExecutionFailure {\n+ /// Python parse error — LLM generated invalid syntax.\n+ SyntaxError,\n+ /// Python runtime error (NameError, TypeError, ValueError, etc.) —\n+ /// LLM logic bug or use of unsupported feature.\n+ RuntimeError,\n+ /// Name lookup failed — function/variable not in scope and not a known tool.\n+ NameLookup,\n+ /// Monty VM panicked (catch_unwind caught it). Indicates a Monty bug,\n+ /// not a user code issue.\n+ VmPanic,\n+ /// Resource limit hit (timeout, memory, or allocation cap).\n+ ResourceLimit,\n+ /// A tool call inside code returned an error.\n+ ToolError,\n+ /// OS operation attempted (blocked by sandbox).\n+ OsDenied,\n+}\n+\n+impl std::fmt::Display for CodeExecutionFailure {\n+ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {\n+ match self {\n+ Self::SyntaxError => write!(f, \"syntax_error\"),\n+ Self::RuntimeError => write!(f, \"runtime_error\"),\n+ Self::NameLookup => write!(f, \"name_lookup\"),\n+ Self::VmPanic => write!(f, \"vm_panic\"),\n+ Self::ResourceLimit => write!(f, \"resource_limit\"),\n+ Self::ToolError => write!(f, \"tool_error\"),\n+ Self::OsDenied => write!(f, \"os_denied\"),\n+ }\n+ }\n+}\n+\n /// Token usage for a single LLM call.\n #[derive(Debug, Clone, Copy, Default, Serialize, Deserialize)]\n pub struct TokenUsage {\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/pulls/2483/files?per_page=100" + }, + "response": { + "status": 200, + "headers": [ + [ + "x-content-type-options", + "nosniff" + ], + [ + "strict-transport-security", + "max-age=31536000; includeSubdomains; preload" + ], + [ + "content-type", + "application/json; charset=utf-8" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "x-xss-protection", + "0" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP,Accept-Encoding, Accept, X-Requested-With" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-github-request-id", + "F535:3D8160:4061E1:4BBA59:69DFAEC3" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:23:08 GMT" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:08 GMT" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-github-media-type", + "github.v3; format=json" + ], + [ + "etag", + "\"684a195e98a258be2a092400e150c420a630966f3f2d516f4f9ff7a9ce4d8fa9\"" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-frame-options", + "deny" + ], + [ + "x-accepted-oauth-scopes", + "" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-ratelimit-remaining", + "4980" + ], + [ + "server", + "github.com" + ], + [ + "content-length", + "49621" + ], + [ + "x-ratelimit-used", + "20" + ], + [ + "x-oauth-scopes", + "repo" + ] + ], + "body": "[{\"sha\":\"76bf321deece47e13ba3044a33ee11da44d6130c\",\"filename\":\"crates/ironclaw_engine/src/executor/orchestrator.rs\",\"status\":\"modified\",\"additions\":118,\"deletions\":4,\"changes\":122,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Forchestrator.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Forchestrator.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Forchestrator.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -720,6 +720,7 @@ async fn handle_execute_code_step(\\n };\\n \\n // Run user code in a nested Monty VM (same pattern as rlm_query)\\n+ let code_start = std::time::Instant::now();\\n match Box::pin(execute_code(\\n &code,\\n thread,\\n@@ -748,10 +749,12 @@ async fn handle_execute_code_step(\\n // error, etc.), surface it as an ActionFailed event so traces and\\n // observers see the failure. Without this, parse errors silently\\n // fall back to the LLM via the result dict and never warn callers.\\n- if result.had_error {\\n+ if let Some(ref category) = result.failure {\\n let error_msg = if !result.stdout.is_empty() {\\n- let snippet: String = result.stdout.chars().take(500).collect();\\n- format!(\\\"CodeAct execution failed: {snippet}\\\")\\n+ format!(\\n+ \\\"CodeAct execution failed: {}\\\",\\n+ tail_chars(&result.stdout, 500)\\n+ )\\n } else {\\n \\\"CodeAct execution failed (no stdout)\\\".to_string()\\n };\\n@@ -774,6 +777,25 @@ async fn handle_execute_code_step(\\n let _ = tx.send(failed_event.clone());\\n }\\n thread.events.push(failed_event);\\n+\\n+ // Emit structured CodeExecutionFailed event for instrumentation.\\n+ // This enables aggregate analysis of WHY code execution fails\\n+ // (Monty limitation vs LLM logic error vs tool dispatch failure).\\n+ let error_text = tail_chars(&result.stdout, 500);\\n+ let instrumentation_event = ThreadEvent::new(\\n+ thread.id,\\n+ EventKind::CodeExecutionFailed {\\n+ step_id: exec_ctx.step_id,\\n+ category: category.clone(),\\n+ error: error_text,\\n+ code_hash: Some(crate::executor::scripting::code_hash(&code)),\\n+ duration_ms: code_start.elapsed().as_millis() as u64,\\n+ },\\n+ );\\n+ if let Some(tx) = event_tx {\\n+ let _ = tx.send(instrumentation_event.clone());\\n+ }\\n+ thread.events.push(instrumentation_event);\\n }\\n thread.updated_at = chrono::Utc::now();\\n \\n@@ -795,7 +817,7 @@ async fn handle_execute_code_step(\\n \\\"stdout\\\": result.stdout,\\n \\\"action_results\\\": action_results,\\n \\\"final_answer\\\": result.final_answer,\\n- \\\"had_error\\\": result.had_error,\\n+ \\\"had_error\\\": result.failure.is_some(),\\n \\\"pending_gate\\\": result.need_approval.as_ref().map(|na| {\\n match na {\\n ThreadOutcome::GatePaused { gate_name, action_name, call_id, parameters, resume_kind, resume_output } => serde_json::json!({\\n@@ -2196,6 +2218,19 @@ fn action_calls_to_python_json(calls: &[ActionCall]) -> Vec {\\n .collect()\\n }\\n \\n+/// Extract the last `n` characters from `s`.\\n+///\\n+/// Error tracebacks appear at the end of stdout, after any `print()` output.\\n+/// Using the head would capture the print statements instead of the error.\\n+fn tail_chars(s: &str, n: usize) -> String {\\n+ let char_count = s.chars().count();\\n+ if char_count > n {\\n+ s.chars().skip(char_count - n).collect()\\n+ } else {\\n+ s.to_owned()\\n+ }\\n+}\\n+\\n /// Build a PII-safe summary of an `action_calls` JSON value for log output.\\n ///\\n /// The action_calls payload contains tool parameters, which can carry user\\n@@ -3860,4 +3895,83 @@ FINAL(batch_error_count)\\n );\\n }\\n }\\n+\\n+ // ── CodeExecutionFailed event emission (caller test) ────────\\n+\\n+ #[tokio::test]\\n+ async fn execute_code_step_emits_code_execution_failed_event() {\\n+ let llm: Arc = Arc::new(ModelCapturingLlm {\\n+ captured: tokio::sync::Mutex::new(Vec::new()),\\n+ });\\n+ let effects: Arc = Arc::new(NoopEffects);\\n+ let leases = Arc::new(LeaseManager::new());\\n+ let policy = Arc::new(PolicyEngine::new());\\n+\\n+ let mut thread = Thread::new(\\n+ \\\"test code execution failure instrumentation\\\",\\n+ crate::types::thread::ThreadType::Foreground,\\n+ ProjectId::new(),\\n+ \\\"test-user\\\",\\n+ crate::types::thread::ThreadConfig::default(),\\n+ );\\n+ thread.transition_to(ThreadState::Running, None).unwrap();\\n+\\n+ // Pass intentionally broken Python code (syntax error)\\n+ let args = &[\\n+ json_to_monty(&serde_json::json!(\\\"def ==\\\")),\\n+ json_to_monty(&serde_json::json!({})),\\n+ ];\\n+\\n+ let (tx, _rx) = tokio::sync::broadcast::channel(16);\\n+ let _result = handle_execute_code_step(\\n+ args,\\n+ &[],\\n+ &mut thread,\\n+ &llm,\\n+ &effects,\\n+ &leases,\\n+ &policy,\\n+ Some(&tx),\\n+ )\\n+ .await;\\n+\\n+ // Verify CodeExecutionFailed event was emitted on thread.events\\n+ let code_failed_events: Vec<_> = thread\\n+ .events\\n+ .iter()\\n+ .filter(|e| matches!(&e.kind, EventKind::CodeExecutionFailed { .. }))\\n+ .collect();\\n+\\n+ assert_eq!(\\n+ code_failed_events.len(),\\n+ 1,\\n+ \\\"expected exactly one CodeExecutionFailed event, got {}\\\",\\n+ code_failed_events.len()\\n+ );\\n+\\n+ if let EventKind::CodeExecutionFailed {\\n+ category,\\n+ code_hash,\\n+ ..\\n+ } = &code_failed_events[0].kind\\n+ {\\n+ assert_eq!(\\n+ *category,\\n+ crate::types::step::CodeExecutionFailure::SyntaxError\\n+ );\\n+ assert!(code_hash.is_some());\\n+ } else {\\n+ panic!(\\\"expected CodeExecutionFailed event kind\\\");\\n+ }\\n+\\n+ // Also verify ActionFailed was emitted (existing behavior)\\n+ let action_failed = thread\\n+ .events\\n+ .iter()\\n+ .any(|e| matches!(&e.kind, EventKind::ActionFailed { .. }));\\n+ assert!(\\n+ action_failed,\\n+ \\\"expected ActionFailed event alongside CodeExecutionFailed\\\"\\n+ );\\n+ }\\n }\"},{\"sha\":\"674fca2750840d6f245ff8676d54df8b83ca74f1\",\"filename\":\"crates/ironclaw_engine/src/executor/scripting.rs\",\"status\":\"modified\",\"additions\":250,\"deletions\":52,\"changes\":302,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Fscripting.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Fscripting.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Fscripting.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -33,7 +33,7 @@ use crate::traits::llm::{LlmBackend, LlmCallConfig};\\n use crate::types::error::EngineError;\\n use crate::types::event::EventKind;\\n use crate::types::message::{MessageRole, ThreadMessage};\\n-use crate::types::step::{ActionResult, LlmResponse, TokenUsage};\\n+use crate::types::step::{ActionResult, CodeExecutionFailure, LlmResponse, TokenUsage};\\n use crate::types::thread::Thread;\\n use ironclaw_common::ValidTimezone;\\n \\n@@ -72,8 +72,10 @@ pub struct CodeExecutionResult {\\n pub recursive_tokens: TokenUsage,\\n /// If set, the code called FINAL() or FINAL_VAR() with this answer.\\n pub final_answer: Option,\\n- /// Whether the code execution hit an error (traceback included in stdout).\\n- pub had_error: bool,\\n+ /// Classified failure category. `None` when execution succeeded or was\\n+ /// paused by a gate. `Some(category)` when code execution failed —\\n+ /// `failure.is_some()` replaces the former `had_error: bool` field.\\n+ pub failure: Option,\\n }\\n \\n /// Build a compact output summary for inclusion in LLM context between steps.\\n@@ -297,7 +299,6 @@ pub async fn execute_code_with_skills(\\n let mut events = Vec::new();\\n let mut recursive_tokens = TokenUsage::default();\\n let mut final_answer: Option = None;\\n- let mut had_error = false;\\n \\n // Build context variables including persisted state from prior steps\\n let (input_names, input_values) = build_context_inputs(thread, persisted_state);\\n@@ -335,12 +336,19 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer: None,\\n- had_error: true,\\n+ failure: Some(CodeExecutionFailure::SyntaxError),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during code parsing\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\\"{stdout}\\\\nVmPanic: Monty VM panicked during code parsing\\\"),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer: None,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n };\\n@@ -356,6 +364,7 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => p,\\n Ok(Err(e)) => {\\n // Runtime error flows back to LLM\\n+ let category = classify_runtime_error(&e.to_string());\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout: format!(\\\"{stdout}\\\\nError: {e}\\\"),\\n@@ -364,12 +373,19 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer: None,\\n- had_error: true,\\n+ failure: Some(category),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during execution start\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\\"{stdout}\\\\nVmPanic: Monty VM panicked during execution start\\\"),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer: None,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n };\\n@@ -392,7 +408,7 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: None,\\n });\\n }\\n \\n@@ -479,7 +495,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -488,12 +503,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(classify_runtime_error(&e.to_string())),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during resume\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during resume\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -509,7 +533,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -518,12 +541,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(classify_runtime_error(&e.to_string())),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during resume_pending\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during resume_pending\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -585,7 +617,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -594,12 +625,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(CodeExecutionFailure::ToolError),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during resume_pending\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during resume_pending\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -612,7 +652,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -621,12 +660,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(CodeExecutionFailure::ToolError),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during resume\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during resume\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -640,7 +688,7 @@ pub async fn execute_code_with_skills(\\n need_approval: Some(outcome),\\n recursive_tokens,\\n final_answer: None,\\n- had_error,\\n+ failure: None,\\n });\\n }\\n }\\n@@ -702,7 +750,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -711,12 +758,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(classify_runtime_error(&e.to_string())),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during ResolveFutures resume\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during ResolveFutures resume\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -747,7 +803,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nNameError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -756,12 +811,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(CodeExecutionFailure::NameLookup),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during name lookup\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during name lookup\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -779,7 +843,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nOSError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -788,12 +851,19 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(CodeExecutionFailure::OsDenied),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during OS call\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\\"{stdout}\\\\nVmPanic: Monty VM panicked during OS call\\\"),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -802,6 +872,52 @@ pub async fn execute_code_with_skills(\\n }\\n }\\n \\n+// ── Error classification ────────────────────────────────────\\n+\\n+/// Classify a runtime error message into a failure category.\\n+///\\n+/// Parses the error text from Monty to distinguish between LLM logic bugs\\n+/// (NameError, TypeError, etc.), resource limit hits, and Monty VM issues.\\n+fn classify_runtime_error(error_msg: &str) -> CodeExecutionFailure {\\n+ let lower = error_msg.to_ascii_lowercase();\\n+\\n+ // Most specific checks first to avoid substring false positives.\\n+ if lower.contains(\\\"timed out\\\")\\n+ || lower.contains(\\\"timeout\\\")\\n+ || lower.contains(\\\"memory limit\\\")\\n+ || lower.contains(\\\"allocation limit\\\")\\n+ || lower.contains(\\\"out of fuel\\\")\\n+ || lower.contains(\\\"fuel exhausted\\\")\\n+ || lower.contains(\\\"resource limit\\\")\\n+ {\\n+ CodeExecutionFailure::ResourceLimit\\n+ } else if lower.contains(\\\"os operations are not permitted\\\") || lower.contains(\\\"oserror\\\") {\\n+ CodeExecutionFailure::OsDenied\\n+ } else if lower.contains(\\\"syntaxerror\\\") {\\n+ CodeExecutionFailure::SyntaxError\\n+ } else {\\n+ // NameError, TypeError, ValueError, AttributeError, IndexError,\\n+ // KeyError, ModuleNotFoundError, NotImplementedError, etc.\\n+ CodeExecutionFailure::RuntimeError\\n+ }\\n+}\\n+\\n+/// Compute a short hash of Python code for dedup/correlation in events.\\n+///\\n+/// Uses FNV-1a (64-bit) which is stable across Rust versions, unlike\\n+/// `DefaultHasher`. Not cryptographic — collision probability is ~2^-32\\n+/// at typical usage levels, sufficient for dedup but not for security.\\n+pub fn code_hash(code: &str) -> String {\\n+ const FNV_OFFSET: u64 = 0xcbf29ce484222325;\\n+ const FNV_PRIME: u64 = 0x00000100000001B3;\\n+ let mut hash = FNV_OFFSET;\\n+ for byte in code.as_bytes() {\\n+ hash ^= *byte as u64;\\n+ hash = hash.wrapping_mul(FNV_PRIME);\\n+ }\\n+ format!(\\\"{hash:016x}\\\")\\n+}\\n+\\n // ── Pending future tracking ─────────────────────────────────\\n \\n /// A deferred computation spawned as a tokio task, pending resolution\\n@@ -1803,7 +1919,7 @@ FINAL(str(result))\\n result.stdout\\n );\\n assert!(\\n- !result.had_error,\\n+ result.failure.is_none(),\\n \\\"should not error, stdout: {}\\\",\\n result.stdout\\n );\\n@@ -1855,7 +1971,7 @@ FINAL(str(a + b))\\n result.stdout\\n );\\n assert_eq!(result.action_results.len(), 2);\\n- assert!(!result.had_error);\\n+ assert!(result.failure.is_none());\\n }\\n \\n // ── asyncio.gather three tools ──────────────────────────\\n@@ -1905,7 +2021,7 @@ FINAL(str(s) + \\\"|\\\" + str(h) + \\\"|\\\" + str(m))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.action_results.len(), 3);\\n let answer = result.final_answer.unwrap();\\n assert!(answer.contains(\\\"search results\\\"), \\\"got: {answer}\\\");\\n@@ -1945,7 +2061,7 @@ FINAL(str(b))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.action_results.len(), 2);\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"final\\\"));\\n }\\n@@ -1980,7 +2096,7 @@ FINAL(\\\"should not reach\\\")\\n let result = run_code(code, effects, &thread).await.unwrap();\\n // Error in gather propagates as exception — code should error\\n assert!(\\n- result.had_error,\\n+ result.failure.is_some(),\\n \\\"should have error, stdout: {}\\\",\\n result.stdout\\n );\\n@@ -2024,7 +2140,7 @@ FINAL(\\\"hello from sync\\\")\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"hello from sync\\\"));\\n- assert!(!result.had_error);\\n+ assert!(result.failure.is_none());\\n }\\n \\n // ── globals() still works ───────────────────────────────\\n@@ -2045,7 +2161,7 @@ FINAL(str(has_search) + \\\"|\\\" + str(has_http))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"True|True\\\"));\\n }\\n \\n@@ -2063,7 +2179,7 @@ FINAL(str(len(results)))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"0\\\"));\\n }\\n \\n@@ -2090,7 +2206,7 @@ FINAL(str(results[0]))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"gathered\\\"));\\n assert_eq!(result.action_results.len(), 1);\\n }\\n@@ -2137,7 +2253,7 @@ while True:\\n // the key assertion is that it DOES NOT run forever.\\n if let Ok(r) = result {\\n assert!(\\n- r.had_error || r.stdout.contains(\\\"Error\\\") || r.stdout.contains(\\\"limit\\\"),\\n+ r.failure.is_some() || r.stdout.contains(\\\"Error\\\") || r.stdout.contains(\\\"limit\\\"),\\n \\\"resource limit should terminate infinite loop, got stdout: {}\\\",\\n truncate_for_assert(&r.stdout, 500),\\n );\\n@@ -2279,7 +2395,7 @@ while True:\\n // Must terminate — either via error or resource limit\\n if let Ok(r) = result {\\n assert!(\\n- r.had_error || r.stdout.contains(\\\"Error\\\") || r.stdout.contains(\\\"limit\\\"),\\n+ r.failure.is_some() || r.stdout.contains(\\\"Error\\\") || r.stdout.contains(\\\"limit\\\"),\\n \\\"cpu-bound loop should be terminated, stdout: {}\\\",\\n truncate_for_assert(&r.stdout, 500),\\n );\\n@@ -2313,7 +2429,7 @@ FINAL(str(x))\\n \\n let code = \\\"def broken(\\\\nFINAL('nope')\\\";\\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(result.had_error, \\\"syntax error should set had_error\\\");\\n+ assert!(result.failure.is_some(), \\\"syntax error should set failure\\\");\\n assert!(\\n result.stdout.contains(\\\"SyntaxError\\\") || result.stdout.contains(\\\"Error\\\"),\\n \\\"should contain SyntaxError, got: {}\\\",\\n@@ -2736,4 +2852,86 @@ FINAL(str(x))\\n assert!(matches!(result, ExtFunctionResult::Error(_)));\\n assert!(llm.calls.lock().await.is_empty());\\n }\\n+\\n+ // ── Error classification tests ──────────────────────────────\\n+\\n+ #[test]\\n+ fn classify_syntax_error() {\\n+ let cat = classify_runtime_error(\\\"SyntaxError: unexpected token\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::SyntaxError);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_timeout() {\\n+ let cat = classify_runtime_error(\\\"execution timed out after 30s\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_memory_limit() {\\n+ let cat = classify_runtime_error(\\\"memory limit exceeded\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_fuel_exhaustion() {\\n+ let cat = classify_runtime_error(\\\"fuel exhausted during execution\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_os_denied() {\\n+ let cat = classify_runtime_error(\\\"OS operations are not permitted in CodeAct scripts\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::OsDenied);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_name_error_as_runtime() {\\n+ // NameError from Monty (not NameLookup) is classified as RuntimeError\\n+ let cat = classify_runtime_error(\\\"NameError: name 'foo' is not defined\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_type_error_as_runtime() {\\n+ let cat = classify_runtime_error(\\\"TypeError: unsupported operand\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_module_not_found_as_runtime() {\\n+ let cat = classify_runtime_error(\\\"ModuleNotFoundError: No module named 'csv'\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_syntax_word_is_not_syntaxerror() {\\n+ // \\\"syntax\\\" alone should not trigger SyntaxError — only \\\"syntaxerror\\\" should.\\n+ let cat = classify_runtime_error(\\\"unexpected syntax in expression\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\\n+ }\\n+\\n+ #[test]\\n+ fn vm_panic_variant_serializes_as_snake_case() {\\n+ // VmPanic is set directly by catch_unwind paths, not by classify_runtime_error.\\n+ // Verify it serializes consistently with Display (both snake_case).\\n+ let failure = CodeExecutionFailure::VmPanic;\\n+ assert_eq!(failure.to_string(), \\\"vm_panic\\\");\\n+ let json = serde_json::to_value(&failure).unwrap();\\n+ assert_eq!(json, serde_json::json!(\\\"vm_panic\\\"));\\n+ }\\n+\\n+ #[test]\\n+ fn code_hash_deterministic() {\\n+ let h1 = code_hash(\\\"print('hello')\\\");\\n+ let h2 = code_hash(\\\"print('hello')\\\");\\n+ assert_eq!(h1, h2);\\n+ }\\n+\\n+ #[test]\\n+ fn code_hash_differs_for_different_code() {\\n+ let h1 = code_hash(\\\"print('hello')\\\");\\n+ let h2 = code_hash(\\\"print('world')\\\");\\n+ assert_ne!(h1, h2);\\n+ }\\n }\"},{\"sha\":\"563d5e8dd6835053c514ff45fcebbf5e94bf283f\",\"filename\":\"crates/ironclaw_engine/src/executor/trace.rs\",\"status\":\"modified\",\"additions\":135,\"deletions\":21,\"changes\":156,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Ftrace.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Ftrace.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Ftrace.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -195,33 +195,76 @@ fn analyze_trace(thread: &Thread) -> Vec {\\n }\\n }\\n \\n- // 4. Check for code execution errors in output messages.\\n- // Code output appears as User-role messages (Monty stdout/stderr) with\\n- // prefixes like \\\"[stdout]\\\" or \\\"[stderr]\\\". Skip the System prompt (index 0)\\n- // and Assistant messages to avoid false positives from example text.\\n- let error_patterns = [\\n- \\\"NameError\\\",\\n- \\\"SyntaxError\\\",\\n- \\\"TypeError\\\",\\n- \\\"NotImplementedError\\\",\\n- ];\\n- for (i, msg) in thread.messages.iter().enumerate() {\\n- let is_code_output = msg.role == crate::types::message::MessageRole::User\\n- && (msg.content.starts_with(\\\"[stdout]\\\")\\n- || msg.content.starts_with(\\\"[stderr]\\\")\\n- || msg.content.starts_with(\\\"[code \\\")\\n- || msg.content.starts_with(\\\"Traceback\\\"));\\n- if is_code_output && error_patterns.iter().any(|p| msg.content.contains(p)) {\\n- let preview: String = msg.content.chars().take(200).collect();\\n+ // 4. Check for code execution errors via structured CodeExecutionFailed events.\\n+ // These carry a classified failure category that tells us exactly what kind\\n+ // of error occurred (syntax, runtime, name lookup, VM panic, resource limit,\\n+ // tool error, OS denied, gate pause).\\n+ let code_failures: Vec<&ThreadEvent> = thread\\n+ .events\\n+ .iter()\\n+ .filter(|e| {\\n+ matches!(\\n+ e.kind,\\n+ crate::types::event::EventKind::CodeExecutionFailed { .. }\\n+ )\\n+ })\\n+ .collect();\\n+ for event in &code_failures {\\n+ if let crate::types::event::EventKind::CodeExecutionFailed {\\n+ category, error, ..\\n+ } = &event.kind\\n+ {\\n+ let preview: String = error.chars().take(200).collect();\\n+ let severity = match category {\\n+ crate::types::step::CodeExecutionFailure::VmPanic => IssueSeverity::Error,\\n+ crate::types::step::CodeExecutionFailure::ResourceLimit => IssueSeverity::Error,\\n+ _ => IssueSeverity::Warning,\\n+ };\\n issues.push(TraceIssue {\\n- severity: IssueSeverity::Warning,\\n- category: \\\"code_error\\\".into(),\\n- description: format!(\\\"Code execution error in message {i}: {preview}\\\"),\\n+ severity,\\n+ category: format!(\\\"code_{category}\\\"),\\n+ description: format!(\\\"Code execution failed ({category}): {preview}\\\"),\\n step: None,\\n });\\n }\\n }\\n \\n+ // Fallback: also check message-level patterns for backward compatibility\\n+ // with threads that ran before the CodeExecutionFailed instrumentation\\n+ // was added (PR #2483). Note: threads from mixed eras (some steps\\n+ // instrumented, some not) will only report structured events when any\\n+ // exist, silently skipping message-level errors from uninstrumented steps.\\n+ if code_failures.is_empty() {\\n+ let error_patterns = [\\n+ \\\"NameError\\\",\\n+ \\\"SyntaxError\\\",\\n+ \\\"TypeError\\\",\\n+ \\\"NotImplementedError\\\",\\n+ \\\"ValueError\\\",\\n+ \\\"AttributeError\\\",\\n+ \\\"IndexError\\\",\\n+ \\\"KeyError\\\",\\n+ \\\"ModuleNotFoundError\\\",\\n+ \\\"RuntimeError\\\",\\n+ ];\\n+ for (i, msg) in thread.messages.iter().enumerate() {\\n+ let is_code_output = msg.role == crate::types::message::MessageRole::User\\n+ && (msg.content.starts_with(\\\"[stdout]\\\")\\n+ || msg.content.starts_with(\\\"[stderr]\\\")\\n+ || msg.content.starts_with(\\\"[code \\\")\\n+ || msg.content.starts_with(\\\"Traceback\\\"));\\n+ if is_code_output && error_patterns.iter().any(|p| msg.content.contains(p)) {\\n+ let preview: String = msg.content.chars().take(200).collect();\\n+ issues.push(TraceIssue {\\n+ severity: IssueSeverity::Warning,\\n+ category: \\\"code_error\\\".into(),\\n+ description: format!(\\\"Code execution error in message {i}: {preview}\\\"),\\n+ step: None,\\n+ });\\n+ }\\n+ }\\n+ }\\n+\\n // 5. Check for empty call_id on ActionResult messages (causes LLM API rejection).\\n for (i, msg) in thread.messages.iter().enumerate() {\\n if msg.role == crate::types::message::MessageRole::ActionResult {\\n@@ -536,6 +579,77 @@ mod tests {\\n );\\n }\\n \\n+ // ── CodeExecutionFailed event detection ────────────────────\\n+\\n+ #[test]\\n+ fn detects_code_execution_failure_from_event() {\\n+ let mut thread = make_thread();\\n+ thread.add_message(ThreadMessage::system(\\\"sys\\\"));\\n+ thread.add_message(ThreadMessage::assistant(\\\"```repl\\\\nimport csv\\\\n```\\\"));\\n+ thread.events.push(ThreadEvent::new(\\n+ thread.id,\\n+ EventKind::CodeExecutionFailed {\\n+ step_id: StepId::new(),\\n+ category: crate::types::step::CodeExecutionFailure::RuntimeError,\\n+ error: \\\"ModuleNotFoundError: No module named 'csv'\\\".into(),\\n+ code_hash: Some(\\\"abc123\\\".into()),\\n+ duration_ms: 42,\\n+ },\\n+ ));\\n+\\n+ let issues = analyze_trace(&thread);\\n+ let code_issues: Vec<_> = issues\\n+ .iter()\\n+ .filter(|i| i.category.starts_with(\\\"code_\\\"))\\n+ .collect();\\n+ assert_eq!(code_issues.len(), 1);\\n+ assert_eq!(code_issues[0].category, \\\"code_runtime_error\\\");\\n+ assert_eq!(code_issues[0].severity, IssueSeverity::Warning);\\n+ assert!(code_issues[0].description.contains(\\\"ModuleNotFoundError\\\"));\\n+ }\\n+\\n+ #[test]\\n+ fn vm_panic_is_error_severity() {\\n+ let mut thread = make_thread();\\n+ thread.add_message(ThreadMessage::system(\\\"sys\\\"));\\n+ thread.add_message(ThreadMessage::assistant(\\\"code\\\"));\\n+ thread.events.push(ThreadEvent::new(\\n+ thread.id,\\n+ EventKind::CodeExecutionFailed {\\n+ step_id: StepId::new(),\\n+ category: crate::types::step::CodeExecutionFailure::VmPanic,\\n+ error: \\\"Monty panicked: unreachable\\\".into(),\\n+ code_hash: None,\\n+ duration_ms: 0,\\n+ },\\n+ ));\\n+\\n+ let issues = analyze_trace(&thread);\\n+ let panic_issues: Vec<_> = issues\\n+ .iter()\\n+ .filter(|i| i.category == \\\"code_vm_panic\\\")\\n+ .collect();\\n+ assert_eq!(panic_issues.len(), 1);\\n+ assert_eq!(panic_issues[0].severity, IssueSeverity::Error);\\n+ }\\n+\\n+ #[test]\\n+ fn fallback_message_detection_when_no_events() {\\n+ // Threads from before instrumentation should still be detected\\n+ let mut thread = make_thread();\\n+ thread.add_message(ThreadMessage::system(\\\"sys\\\"));\\n+ thread.add_message(ThreadMessage::assistant(\\\"code\\\"));\\n+ thread.add_message(ThreadMessage::user(\\n+ \\\"[stdout]\\\\nNameError: name 'foo' is not defined\\\",\\n+ ));\\n+\\n+ let issues = analyze_trace(&thread);\\n+ assert!(\\n+ issues.iter().any(|i| i.category == \\\"code_error\\\"),\\n+ \\\"should detect code error from message when no CodeExecutionFailed events exist\\\"\\n+ );\\n+ }\\n+\\n #[test]\\n fn trace_serializes_approval_request_payload() {\\n let mut thread = make_thread();\"},{\"sha\":\"9ecaa6b575a369535e657b5f922187d3ce5149fa\",\"filename\":\"crates/ironclaw_engine/src/lib.rs\",\"status\":\"modified\",\"additions\":2,\"deletions\":1,\"changes\":3,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Flib.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Flib.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Flib.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -46,7 +46,8 @@ pub use types::mission::{Mission, MissionCadence, MissionId, MissionStatus, Vali\\n pub use types::project::{Project, ProjectId};\\n pub use types::provenance::Provenance;\\n pub use types::step::{\\n- ActionCall, ActionResult, ExecutionTier, LlmResponse, Step, StepId, StepStatus, TokenUsage,\\n+ ActionCall, ActionResult, CodeExecutionFailure, ExecutionTier, LlmResponse, Step, StepId,\\n+ StepStatus, TokenUsage,\\n };\\n pub use types::thread::{\\n ActiveSkillProvenance, Thread, ThreadConfig, ThreadId, ThreadState, ThreadType,\"},{\"sha\":\"b61185f07131d737272f1068c899f7f6b5821960\",\"filename\":\"crates/ironclaw_engine/src/types/event.rs\",\"status\":\"modified\",\"additions\":25,\"deletions\":0,\"changes\":25,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fevent.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fevent.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fevent.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -215,10 +215,35 @@ pub enum EventKind {\\n skill_names: Vec,\\n },\\n \\n+ // ── Code execution instrumentation ────────────────────────\\n+ /// Emitted when a code (REPL) execution attempt fails. Enables aggregate\\n+ /// analysis of code execution failure modes to determine whether the\\n+ /// runtime (Monty), the LLM, or tool dispatch is the primary source of\\n+ /// failures.\\n+ CodeExecutionFailed {\\n+ step_id: StepId,\\n+ /// Classified failure category.\\n+ category: crate::types::step::CodeExecutionFailure,\\n+ /// The error message text (truncated to 500 chars).\\n+ error: String,\\n+ /// Hash of the Python code that was executed, for dedup/correlation.\\n+ #[serde(default, skip_serializing_if = \\\"Option::is_none\\\")]\\n+ code_hash: Option,\\n+ /// Duration of the code execution attempt in milliseconds.\\n+ #[serde(default)]\\n+ duration_ms: u64,\\n+ },\\n+\\n // ── Orchestrator versioning ───────────────────────────────\\n OrchestratorRollback {\\n from_version: u64,\\n to_version: u64,\\n reason: String,\\n },\\n+\\n+ /// Unknown event kind — catch-all for forward compatibility during\\n+ /// rolling deploys. Older binaries deserializing events written by\\n+ /// newer binaries will produce this variant instead of failing.\\n+ #[serde(other)]\\n+ Unknown,\\n }\"},{\"sha\":\"161096e9b02238d78d86f611befa72d32fbe0a76\",\"filename\":\"crates/ironclaw_engine/src/types/step.rs\",\"status\":\"modified\",\"additions\":40,\"deletions\":0,\"changes\":40,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fstep.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fstep.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fstep.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -132,6 +132,46 @@ pub struct ActionResult {\\n pub duration: Duration,\\n }\\n \\n+/// Classification of code execution failures.\\n+///\\n+/// Used by the instrumentation layer to distinguish Monty VM limitations\\n+/// from LLM logic errors, tool dispatch failures, and resource exhaustion.\\n+/// This data enables informed decisions about runtime alternatives.\\n+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]\\n+#[serde(rename_all = \\\"snake_case\\\")]\\n+pub enum CodeExecutionFailure {\\n+ /// Python parse error — LLM generated invalid syntax.\\n+ SyntaxError,\\n+ /// Python runtime error (NameError, TypeError, ValueError, etc.) —\\n+ /// LLM logic bug or use of unsupported feature.\\n+ RuntimeError,\\n+ /// Name lookup failed — function/variable not in scope and not a known tool.\\n+ NameLookup,\\n+ /// Monty VM panicked (catch_unwind caught it). Indicates a Monty bug,\\n+ /// not a user code issue.\\n+ VmPanic,\\n+ /// Resource limit hit (timeout, memory, or allocation cap).\\n+ ResourceLimit,\\n+ /// A tool call inside code returned an error.\\n+ ToolError,\\n+ /// OS operation attempted (blocked by sandbox).\\n+ OsDenied,\\n+}\\n+\\n+impl std::fmt::Display for CodeExecutionFailure {\\n+ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {\\n+ match self {\\n+ Self::SyntaxError => write!(f, \\\"syntax_error\\\"),\\n+ Self::RuntimeError => write!(f, \\\"runtime_error\\\"),\\n+ Self::NameLookup => write!(f, \\\"name_lookup\\\"),\\n+ Self::VmPanic => write!(f, \\\"vm_panic\\\"),\\n+ Self::ResourceLimit => write!(f, \\\"resource_limit\\\"),\\n+ Self::ToolError => write!(f, \\\"tool_error\\\"),\\n+ Self::OsDenied => write!(f, \\\"os_denied\\\"),\\n+ }\\n+ }\\n+}\\n+\\n /// Token usage for a single LLM call.\\n #[derive(Debug, Clone, Copy, Default, Serialize, Deserialize)]\\n pub struct TokenUsage {\"}]" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/executor/orchestrator.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "x-github-request-id", + "F54A:13AC25:3F9C3F:4AF4D2:69DFAEC4" + ], + [ + "x-xss-protection", + "0" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "x-ratelimit-used", + "21" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:08 GMT" + ], + [ + "content-length", + "152011" + ], + [ + "etag", + "\"76bf321deece47e13ba3044a33ee11da44d6130c\"" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "server", + "github.com" + ], + [ + "x-ratelimit-remaining", + "4979" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-frame-options", + "deny" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "x-ratelimit-limit", + "5000" + ] + ], + "body": "//! Python orchestrator — the self-modifiable execution loop.\n//!\n//! Replaces the Rust `ExecutionLoop::run()` with versioned Python code\n//! executed via Monty. The orchestrator is the \"glue layer\" between the\n//! LLM and tools — tool dispatch, output formatting, state management,\n//! truncation — all in Python, patchable by the self-improvement Mission.\n//!\n//! Host functions exposed to the orchestrator Python:\n//! - `__llm_complete__` — make an LLM call\n//! - `__execute_code_step__` — run user CodeAct code in a nested Monty VM\n//! - `__execute_action__` — execute a single tool action\n//! - `__execute_actions_parallel__` — execute multiple tool actions concurrently\n//! - `__check_signals__` — poll for stop/inject signals\n//! - `__emit_event__` — broadcast a ThreadEvent\n//! - `__save_checkpoint__` — persist thread state\n//! - `__transition_to__` — change thread state (validated)\n//! - `__retrieve_docs__` — query memory docs\n//! - `__check_budget__` — remaining tokens/time/USD\n//! - `__get_actions__` — available tool definitions\n\nuse std::sync::Arc;\nuse std::sync::atomic::{AtomicU64, Ordering};\n\nuse std::collections::HashMap;\n\nuse monty::{\n ExtFunctionResult, LimitedTracker, MontyObject, MontyRun, NameLookupResult, PrintWriter,\n ResourceLimits, RunProgress,\n};\nuse tracing::{debug, warn};\n\nuse crate::capability::lease::LeaseManager;\nuse crate::capability::policy::PolicyEngine;\nuse crate::memory::RetrievalEngine;\nuse crate::runtime::lease_refresh::reconcile_dynamic_tool_lease;\nuse crate::runtime::messaging::{SignalReceiver, ThreadOutcome, ThreadSignal};\nuse crate::traits::effect::{EffectExecutor, ThreadExecutionContext};\nuse crate::traits::llm::{LlmBackend, LlmCallConfig};\nuse crate::traits::store::Store;\nuse crate::types::error::EngineError;\nuse crate::types::event::{EventKind, ThreadEvent, summarize_params};\nuse crate::types::message::ThreadMessage;\nuse crate::types::project::ProjectId;\nuse crate::types::shared_owner_id;\nuse crate::types::step::{ActionCall, StepId, TokenUsage};\nuse crate::types::thread::{ActiveSkillProvenance, Thread, ThreadState};\nuse ironclaw_common::ValidTimezone;\n\nuse super::scripting::{execute_code, json_to_monty, monty_to_json, monty_to_string};\n\n/// The compiled-in default orchestrator (v0).\npub(crate) const DEFAULT_ORCHESTRATOR: &str = include_str!(\"../../orchestrator/default.py\");\n\n/// Well-known title for orchestrator code in the Store.\npub const ORCHESTRATOR_TITLE: &str = \"orchestrator:main\";\n\n/// Well-known tag for orchestrator code docs.\npub const ORCHESTRATOR_TAG: &str = \"orchestrator_code\";\n\n/// Result of running the orchestrator.\npub struct OrchestratorResult {\n /// The thread outcome parsed from the orchestrator's return value.\n pub outcome: ThreadOutcome,\n /// Total tokens used by LLM calls within the orchestrator.\n pub tokens_used: TokenUsage,\n}\n\n/// Extract source_channel from thread metadata (set by ConversationManager).\nfn thread_source_channel(thread: &Thread) -> Option {\n thread\n .metadata\n .get(\"source_channel\")\n .and_then(|v| v.as_str())\n .map(String::from)\n}\n\n/// Extract and validate user_timezone from thread metadata (set by bridge router).\nfn thread_user_timezone(thread: &Thread) -> Option {\n thread\n .metadata\n .get(\"user_timezone\")\n .and_then(|v| v.as_str())\n .and_then(ValidTimezone::parse)\n}\n\nfn normalize_pause_outcome(\n thread: &mut Thread,\n outcome: &ThreadOutcome,\n) -> Result<(), EngineError> {\n if matches!(outcome, ThreadOutcome::GatePaused { .. }) && thread.state != ThreadState::Waiting {\n thread.transition_to(\n ThreadState::Waiting,\n Some(\"waiting on external gate resolution\".into()),\n )?;\n }\n Ok(())\n}\n\n/// Resource limits for the orchestrator VM.\nfn orchestrator_limits() -> ResourceLimits {\n ResourceLimits::new()\n .max_duration(std::time::Duration::from_secs(300)) // 5 min (longer than user code)\n .max_allocations(5_000_000)\n .max_memory(128 * 1024 * 1024) // 128 MB\n}\n\n/// Maximum consecutive failures before auto-rollback.\nconst MAX_FAILURES_BEFORE_ROLLBACK: u64 = 3;\n\n/// Well-known title for orchestrator failure tracking.\nconst FAILURE_TRACKER_TITLE: &str = \"orchestrator:failures\";\nconst LEASE_REFRESH_WARN_INTERVAL_SECS: u64 = 60;\n\nfn warn_on_lease_refresh_failure(context: &'static str, error: &crate::types::error::EngineError) {\n static LAST_WARN_TS: AtomicU64 = AtomicU64::new(0);\n\n let now = chrono::Utc::now().timestamp().max(0) as u64;\n let last = LAST_WARN_TS.load(Ordering::Relaxed);\n if now.saturating_sub(last) >= LEASE_REFRESH_WARN_INTERVAL_SECS\n && LAST_WARN_TS\n .compare_exchange(last, now, Ordering::Relaxed, Ordering::Relaxed)\n .is_ok()\n {\n warn!(context, error = %error, \"dynamic lease refresh failed\");\n } else {\n debug!(context, error = %error, \"dynamic lease refresh failed\");\n }\n}\n\n/// Load orchestrator code: runtime version from Store, or compiled-in default.\n///\n/// When `allow_self_modify` is false, always uses the compiled-in default\n/// regardless of any runtime versions in the Store. This is the safe default\n/// for production — runtime orchestrator patching is opt-in.\n///\n/// Checks the failure tracker — if the latest version has >= 3 consecutive\n/// failures, falls back to the previous version (or compiled-in default).\npub async fn load_orchestrator(\n store: Option<&Arc>,\n project_id: ProjectId,\n allow_self_modify: bool,\n) -> (String, u64) {\n if !allow_self_modify {\n debug!(\"orchestrator self-modification disabled, using compiled-in default (v0)\");\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n }\n\n let Some(store) = store else {\n debug!(\"using compiled-in default orchestrator (v0, no store)\");\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n };\n\n let docs = match store.list_shared_memory_docs(project_id).await {\n Ok(d) => d,\n Err(_) => {\n debug!(\"using compiled-in default orchestrator (v0, store error)\");\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n }\n };\n\n load_orchestrator_from_docs(&docs, allow_self_modify)\n}\n\n/// Load orchestrator from pre-fetched system memory docs.\n///\n/// When the caller already has the `list_memory_docs` result, use this to\n/// avoid a duplicate Store query. Returns `(code, version)`.\n///\n/// Respects `allow_self_modify` — when false, always returns the compiled-in\n/// default. The caller in `loop_engine.rs` passes this from engine config.\npub fn load_orchestrator_from_docs(\n docs: &[crate::types::memory::MemoryDoc],\n allow_self_modify: bool,\n) -> (String, u64) {\n if !allow_self_modify {\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n }\n\n // Find all orchestrator versions, sorted by version number descending\n let mut versions: Vec<_> = docs\n .iter()\n .filter(|d| d.title == ORCHESTRATOR_TITLE && d.tags.contains(&ORCHESTRATOR_TAG.to_string()))\n .collect();\n versions.sort_by(|a, b| {\n let va = a\n .metadata\n .get(\"version\")\n .and_then(|v| v.as_u64())\n .unwrap_or(0);\n let vb = b\n .metadata\n .get(\"version\")\n .and_then(|v| v.as_u64())\n .unwrap_or(0);\n vb.cmp(&va) // descending\n });\n\n if versions.is_empty() {\n debug!(\"using compiled-in default orchestrator (v0)\");\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n }\n\n // Check failure count for the latest version\n let failures = load_failure_count(docs);\n\n for doc in &versions {\n let version = doc\n .metadata\n .get(\"version\")\n .and_then(|v| v.as_u64())\n .unwrap_or(1);\n\n // Skip versions with too many failures (only check the latest)\n if version\n == versions[0]\n .metadata\n .get(\"version\")\n .and_then(|v| v.as_u64())\n .unwrap_or(1)\n && failures >= MAX_FAILURES_BEFORE_ROLLBACK\n {\n debug!(\n version,\n failures, \"orchestrator version has too many failures, skipping\"\n );\n continue;\n }\n\n debug!(version, \"loaded runtime orchestrator\");\n return (doc.content.clone(), version);\n }\n\n // All versions failed — fall back to compiled-in default\n debug!(\"all orchestrator versions failed, using compiled-in default (v0)\");\n (DEFAULT_ORCHESTRATOR.to_string(), 0)\n}\n\n/// Record a failure for the current orchestrator version.\npub async fn record_orchestrator_failure(\n store: &Arc,\n project_id: ProjectId,\n version: u64,\n) {\n use crate::types::memory::{DocType, MemoryDoc};\n\n let docs = match store.list_shared_memory_docs(project_id).await {\n Ok(docs) => docs,\n Err(e) => {\n debug!(\"failed to list memory docs for failure tracker: {e}\");\n return;\n }\n };\n let existing = docs.iter().find(|d| d.title == FAILURE_TRACKER_TITLE);\n\n let mut tracker = if let Some(doc) = existing {\n doc.clone()\n } else {\n MemoryDoc::new(\n project_id,\n shared_owner_id(),\n DocType::Note,\n FAILURE_TRACKER_TITLE,\n \"\",\n )\n .with_tags(vec![\"orchestrator_meta\".to_string()])\n };\n\n // Store failure count as JSON in content: {\"version\": N, \"count\": M}\n let current: serde_json::Value =\n serde_json::from_str(&tracker.content).unwrap_or(serde_json::json!({}));\n let current_version = current.get(\"version\").and_then(|v| v.as_u64()).unwrap_or(0);\n let current_count = current.get(\"count\").and_then(|v| v.as_u64()).unwrap_or(0);\n\n let new_count = if current_version == version {\n current_count + 1\n } else {\n 1 // new version, reset count\n };\n\n tracker.content = serde_json::json!({\n \"version\": version,\n \"count\": new_count,\n })\n .to_string();\n tracker.updated_at = chrono::Utc::now();\n\n if let Err(e) = store.save_memory_doc(&tracker).await {\n debug!(\"failed to save orchestrator failure tracker: {e}\");\n }\n\n debug!(version, count = new_count, \"recorded orchestrator failure\");\n}\n\n/// Reset the failure counter (called after successful execution).\npub async fn reset_orchestrator_failures(store: &Arc, project_id: ProjectId) {\n let docs = store\n .list_shared_memory_docs(project_id)\n .await\n .unwrap_or_default();\n let existing = docs.iter().find(|d| d.title == FAILURE_TRACKER_TITLE);\n\n if let Some(doc) = existing {\n let mut tracker = doc.clone();\n tracker.content = serde_json::json!({\"version\": 0, \"count\": 0}).to_string();\n tracker.updated_at = chrono::Utc::now();\n let _ = store.save_memory_doc(&tracker).await;\n }\n}\n\n/// Load failure count for the latest orchestrator version.\nfn load_failure_count(docs: &[crate::types::memory::MemoryDoc]) -> u64 {\n docs.iter()\n .find(|d| d.title == FAILURE_TRACKER_TITLE)\n .and_then(|d| serde_json::from_str::(&d.content).ok())\n .and_then(|v| v.get(\"count\").and_then(|c| c.as_u64()))\n .unwrap_or(0)\n}\n\n/// Execute the orchestrator Python code with host function dispatch.\n///\n/// This is the core function that replaces `ExecutionLoop::run()`'s inner loop.\n/// The orchestrator Python calls host functions via Monty's suspension mechanism,\n/// and this function handles each suspension by delegating to the appropriate\n/// Rust implementation.\n#[allow(clippy::too_many_arguments)]\npub async fn execute_orchestrator(\n code: &str,\n thread: &mut Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &Arc,\n policy: &Arc,\n signal_rx: &mut SignalReceiver,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n retrieval: Option<&RetrievalEngine>,\n store: Option<&Arc>,\n persisted_state: &serde_json::Value,\n) -> Result {\n let mut total_tokens = TokenUsage::default();\n\n // Build context variables for the orchestrator\n let (input_names, input_values) = build_orchestrator_inputs(thread, persisted_state);\n\n // Parse and compile\n let runner = match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n MontyRun::new(code.to_string(), \"orchestrator.py\", input_names)\n })) {\n Ok(Ok(runner)) => runner,\n Ok(Err(e)) => {\n return Err(EngineError::Effect {\n reason: format!(\"Orchestrator parse error: {e}\"),\n });\n }\n Err(_) => {\n return Err(EngineError::Effect {\n reason: \"Monty VM panicked during orchestrator parsing\".into(),\n });\n }\n };\n\n // Start execution\n let mut stdout = String::new();\n let tracker = LimitedTracker::new(orchestrator_limits());\n\n let run_result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n runner.start(input_values, tracker, PrintWriter::Collect(&mut stdout))\n }));\n\n let mut progress = match run_result {\n Ok(Ok(p)) => p,\n Ok(Err(e)) => {\n return Err(EngineError::Effect {\n reason: format!(\"Orchestrator runtime error: {e}\"),\n });\n }\n Err(_) => {\n return Err(EngineError::Effect {\n reason: \"Monty VM panicked during orchestrator start\".into(),\n });\n }\n };\n\n // Drive the orchestrator dispatch loop\n let mut final_result: Option = None;\n\n loop {\n match progress {\n RunProgress::Complete(obj) => {\n // Use FINAL result if set, otherwise fall back to VM return value\n let result = if let Some(ref fr) = final_result {\n fr.clone()\n } else {\n monty_to_json(&obj)\n };\n sync_runtime_state(thread, result.get(\"state\"));\n let outcome = parse_outcome(&result);\n sync_visible_outcome(thread, &outcome);\n normalize_pause_outcome(thread, &outcome)?;\n return Ok(OrchestratorResult {\n outcome,\n tokens_used: total_tokens,\n });\n }\n\n RunProgress::FunctionCall(call) => {\n let action_name = call.function_name.clone();\n let args = &call.args;\n let kwargs = &call.kwargs;\n\n debug!(action = %action_name, \"orchestrator: host function call\");\n\n let ext_result = match action_name.as_str() {\n // FINAL(result) — orchestrator returns its outcome\n \"FINAL\" => {\n let val = args.first().map(monty_to_json).unwrap_or_default();\n final_result = Some(val);\n ExtFunctionResult::Return(MontyObject::None)\n }\n\n // __llm_complete__(messages, actions, config)\n \"__llm_complete__\" => {\n handle_llm_complete(\n args,\n kwargs,\n thread,\n LlmCompleteDeps {\n llm,\n effects,\n leases,\n store,\n },\n &mut total_tokens,\n )\n .await\n }\n\n // __execute_code_step__(code, state)\n \"__execute_code_step__\" => {\n handle_execute_code_step(\n args, kwargs, thread, llm, effects, leases, policy, event_tx,\n )\n .await\n }\n\n // __execute_action__(name, params, call_id=...)\n \"__execute_action__\" => {\n handle_execute_action(\n args, kwargs, thread, effects, leases, policy, event_tx,\n )\n .await\n }\n\n // __execute_actions_parallel__(calls)\n \"__execute_actions_parallel__\" => {\n handle_execute_actions_parallel(\n args, thread, effects, leases, policy, event_tx,\n )\n .await\n }\n\n // __check_signals__()\n \"__check_signals__\" => handle_check_signals(signal_rx, thread),\n\n // __emit_event__(kind, **data)\n \"__emit_event__\" => handle_emit_event(args, kwargs, thread, event_tx),\n\n // __save_checkpoint__(state, counters)\n \"__save_checkpoint__\" => handle_save_checkpoint(args, kwargs, thread),\n\n // __transition_to__(state, reason)\n \"__transition_to__\" => handle_transition_to(args, kwargs, thread),\n\n // __retrieve_docs__(goal, max_docs)\n \"__retrieve_docs__\" => {\n handle_retrieve_docs(args, kwargs, thread, retrieval).await\n }\n\n // __check_budget__()\"\n \"__check_budget__\" => handle_check_budget(thread),\n\n // __get_actions__()\n \"__get_actions__\" => handle_get_actions(thread, effects, leases, store).await,\n\n // __list_skills__(max_candidates, max_tokens)\n \"__list_skills__\" => handle_list_skills(args, thread, store).await,\n\n // __record_skill_usage__(doc_id, success)\n \"__record_skill_usage__\" => handle_record_skill_usage(args, store).await,\n\n // __regex_match__(pattern, text) -> bool\n // Evaluates a regex against text using Rust's regex crate.\n // Invalid patterns return False silently. Monty has no `re`\n // module, so this host function bridges the gap for the\n // skill selector's pattern-based scoring.\n \"__regex_match__\" => handle_regex_match(args),\n\n // __set_active_skills__(skills)\n \"__set_active_skills__\" => handle_set_active_skills(args, thread),\n\n // Unknown — let Monty resolve it (user-defined functions, builtins)\n other => ExtFunctionResult::NotFound(other.to_string()),\n };\n\n // Resume the orchestrator VM\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume(ext_result, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n return Err(EngineError::Effect {\n reason: format!(\"Orchestrator error after resume: {e}\"),\n });\n }\n Err(_) => {\n return Err(EngineError::Effect {\n reason: \"Monty VM panicked during orchestrator resume\".into(),\n });\n }\n }\n\n // If FINAL was called, the VM should complete on next iteration\n if final_result.is_some() {\n continue;\n }\n }\n\n RunProgress::NameLookup(lookup) => {\n // Undefined variable — resume with NameError\n let name = lookup.name.clone();\n debug!(name = %name, \"orchestrator: unresolved name\");\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n lookup.resume(\n NameLookupResult::Undefined,\n PrintWriter::Collect(&mut stdout),\n )\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n return Err(EngineError::Effect {\n reason: format!(\"Orchestrator NameError '{name}': {e}\"),\n });\n }\n Err(_) => {\n return Err(EngineError::Effect {\n reason: format!(\"Monty panic on NameLookup '{name}'\"),\n });\n }\n }\n }\n\n RunProgress::OsCall(_) => {\n return Err(EngineError::Effect {\n reason: \"Orchestrator attempted OS call (blocked)\".into(),\n });\n }\n\n RunProgress::ResolveFutures(_) => {\n return Err(EngineError::Effect {\n reason: \"Orchestrator attempted async (not supported)\".into(),\n });\n }\n }\n }\n}\n\n// ── Host function handlers ──────────────────────────────────\n\nstruct LlmCompleteDeps<'a> {\n llm: &'a Arc,\n effects: &'a Arc,\n leases: &'a Arc,\n store: Option<&'a Arc>,\n}\n\n/// Handle `__llm_complete__(messages, actions, config)`.\n///\n/// Calls the LLM and returns the response as a dict:\n/// `{type: \"text\"|\"code\"|\"actions\", content/code/calls: ..., usage: {...}}`\n///\nasync fn handle_llm_complete(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n deps: LlmCompleteDeps<'_>,\n total_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n use crate::types::step::LlmResponse;\n\n let explicit_messages = args.first().map(monty_to_json).filter(|v| !v.is_null());\n let explicit_config = args.get(2).map(monty_to_json).filter(|v| !v.is_null());\n let messages = explicit_messages\n .as_ref()\n .and_then(json_to_thread_messages)\n .unwrap_or_else(|| thread.messages.clone());\n\n if let Err(e) = reconcile_dynamic_tool_lease(\n thread,\n deps.effects,\n deps.leases,\n deps.store,\n &crate::LeasePlanner::new(),\n )\n .await\n {\n warn_on_lease_refresh_failure(\"llm_complete\", &e);\n }\n\n let active_leases = deps.leases.active_for_thread(thread.id).await;\n let actions = deps\n .effects\n .available_actions(&active_leases)\n .await\n .unwrap_or_default();\n\n let config = LlmCallConfig {\n max_tokens: explicit_config\n .as_ref()\n .and_then(|cfg| cfg.get(\"max_tokens\"))\n .and_then(|v| v.as_u64())\n .and_then(|v| u32::try_from(v).ok()),\n temperature: explicit_config\n .as_ref()\n .and_then(|cfg| cfg.get(\"temperature\"))\n .and_then(|v| v.as_f64())\n .map(|v| v as f32),\n force_text: explicit_config\n .as_ref()\n .and_then(|cfg| cfg.get(\"force_text\"))\n .and_then(|v| v.as_bool())\n .unwrap_or(false),\n depth: thread.config.depth,\n model: explicit_config\n .as_ref()\n .and_then(|cfg| cfg.get(\"model\"))\n .and_then(|v| v.as_str())\n .map(String::from),\n metadata: HashMap::new(),\n };\n\n match deps.llm.complete(&messages, &actions, &config).await {\n Ok(output) => {\n total_tokens.input_tokens += output.usage.input_tokens;\n total_tokens.output_tokens += output.usage.output_tokens;\n total_tokens.cost_usd += output.usage.cost_usd;\n\n let usage = serde_json::json!({\n \"input_tokens\": output.usage.input_tokens,\n \"output_tokens\": output.usage.output_tokens,\n \"cost_usd\": output.usage.cost_usd,\n });\n\n let result = match output.response {\n LlmResponse::Text(text) => {\n serde_json::json!({\"type\": \"text\", \"content\": text, \"usage\": usage})\n }\n LlmResponse::Code { code, .. } => {\n serde_json::json!({\"type\": \"code\", \"code\": code, \"usage\": usage})\n }\n LlmResponse::ActionCalls { calls, content } => {\n // Single source of truth for the Python interchange\n // shape — must round-trip via `python_json_to_action_calls`.\n let calls_json = action_calls_to_python_json(&calls);\n serde_json::json!({\n \"type\": \"actions\",\n \"content\": content,\n \"calls\": calls_json,\n \"usage\": usage\n })\n }\n };\n\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n Err(e) => ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::RuntimeError,\n Some(format!(\"LLM call failed: {e}\")),\n )),\n }\n}\n\n/// Handle `__execute_code_step__(code, state)`.\n///\n/// Runs user CodeAct code in a nested Monty VM with full tool dispatch.\n/// Returns a dict with stdout, return_value, action_results, etc.\n#[allow(clippy::too_many_arguments)]\nasync fn handle_execute_code_step(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &Arc,\n policy: &Arc,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n) -> ExtFunctionResult {\n let code = match args.first() {\n Some(obj) => monty_to_string(obj),\n None => {\n return ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::TypeError,\n Some(\"__execute_code_step__ requires a code string\".into()),\n ));\n }\n };\n\n let state = args\n .get(1)\n .map(monty_to_json)\n .unwrap_or(serde_json::json!({}));\n\n let exec_ctx = ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: thread.user_id.clone(),\n step_id: StepId::new(),\n current_call_id: None,\n source_channel: thread_source_channel(thread),\n user_timezone: thread_user_timezone(thread),\n };\n\n // Run user code in a nested Monty VM (same pattern as rlm_query)\n let code_start = std::time::Instant::now();\n match Box::pin(execute_code(\n &code,\n thread,\n llm,\n effects,\n leases,\n policy,\n &exec_ctx,\n &[],\n &state,\n ))\n .await\n {\n Ok(result) => {\n // Broadcast events from code execution to the thread and event channel.\n // Without this, ActionExecuted events from CodeAct tool calls are lost\n // and never appear in traces.\n for event_kind in &result.events {\n let event = ThreadEvent::new(thread.id, event_kind.clone());\n if let Some(tx) = event_tx {\n let _ = tx.send(event.clone());\n }\n thread.events.push(event);\n }\n // If the CodeAct snippet itself failed (Python SyntaxError, runtime\n // error, etc.), surface it as an ActionFailed event so traces and\n // observers see the failure. Without this, parse errors silently\n // fall back to the LLM via the result dict and never warn callers.\n if let Some(ref category) = result.failure {\n let error_msg = if !result.stdout.is_empty() {\n format!(\n \"CodeAct execution failed: {}\",\n tail_chars(&result.stdout, 500)\n )\n } else {\n \"CodeAct execution failed (no stdout)\".to_string()\n };\n let failed_event = ThreadEvent::new(\n thread.id,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: \"__codeact__\".to_string(),\n // Synthetic call_id derived from the step id —\n // CodeAct snippet failures don't have an LLM-provided\n // call_id, but `loop_engine.rs:1277` asserts that\n // ActionFailed events carry a non-empty call_id for\n // trace correlation.\n call_id: format!(\"codeact-step-{}\", exec_ctx.step_id.0),\n error: error_msg,\n params_summary: None,\n },\n );\n if let Some(tx) = event_tx {\n let _ = tx.send(failed_event.clone());\n }\n thread.events.push(failed_event);\n\n // Emit structured CodeExecutionFailed event for instrumentation.\n // This enables aggregate analysis of WHY code execution fails\n // (Monty limitation vs LLM logic error vs tool dispatch failure).\n let error_text = tail_chars(&result.stdout, 500);\n let instrumentation_event = ThreadEvent::new(\n thread.id,\n EventKind::CodeExecutionFailed {\n step_id: exec_ctx.step_id,\n category: category.clone(),\n error: error_text,\n code_hash: Some(crate::executor::scripting::code_hash(&code)),\n duration_ms: code_start.elapsed().as_millis() as u64,\n },\n );\n if let Some(tx) = event_tx {\n let _ = tx.send(instrumentation_event.clone());\n }\n thread.events.push(instrumentation_event);\n }\n thread.updated_at = chrono::Utc::now();\n\n let action_results: Vec = result\n .action_results\n .iter()\n .map(|r| {\n serde_json::json!({\n \"action_name\": r.action_name,\n \"output\": r.output,\n \"is_error\": r.is_error,\n \"duration_ms\": r.duration.as_millis(),\n })\n })\n .collect();\n\n let result_json = serde_json::json!({\n \"return_value\": result.return_value,\n \"stdout\": result.stdout,\n \"action_results\": action_results,\n \"final_answer\": result.final_answer,\n \"had_error\": result.failure.is_some(),\n \"pending_gate\": result.need_approval.as_ref().map(|na| {\n match na {\n ThreadOutcome::GatePaused { gate_name, action_name, call_id, parameters, resume_kind, resume_output } => serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": gate_name,\n \"action_name\": action_name,\n \"call_id\": call_id,\n \"parameters\": parameters,\n \"resume_kind\": serde_json::to_value(resume_kind).unwrap_or_default(),\n \"resume_output\": resume_output,\n }),\n _ => serde_json::Value::Null,\n }\n }),\n });\n\n ExtFunctionResult::Return(json_to_monty(&result_json))\n }\n Err(e) => ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::RuntimeError,\n Some(format!(\"Code execution failed: {e}\")),\n )),\n }\n}\n\n/// Handle `__execute_action__(name, params, call_id=...)`.\n///\n/// Single source of truth for action execution. Performs:\n/// 1. Lease lookup\n/// 2. Policy check\n/// 3. Lease consumption\n/// 4. Action execution via EffectExecutor\n/// 5. Event emission (ActionExecuted/ActionFailed)\n///\n/// Python owns the working transcript and decides how tool outputs are\n/// represented in internal message history.\nasync fn handle_execute_action(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n effects: &Arc,\n leases: &Arc,\n policy: &Arc,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n) -> ExtFunctionResult {\n let name = match extract_string_arg(args, kwargs, \"name\", 0) {\n Some(n) => n,\n None => {\n return ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::TypeError,\n Some(\"__execute_action__ requires a name argument\".into()),\n ));\n }\n };\n\n let params = args\n .get(1)\n .map(monty_to_json)\n .unwrap_or(serde_json::json!({}));\n\n let call_id = extract_string_kwarg(kwargs, \"call_id\").unwrap_or_default();\n\n let exec_ctx = ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: thread.user_id.clone(),\n step_id: StepId::new(),\n current_call_id: Some(call_id.clone()),\n source_channel: thread_source_channel(thread),\n user_timezone: thread_user_timezone(thread),\n };\n\n // Helper: emit event only. The orchestrator owns transcript recording.\n let emit_and_record = |thread: &mut Thread,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n event_kind: EventKind,\n _call_id: &str,\n _action_name: &str,\n _output: &serde_json::Value| {\n let event = ThreadEvent::new(thread.id, event_kind);\n if let Some(tx) = event_tx {\n let _ = tx.send(event.clone());\n }\n thread.events.push(event);\n thread.updated_at = chrono::Utc::now();\n };\n\n // 1. Find lease for this action\n let lease = match leases.find_lease_for_action(thread.id, &name).await {\n Some(l) => l,\n None => {\n let error = format!(\"No lease for action '{name}'\");\n let output = serde_json::json!({\"error\": &error});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error,\n params_summary: None,\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"output\": output,\n \"is_error\": true,\n });\n return ExtFunctionResult::Return(json_to_monty(&result));\n }\n };\n\n // 2. Check policy\n let action_def = effects\n .available_actions(std::slice::from_ref(&lease))\n .await\n .ok()\n .and_then(|actions| actions.into_iter().find(|a| a.name == name));\n\n if let Some(ref ad) = action_def {\n match policy.evaluate(ad, &lease, &[]) {\n crate::capability::policy::PolicyDecision::Deny { reason } => {\n let output = serde_json::json!({\"error\": format!(\"Denied: {reason}\")});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error: reason,\n params_summary: None,\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"output\": output,\n \"is_error\": true,\n });\n return ExtFunctionResult::Return(json_to_monty(&result));\n }\n crate::capability::policy::PolicyDecision::RequireApproval { .. } => {\n let output = serde_json::json!({\"status\": \"gate_paused\", \"gate_name\": \"approval\"});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ApprovalRequested {\n action_name: name.clone(),\n call_id: call_id.clone(),\n parameters: Some(params.clone()),\n description: None,\n allow_always: None,\n gate_name: None,\n params_summary: summarize_params(&name, ¶ms),\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": \"approval\",\n \"action_name\": name,\n \"call_id\": call_id,\n \"parameters\": params,\n \"resume_kind\": serde_json::to_value(crate::gate::ResumeKind::Approval {\n allow_always: true,\n })\n .unwrap_or_default(),\n });\n return ExtFunctionResult::Return(json_to_monty(&result));\n }\n crate::capability::policy::PolicyDecision::Allow => {}\n }\n }\n\n // 3. Atomically re-find + consume a lease use under a single write\n // lock. This closes the TOCTOU window between the read-only\n // `find_lease_for_action` (used above for the policy check) and the\n // consume — without it, two concurrent calls could both observe a\n // lease with one remaining use and both proceed to execute. Mirrors\n // `structured.rs::execute_action_batch_with_results`.\n let lease = match leases.find_and_consume(thread.id, &name).await {\n Ok(l) => l,\n Err(e) => {\n debug!(error = %e, \"atomic lease find_and_consume failed\");\n let error = format!(\"lease consumption failed for action '{name}': {e}\");\n let output = serde_json::json!({\"error\": &error});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error,\n params_summary: None,\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"output\": output,\n \"is_error\": true,\n });\n return ExtFunctionResult::Return(json_to_monty(&result));\n }\n };\n\n // 4. Execute\n let ps = summarize_params(&name, ¶ms);\n match effects\n .execute_action(&name, params, &lease, &exec_ctx)\n .await\n {\n Ok(r) => {\n // Effect adapters wrap tool errors as `Ok(ActionResult { is_error: true })`\n // — surface them as `ActionFailed` so traces and observers see the\n // failure. See `resolve_tool_future` in `scripting.rs` for the same\n // pattern on the structured-tool path.\n if r.is_error {\n let error_msg = r\n .output\n .get(\"error\")\n .and_then(|v| v.as_str())\n .map(String::from)\n .unwrap_or_else(|| r.output.to_string());\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error: error_msg,\n params_summary: ps.clone(),\n },\n &call_id,\n &name,\n &r.output,\n );\n } else {\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionExecuted {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n duration_ms: r.duration.as_millis() as u64,\n params_summary: ps.clone(),\n },\n &call_id,\n &name,\n &r.output,\n );\n }\n let result = serde_json::json!({\n \"action_name\": r.action_name,\n \"output\": r.output,\n \"is_error\": r.is_error,\n \"duration_ms\": r.duration.as_millis(),\n });\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n Err(EngineError::GatePaused {\n gate_name,\n action_name: _,\n call_id: _,\n parameters,\n resume_kind,\n resume_output,\n }) => {\n let _ = leases.refund_use(lease.id).await;\n let output = serde_json::json!({\"status\": \"gate_paused\", \"gate_name\": gate_name});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ApprovalRequested {\n action_name: name.clone(),\n call_id: call_id.clone(),\n parameters: Some((*parameters).clone()),\n description: None,\n allow_always: match resume_kind.as_ref() {\n crate::gate::ResumeKind::Approval { allow_always } => Some(*allow_always),\n _ => None,\n },\n gate_name: Some(gate_name.clone()),\n params_summary: summarize_params(&name, ¶meters),\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": gate_name,\n \"action_name\": name,\n \"call_id\": call_id,\n \"parameters\": parameters,\n \"resume_kind\": serde_json::to_value(&*resume_kind).unwrap_or_default(),\n \"resume_output\": resume_output,\n });\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n Err(e) => {\n let output = serde_json::json!({\"error\": e.to_string()});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error: e.to_string(),\n params_summary: ps,\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"output\": output,\n \"is_error\": true,\n });\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n }\n}\n\n/// Handle `__execute_actions_parallel__(calls)`.\n///\n/// Batch host function that receives a list of action calls and executes them\n/// concurrently. Each call is a dict with `name`, `params`, and optionally `call_id`.\n///\n/// Returns a list of result dicts (one per call, in order). Each result has the\n/// same shape as `__execute_action__` output, plus an optional gate pause payload.\n///\n/// Events are emitted in original call order after all parallel executions complete.\nasync fn handle_execute_actions_parallel(\n args: &[MontyObject],\n thread: &mut Thread,\n effects: &Arc,\n leases: &Arc,\n policy: &Arc,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n) -> ExtFunctionResult {\n // Parse the calls list from the first argument (list of dicts)\n let calls_json = args\n .first()\n .map(monty_to_json)\n .unwrap_or(serde_json::json!([]));\n let calls_array = match calls_json.as_array() {\n Some(arr) => arr.clone(),\n None => {\n return ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::TypeError,\n Some(\"__execute_actions_parallel__ requires a list of call dicts\".into()),\n ));\n }\n };\n\n if calls_array.is_empty() {\n return ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])));\n }\n\n // Parse each call dict into (name, params, call_id)\n struct ParsedCall {\n name: String,\n params: serde_json::Value,\n call_id: String,\n }\n\n let mut parsed: Vec = Vec::with_capacity(calls_array.len());\n for c in &calls_array {\n let name = c\n .get(\"name\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"\")\n .to_string();\n let params = c.get(\"params\").cloned().unwrap_or(serde_json::json!({}));\n let call_id = c\n .get(\"call_id\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"\")\n .to_string();\n parsed.push(ParsedCall {\n name,\n params,\n call_id,\n });\n }\n\n let step_id = StepId::new();\n\n // ── Phase 1: Preflight (sequential) ─────────────────────────\n // Check leases and policies. Denied → error result. Approval → interrupt.\n\n enum PfOutcome {\n Runnable {\n lease: crate::types::capability::CapabilityLease,\n },\n Error {\n result_json: serde_json::Value,\n event: EventKind,\n output: serde_json::Value,\n },\n }\n\n let mut preflight: Vec> = Vec::with_capacity(parsed.len());\n\n for pc in &parsed {\n // Find lease\n let lease = match leases.find_lease_for_action(thread.id, &pc.name).await {\n Some(l) => l,\n None => {\n let error = format!(\"No lease for action '{}'\", pc.name);\n let output = serde_json::json!({\"error\": &error});\n let result_json = serde_json::json!({\n \"output\": &output,\n \"is_error\": true,\n });\n let event = EventKind::ActionFailed {\n step_id,\n action_name: pc.name.clone(),\n call_id: pc.call_id.clone(),\n error,\n params_summary: None,\n };\n preflight.push(Some(PfOutcome::Error {\n result_json,\n event,\n output,\n }));\n continue;\n }\n };\n\n // Check policy\n let action_def = effects\n .available_actions(std::slice::from_ref(&lease))\n .await\n .ok()\n .and_then(|actions| actions.into_iter().find(|a| a.name == pc.name));\n\n if let Some(ref ad) = action_def {\n match policy.evaluate(ad, &lease, &[]) {\n crate::capability::policy::PolicyDecision::Deny { reason } => {\n let output = serde_json::json!({\"error\": format!(\"Denied: {reason}\")});\n let result_json = serde_json::json!({\n \"output\": &output,\n \"is_error\": true,\n });\n let event = EventKind::ActionFailed {\n step_id,\n action_name: pc.name.clone(),\n call_id: pc.call_id.clone(),\n error: reason,\n params_summary: None,\n };\n preflight.push(Some(PfOutcome::Error {\n result_json,\n event,\n output,\n }));\n continue;\n }\n crate::capability::policy::PolicyDecision::RequireApproval { .. } => {\n // Emit events for earlier errors, then interrupt\n let mut results_json = Vec::with_capacity(preflight.len() + 1);\n for pf in preflight {\n match pf {\n Some(PfOutcome::Error {\n result_json,\n event,\n output: _,\n }) => {\n let ev = ThreadEvent::new(thread.id, event);\n if let Some(tx) = event_tx {\n let _ = tx.send(ev.clone());\n }\n thread.events.push(ev);\n results_json.push(result_json);\n }\n Some(PfOutcome::Runnable { .. }) | None => {\n results_json.push(serde_json::json!(null));\n }\n }\n }\n // Add the approval entry\n let ev = ThreadEvent::new(\n thread.id,\n EventKind::ApprovalRequested {\n action_name: pc.name.clone(),\n call_id: pc.call_id.clone(),\n parameters: Some(pc.params.clone()),\n description: None,\n allow_always: None,\n gate_name: None,\n params_summary: summarize_params(&pc.name, &pc.params),\n },\n );\n if let Some(tx) = event_tx {\n let _ = tx.send(ev.clone());\n }\n thread.events.push(ev);\n thread.updated_at = chrono::Utc::now();\n\n results_json.push(serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": \"approval\",\n \"action_name\": &pc.name,\n \"call_id\": &pc.call_id,\n \"parameters\": &pc.params,\n \"resume_kind\": serde_json::to_value(crate::gate::ResumeKind::Approval {\n allow_always: true,\n })\n .unwrap_or_default(),\n }));\n // Pad with nulls for calls that weren't reached so the\n // Python-side loop can emit ActionResult placeholders for\n // every tool call in the assistant message.\n while results_json.len() < parsed.len() {\n results_json.push(serde_json::json!(null));\n }\n return ExtFunctionResult::Return(json_to_monty(&serde_json::json!(\n results_json\n )));\n }\n crate::capability::policy::PolicyDecision::Allow => {}\n }\n }\n\n // Atomically re-find + consume a lease use under a single write\n // lock, closing the TOCTOU window between the read-only\n // `find_lease_for_action` above and the consume. Mirrors\n // `structured.rs::execute_action_batch_with_results`.\n let lease = match leases.find_and_consume(thread.id, &pc.name).await {\n Ok(l) => l,\n Err(e) => {\n debug!(error = %e, \"atomic lease find_and_consume failed\");\n let error = format!(\"lease consumption failed for action '{}': {e}\", pc.name);\n let output = serde_json::json!({\"error\": &error});\n let result_json = serde_json::json!({\n \"output\": &output,\n \"is_error\": true,\n });\n let event = EventKind::ActionFailed {\n step_id,\n action_name: pc.name.clone(),\n call_id: pc.call_id.clone(),\n error,\n params_summary: None,\n };\n preflight.push(Some(PfOutcome::Error {\n result_json,\n event,\n output,\n }));\n continue;\n }\n };\n\n preflight.push(Some(PfOutcome::Runnable { lease }));\n }\n\n // ── Phase 2: Execute in parallel ────────────────────────────\n\n // Slot array: index → execution result\n let mut slot_results: Vec> = vec![None; parsed.len()];\n let mut slot_events: Vec> = vec![None; parsed.len()];\n let mut slot_outputs: Vec> = vec![None; parsed.len()];\n\n // Separate runnable from errors\n let mut runnable: Vec<(usize, crate::types::capability::CapabilityLease)> = Vec::new();\n for (idx, pf) in preflight.into_iter().enumerate() {\n match pf {\n Some(PfOutcome::Error {\n result_json,\n event,\n output,\n }) => {\n slot_results[idx] = Some(result_json);\n slot_events[idx] = Some(event);\n slot_outputs[idx] = Some(output);\n }\n Some(PfOutcome::Runnable { lease }) => {\n runnable.push((idx, lease));\n }\n None => {}\n }\n }\n\n if runnable.len() == 1 {\n // Single call: execute directly\n let (idx, lease) = runnable.into_iter().next().unwrap(); // safety: len()==1 checked above\n let pc = &parsed[idx];\n let exec_ctx = ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: thread.user_id.clone(),\n step_id,\n current_call_id: Some(pc.call_id.clone()),\n // Read source_channel from thread metadata so downstream tools\n // (e.g. mission_create) can default notify_channels to the\n // originating channel. Hardcoding `None` here was a bug — it\n // silently dropped the gateway routing for any tool dispatched\n // through the parallel batch path.\n source_channel: thread_source_channel(thread),\n user_timezone: thread_user_timezone(thread),\n };\n let ps = summarize_params(&pc.name, &pc.params);\n let (result_json, event, output) = execute_single_action(\n effects,\n &pc.name,\n pc.params.clone(),\n &pc.call_id,\n &lease,\n &exec_ctx,\n ps,\n )\n .await;\n if interrupted_result_needs_refund(&result_json) {\n let _ = leases.refund_use(lease.id).await;\n }\n slot_results[idx] = Some(result_json);\n slot_events[idx] = Some(event);\n slot_outputs[idx] = Some(output);\n } else if runnable.len() > 1 {\n // Multiple calls: execute in parallel via JoinSet\n let mut join_set = tokio::task::JoinSet::new();\n let effects = effects.clone();\n // Capture once outside the loop — the thread's metadata is stable\n // for the duration of the parallel batch.\n let parallel_source_channel = thread_source_channel(thread);\n let parallel_user_timezone = thread_user_timezone(thread);\n\n for (idx, lease) in runnable {\n let pc_name = parsed[idx].name.clone();\n let pc_params = parsed[idx].params.clone();\n let pc_call_id = parsed[idx].call_id.clone();\n let effects = effects.clone();\n let lease = lease.clone();\n let exec_ctx = ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: thread.user_id.clone(),\n step_id,\n current_call_id: Some(pc_call_id.clone()),\n // See comment above — read from thread metadata, not None.\n source_channel: parallel_source_channel.clone(),\n user_timezone: parallel_user_timezone,\n };\n let ps = summarize_params(&pc_name, &pc_params);\n\n join_set.spawn(async move {\n let (result_json, event, output) = execute_single_action(\n &effects,\n &pc_name,\n pc_params,\n &pc_call_id,\n &lease,\n &exec_ctx,\n ps,\n )\n .await;\n (idx, lease.id, result_json, event, output)\n });\n }\n\n while let Some(join_result) = join_set.join_next().await {\n match join_result {\n Ok((idx, lease_id, result_json, event, output)) => {\n if interrupted_result_needs_refund(&result_json) {\n let _ = leases.refund_use(lease_id).await;\n }\n slot_results[idx] = Some(result_json);\n slot_events[idx] = Some(event);\n slot_outputs[idx] = Some(output);\n }\n Err(e) => {\n debug!(\"parallel action execution task panicked: {e}\");\n }\n }\n }\n }\n\n // ── Phase 3: Emit events in order ───────────────────────────\n\n let mut results_json = Vec::with_capacity(parsed.len());\n for idx in 0..parsed.len() {\n let result_json = slot_results[idx].take().unwrap_or(\n serde_json::json!({\"is_error\": true, \"output\": {\"error\": \"execution slot empty\"}}),\n );\n let _output = slot_outputs[idx]\n .take()\n .unwrap_or(serde_json::json!({\"error\": \"no output\"}));\n\n if let Some(event) = slot_events[idx].take() {\n let ev = ThreadEvent::new(thread.id, event);\n if let Some(tx) = event_tx {\n let _ = tx.send(ev.clone());\n }\n thread.events.push(ev);\n }\n\n results_json.push(result_json.clone());\n }\n\n thread.updated_at = chrono::Utc::now();\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!(results_json)))\n}\n\n/// Execute a single action and return (result_json, event, output) for the\n/// batch handler to record. Shared by both single-call and parallel paths.\nasync fn execute_single_action(\n effects: &Arc,\n name: &str,\n params: serde_json::Value,\n call_id: &str,\n lease: &crate::types::capability::CapabilityLease,\n exec_ctx: &ThreadExecutionContext,\n params_summary: Option,\n) -> (serde_json::Value, EventKind, serde_json::Value) {\n match effects.execute_action(name, params, lease, exec_ctx).await {\n Ok(r) => {\n // Surface wrapped errors as ActionFailed (see resolve_tool_future\n // and the parallel execute path for the same pattern).\n let event = if r.is_error {\n let error_msg = r\n .output\n .get(\"error\")\n .and_then(|v| v.as_str())\n .map(String::from)\n .unwrap_or_else(|| r.output.to_string());\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.to_string(),\n call_id: call_id.to_string(),\n error: error_msg,\n params_summary: params_summary.clone(),\n }\n } else {\n EventKind::ActionExecuted {\n step_id: exec_ctx.step_id,\n action_name: name.to_string(),\n call_id: call_id.to_string(),\n duration_ms: r.duration.as_millis() as u64,\n params_summary: params_summary.clone(),\n }\n };\n let result_json = serde_json::json!({\n \"action_name\": r.action_name,\n \"output\": r.output,\n \"is_error\": r.is_error,\n \"duration_ms\": r.duration.as_millis(),\n });\n (result_json, event, r.output)\n }\n Err(EngineError::GatePaused {\n gate_name,\n action_name: _,\n call_id: _,\n parameters,\n resume_kind,\n resume_output,\n }) => {\n let output = serde_json::json!({\"status\": \"gate_paused\", \"gate_name\": &gate_name});\n let event = EventKind::ApprovalRequested {\n action_name: name.to_string(),\n call_id: call_id.to_string(),\n parameters: Some((*parameters).clone()),\n description: None,\n allow_always: match resume_kind.as_ref() {\n crate::gate::ResumeKind::Approval { allow_always } => Some(*allow_always),\n _ => None,\n },\n gate_name: Some(gate_name.clone()),\n params_summary: summarize_params(name, ¶meters),\n };\n let result_json = serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": gate_name,\n \"action_name\": name,\n \"call_id\": call_id,\n \"parameters\": parameters,\n \"resume_kind\": serde_json::to_value(&*resume_kind).unwrap_or_default(),\n \"resume_output\": resume_output,\n });\n (result_json, event, output)\n }\n Err(e) => {\n let output = serde_json::json!({\"error\": e.to_string()});\n let event = EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.to_string(),\n call_id: call_id.to_string(),\n error: e.to_string(),\n params_summary,\n };\n let result_json = serde_json::json!({\n \"output\": &output,\n \"is_error\": true,\n });\n (result_json, event, output)\n }\n }\n}\n\nfn interrupted_result_needs_refund(result: &serde_json::Value) -> bool {\n result.get(\"gate_paused\").and_then(|v| v.as_bool()) == Some(true)\n}\n\n/// Handle `__check_signals__()`.\nfn handle_check_signals(signal_rx: &mut SignalReceiver, thread: &mut Thread) -> ExtFunctionResult {\n match signal_rx.try_recv() {\n Ok(ThreadSignal::Stop) | Ok(ThreadSignal::Suspend) => {\n ExtFunctionResult::Return(MontyObject::String(\"stop\".into()))\n }\n Ok(ThreadSignal::InjectMessage(msg)) => {\n thread.add_message(msg.clone());\n let result = serde_json::json!({\"inject\": msg.content});\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n Ok(ThreadSignal::Resume) | Ok(ThreadSignal::ChildCompleted { .. }) => {\n ExtFunctionResult::Return(MontyObject::None)\n }\n Err(_) => ExtFunctionResult::Return(MontyObject::None),\n }\n}\n\n/// Handle `__emit_event__(kind, **data)`.\nfn handle_emit_event(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n) -> ExtFunctionResult {\n let kind_str = args.first().map(monty_to_string).unwrap_or_default();\n\n let kind = match kind_str.as_str() {\n \"step_started\" => {\n let _step = extract_u64_kwarg(kwargs, \"step\").unwrap_or(0);\n EventKind::StepStarted {\n step_id: StepId::new(),\n }\n }\n \"step_completed\" => {\n let input = extract_u64_kwarg(kwargs, \"input_tokens\").unwrap_or(0);\n let output = extract_u64_kwarg(kwargs, \"output_tokens\").unwrap_or(0);\n // Increment step count (mirrors the old Rust loop's step_count += 1)\n thread.step_count += 1;\n // Track token usage\n thread.total_tokens_used += input + output;\n EventKind::StepCompleted {\n step_id: StepId::new(),\n tokens: TokenUsage {\n input_tokens: input,\n output_tokens: output,\n ..Default::default()\n },\n }\n }\n \"action_executed\" => {\n let action_name = extract_string_kwarg(kwargs, \"action_name\").unwrap_or_default();\n let call_id = extract_string_kwarg(kwargs, \"call_id\").unwrap_or_default();\n EventKind::ActionExecuted {\n step_id: StepId::new(),\n action_name,\n call_id,\n duration_ms: 0,\n params_summary: None,\n }\n }\n \"action_failed\" => {\n let action_name = extract_string_kwarg(kwargs, \"action_name\").unwrap_or_default();\n let call_id = extract_string_kwarg(kwargs, \"call_id\").unwrap_or_default();\n let error = extract_string_kwarg(kwargs, \"error\").unwrap_or_default();\n EventKind::ActionFailed {\n step_id: StepId::new(),\n action_name,\n call_id,\n error,\n params_summary: None,\n }\n }\n \"skill_activated\" => {\n let names_str = extract_string_kwarg(kwargs, \"skill_names\").unwrap_or_default();\n let skill_names: Vec = names_str\n .split(',')\n .map(|s| s.trim().to_string())\n .filter(|s| !s.is_empty())\n .collect();\n EventKind::SkillActivated { skill_names }\n }\n _ => {\n debug!(kind = %kind_str, \"orchestrator: unknown event kind, skipping\");\n return ExtFunctionResult::Return(MontyObject::None);\n }\n };\n\n let event = ThreadEvent::new(thread.id, kind);\n if let Some(tx) = event_tx {\n let _ = tx.send(event.clone());\n }\n thread.events.push(event);\n thread.updated_at = chrono::Utc::now();\n\n ExtFunctionResult::Return(MontyObject::None)\n}\n\n/// Handle `__save_checkpoint__(state, counters)`.\nfn handle_save_checkpoint(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n) -> ExtFunctionResult {\n let state = args\n .first()\n .map(monty_to_json)\n .unwrap_or(serde_json::json!({}));\n let counters = args\n .get(1)\n .map(monty_to_json)\n .unwrap_or(serde_json::json!({}));\n\n sync_runtime_state(thread, Some(&state));\n\n if let Some(metadata) = thread.metadata.as_object_mut() {\n metadata.insert(\n \"runtime_checkpoint\".into(),\n serde_json::json!({\n \"persisted_state\": state,\n \"nudge_count\": counters.get(\"nudge_count\").and_then(|v| v.as_u64()).unwrap_or(0),\n \"consecutive_errors\": counters.get(\"consecutive_errors\").and_then(|v| v.as_u64()).unwrap_or(0),\n \"consecutive_action_errors\": counters.get(\"consecutive_action_errors\").and_then(|v| v.as_u64()).unwrap_or(0),\n \"compaction_count\": counters.get(\"compaction_count\").and_then(|v| v.as_u64()).unwrap_or(0),\n }),\n );\n }\n thread.updated_at = chrono::Utc::now();\n\n ExtFunctionResult::Return(MontyObject::None)\n}\n\n/// Handle `__transition_to__(state, reason)`.\nfn handle_transition_to(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n) -> ExtFunctionResult {\n let state_str = args.first().map(monty_to_string).unwrap_or_default();\n let reason = args.get(1).map(monty_to_string);\n\n let target = match state_str.as_str() {\n \"running\" => crate::types::thread::ThreadState::Running,\n \"completed\" => crate::types::thread::ThreadState::Completed,\n \"failed\" => crate::types::thread::ThreadState::Failed,\n \"waiting\" => crate::types::thread::ThreadState::Waiting,\n \"suspended\" => crate::types::thread::ThreadState::Suspended,\n other => {\n return ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::ValueError,\n Some(format!(\"Unknown thread state: {other}\")),\n ));\n }\n };\n\n match thread.transition_to(target, reason) {\n Ok(()) => ExtFunctionResult::Return(MontyObject::None),\n Err(e) => ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::RuntimeError,\n Some(format!(\"State transition failed: {e}\")),\n )),\n }\n}\n\n/// Handle `__retrieve_docs__(goal, max_docs)`.\nasync fn handle_retrieve_docs(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &Thread,\n retrieval: Option<&RetrievalEngine>,\n) -> ExtFunctionResult {\n let retrieval = match retrieval {\n Some(r) => r,\n None => return ExtFunctionResult::Return(json_to_monty(&serde_json::json!([]))),\n };\n\n let goal = args.first().map(monty_to_string).unwrap_or_default();\n let max_docs = args\n .get(1)\n .and_then(|v| match v {\n MontyObject::Int(i) => Some(*i as usize),\n _ => None,\n })\n .unwrap_or(5);\n\n match retrieval\n .retrieve_context(thread.project_id, &thread.user_id, &goal, max_docs)\n .await\n {\n Ok(docs) => {\n let docs_json: Vec = docs\n .iter()\n .map(|d| {\n serde_json::json!({\n \"type\": format!(\"{:?}\", d.doc_type),\n \"title\": d.title,\n \"content\": d.content,\n })\n })\n .collect();\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!(docs_json)))\n }\n Err(e) => {\n debug!(\"retrieve_docs failed: {e}\");\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])))\n }\n }\n}\n\n/// Handle `__check_budget__()`.\nfn handle_check_budget(thread: &Thread) -> ExtFunctionResult {\n let tokens_remaining = thread\n .config\n .max_tokens_total\n .map(|max| max.saturating_sub(thread.total_tokens_used))\n .unwrap_or(u64::MAX);\n\n let time_remaining_ms = thread\n .config\n .max_duration\n .map(|dur| {\n let elapsed = chrono::Utc::now()\n .signed_duration_since(thread.created_at)\n .num_milliseconds()\n .max(0) as u64;\n dur.as_millis() as u64 - elapsed.min(dur.as_millis() as u64)\n })\n .unwrap_or(u64::MAX);\n\n let usd_remaining = thread\n .config\n .max_budget_usd\n .map(|max| (max - thread.total_cost_usd).max(0.0));\n\n let result = serde_json::json!({\n \"tokens_remaining\": tokens_remaining,\n \"time_remaining_ms\": time_remaining_ms,\n \"usd_remaining\": usd_remaining,\n });\n\n ExtFunctionResult::Return(json_to_monty(&result))\n}\n\n/// Handle `__get_actions__()`.\nasync fn handle_get_actions(\n thread: &mut Thread,\n effects: &Arc,\n leases: &Arc,\n store: Option<&Arc>,\n) -> ExtFunctionResult {\n if let Err(e) =\n reconcile_dynamic_tool_lease(thread, effects, leases, store, &crate::LeasePlanner::new())\n .await\n {\n warn_on_lease_refresh_failure(\"get_actions\", &e);\n }\n\n let active_leases = leases.active_for_thread(thread.id).await;\n match effects.available_actions(&active_leases).await {\n Ok(actions) => {\n let actions_json: Vec = actions\n .iter()\n .map(|a| {\n serde_json::json!({\n \"name\": a.name,\n \"description\": a.description,\n \"params\": a.parameters_schema,\n })\n })\n .collect();\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!(actions_json)))\n }\n Err(e) => {\n debug!(\"get_actions failed: {e}\");\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])))\n }\n }\n}\n\n/// Handle `__list_skills__()`.\n///\n/// Loads all `DocType::Skill` MemoryDocs from the project and returns them\n/// as a list of Python dicts. The Python orchestrator handles scoring,\n/// selection, and injection — Rust just provides data access.\nasync fn handle_list_skills(\n _args: &[MontyObject],\n thread: &Thread,\n store: Option<&Arc>,\n) -> ExtFunctionResult {\n let Some(store) = store else {\n return ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])));\n };\n\n // Use shared listing: user's own skills + system/admin-installed skills.\n let docs = match store\n .list_memory_docs_with_shared(thread.project_id, &thread.user_id)\n .await\n {\n Ok(docs) => docs,\n Err(e) => {\n debug!(\"__list_skills__: failed to load docs: {e}\");\n return ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])));\n }\n };\n\n let skills: Vec = docs\n .into_iter()\n .filter(|d| d.doc_type == crate::types::memory::DocType::Skill)\n .map(|d| {\n serde_json::json!({\n \"doc_id\": d.id.0.to_string(),\n \"title\": d.title,\n \"content\": d.content,\n \"metadata\": d.metadata,\n })\n })\n .collect();\n\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!(skills)))\n}\n\n/// Handle `__record_skill_usage__(doc_id, success)`.\n///\n/// Records that a skill was used in this thread. Called by the Python\n/// orchestrator after skill-assisted execution completes.\nasync fn handle_record_skill_usage(\n args: &[MontyObject],\n store: Option<&Arc>,\n) -> ExtFunctionResult {\n let Some(store) = store else {\n return ExtFunctionResult::Return(MontyObject::None);\n };\n\n let doc_id_str = args.first().map(monty_to_string).unwrap_or_default();\n let success = args\n .get(1)\n .map(|o| matches!(o, MontyObject::Bool(true)))\n .unwrap_or(false);\n\n let Ok(uuid) = uuid::Uuid::parse_str(&doc_id_str) else {\n debug!(\"__record_skill_usage__: invalid doc_id: {doc_id_str}\");\n return ExtFunctionResult::Return(MontyObject::None);\n };\n\n let tracker = crate::memory::SkillTracker::new(Arc::clone(store));\n if let Err(e) = tracker\n .record_usage(crate::types::memory::DocId(uuid), success)\n .await\n {\n debug!(\"__record_skill_usage__: failed: {e}\");\n }\n\n ExtFunctionResult::Return(MontyObject::None)\n}\n\n/// Handle `__regex_match__(pattern, text) -> bool`.\n///\n/// Compiles `pattern` with a bounded size limit and returns whether it\n/// matches anywhere in `text`. Invalid regex or a size-limit violation\n/// returns `False` silently. Used by the Python skill selector for regex\n/// pattern scoring (Monty has no `re` module).\n///\n/// **Security: ReDoS safety.** This handler accepts arbitrary patterns from\n/// the Python orchestrator (which itself receives them from skill manifests)\n/// and runs them on user-supplied text. Safety relies on the `regex` crate's\n/// linear-time matching guarantee (no backreferences, no lookaround) plus the\n/// 64 KiB compiled-size cap and DFA-size cap below. If the `regex` crate is\n/// ever swapped for `fancy-regex` (which supports backreferences and is NOT\n/// linear-time), this becomes a real ReDoS vector. This is enforced by\n/// convention and documentation only — see the top-of-crate comment in\n/// `crates/ironclaw_engine/src/lib.rs`. (A `#[cfg(feature = \"fancy-regex\")]\n/// compile_error!` tripwire was evaluated but conflicts with\n/// `cargo clippy --all-features` which is the standard CI command.)\nfn handle_regex_match(args: &[MontyObject]) -> ExtFunctionResult {\n let pattern = args.first().map(monty_to_string).unwrap_or_default();\n let text = args.get(1).map(monty_to_string).unwrap_or_default();\n if pattern.is_empty() {\n return ExtFunctionResult::Return(MontyObject::Bool(false));\n }\n // Cap compiled regex size to prevent ReDoS (matches the 64 KiB limit used\n // by `LoadedSkill::compile_patterns` in `ironclaw_skills`). Also cap the\n // lazy-DFA cache: the `regex` crate's DFA can grow beyond `size_limit`\n // during matching, so `dfa_size_limit` is a separate defensive cap on\n // memory allocation from a crafted pattern over untrusted skill manifests.\n const MAX_REGEX_SIZE: usize = 1 << 16;\n let matched = match regex::RegexBuilder::new(&pattern)\n .size_limit(MAX_REGEX_SIZE)\n .dfa_size_limit(MAX_REGEX_SIZE)\n .build()\n {\n Ok(re) => re.is_match(&text),\n Err(e) => {\n debug!(\"__regex_match__: invalid pattern '{pattern}': {e}\");\n false\n }\n };\n ExtFunctionResult::Return(MontyObject::Bool(matched))\n}\n\n/// Handle `__set_active_skills__(skills)`.\n///\n/// Persists the selected skill provenance onto the thread so post-run learning\n/// flows can reason about the exact skill versions and snippets that were active.\nfn handle_set_active_skills(args: &[MontyObject], thread: &mut Thread) -> ExtFunctionResult {\n let skills_json = args\n .first()\n .map(monty_to_json)\n .unwrap_or_else(|| serde_json::json!([]));\n\n let skills = match serde_json::from_value::>(skills_json) {\n Ok(skills) => skills,\n Err(e) => {\n debug!(\"__set_active_skills__: invalid payload: {e}\");\n return ExtFunctionResult::Return(MontyObject::None);\n }\n };\n\n if let Err(e) = thread.set_active_skills(&skills) {\n debug!(\"__set_active_skills__: failed to persist active skills: {e}\");\n }\n\n ExtFunctionResult::Return(MontyObject::None)\n}\n\n// ── Helpers ─────────────────────────────────────────────────\n\n/// Build the context variables injected into the orchestrator Python.\nfn build_orchestrator_inputs(\n thread: &Thread,\n persisted_state: &serde_json::Value,\n) -> (Vec, Vec) {\n let names = vec![\n \"context\".into(),\n \"goal\".into(),\n \"actions\".into(),\n \"state\".into(),\n \"config\".into(),\n ];\n\n // Build orchestrator bootstrap context. Prefer the internal execution\n // transcript when present, otherwise fall back to the user-visible transcript.\n let bootstrap_messages = if thread.internal_messages.is_empty() {\n &thread.messages\n } else {\n &thread.internal_messages\n };\n let context: Vec = bootstrap_messages\n .iter()\n .map(|m| {\n // Serialize action_calls through the Python interchange shape\n // (`{name, call_id, params}`) so the bootstrap context is\n // round-trip compatible with `python_json_to_action_calls`.\n // Using bare `m.action_calls` here produces the canonical Rust\n // serde format (`{action_name, id, parameters}`), which the\n // Python orchestrator passes back verbatim on the next\n // `__llm_complete__` call — and `python_json_to_action_calls`\n // then fails with \"missing field `name`\", orphaning every\n // subsequent tool result. This is the SECOND code path (after\n // `handle_llm_complete`) that feeds action_calls into the\n // Python working transcript; both must use the same shape.\n let calls_json = m\n .action_calls\n .as_ref()\n .map(|calls| serde_json::Value::Array(action_calls_to_python_json(calls)));\n serde_json::json!({\n \"role\": format!(\"{:?}\", m.role),\n \"content\": m.content,\n \"action_name\": m.action_name,\n \"action_call_id\": m.action_call_id,\n \"action_calls\": calls_json,\n })\n })\n .collect();\n\n // Build config\n let config = serde_json::json!({\n \"max_iterations\": thread.config.max_iterations,\n \"max_tool_intent_nudges\": thread.config.max_tool_intent_nudges,\n \"enable_tool_intent_nudge\": thread.config.enable_tool_intent_nudge,\n \"max_consecutive_errors\": thread.config.max_consecutive_errors,\n \"max_tokens_total\": thread.config.max_tokens_total,\n \"max_budget_usd\": thread.config.max_budget_usd,\n \"model_context_limit\": thread.config.model_context_limit,\n \"enable_compaction\": thread.config.enable_compaction,\n \"compaction_threshold\": thread.config.compaction_threshold,\n \"depth\": thread.config.depth,\n \"max_depth\": thread.config.max_depth,\n \"step_count\": thread.step_count,\n });\n\n let values = vec![\n json_to_monty(&serde_json::json!(context)),\n MontyObject::String(thread.goal.clone()),\n json_to_monty(&serde_json::json!([])), // actions loaded dynamically via __get_actions__\n json_to_monty(persisted_state),\n json_to_monty(&config),\n ];\n\n (names, values)\n}\n\n/// JSON shape used to interchange `ActionCall`s with the Python orchestrator.\n///\n/// This is the *single* place that defines the field naming convention used\n/// across the Python boundary. It is intentionally separate from the\n/// canonical `ActionCall` type because:\n///\n/// - `ActionCall` uses Rust-idiomatic field names (`id`, `action_name`,\n/// `parameters`) and is also persisted into Step records and ThreadEvents.\n/// Renaming its serde fields would invalidate every existing row.\n/// - The Python orchestrator uses friendlier names (`call_id`, `name`,\n/// `params`) that read naturally in CodeAct prompts and `default.py`.\n///\n/// Without this type, the round-trip is asymmetric: Rust → Python uses one\n/// shape, Python → Rust used `serde_json::from_value::>`\n/// which silently fails (`.ok()` swallows the error) and produces `None`,\n/// which means assistant messages came back without `action_calls`. The\n/// downstream effect is that every tool result looks orphaned to\n/// `sanitize_tool_messages` and gets rewritten as a user message — losing\n/// the assistant ↔ tool_result linkage the LLM needs to reason about prior\n/// tool calls.\n#[derive(Debug, serde::Serialize, serde::Deserialize)]\nstruct PythonActionCall {\n name: String,\n call_id: String,\n params: serde_json::Value,\n}\n\nimpl From<&ActionCall> for PythonActionCall {\n fn from(c: &ActionCall) -> Self {\n Self {\n name: c.action_name.clone(),\n call_id: c.id.clone(),\n params: c.parameters.clone(),\n }\n }\n}\n\nimpl From for ActionCall {\n fn from(p: PythonActionCall) -> Self {\n Self {\n id: p.call_id,\n action_name: p.name,\n parameters: p.params,\n }\n }\n}\n\n/// Serialize a slice of `ActionCall`s into the Python interchange shape.\n///\n/// On serialization failure (essentially unreachable for `String + String +\n/// Value`, but still possible if the `serde_json::Value` parameters tree\n/// contains a key whose stringification fails), the entry is **dropped**\n/// from the output rather than replaced with `Value::Null`. The previous\n/// `unwrap_or_else(|_| Value::Null)` corrupted the array — Python's\n/// `default.py` accesses `c.get(\"name\")` / `c.get(\"call_id\")` /\n/// `c.get(\"params\")` on each entry, so a `null` would crash with a Python\n/// `AttributeError` and lose the entire LLM step. `filter_map` produces a\n/// shorter array, which Python's tool-result loop handles correctly because\n/// it iterates `range(len(results))` against the shortened call list. The\n/// warn log is preserved so operators have a breadcrumb if it ever fires.\nfn action_calls_to_python_json(calls: &[ActionCall]) -> Vec {\n calls\n .iter()\n .filter_map(|c| match serde_json::to_value(PythonActionCall::from(c)) {\n Ok(value) => Some(value),\n Err(e) => {\n warn!(\n error = %e,\n action_name = %c.action_name,\n \"Failed to serialize ActionCall for Python orchestrator — dropping entry\"\n );\n None\n }\n })\n .collect()\n}\n\n/// Extract the last `n` characters from `s`.\n///\n/// Error tracebacks appear at the end of stdout, after any `print()` output.\n/// Using the head would capture the print statements instead of the error.\nfn tail_chars(s: &str, n: usize) -> String {\n let char_count = s.chars().count();\n if char_count > n {\n s.chars().skip(char_count - n).collect()\n } else {\n s.to_owned()\n }\n}\n\n/// Build a PII-safe summary of an `action_calls` JSON value for log output.\n///\n/// The action_calls payload contains tool parameters, which can carry user\n/// PII (search queries, file names, email content, conversation text).\n/// Dumping the full value into a `warn!` log would leak that PII to log\n/// aggregation systems (Datadog, CloudWatch, Sentry) the moment the parser\n/// fails — and the parser only fails when the Python ↔ Rust shape drifts,\n/// which is exactly when an operator is most likely to be grepping logs.\n///\n/// We emit only the structural information operators actually need to\n/// debug a shape drift: array length and the keys of the first entry. The\n/// keys themselves are not user data — they're field names like\n/// `name`/`call_id`/`params` that are static across all calls.\nfn summarize_action_calls_for_log(value: &serde_json::Value) -> String {\n match value.as_array() {\n Some(arr) if arr.is_empty() => \"empty array\".to_string(),\n Some(arr) => {\n let first_keys = arr\n .first()\n .and_then(|v| v.as_object())\n .map(|obj| {\n let mut keys: Vec<&str> = obj.keys().map(String::as_str).collect();\n keys.sort_unstable();\n keys.join(\",\")\n })\n .unwrap_or_else(|| \"\".to_string());\n format!(\n \"array of {} entries; first entry keys: [{}]\",\n arr.len(),\n first_keys\n )\n }\n None => format!(\"non-array value of type {}\", json_value_type_name(value)),\n }\n}\n\n/// Cheap type-name string for a `serde_json::Value`. Used by\n/// `summarize_action_calls_for_log` to surface the wrong-shape case\n/// (e.g. Python passed a string instead of an array) without leaking the\n/// actual contents.\nfn json_value_type_name(value: &serde_json::Value) -> &'static str {\n match value {\n serde_json::Value::Null => \"null\",\n serde_json::Value::Bool(_) => \"bool\",\n serde_json::Value::Number(_) => \"number\",\n serde_json::Value::String(_) => \"string\",\n serde_json::Value::Array(_) => \"array\",\n serde_json::Value::Object(_) => \"object\",\n }\n}\n\n/// Deserialize an `action_calls` JSON array (in Python interchange shape)\n/// back into canonical `ActionCall`s.\n///\n/// Logs a warning on failure rather than swallowing silently. The whole\n/// commit that introduced this helper exists to undo a `.ok()` swallow that\n/// dropped action_calls without any signal — replacing it with another\n/// `.ok()?` would re-introduce the same trap, just one layer deeper. If the\n/// shape ever drifts again (Python orchestrator field rename, extra\n/// required field, partial migration), the warning is the operator-visible\n/// breadcrumb that explains why subsequent tool results suddenly look\n/// orphaned to `sanitize_tool_messages`.\n///\n/// The warn log emits a structural summary (`summarize_action_calls_for_log`)\n/// instead of the raw value because tool parameters can contain user PII.\nfn python_json_to_action_calls(value: &serde_json::Value) -> Option> {\n match serde_json::from_value::>(value.clone()) {\n Ok(parsed) => Some(parsed.into_iter().map(ActionCall::from).collect()),\n Err(e) => {\n warn!(\n error = %e,\n shape = %summarize_action_calls_for_log(value),\n \"Failed to parse action_calls from Python orchestrator — \\\n assistant message will lose tool_call linkage and downstream \\\n tool results will be rewritten as user messages\"\n );\n None\n }\n }\n}\n\nfn json_to_thread_messages(value: &serde_json::Value) -> Option> {\n let arr = value.as_array()?;\n let mut messages = Vec::with_capacity(arr.len());\n\n for item in arr {\n let role = item.get(\"role\").and_then(|v| v.as_str()).unwrap_or(\"User\");\n let content = item\n .get(\"content\")\n .and_then(|v| v.as_str())\n .unwrap_or_default();\n // Filter out null before calling the parser — `action_calls: null`\n // is Python's legitimate \"this message has no tool calls\" signal (text\n // response), not a parse error. Without this filter, the warn log in\n // python_json_to_action_calls fires on every text-only assistant\n // message with \"invalid type: null, expected a sequence\".\n let action_calls = item\n .get(\"action_calls\")\n .filter(|v| !v.is_null())\n .and_then(python_json_to_action_calls);\n\n let message = match role {\n \"System\" | \"system\" => ThreadMessage::system(content),\n \"Assistant\" | \"assistant\" => {\n if let Some(calls) = action_calls {\n ThreadMessage::assistant_with_actions(Some(content.to_string()), calls)\n } else {\n ThreadMessage::assistant(content)\n }\n }\n \"ActionResult\" | \"action_result\" => ThreadMessage::action_result(\n item.get(\"action_call_id\")\n .and_then(|v| v.as_str())\n .unwrap_or_default(),\n item.get(\"action_name\")\n .and_then(|v| v.as_str())\n .unwrap_or_default(),\n content,\n ),\n _ => ThreadMessage::user(content),\n };\n messages.push(message);\n }\n\n Some(messages)\n}\n\nfn sync_runtime_state(thread: &mut Thread, state: Option<&serde_json::Value>) {\n let Some(state) = state else {\n return;\n };\n if let Some(messages) = state\n .get(\"working_messages\")\n .and_then(json_to_thread_messages)\n {\n thread.internal_messages = messages;\n thread.updated_at = chrono::Utc::now();\n }\n}\n\nfn sync_visible_outcome(thread: &mut Thread, outcome: &ThreadOutcome) {\n if let ThreadOutcome::Completed {\n response: Some(response),\n } = outcome\n {\n let already_present = thread\n .messages\n .last()\n .map(|msg| {\n msg.role == crate::types::message::MessageRole::Assistant\n && msg.content == *response\n })\n .unwrap_or(false);\n if !already_present {\n thread.add_message(ThreadMessage::assistant(response));\n }\n }\n}\n\n/// Parse the orchestrator's return value into a ThreadOutcome.\nfn parse_outcome(result: &serde_json::Value) -> ThreadOutcome {\n let outcome = result\n .get(\"outcome\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"completed\");\n\n match outcome {\n \"completed\" => ThreadOutcome::Completed {\n response: result\n .get(\"response\")\n .and_then(|v| v.as_str())\n .map(String::from),\n },\n \"stopped\" => ThreadOutcome::Stopped,\n \"max_iterations\" => ThreadOutcome::MaxIterations,\n \"failed\" => ThreadOutcome::Failed {\n error: result\n .get(\"error\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"unknown error\")\n .to_string(),\n },\n \"gate_paused\" => {\n let resume_kind_value = result\n .get(\"resume_kind\")\n .cloned()\n .unwrap_or(serde_json::json!({}));\n let resume_kind = serde_json::from_value(resume_kind_value).unwrap_or(\n crate::gate::ResumeKind::Approval {\n allow_always: false,\n },\n );\n ThreadOutcome::GatePaused {\n gate_name: result\n .get(\"gate_name\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"unknown\")\n .to_string(),\n action_name: result\n .get(\"action_name\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"\")\n .to_string(),\n call_id: result\n .get(\"call_id\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"\")\n .to_string(),\n parameters: result\n .get(\"parameters\")\n .cloned()\n .unwrap_or(serde_json::json!({})),\n resume_kind,\n resume_output: result.get(\"resume_output\").cloned(),\n }\n }\n _ => ThreadOutcome::Completed { response: None },\n }\n}\n\nfn extract_string_arg(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n name: &str,\n position: usize,\n) -> Option {\n for (k, v) in kwargs {\n if let MontyObject::String(key) = k\n && key == name\n {\n return Some(monty_to_string(v));\n }\n }\n args.get(position).map(monty_to_string)\n}\n\nfn extract_string_kwarg(kwargs: &[(MontyObject, MontyObject)], name: &str) -> Option {\n for (k, v) in kwargs {\n if let MontyObject::String(key) = k\n && key == name\n {\n return Some(monty_to_string(v));\n }\n }\n None\n}\n\nfn extract_u64_kwarg(kwargs: &[(MontyObject, MontyObject)], name: &str) -> Option {\n for (k, v) in kwargs {\n if let MontyObject::String(key) = k\n && key == name\n && let MontyObject::Int(i) = v\n {\n return Some(*i as u64);\n }\n }\n None\n}\n\n#[cfg(test)]\nmod tests {\n use super::*;\n use crate::types::memory::{DocType, MemoryDoc};\n use crate::types::project::ProjectId;\n\n // ── Python helper unit tests via Monty ──────────────────────\n //\n // Extracts the helper functions from the default orchestrator and\n // evaluates `signals_tool_intent(text)` directly, mirroring the V1\n // Rust unit test suite in src/llm/reasoning.rs.\n\n /// Run a Python expression that returns a bool by prepending the\n /// orchestrator helper definitions and wrapping in `FINAL(expr)`.\n /// Run a Python snippet and drive the Monty VM, returning the FINAL()\n /// value as a `MontyObject`. This is the common core for `eval_python_bool`\n /// and `eval_python_int`.\n fn run_python_final(code: String) -> MontyObject {\n let runner =\n MontyRun::new(code, \"test.py\", vec![]).expect(\"Failed to parse orchestrator helpers\");\n let mut stdout = String::new();\n let tracker = LimitedTracker::new(ResourceLimits::new().max_allocations(500_000));\n\n let mut progress = runner\n .start(vec![], tracker, PrintWriter::Collect(&mut stdout))\n .expect(\"Failed to start orchestrator test\");\n\n loop {\n match progress {\n RunProgress::Complete(obj) => return obj,\n RunProgress::FunctionCall(call) => {\n if call.function_name == \"FINAL\" {\n let val = call.args.first().cloned().unwrap_or(MontyObject::None);\n let _ = call.resume(\n ExtFunctionResult::Return(MontyObject::None),\n PrintWriter::Collect(&mut stdout),\n );\n return val;\n }\n let ext_result = match call.function_name.as_str() {\n \"__regex_match__\" => handle_regex_match(&call.args),\n _ => ExtFunctionResult::Return(MontyObject::None),\n };\n progress = call\n .resume(ext_result, PrintWriter::Collect(&mut stdout))\n .expect(\"resume failed\");\n }\n RunProgress::NameLookup(lookup) => {\n progress = lookup\n .resume(\n NameLookupResult::Undefined,\n PrintWriter::Collect(&mut stdout),\n )\n .expect(\"name lookup resume failed\");\n }\n _ => panic!(\"Unexpected RunProgress variant in test\"),\n }\n }\n }\n\n fn eval_python_bool(expr: &str) -> bool {\n // Extract only the helper functions (everything before run_loop)\n let helpers_end = DEFAULT_ORCHESTRATOR\n .find(\"\\ndef run_loop(\")\n .unwrap_or(DEFAULT_ORCHESTRATOR.len());\n let helpers = &DEFAULT_ORCHESTRATOR[..helpers_end]; // safety: find() returns a char boundary on this ASCII-only constant\n\n let code = format!(\"{helpers}\\nFINAL({expr})\");\n match run_python_final(code) {\n MontyObject::Bool(v) => v,\n other => panic!(\"Expected bool, got: {other:?}\"),\n }\n }\n\n /// Run a Python program (with orchestrator helpers in scope) that ends\n /// with `FINAL(int_expr)` and return the integer value.\n fn eval_python_int(program: &str) -> i64 {\n let helpers_end = DEFAULT_ORCHESTRATOR\n .find(\"\\ndef run_loop(\")\n .unwrap_or(DEFAULT_ORCHESTRATOR.len());\n let helpers = &DEFAULT_ORCHESTRATOR[..helpers_end];\n\n let code = format!(\"{helpers}\\n{program}\");\n match run_python_final(code) {\n MontyObject::Int(v) => v,\n other => panic!(\"Expected int, got: {other:?}\"),\n }\n }\n\n // ── __regex_match__ host function reachability ───────────────\n\n #[test]\n fn regex_match_host_function_is_callable_from_monty() {\n // Regression test for PR #1736 review (serrrfirat, 3059161877):\n // verify that Monty's NameLookup + FunctionCall dispatch actually\n // reaches `handle_regex_match` when default.py calls\n // `__regex_match__(...)`. If Monty ever starts resolving the name\n // before the call, this test will fail with a NameError.\n assert!(eval_python_bool(\n r#\"bool(__regex_match__(\"abc\", \"xxabcxx\"))\"#\n ));\n assert!(!eval_python_bool(\n r#\"bool(__regex_match__(\"zzz\", \"xxabcxx\"))\"#\n ));\n // Invalid pattern should return false silently (the host function\n // swallows the compile error).\n assert!(!eval_python_bool(r#\"bool(__regex_match__(\"[\", \"abc\"))\"#));\n }\n\n // ── True positives (should trigger nudge) ───────────────────\n\n #[test]\n fn signals_tool_intent_true_positives() {\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"Let me search for that file.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'll fetch the data now.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'm going to check the logs.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"Let me add it now.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I will run the tests to verify.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'll look up the documentation.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"Let me read the file contents.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'm going to execute the command.\")\"#\n ));\n }\n\n // ── True negatives: conversational phrases ──────────────────\n\n #[test]\n fn signals_tool_intent_true_negatives_conversational() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me explain how this works.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me know if you need anything.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me think about this.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me summarize the findings.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me clarify what I mean.\")\"#\n ));\n }\n\n // ── Exclusion takes precedence ──────────────────────────────\n\n #[test]\n fn signals_tool_intent_exclusion_takes_precedence() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me explain the approach, then I'll search for the file.\")\"#\n ));\n }\n\n // ── Code blocks are stripped ────────────────────────────────\n\n #[test]\n fn signals_tool_intent_ignores_code_blocks() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Here's the code:\\n\\n```\\nfn main() {\\n println!(\\\"Let me search the database\\\");\\n}\\n```\")\"#\n ));\n }\n\n #[test]\n fn signals_tool_intent_ignores_indented_code() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Here's the code:\\n\\n println!(\\\"I'll fetch the data\\\");\\n\\nThat's it.\")\"#\n ));\n }\n\n // ── Plain informational text ────────────────────────────────\n\n #[test]\n fn signals_tool_intent_ignores_plain_text() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"The task is complete.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Here are the results you asked for.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"I found 3 matching files.\")\"#\n ));\n }\n\n // ── Quoted strings are stripped ─────────────────────────────\n\n #[test]\n fn signals_tool_intent_ignores_quoted_strings() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"The button says \\\"Let me search the database\\\" to the user.\")\"#\n ));\n // But unquoted intent should still trigger\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'll fetch the results for you.\")\"#\n ));\n }\n\n // ── Shadowed prefix (exclusion cancels all) ─────────────────\n\n #[test]\n fn signals_tool_intent_shadowed_prefix() {\n // \"let me think\" is an exclusion → entire text returns false\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Sure, let me think about it. Actually, let me search for the file.\")\"#\n ));\n }\n\n // ── Regression: trace false positive (news content) ─────────\n\n #[test]\n fn signals_tool_intent_no_false_positive_news_content() {\n // \"I can\" + \"call\" in news content triggered false positive in old code\n let news_response = concat!(\n \"The latest headlines suggest this is a fast-moving war.\\n\",\n \"- Reuters: Iran is calling US peace proposals unrealistic.\\n\",\n \"If you want, I can do one of these next:\\n\",\n \"1. give you a 5-bullet update\\n\",\n \"2. focus just on military developments\",\n );\n assert!(!eval_python_bool(&format!(\n \"signals_tool_intent({news_response:?})\"\n )));\n }\n\n #[test]\n fn signals_tool_intent_no_false_positive_past_tense() {\n // \"I fetched\" / \"I already called\" should not trigger\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"I already completed the needed action call by fetching current news feeds.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Current status from the live feeds I fetched:\")\"#\n ));\n }\n\n #[test]\n fn signals_tool_intent_no_false_positive_offer() {\n // \"If you want, I can fetch...\" uses \"I can\" which is not a V1 prefix\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"If you want, I can next fetch a cleaner update.\")\"#\n ));\n }\n\n #[tokio::test]\n async fn load_orchestrator_without_store_returns_default() {\n let (code, version) = load_orchestrator(None, ProjectId::new(), true).await;\n assert_eq!(version, 0);\n assert!(code.contains(\"run_loop\"));\n assert!(code.contains(\"__llm_complete__\"));\n }\n\n #[tokio::test]\n async fn load_orchestrator_with_runtime_version() {\n let project_id = ProjectId::new();\n let mut doc = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"custom_orchestrator_code()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc.metadata = serde_json::json!({\"version\": 1});\n\n let store = Arc::new(crate::tests::InMemoryStore::with_docs(vec![doc]));\n let (code, version) =\n load_orchestrator(Some(&(store as Arc)), project_id, true).await;\n assert_eq!(version, 1);\n assert!(code.contains(\"custom_orchestrator_code\"));\n }\n\n #[tokio::test]\n async fn load_orchestrator_picks_highest_version() {\n let project_id = ProjectId::new();\n let mut doc_v1 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v1_code()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v1.metadata = serde_json::json!({\"version\": 1});\n\n let mut doc_v3 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v3_code()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v3.metadata = serde_json::json!({\"version\": 3});\n\n let mut doc_v2 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v2_code()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v2.metadata = serde_json::json!({\"version\": 2});\n\n let store = Arc::new(crate::tests::InMemoryStore::with_docs(vec![\n doc_v1, doc_v3, doc_v2,\n ]));\n let (code, version) =\n load_orchestrator(Some(&(store as Arc)), project_id, true).await;\n assert_eq!(version, 3);\n assert!(code.contains(\"v3_code\"));\n }\n\n #[tokio::test]\n async fn rollback_after_max_failures() {\n let project_id = ProjectId::new();\n\n // Create v2 orchestrator\n let mut doc_v2 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v2_buggy()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v2.metadata = serde_json::json!({\"version\": 2});\n\n // Create v1 orchestrator (fallback)\n let mut doc_v1 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v1_stable()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v1.metadata = serde_json::json!({\"version\": 1});\n\n // Create failure tracker showing v2 has 3 failures\n let tracker = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n FAILURE_TRACKER_TITLE,\n r#\"{\"version\": 2, \"count\": 3}\"#,\n )\n .with_tags(vec![\"orchestrator_meta\".to_string()]);\n\n let store = Arc::new(crate::tests::InMemoryStore::with_docs(vec![\n doc_v2, doc_v1, tracker,\n ]));\n let (code, version) =\n load_orchestrator(Some(&(store as Arc)), project_id, true).await;\n\n // Should skip v2 (too many failures) and load v1\n assert_eq!(version, 1);\n assert!(code.contains(\"v1_stable\"));\n }\n\n #[tokio::test]\n async fn rollback_to_default_when_all_versions_fail() {\n let project_id = ProjectId::new();\n\n // Single version with 3 failures\n let mut doc_v1 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v1_broken()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v1.metadata = serde_json::json!({\"version\": 1});\n\n let tracker = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n FAILURE_TRACKER_TITLE,\n r#\"{\"version\": 1, \"count\": 5}\"#,\n )\n .with_tags(vec![\"orchestrator_meta\".to_string()]);\n\n let store = Arc::new(crate::tests::InMemoryStore::with_docs(vec![\n doc_v1, tracker,\n ]));\n let (code, version) =\n load_orchestrator(Some(&(store as Arc)), project_id, true).await;\n\n // Should fall back to compiled-in default (v0)\n assert_eq!(version, 0);\n assert!(code.contains(\"run_loop\"));\n }\n\n #[tokio::test]\n async fn record_and_reset_failures() {\n let project_id = ProjectId::new();\n let store: Arc = Arc::new(crate::tests::InMemoryStore::with_docs(vec![]));\n\n // Record 3 failures\n record_orchestrator_failure(&store, project_id, 2).await;\n record_orchestrator_failure(&store, project_id, 2).await;\n record_orchestrator_failure(&store, project_id, 2).await;\n\n let docs = store.list_shared_memory_docs(project_id).await.unwrap();\n let count = load_failure_count(&docs);\n assert_eq!(count, 3);\n\n // Reset\n reset_orchestrator_failures(&store, project_id).await;\n let docs = store.list_shared_memory_docs(project_id).await.unwrap();\n let count = load_failure_count(&docs);\n assert_eq!(count, 0);\n }\n\n #[tokio::test]\n async fn failure_count_resets_on_new_version() {\n let project_id = ProjectId::new();\n let store: Arc = Arc::new(crate::tests::InMemoryStore::with_docs(vec![]));\n\n // Record failures for version 1\n record_orchestrator_failure(&store, project_id, 1).await;\n record_orchestrator_failure(&store, project_id, 1).await;\n\n // Switch to version 2 — count should reset to 1\n record_orchestrator_failure(&store, project_id, 2).await;\n\n let docs = store.list_shared_memory_docs(project_id).await.unwrap();\n let count = load_failure_count(&docs);\n assert_eq!(count, 1);\n }\n\n #[test]\n fn normalize_pause_outcome_transitions_thread_to_waiting() {\n let mut thread = Thread::new(\n \"goal\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n let outcome = ThreadOutcome::GatePaused {\n gate_name: \"approval\".into(),\n action_name: \"shell\".into(),\n call_id: \"call-1\".into(),\n parameters: serde_json::json!({\"cmd\":\"ls\"}),\n resume_kind: crate::gate::ResumeKind::Approval { allow_always: true },\n resume_output: None,\n };\n normalize_pause_outcome(&mut thread, &outcome).unwrap();\n assert_eq!(thread.state, ThreadState::Waiting);\n }\n\n #[test]\n fn parse_outcome_completed() {\n let result = serde_json::json!({\"outcome\": \"completed\", \"response\": \"Hello!\"});\n let outcome = parse_outcome(&result);\n assert!(matches!(outcome, ThreadOutcome::Completed { response: Some(r) } if r == \"Hello!\"));\n }\n\n #[test]\n fn parse_outcome_failed() {\n let result = serde_json::json!({\"outcome\": \"failed\", \"error\": \"boom\"});\n let outcome = parse_outcome(&result);\n assert!(matches!(outcome, ThreadOutcome::Failed { error } if error == \"boom\"));\n }\n\n #[test]\n fn parse_outcome_gate_paused() {\n let result = serde_json::json!({\n \"outcome\": \"gate_paused\",\n \"gate_name\": \"approval\",\n \"action_name\": \"shell\",\n \"call_id\": \"abc\",\n \"parameters\": {\"cmd\": \"rm -rf /\"},\n \"resume_kind\": {\"Approval\": {\"allow_always\": true}}\n });\n let outcome = parse_outcome(&result);\n assert!(\n matches!(outcome, ThreadOutcome::GatePaused { action_name, .. } if action_name == \"shell\")\n );\n }\n\n #[test]\n fn parse_outcome_max_iterations() {\n let result = serde_json::json!({\"outcome\": \"max_iterations\"});\n let outcome = parse_outcome(&result);\n assert!(matches!(outcome, ThreadOutcome::MaxIterations));\n }\n\n #[test]\n fn parse_outcome_stopped() {\n let result = serde_json::json!({\"outcome\": \"stopped\"});\n let outcome = parse_outcome(&result);\n assert!(matches!(outcome, ThreadOutcome::Stopped));\n }\n\n // ── handle_llm_complete model forwarding ────────────────────\n\n /// LLM backend that records the model from each `complete()` call.\n /// Used to verify the orchestrator's __llm_complete__ host fn forwards\n /// `explicit_config[\"model\"]` onto `LlmCallConfig.model`.\n struct ModelCapturingLlm {\n captured: tokio::sync::Mutex>>,\n }\n\n #[async_trait::async_trait]\n impl LlmBackend for ModelCapturingLlm {\n fn model_name(&self) -> &str {\n \"capturing\"\n }\n\n async fn complete(\n &self,\n _messages: &[ThreadMessage],\n _actions: &[crate::types::capability::ActionDef],\n config: &LlmCallConfig,\n ) -> Result {\n self.captured.lock().await.push(config.model.clone());\n Ok(crate::traits::llm::LlmOutput {\n response: crate::types::step::LlmResponse::Text(\"ok\".into()),\n usage: crate::types::step::TokenUsage::default(),\n })\n }\n }\n\n /// No-op effect executor — handle_llm_complete only consults it for\n /// `available_actions(...)`, which we satisfy with an empty list.\n struct NoopEffects;\n\n #[async_trait::async_trait]\n impl EffectExecutor for NoopEffects {\n async fn execute_action(\n &self,\n _: &str,\n _: serde_json::Value,\n _: &crate::types::capability::CapabilityLease,\n _: &ThreadExecutionContext,\n ) -> Result {\n Ok(crate::types::step::ActionResult {\n call_id: String::new(),\n action_name: String::new(),\n output: serde_json::json!({}),\n is_error: false,\n duration: std::time::Duration::from_millis(1),\n })\n }\n\n async fn available_actions(\n &self,\n _: &[crate::types::capability::CapabilityLease],\n ) -> Result, EngineError> {\n Ok(vec![])\n }\n }\n\n #[tokio::test]\n async fn llm_complete_forwards_model_from_explicit_config() {\n let concrete = Arc::new(ModelCapturingLlm {\n captured: tokio::sync::Mutex::new(Vec::new()),\n });\n let llm: Arc = Arc::clone(&concrete) as Arc;\n let effects: Arc = Arc::new(NoopEffects);\n let leases = Arc::new(LeaseManager::new());\n let store: Arc = Arc::new(crate::tests::InMemoryStore::with_docs(vec![]));\n\n let mut thread = Thread::new(\n \"goal\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n // Build the args __llm_complete__ receives from Python:\n // (messages, actions, config). config = {\"model\": \"gpt-4o\"}.\n let mut total_tokens = TokenUsage::default();\n let result = handle_llm_complete(\n &[\n json_to_monty(&serde_json::json!([{\"role\":\"user\",\"content\":\"hi\"}])),\n json_to_monty(&serde_json::json!([])),\n json_to_monty(&serde_json::json!({\"model\": \"gpt-4o\"})),\n ],\n &[],\n &mut thread,\n LlmCompleteDeps {\n llm: &llm,\n effects: &effects,\n leases: &leases,\n store: Some(&store),\n },\n &mut total_tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Return(_)));\n let captured = concrete.captured.lock().await;\n assert_eq!(captured.len(), 1);\n assert_eq!(captured[0].as_deref(), Some(\"gpt-4o\"));\n }\n\n #[tokio::test]\n async fn llm_complete_without_model_passes_none() {\n let concrete = Arc::new(ModelCapturingLlm {\n captured: tokio::sync::Mutex::new(Vec::new()),\n });\n let llm: Arc = Arc::clone(&concrete) as Arc;\n let effects: Arc = Arc::new(NoopEffects);\n let leases = Arc::new(LeaseManager::new());\n let store: Arc = Arc::new(crate::tests::InMemoryStore::with_docs(vec![]));\n\n let mut thread = Thread::new(\n \"goal\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n let mut total_tokens = TokenUsage::default();\n let _ = handle_llm_complete(\n &[\n json_to_monty(&serde_json::json!([{\"role\":\"user\",\"content\":\"hi\"}])),\n json_to_monty(&serde_json::json!([])),\n json_to_monty(&serde_json::json!({\"max_tokens\": 100})),\n ],\n &[],\n &mut thread,\n LlmCompleteDeps {\n llm: &llm,\n effects: &effects,\n leases: &leases,\n store: Some(&store),\n },\n &mut total_tokens,\n )\n .await;\n\n let captured = concrete.captured.lock().await;\n assert_eq!(captured.len(), 1);\n assert_eq!(captured[0], None);\n }\n\n // ── Python ↔ Rust ActionCall round-trip ───────────────────────────────\n //\n // Regression tests for the orphaned-tool-result bug. The Python\n // orchestrator stores `action_calls` on assistant messages using the\n // shape `{name, call_id, params}`, but the canonical Rust `ActionCall`\n // uses `{action_name, id, parameters}`. Without the explicit\n // `PythonActionCall` interchange type, `serde_json::from_value` would\n // silently fail (`.ok()` swallows the error) and the Python-shaped\n // assistant message would be parsed back as a plain assistant message\n // with no tool calls, causing every subsequent ActionResult to be\n // detected as orphaned by `sanitize_tool_messages` in the host crate.\n\n #[test]\n fn python_action_call_round_trips_through_serde() {\n let original = ActionCall {\n id: \"call_abc123\".to_string(),\n action_name: \"google_drive_tool\".to_string(),\n parameters: serde_json::json!({\"query\": \"expenses\"}),\n };\n\n let python_json = serde_json::to_value(PythonActionCall::from(&original))\n .expect(\"PythonActionCall must serialize\");\n // Python-friendly field names — match what default.py reads.\n assert_eq!(python_json[\"name\"], \"google_drive_tool\");\n assert_eq!(python_json[\"call_id\"], \"call_abc123\");\n assert_eq!(\n python_json[\"params\"],\n serde_json::json!({\"query\": \"expenses\"})\n );\n\n let parsed: PythonActionCall =\n serde_json::from_value(python_json).expect(\"must deserialize\");\n let round_tripped: ActionCall = parsed.into();\n assert_eq!(round_tripped.id, original.id);\n assert_eq!(round_tripped.action_name, original.action_name);\n assert_eq!(round_tripped.parameters, original.parameters);\n }\n\n #[test]\n fn action_calls_to_python_json_uses_python_field_names() {\n let calls = vec![\n ActionCall {\n id: \"call_1\".to_string(),\n action_name: \"notion_notion_search\".to_string(),\n parameters: serde_json::json!({\"query\": \"name\"}),\n },\n ActionCall {\n id: \"call_2\".to_string(),\n action_name: \"google_drive_tool\".to_string(),\n parameters: serde_json::json!({\"action\": \"list\"}),\n },\n ];\n let json = action_calls_to_python_json(&calls);\n assert_eq!(json.len(), 2);\n assert_eq!(json[0][\"name\"], \"notion_notion_search\");\n assert_eq!(json[0][\"call_id\"], \"call_1\");\n assert_eq!(json[1][\"name\"], \"google_drive_tool\");\n assert_eq!(json[1][\"call_id\"], \"call_2\");\n }\n\n #[test]\n fn python_json_to_action_calls_parses_python_field_names() {\n // The exact shape default.py produces (and stores on assistant\n // messages via `append_message(..., action_calls=calls)`).\n let python_json = serde_json::json!([\n {\"name\": \"notion_notion_search\", \"call_id\": \"call_xyz\", \"params\": {\"q\": \"foo\"}},\n {\"name\": \"google_drive_tool\", \"call_id\": \"call_abc\", \"params\": {\"action\": \"list\"}},\n ]);\n let parsed = python_json_to_action_calls(&python_json).expect(\"must parse\");\n assert_eq!(parsed.len(), 2);\n assert_eq!(parsed[0].action_name, \"notion_notion_search\");\n assert_eq!(parsed[0].id, \"call_xyz\");\n assert_eq!(parsed[0].parameters, serde_json::json!({\"q\": \"foo\"}));\n assert_eq!(parsed[1].action_name, \"google_drive_tool\");\n assert_eq!(parsed[1].id, \"call_abc\");\n }\n\n #[test]\n fn python_json_to_action_calls_rejects_canonical_field_names() {\n // Sanity check: the parser is strict about Python field names.\n // If `default.py` ever changes the shape, the test must catch it.\n let canonical_json = serde_json::json!([\n {\"action_name\": \"search\", \"id\": \"call_x\", \"parameters\": {}}\n ]);\n // Missing \"name\", \"call_id\", \"params\" → returns None.\n assert!(python_json_to_action_calls(&canonical_json).is_none());\n }\n\n #[test]\n fn summarize_action_calls_for_log_does_not_leak_user_pii() {\n // The whole point of this helper is that the warn log path on a\n // shape-drift failure must NOT dump tool parameters (which can\n // contain user PII like search queries, file names, email content)\n // into log aggregation systems. The summary should expose only\n // structural information: array length and the keys of the first\n // entry. The keys themselves are static (`name`, `call_id`,\n // `params`), not user data.\n let pii_value = serde_json::json!([\n {\n \"name\": \"google_drive_tool\",\n \"call_id\": \"call_xyz\",\n \"params\": {\n \"query\": \"salary spreadsheet for joe\",\n \"secret_token\": \"very-sensitive-token-do-not-log\"\n }\n },\n {\n \"name\": \"gmail\",\n \"call_id\": \"call_abc\",\n \"params\": {\n \"subject\": \"private message about layoffs\"\n }\n }\n ]);\n let summary = summarize_action_calls_for_log(&pii_value);\n\n // Structural info present.\n assert!(summary.contains(\"array of 2 entries\"));\n assert!(summary.contains(\"call_id\"));\n assert!(summary.contains(\"name\"));\n assert!(summary.contains(\"params\"));\n\n // PII fields and their values must NOT appear.\n assert!(\n !summary.contains(\"salary\"),\n \"summary must not leak user PII from params: {summary}\"\n );\n assert!(\n !summary.contains(\"very-sensitive-token\"),\n \"summary must not leak credential-shaped values: {summary}\"\n );\n assert!(\n !summary.contains(\"layoffs\"),\n \"summary must not leak free-text content: {summary}\"\n );\n assert!(\n !summary.contains(\"google_drive_tool\"),\n \"summary must not leak the tool name itself (could expose intent): {summary}\"\n );\n }\n\n #[test]\n fn summarize_action_calls_for_log_handles_edge_cases() {\n assert_eq!(\n summarize_action_calls_for_log(&serde_json::json!([])),\n \"empty array\"\n );\n assert!(\n summarize_action_calls_for_log(&serde_json::json!(\"not an array\")).contains(\"string\")\n );\n assert!(\n summarize_action_calls_for_log(&serde_json::json!({\"foo\": \"bar\"})).contains(\"object\")\n );\n assert!(summarize_action_calls_for_log(&serde_json::json!(null)).contains(\"null\"));\n }\n\n /// Caller-level regression test: feeds `json_to_thread_messages` the\n /// exact JSON shape that `default.py` produces for an assistant message\n /// with tool calls followed by tool results, and asserts that the\n /// resulting `ThreadMessage`s preserve the `action_calls` ↔\n /// `action_call_id` linkage. Without the `PythonActionCall` parser the\n /// assistant message would come back with `action_calls = None` and\n /// every following ActionResult would look orphaned to the bridge.\n #[test]\n fn json_to_thread_messages_preserves_action_calls_from_python_orchestrator() {\n // This is the literal shape `default.py` writes into\n // `state[\"working_messages\"]` after a Tier 0 step:\n //\n // append_message(working_messages, \"Assistant\", \"...\", action_calls=calls)\n // append_message(working_messages, \"ActionResult\", \"...\", action_name=..., action_call_id=...)\n //\n // where `calls` came from the LLM response and has shape\n // `[{\"name\": ..., \"call_id\": ..., \"params\": ...}]`.\n let working_messages = serde_json::json!([\n {\"role\": \"User\", \"content\": \"search in notion for my name\"},\n {\n \"role\": \"Assistant\",\n \"content\": \"\",\n \"action_calls\": [\n {\n \"name\": \"notion_notion_search\",\n \"call_id\": \"call_xyz\",\n \"params\": {\"query\": \"Illia\"}\n }\n ]\n },\n {\n \"role\": \"ActionResult\",\n \"content\": \"found 3 results\",\n \"action_name\": \"notion_notion_search\",\n \"action_call_id\": \"call_xyz\"\n }\n ]);\n\n let messages = json_to_thread_messages(&working_messages).expect(\"must parse\");\n assert_eq!(messages.len(), 3);\n\n // The assistant message MUST have action_calls populated, with\n // matching call_id. If this assertion fails, the bridge layer\n // will treat the following ActionResult as orphaned and rewrite\n // it as a user message — losing the model's ability to reason\n // about prior tool output.\n let assistant = &messages[1];\n assert_eq!(\n assistant.role,\n crate::types::message::MessageRole::Assistant\n );\n let calls = assistant\n .action_calls\n .as_ref()\n .expect(\"assistant message must carry action_calls after round-trip\");\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].id, \"call_xyz\");\n assert_eq!(calls[0].action_name, \"notion_notion_search\");\n assert_eq!(calls[0].parameters, serde_json::json!({\"query\": \"Illia\"}));\n\n // The ActionResult must reference the same call_id so the bridge\n // can pair them.\n let result = &messages[2];\n assert_eq!(\n result.role,\n crate::types::message::MessageRole::ActionResult\n );\n assert_eq!(result.action_call_id.as_deref(), Some(\"call_xyz\"));\n assert_eq!(result.action_name.as_deref(), Some(\"notion_notion_search\"));\n }\n\n /// Regression for the gate-resume / bootstrap path: when a thread\n /// resumes after approval or auth, `build_orchestrator_inputs`\n /// serializes `thread.internal_messages` into the bootstrap context\n /// that Python reads into `working_messages`. If `action_calls` is\n /// serialized with canonical `ActionCall` field names (`action_name`,\n /// `id`, `parameters`) instead of the Python interchange names\n /// (`name`, `call_id`, `params`), the next `__llm_complete__` call\n /// passes them back through `json_to_thread_messages` which fails\n /// with \"missing field `name`\" and orphans every subsequent tool\n /// result.\n ///\n /// This test simulates the full round-trip: build a `ThreadMessage`\n /// with action_calls → serialize through `build_orchestrator_inputs`'s\n /// exact serialization pattern → parse back through\n /// `json_to_thread_messages` → assert the calls survive. If anyone\n /// adds a THIRD serialization path in the future and uses canonical\n /// names, this test documents the pattern they should follow.\n #[test]\n fn bootstrap_context_action_calls_round_trip_through_python_interchange() {\n // Build a thread message the way the engine does: an assistant\n // message with action_calls in canonical ActionCall format (the\n // shape stored in the DB / internal_messages).\n let msg = ThreadMessage::assistant_with_actions(\n Some(\"I'll search for that\".to_string()),\n vec![ActionCall {\n id: \"call_resume_test\".to_string(),\n action_name: \"google_drive_tool\".to_string(),\n parameters: serde_json::json!({\"query\": \"budget\"}),\n }],\n );\n\n // Serialize through the SAME pattern `build_orchestrator_inputs`\n // uses. This is the exact code path that was broken before the\n // fix — it was using `\"action_calls\": m.action_calls` which\n // produced canonical field names.\n let calls_json = msg\n .action_calls\n .as_ref()\n .map(|calls| serde_json::Value::Array(action_calls_to_python_json(calls)));\n let serialized = serde_json::json!([{\n \"role\": \"Assistant\",\n \"content\": msg.content,\n \"action_name\": msg.action_name,\n \"action_call_id\": msg.action_call_id,\n \"action_calls\": calls_json,\n }]);\n\n // Parse back through the same path Python's working_messages\n // takes when it calls __llm_complete__.\n let parsed = json_to_thread_messages(&serialized).expect(\"must parse\");\n assert_eq!(parsed.len(), 1);\n\n let assistant = &parsed[0];\n let calls = assistant.action_calls.as_ref().expect(\n \"bootstrap context action_calls must survive the round-trip. \\\n If this fails, a serialization path is using canonical ActionCall \\\n field names instead of PythonActionCall interchange names.\",\n );\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].id, \"call_resume_test\");\n assert_eq!(calls[0].action_name, \"google_drive_tool\");\n assert_eq!(calls[0].parameters, serde_json::json!({\"query\": \"budget\"}));\n }\n\n /// Negative regression: verify that canonical ActionCall field names\n /// do NOT round-trip. If this test ever PASSES, it means someone\n /// added `#[serde(rename)]` to ActionCall or changed the parser to\n /// accept both formats — which is fine, but the PythonActionCall\n /// interchange type can then be removed. This test documents the\n /// current contract: canonical names are rejected by the parser.\n #[test]\n fn canonical_action_call_field_names_do_not_round_trip() {\n let serialized_with_canonical_names = serde_json::json!([{\n \"role\": \"Assistant\",\n \"content\": \"\",\n \"action_calls\": [{\n \"action_name\": \"search\",\n \"id\": \"call_x\",\n \"parameters\": {}\n }],\n }]);\n let parsed =\n json_to_thread_messages(&serialized_with_canonical_names).expect(\"messages parse\");\n // The assistant message should have NO action_calls because the\n // parser rejects canonical field names.\n assert!(\n parsed[0].action_calls.is_none(),\n \"canonical ActionCall field names must NOT parse as action_calls. \\\n If this assertion fails, the PythonActionCall interchange type \\\n is no longer needed — either remove it or update the contract.\"\n );\n }\n\n /// Regression: `action_calls: null` is Python's legitimate \"this\n /// message has no tool calls\" signal (text-only response). Before the\n /// null filter, `python_json_to_action_calls` would fire a warn log\n /// with \"invalid type: null, expected a sequence\" on every text-only\n /// assistant message — a false alarm that masked real drift issues.\n #[test]\n fn json_to_thread_messages_handles_null_action_calls_gracefully() {\n let messages = serde_json::json!([\n {\n \"role\": \"Assistant\",\n \"content\": \"Here is your answer.\",\n \"action_calls\": null\n }\n ]);\n let parsed = json_to_thread_messages(&messages).expect(\"must parse\");\n assert_eq!(parsed.len(), 1);\n assert_eq!(\n parsed[0].role,\n crate::types::message::MessageRole::Assistant\n );\n assert_eq!(parsed[0].content, \"Here is your answer.\");\n assert!(\n parsed[0].action_calls.is_none(),\n \"null action_calls must produce None, not a parse error\"\n );\n }\n\n /// Verify that messages WITHOUT the action_calls key at all (the most\n /// common case for text responses) also parse correctly — this is the\n /// baseline that the null-filtering regression test extends.\n #[test]\n fn json_to_thread_messages_handles_absent_action_calls() {\n let messages = serde_json::json!([\n {\"role\": \"Assistant\", \"content\": \"Just text, no tools.\"}\n ]);\n let parsed = json_to_thread_messages(&messages).expect(\"must parse\");\n assert_eq!(parsed.len(), 1);\n assert!(parsed[0].action_calls.is_none());\n }\n\n /// Empty action_calls array is valid (LLM decided not to call any\n /// tools this turn but the response still has the array field). Must\n /// produce `Some(vec![])`, not `None`.\n #[test]\n fn json_to_thread_messages_handles_empty_action_calls_array() {\n let messages = serde_json::json!([\n {\n \"role\": \"Assistant\",\n \"content\": \"No tools needed.\",\n \"action_calls\": []\n }\n ]);\n let parsed = json_to_thread_messages(&messages).expect(\"must parse\");\n assert_eq!(parsed.len(), 1);\n let calls = parsed[0]\n .action_calls\n .as_ref()\n .expect(\"empty array should produce Some(vec![])\");\n assert!(calls.is_empty());\n }\n\n // ── Consecutive action error counting (issue #2325) ──────────\n //\n // The run_loop tracks `consecutive_action_errors` for Tier 0 (structured\n // action calls). These tests exercise the counting logic extracted from\n // run_loop into small Python snippets that simulate batch outcomes.\n\n #[test]\n fn action_errors_increment_when_all_actions_fail() {\n // Simulate 3 consecutive batches where all actions fail.\n let count = eval_python_int(\n r#\"\nconsecutive_action_errors = 0\nfor _ in range(3):\n batch_error_count = 2\n batch_success_count = 0\n if batch_success_count > 0:\n consecutive_action_errors = 0\n elif batch_error_count > 0:\n consecutive_action_errors += 1\nFINAL(consecutive_action_errors)\n\"#,\n );\n assert_eq!(count, 3);\n }\n\n #[test]\n fn action_errors_reset_when_any_action_succeeds() {\n // 2 all-fail batches, then 1 batch with a success => resets to 0.\n let count = eval_python_int(\n r#\"\nconsecutive_action_errors = 0\nfor batch in [(0, 2), (0, 1), (1, 1)]:\n batch_success_count = batch[0]\n batch_error_count = batch[1]\n if batch_success_count > 0:\n consecutive_action_errors = 0\n elif batch_error_count > 0:\n consecutive_action_errors += 1\nFINAL(consecutive_action_errors)\n\"#,\n );\n assert_eq!(count, 0);\n }\n\n #[test]\n fn action_errors_partial_success_resets_counter() {\n // A batch with mixed results (some succeed, some fail) should reset.\n let count = eval_python_int(\n r#\"\nconsecutive_action_errors = 5\nbatch_success_count = 1\nbatch_error_count = 3\nif batch_success_count > 0:\n consecutive_action_errors = 0\nelif batch_error_count > 0:\n consecutive_action_errors += 1\nFINAL(consecutive_action_errors)\n\"#,\n );\n assert_eq!(count, 0);\n }\n\n #[test]\n fn action_errors_nudge_injected_at_threshold() {\n // When consecutive_action_errors reaches max_consecutive_errors,\n // a nudge message should be appended. We simulate the branching\n // logic and check whether a nudge would fire.\n // Returns 1 if nudge fires (not failure), 0 otherwise.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = 5\nconsecutive_action_errors = 5\nnudge = False\nfailed = False\nif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nelif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors:\n nudge = True\nif nudge and not failed:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 1, \"nudge should fire at threshold\");\n }\n\n #[test]\n fn action_errors_no_nudge_below_threshold() {\n // Returns 1 if nudge fires, 0 if not.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = 5\nconsecutive_action_errors = 4\nnudge = False\nfailed = False\nif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nelif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors:\n nudge = True\nif nudge:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 0, \"nudge should not fire below threshold\");\n }\n\n #[test]\n fn action_errors_failure_at_threshold_plus_two() {\n // At max_consecutive_errors + 2, the thread should transition to failed.\n // Returns 1 if failed, 0 if not.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = 5\nconsecutive_action_errors = 7\nfailed = False\nif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nif failed:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 1, \"should fail at threshold + 2\");\n }\n\n #[test]\n fn action_errors_nudge_at_threshold_not_failure() {\n // At exactly max_consecutive_errors + 1, we get a nudge but not failure.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = 5\nconsecutive_action_errors = 6\nnudge = False\nfailed = False\nif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nelif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors:\n nudge = True\n# Return 0=nothing, 1=nudge, 2=failed\nif failed:\n FINAL(2)\nelif nudge:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 1, \"should nudge at threshold + 1, not fail\");\n }\n\n #[test]\n fn action_errors_none_limit_skips_check_without_typeerror() {\n // Regression: when max_consecutive_errors is None (meaning \"no limit\"),\n // the arithmetic `max_consecutive_errors + 2` used to crash with\n // TypeError on the first action error. The guard must short-circuit\n // on None and leave both the nudge and failure branches untaken.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = None\nconsecutive_action_errors = 1\nnudge = False\nfailed = False\nif max_consecutive_errors is not None and consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nelif max_consecutive_errors is not None and consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors:\n nudge = True\n# Return 0=nothing, 1=nudge, 2=failed\nif failed:\n FINAL(2)\nelif nudge:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 0, \"None limit should disable the guard entirely\");\n }\n\n #[test]\n fn code_errors_none_limit_skips_failure_check() {\n // Regression: same None-guard for the code-error branch at line 660.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = None\nconsecutive_errors = 99\nfailed = False\nif max_consecutive_errors is not None and consecutive_errors >= max_consecutive_errors:\n failed = True\nif failed:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(\n result, 0,\n \"None limit should not trigger failure regardless of consecutive_errors\"\n );\n }\n\n #[test]\n fn action_error_prefix_added_to_error_output() {\n // Verify that [ACTION FAILED] prefix is prepended to error outputs.\n // Returns 1 if prefix present, 0 if not.\n let result = eval_python_int(\n r#\"\nr = {\"action_name\": \"http\", \"output\": \"connection refused\", \"is_error\": True}\noutput = r.get(\"output\")\noutput_str = str(output) if output is not None else \"[no output]\"\nif r.get(\"is_error\"):\n output_str = \"[ACTION FAILED] \" + output_str\nif output_str.startswith(\"[ACTION FAILED]\"):\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 1, \"error outputs must get [ACTION FAILED] prefix\");\n }\n\n #[test]\n fn action_error_skipped_calls_count_as_errors() {\n // When a call has no result (r is None), it should count as an error.\n let count = eval_python_int(\n r#\"\nbatch_error_count = 0\nbatch_success_count = 0\nr = None\nif r is not None:\n if r.get(\"is_error\"):\n batch_error_count += 1\n else:\n batch_success_count += 1\nelse:\n batch_error_count += 1\nFINAL(batch_error_count)\n\"#,\n );\n assert_eq!(count, 1, \"skipped calls must count as batch errors\");\n }\n\n #[test]\n fn checkpoint_includes_consecutive_action_errors() {\n // Test that handle_save_checkpoint persists consecutive_action_errors\n // in the thread metadata.\n let mut thread = Thread::new(\n \"goal\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n let state = json_to_monty(&serde_json::json!({}));\n let counters = json_to_monty(&serde_json::json!({\n \"nudge_count\": 0,\n \"consecutive_errors\": 1,\n \"consecutive_action_errors\": 4,\n \"compaction_count\": 2,\n }));\n\n handle_save_checkpoint(&[state, counters], &[], &mut thread);\n\n let checkpoint = thread\n .metadata\n .get(\"runtime_checkpoint\")\n .expect(\"checkpoint must exist\");\n assert_eq!(\n checkpoint\n .get(\"consecutive_action_errors\")\n .and_then(|v| v.as_u64()),\n Some(4),\n \"consecutive_action_errors must be persisted in checkpoint\"\n );\n assert_eq!(\n checkpoint\n .get(\"consecutive_errors\")\n .and_then(|v| v.as_u64()),\n Some(1),\n );\n assert_eq!(\n checkpoint.get(\"compaction_count\").and_then(|v| v.as_u64()),\n Some(2),\n );\n }\n\n /// Regression test: every assistant tool_call must have a matching\n /// ActionResult after parsing. If an ActionResult is missing, the LLM\n /// API rejects with \"No tool output found for function call \".\n ///\n /// This was the root cause of the HTTP 400 from the OpenAI Codex\n /// provider: a tool returning null output caused the Python\n /// orchestrator to skip appending the ActionResult.\n #[test]\n fn json_to_thread_messages_every_tool_call_has_action_result() {\n // Simulate working_messages after the Python fix: every call gets\n // an ActionResult, even when the original output was null.\n let messages = serde_json::json!([\n {\"role\": \"System\", \"content\": \"You are a helpful assistant.\"},\n {\"role\": \"User\", \"content\": \"Update all tools.\"},\n {\n \"role\": \"Assistant\",\n \"content\": \"\",\n \"action_calls\": [\n {\"call_id\": \"call_AAA\", \"name\": \"tool_a\", \"params\": {}},\n {\"call_id\": \"call_BBB\", \"name\": \"tool_b\", \"params\": {}},\n {\"call_id\": \"call_CCC\", \"name\": \"tool_c\", \"params\": {}}\n ]\n },\n {\n \"role\": \"ActionResult\",\n \"content\": \"{\\\"ok\\\": true}\",\n \"action_name\": \"tool_a\",\n \"action_call_id\": \"call_AAA\"\n },\n {\n \"role\": \"ActionResult\",\n \"content\": \"[no output]\",\n \"action_name\": \"tool_b\",\n \"action_call_id\": \"call_BBB\"\n },\n {\n \"role\": \"ActionResult\",\n \"content\": \"{\\\"done\\\": true}\",\n \"action_name\": \"tool_c\",\n \"action_call_id\": \"call_CCC\"\n }\n ]);\n\n let parsed = json_to_thread_messages(&messages).expect(\"must parse\");\n assert_eq!(parsed.len(), 6);\n\n // Extract call IDs from the assistant message\n let assistant_calls: std::collections::HashSet = parsed\n .iter()\n .filter_map(|m| m.action_calls.as_ref())\n .flat_map(|calls| calls.iter().map(|c| c.id.clone()))\n .collect();\n\n // Extract call IDs from ActionResult messages\n let result_call_ids: std::collections::HashSet = parsed\n .iter()\n .filter(|m| m.role == crate::types::message::MessageRole::ActionResult)\n .filter_map(|m| m.action_call_id.clone())\n .collect();\n\n // Every tool_call must have a matching ActionResult\n for call_id in &assistant_calls {\n assert!(\n result_call_ids.contains(call_id),\n \"tool_call {call_id} has no matching ActionResult — \\\n this would cause 'No tool output found' from the LLM API\"\n );\n }\n }\n\n // ── CodeExecutionFailed event emission (caller test) ────────\n\n #[tokio::test]\n async fn execute_code_step_emits_code_execution_failed_event() {\n let llm: Arc = Arc::new(ModelCapturingLlm {\n captured: tokio::sync::Mutex::new(Vec::new()),\n });\n let effects: Arc = Arc::new(NoopEffects);\n let leases = Arc::new(LeaseManager::new());\n let policy = Arc::new(PolicyEngine::new());\n\n let mut thread = Thread::new(\n \"test code execution failure instrumentation\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n // Pass intentionally broken Python code (syntax error)\n let args = &[\n json_to_monty(&serde_json::json!(\"def ==\")),\n json_to_monty(&serde_json::json!({})),\n ];\n\n let (tx, _rx) = tokio::sync::broadcast::channel(16);\n let _result = handle_execute_code_step(\n args,\n &[],\n &mut thread,\n &llm,\n &effects,\n &leases,\n &policy,\n Some(&tx),\n )\n .await;\n\n // Verify CodeExecutionFailed event was emitted on thread.events\n let code_failed_events: Vec<_> = thread\n .events\n .iter()\n .filter(|e| matches!(&e.kind, EventKind::CodeExecutionFailed { .. }))\n .collect();\n\n assert_eq!(\n code_failed_events.len(),\n 1,\n \"expected exactly one CodeExecutionFailed event, got {}\",\n code_failed_events.len()\n );\n\n if let EventKind::CodeExecutionFailed {\n category,\n code_hash,\n ..\n } = &code_failed_events[0].kind\n {\n assert_eq!(\n *category,\n crate::types::step::CodeExecutionFailure::SyntaxError\n );\n assert!(code_hash.is_some());\n } else {\n panic!(\"expected CodeExecutionFailed event kind\");\n }\n\n // Also verify ActionFailed was emitted (existing behavior)\n let action_failed = thread\n .events\n .iter()\n .any(|e| matches!(&e.kind, EventKind::ActionFailed { .. }));\n assert!(\n action_failed,\n \"expected ActionFailed event alongside CodeExecutionFailed\"\n );\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/executor/scripting.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-ratelimit-used", + "22" + ], + [ + "x-github-request-id", + "F564:3BD57F:41F8D2:4D51E5:69DFAEC5" + ], + [ + "server", + "github.com" + ], + [ + "x-frame-options", + "deny" + ], + [ + "x-ratelimit-remaining", + "4978" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-xss-protection", + "0" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:10 GMT" + ], + [ + "etag", + "\"674fca2750840d6f245ff8676d54df8b83ca74f1\"" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "content-length", + "113551" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ] + ], + "body": "//! Tier 1 executor: embedded Python via Monty.\n//!\n//! Executes LLM-generated Python code using the Monty interpreter. Tool\n//! calls use **async dispatch**: each tool call returns a Monty `ExternalFuture`\n//! via `resume_pending()`, allowing Python code to use `await` and\n//! `asyncio.gather()` for parallel execution. When all tasks are blocked,\n//! Monty yields `ResolveFutures` and we execute pending tools concurrently\n//! via `JoinSet`.\n//!\n//! Follows the RLM (Recursive Language Model) pattern:\n//! - Thread context injected as Python variables (not LLM attention input)\n//! - `llm_query()` / `llm_query_batched()` for recursive subagent spawning\n//! - `FINAL(answer)` / `FINAL_VAR(name)` for explicit termination\n//! - Step 0 orientation preamble for context awareness\n//! - Errors flow back to LLM for self-correction (not step termination)\n//! - Output truncated to configurable limit with variable listing\n//! - `asyncio.gather()` for parallel tool execution (via ResolveFutures)\n\nuse std::collections::HashMap;\nuse std::sync::Arc;\nuse std::time::Duration;\n\nuse monty::{\n ExcType, ExtFunctionResult, LimitedTracker, MontyException, MontyObject, MontyRun,\n NameLookupResult, PrintWriter, ResourceLimits, RunProgress,\n};\nuse tracing::debug;\n\nuse crate::capability::lease::LeaseManager;\nuse crate::capability::policy::{PolicyDecision, PolicyEngine};\nuse crate::traits::effect::{EffectExecutor, ThreadExecutionContext};\nuse crate::traits::llm::{LlmBackend, LlmCallConfig};\nuse crate::types::error::EngineError;\nuse crate::types::event::EventKind;\nuse crate::types::message::{MessageRole, ThreadMessage};\nuse crate::types::step::{ActionResult, CodeExecutionFailure, LlmResponse, TokenUsage};\nuse crate::types::thread::Thread;\nuse ironclaw_common::ValidTimezone;\n\n// ── Configuration ───────────────────────────────────────────\n\n/// Maximum characters of output to include in LLM context between steps.\n/// Matches Prime Intellect's default. Configurable per thread in the future.\nconst OUTPUT_TRUNCATE_LEN: usize = 8_000;\n\n/// Maximum characters for a preview prefix in compact metadata.\nconst OUTPUT_PREVIEW_LEN: usize = 200;\n\n/// Default resource limits for Monty execution.\nfn default_limits() -> ResourceLimits {\n ResourceLimits::new()\n .max_duration(Duration::from_secs(30))\n .max_allocations(1_000_000)\n .max_memory(64 * 1024 * 1024) // 64 MB\n}\n\n// ── Result types ────────────────────────────────────────────\n\n/// Result of executing a code block.\npub struct CodeExecutionResult {\n /// The Python return value, converted to JSON.\n pub return_value: serde_json::Value,\n /// Captured print output.\n pub stdout: String,\n /// All action calls that were made during execution.\n pub action_results: Vec,\n /// Events generated during execution.\n pub events: Vec,\n /// If set, execution was interrupted for approval.\n pub need_approval: Option,\n /// Tokens used by recursive llm_query() calls.\n pub recursive_tokens: TokenUsage,\n /// If set, the code called FINAL() or FINAL_VAR() with this answer.\n pub final_answer: Option,\n /// Classified failure category. `None` when execution succeeded or was\n /// paused by a gate. `Some(category)` when code execution failed —\n /// `failure.is_some()` replaces the former `had_error: bool` field.\n pub failure: Option,\n}\n\n/// Build a compact output summary for inclusion in LLM context between steps.\n///\n/// Truncates to `OUTPUT_TRUNCATE_LEN` (last N chars shown, like fast-rlm).\n/// Includes a list of REPL variable names if available.\npub fn compact_output_metadata(stdout: &str, return_value: &serde_json::Value) -> String {\n let mut parts = Vec::new();\n\n if !stdout.is_empty() {\n let char_count = stdout.chars().count();\n if char_count > OUTPUT_TRUNCATE_LEN {\n let truncated: String = stdout\n .chars()\n .skip(char_count - OUTPUT_TRUNCATE_LEN)\n .collect();\n parts.push(format!(\n \"[TRUNCATED: last {OUTPUT_TRUNCATE_LEN} of {char_count} chars shown]\\n{truncated}\",\n ));\n } else {\n parts.push(format!(\"[FULL OUTPUT: {char_count} chars]\\n{stdout}\"));\n }\n }\n\n if *return_value != serde_json::Value::Null {\n let val_str = serde_json::to_string_pretty(return_value).unwrap_or_default();\n let val_char_count = val_str.chars().count();\n if val_char_count > OUTPUT_PREVIEW_LEN {\n let preview: String = val_str.chars().take(OUTPUT_PREVIEW_LEN).collect();\n parts.push(format!(\n \"Return value ({val_char_count} chars): {preview}...\",\n ));\n } else {\n parts.push(format!(\"Return value: {val_str}\"));\n }\n }\n\n if parts.is_empty() {\n \"[code executed, no output]\".into()\n } else {\n parts.join(\"\\n\")\n }\n}\n\n// ── Step 0 orientation preamble ─────────────────────────────\n\n/// Build the Step 0 orientation preamble that auto-executes before the\n/// first LLM call to give the model structural awareness of the context.\npub fn build_orientation_preamble(thread: &Thread) -> String {\n let msg_count = thread.messages.len();\n let total_chars: usize = thread.messages.iter().map(|m| m.content.len()).sum();\n let user_msgs = thread\n .messages\n .iter()\n .filter(|m| m.role == MessageRole::User)\n .count();\n\n let mut preview = String::new();\n if let Some(last_user) = thread\n .messages\n .iter()\n .rev()\n .find(|m| m.role == MessageRole::User)\n {\n let content_preview: String = last_user.content.chars().take(500).collect();\n let truncated = if last_user.content.chars().count() > 500 {\n \"...\"\n } else {\n \"\"\n };\n preview = format!(\"\\nLast user message preview: {content_preview}{truncated}\");\n }\n\n format!(\n \"[Step 0 — Context Orientation]\\n\\\n Goal: {goal}\\n\\\n Context: {msg_count} messages, {total_chars} total chars, {user_msgs} from user\\n\\\n Step: {step}{preview}\",\n goal = thread.goal,\n step = thread.step_count + 1,\n )\n}\n\n// ── Context injection (RLM 3.4) ────────────────────────────\n\n/// Build Monty input variables from thread state.\n///\n/// `persisted_state` carries variables from previous code steps so the\n/// REPL feels persistent even though each step creates a fresh MontyRun.\nfn build_context_inputs(\n thread: &Thread,\n persisted_state: &serde_json::Value,\n) -> (Vec, Vec) {\n let mut names = Vec::new();\n let mut values = Vec::new();\n\n // `context` — thread messages as a list of dicts\n let messages: Vec = thread\n .messages\n .iter()\n .map(|msg| {\n let mut pairs = vec![\n (\n MontyObject::String(\"role\".into()),\n MontyObject::String(format!(\"{:?}\", msg.role)),\n ),\n (\n MontyObject::String(\"content\".into()),\n MontyObject::String(msg.content.clone()),\n ),\n ];\n if let Some(ref name) = msg.action_name {\n pairs.push((\n MontyObject::String(\"action_name\".into()),\n MontyObject::String(name.clone()),\n ));\n }\n MontyObject::dict(pairs)\n })\n .collect();\n names.push(\"context\".into());\n values.push(MontyObject::List(messages));\n\n // `goal` — the thread's goal string\n names.push(\"goal\".into());\n values.push(MontyObject::String(thread.goal.clone()));\n\n // `step_number` — current step index\n names.push(\"step_number\".into());\n values.push(MontyObject::Int(thread.step_count as i64));\n\n // `state` — persisted variables from previous code steps.\n // This is a dict that accumulates: return values, tool results, etc.\n // The model can read `state[\"results\"]`, `state[\"prev_return\"]`, etc.\n names.push(\"state\".into());\n values.push(json_to_monty(persisted_state));\n\n // `previous_results` — dict of {call_id: result_json} from prior steps\n let result_pairs: Vec<(MontyObject, MontyObject)> = thread\n .messages\n .iter()\n .filter(|m| m.role == MessageRole::ActionResult)\n .filter_map(|m| {\n let call_id = m.action_call_id.as_ref()?;\n Some((\n MontyObject::String(call_id.clone()),\n MontyObject::String(m.content.clone()),\n ))\n })\n .collect();\n names.push(\"previous_results\".into());\n values.push(MontyObject::dict(result_pairs));\n\n // `user_timezone` — validated IANA timezone from the user's channel (e.g. \"America/New_York\")\n let tz = thread\n .metadata\n .get(\"user_timezone\")\n .and_then(|v| v.as_str())\n .and_then(ValidTimezone::parse)\n .map(|vtz| vtz.name().to_string())\n .unwrap_or_else(|| \"UTC\".into());\n names.push(\"user_timezone\".into());\n values.push(MontyObject::String(tz));\n\n (names, values)\n}\n\n// ── Main execution function ─────────────────────────────────\n\n/// Execute a Python code block using Monty.\n///\n/// Handles the full RLM execution pattern: context-as-variables, FINAL()\n/// termination, llm_query() recursive calls, error-to-LLM flow, and\n/// output truncation.\n#[allow(clippy::too_many_arguments)]\npub async fn execute_code(\n code: &str,\n thread: &Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &LeaseManager,\n policy: &PolicyEngine,\n context: &ThreadExecutionContext,\n capability_policies: &[crate::types::capability::PolicyRule],\n persisted_state: &serde_json::Value,\n) -> Result {\n execute_code_with_skills(\n code,\n thread,\n llm,\n effects,\n leases,\n policy,\n context,\n capability_policies,\n persisted_state,\n &[],\n )\n .await\n}\n\n/// Execute a Python code block with optional skill code snippets.\n///\n/// `skill_snippet_names` are registered as additional known functions in the\n/// Monty NameLookup, alongside tool names from capability leases.\n#[allow(clippy::too_many_arguments)]\npub async fn execute_code_with_skills(\n code: &str,\n thread: &Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &LeaseManager,\n policy: &PolicyEngine,\n context: &ThreadExecutionContext,\n capability_policies: &[crate::types::capability::PolicyRule],\n persisted_state: &serde_json::Value,\n skill_snippet_names: &[String],\n) -> Result {\n let mut stdout = String::new();\n let mut action_results = Vec::new();\n let mut events = Vec::new();\n let mut recursive_tokens = TokenUsage::default();\n let mut final_answer: Option = None;\n\n // Build context variables including persisted state from prior steps\n let (input_names, input_values) = build_context_inputs(thread, persisted_state);\n\n // Collect known tool names so NameLookup can return callable stubs.\n // Without this, `mission_list()` in code raises NameError because Monty\n // resolves the name before calling it, and Undefined → NameError.\n let active_leases = leases.active_for_thread(thread.id).await;\n let mut known_actions: std::collections::HashSet = effects\n .available_actions(&active_leases)\n .await\n .unwrap_or_default()\n .into_iter()\n .map(|a| a.name)\n .collect();\n\n // Register skill code snippet function names as additional known actions.\n // These resolve in NameLookup so the LLM can call them as Python functions.\n for name in skill_snippet_names {\n known_actions.insert(name.clone());\n }\n\n // Parse and compile (wrap in catch_unwind — Monty 0.0.x can panic)\n let runner = match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n MontyRun::new(code.to_string(), \"step.py\", input_names)\n })) {\n Ok(Ok(runner)) => runner,\n Ok(Err(e)) => {\n // Parse error flows back to LLM (not a termination)\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"SyntaxError: {e}\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n failure: Some(CodeExecutionFailure::SyntaxError),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"{stdout}\\nVmPanic: Monty VM panicked during code parsing\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n };\n\n // Start execution with resource limits and context inputs\n let tracker = LimitedTracker::new(default_limits());\n\n let run_result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n runner.start(input_values, tracker, PrintWriter::Collect(&mut stdout))\n }));\n\n let mut progress = match run_result {\n Ok(Ok(p)) => p,\n Ok(Err(e)) => {\n // Runtime error flows back to LLM\n let category = classify_runtime_error(&e.to_string());\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"{stdout}\\nError: {e}\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n failure: Some(category),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"{stdout}\\nVmPanic: Monty VM panicked during execution start\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n };\n\n // Pending async tool executions keyed by Monty call_id.\n // When a tool FunctionCall comes in, we spawn a tokio task and store\n // the JoinHandle here. When ResolveFutures yields, we await them.\n let mut pending_futures: HashMap = HashMap::new();\n\n // Drive the execution loop\n let mut call_counter = 0u32;\n loop {\n match progress {\n RunProgress::Complete(obj) => {\n return Ok(CodeExecutionResult {\n return_value: monty_to_json(&obj),\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: None,\n });\n }\n\n RunProgress::FunctionCall(call) => {\n call_counter += 1;\n let str_call_id = format!(\"code_call_{call_counter}\");\n let monty_call_id = call.call_id;\n let action_name = call.function_name.clone();\n let params = monty_args_to_json(&call.args, &call.kwargs);\n\n debug!(action = %action_name, call_id = %str_call_id, monty_id = monty_call_id, \"Monty: function call\");\n\n // Builtins that need synchronous results — resume with value.\n let sync_result = match action_name.as_str() {\n \"FINAL\" => {\n let answer = call.args.first().map(monty_to_string).unwrap_or_default();\n final_answer = Some(answer);\n Some(ExtFunctionResult::Return(MontyObject::None))\n }\n \"FINAL_VAR\" => {\n let var_name = call\n .args\n .first()\n .map(monty_to_string)\n .unwrap_or_else(|| \"result\".into());\n final_answer = Some(format!(\"[FINAL_VAR: {var_name}]\"));\n Some(ExtFunctionResult::Return(MontyObject::None))\n }\n // LLM calls are async — spawn tokio task, resume_pending.\n // This allows asyncio.gather(llm_query(...), tool(...))\n // to run the LLM call and tool call concurrently.\n \"llm_query\" => {\n let args = call.args.clone();\n let kwargs = call.kwargs.clone();\n let llm = llm.clone();\n let handle = tokio::spawn(async move {\n handle_llm_query_standalone(&args, &kwargs, &llm).await\n });\n pending_futures.insert(monty_call_id, PendingFuture::Llm { handle });\n None // handled as async below\n }\n \"llm_query_batched\" => {\n let args = call.args.clone();\n let kwargs = call.kwargs.clone();\n let llm = llm.clone();\n let handle = tokio::spawn(async move {\n handle_llm_query_batched_standalone(&args, &kwargs, &llm).await\n });\n pending_futures.insert(monty_call_id, PendingFuture::Llm { handle });\n None\n }\n // rlm_query stays synchronous — it spawns a child Monty VM\n // which isn't Send, so it can't run in tokio::spawn.\n \"rlm_query\" => Some(\n handle_rlm_query(\n &call.args,\n &call.kwargs,\n thread,\n llm,\n effects,\n leases,\n policy,\n &mut recursive_tokens,\n )\n .await,\n ),\n \"globals\" | \"locals\" => {\n let entries: Vec<(MontyObject, MontyObject)> = known_actions\n .iter()\n .map(|name| {\n (MontyObject::String(name.clone()), MontyObject::Bool(true))\n })\n .collect();\n Some(ExtFunctionResult::Return(MontyObject::Dict(entries.into())))\n }\n _ => None, // tool call — handled async below\n };\n\n if let Some(ext_result) = sync_result {\n // Sync resume for builtins\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume(ext_result, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(classify_runtime_error(&e.to_string())),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during resume\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n continue;\n }\n\n // If an LLM call already inserted a pending future, just\n // resume_pending and continue — no preflight needed.\n if pending_futures.contains_key(&monty_call_id) {\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume_pending(PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(classify_runtime_error(&e.to_string())),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during resume_pending\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n continue;\n }\n\n // ── Async tool dispatch ─────────────────────────────\n // Preflight (lease + policy) is sync. If denied or\n // needs approval, resume with error immediately.\n // If approved, spawn tokio task and resume_pending().\n\n let preflight = preflight_action(\n &action_name,\n ¶ms,\n thread,\n effects,\n leases,\n policy,\n context,\n capability_policies,\n &str_call_id,\n &mut events,\n )\n .await;\n\n match preflight {\n PreflightResult::Approved(lease) => {\n // Spawn async execution\n let effects = effects.clone();\n let name = action_name.clone();\n let params_clone = params.clone();\n let lease_clone = lease.clone();\n let mut ctx = context.clone();\n ctx.current_call_id = Some(str_call_id.clone());\n let ps = crate::types::event::summarize_params(&name, ¶ms);\n\n let handle = tokio::spawn(async move {\n effects\n .execute_action(&name, params_clone, &lease_clone, &ctx)\n .await\n });\n\n pending_futures.insert(\n monty_call_id,\n PendingFuture::Tool {\n handle,\n action_name,\n call_id: str_call_id,\n lease_id: lease.id,\n parameters: params.clone(),\n params_summary: ps,\n },\n );\n\n // Resume with pending future — Python gets ExternalFuture\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume_pending(PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::ToolError),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during resume_pending\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n PreflightResult::Denied(ext_result) => {\n // Resume with error — Python sees an exception\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume(ext_result, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::ToolError),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during resume\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n PreflightResult::GatePaused(outcome) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: Some(outcome),\n recursive_tokens,\n final_answer: None,\n failure: None,\n });\n }\n }\n }\n\n // ── ResolveFutures: parallel execution ────────────────\n // Resolves both tool calls and LLM calls that were deferred\n // via resume_pending(). All pending tokio tasks are awaited\n // and their results fed back to Monty.\n RunProgress::ResolveFutures(resolve) => {\n let pending_ids = resolve.pending_call_ids().to_vec();\n debug!(pending = ?pending_ids, \"Monty: ResolveFutures — resolving {} pending futures\", pending_ids.len());\n\n let mut results: Vec<(u32, ExtFunctionResult)> =\n Vec::with_capacity(pending_ids.len());\n\n for &mid in &pending_ids {\n let ext_result = if let Some(pf) = pending_futures.remove(&mid) {\n match pf {\n PendingFuture::Tool {\n handle,\n action_name,\n call_id,\n lease_id,\n parameters,\n params_summary,\n } => {\n resolve_tool_future(\n handle,\n &action_name,\n &call_id,\n lease_id,\n parameters,\n params_summary,\n leases,\n context,\n &mut action_results,\n &mut events,\n )\n .await\n }\n PendingFuture::Llm { handle } => {\n resolve_llm_future(handle, &mut recursive_tokens).await\n }\n }\n } else {\n debug!(call_id = mid, \"ResolveFutures: unknown pending call_id\");\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"unknown pending call_id {mid}\")),\n ))\n };\n results.push((mid, ext_result));\n }\n\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n resolve.resume(results, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(classify_runtime_error(&e.to_string())),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during ResolveFutures resume\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n\n RunProgress::NameLookup(lookup) => {\n let name = lookup.name.clone();\n\n let result = if known_actions.contains(&name) {\n debug!(name = %name, \"Monty: resolved as tool function\");\n NameLookupResult::Value(MontyObject::Function {\n name: name.clone(),\n docstring: None,\n })\n } else if name == \"globals\" || name == \"locals\" {\n NameLookupResult::Value(MontyObject::Function {\n name: name.clone(),\n docstring: None,\n })\n } else {\n debug!(name = %name, \"Monty: unresolved name\");\n NameLookupResult::Undefined\n };\n\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n lookup.resume(result, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nNameError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::NameLookup),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during name lookup\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n\n RunProgress::OsCall(os_call) => {\n debug!(function = ?os_call.function, \"Monty: OS call denied\");\n let err = ExtFunctionResult::Error(MontyException::new(\n ExcType::OSError,\n Some(\"OS operations are not permitted in CodeAct scripts\".into()),\n ));\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n os_call.resume(err, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nOSError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::OsDenied),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"{stdout}\\nVmPanic: Monty VM panicked during OS call\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n }\n }\n}\n\n// ── Error classification ────────────────────────────────────\n\n/// Classify a runtime error message into a failure category.\n///\n/// Parses the error text from Monty to distinguish between LLM logic bugs\n/// (NameError, TypeError, etc.), resource limit hits, and Monty VM issues.\nfn classify_runtime_error(error_msg: &str) -> CodeExecutionFailure {\n let lower = error_msg.to_ascii_lowercase();\n\n // Most specific checks first to avoid substring false positives.\n if lower.contains(\"timed out\")\n || lower.contains(\"timeout\")\n || lower.contains(\"memory limit\")\n || lower.contains(\"allocation limit\")\n || lower.contains(\"out of fuel\")\n || lower.contains(\"fuel exhausted\")\n || lower.contains(\"resource limit\")\n {\n CodeExecutionFailure::ResourceLimit\n } else if lower.contains(\"os operations are not permitted\") || lower.contains(\"oserror\") {\n CodeExecutionFailure::OsDenied\n } else if lower.contains(\"syntaxerror\") {\n CodeExecutionFailure::SyntaxError\n } else {\n // NameError, TypeError, ValueError, AttributeError, IndexError,\n // KeyError, ModuleNotFoundError, NotImplementedError, etc.\n CodeExecutionFailure::RuntimeError\n }\n}\n\n/// Compute a short hash of Python code for dedup/correlation in events.\n///\n/// Uses FNV-1a (64-bit) which is stable across Rust versions, unlike\n/// `DefaultHasher`. Not cryptographic — collision probability is ~2^-32\n/// at typical usage levels, sufficient for dedup but not for security.\npub fn code_hash(code: &str) -> String {\n const FNV_OFFSET: u64 = 0xcbf29ce484222325;\n const FNV_PRIME: u64 = 0x00000100000001B3;\n let mut hash = FNV_OFFSET;\n for byte in code.as_bytes() {\n hash ^= *byte as u64;\n hash = hash.wrapping_mul(FNV_PRIME);\n }\n format!(\"{hash:016x}\")\n}\n\n// ── Pending future tracking ─────────────────────────────────\n\n/// A deferred computation spawned as a tokio task, pending resolution\n/// via `ResolveFutures`. Can be a tool execution or an LLM call.\nenum PendingFuture {\n /// Tool action execution.\n Tool {\n handle: tokio::task::JoinHandle>,\n action_name: String,\n call_id: String,\n lease_id: crate::types::capability::LeaseId,\n parameters: serde_json::Value,\n params_summary: Option,\n },\n /// LLM call (llm_query / llm_query_batched / rlm_query).\n Llm {\n handle: tokio::task::JoinHandle<(ExtFunctionResult, TokenUsage)>,\n },\n}\n\n/// Result of preflight checks (lease + policy) for a tool call.\nenum PreflightResult {\n /// Tool approved — lease is consumed, ready to execute.\n Approved(crate::types::capability::CapabilityLease),\n /// Tool denied — return this error to Monty.\n Denied(ExtFunctionResult),\n /// Tool is paused by a gate — interrupt the batch.\n GatePaused(crate::runtime::messaging::ThreadOutcome),\n}\n\n/// Run preflight checks for a tool call: find lease, check policy, consume use.\n#[allow(clippy::too_many_arguments)]\nasync fn preflight_action(\n action_name: &str,\n params: &serde_json::Value,\n thread: &Thread,\n effects: &Arc,\n leases: &LeaseManager,\n policy: &PolicyEngine,\n context: &ThreadExecutionContext,\n capability_policies: &[crate::types::capability::PolicyRule],\n call_id: &str,\n events: &mut Vec,\n) -> PreflightResult {\n let lease = match leases.find_lease_for_action(thread.id, action_name).await {\n Some(l) => l,\n None => {\n events.push(EventKind::ActionFailed {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n error: format!(\"no lease for action '{action_name}'\"),\n params_summary: None,\n });\n return PreflightResult::Denied(ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"no lease for action '{action_name}'\")),\n )));\n }\n };\n\n let action_def = effects\n .available_actions(std::slice::from_ref(&lease))\n .await\n .ok()\n .and_then(|actions| actions.into_iter().find(|a| a.name == action_name));\n\n if let Some(ref action_def) = action_def {\n match policy.evaluate(action_def, &lease, capability_policies) {\n PolicyDecision::Deny { reason } => {\n events.push(EventKind::ActionFailed {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n error: reason.clone(),\n params_summary: None,\n });\n return PreflightResult::Denied(ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"denied: {reason}\")),\n )));\n }\n PolicyDecision::RequireApproval { .. } => {\n events.push(EventKind::ApprovalRequested {\n action_name: action_name.into(),\n call_id: call_id.into(),\n parameters: Some(params.clone()),\n description: None,\n allow_always: None,\n gate_name: None,\n params_summary: crate::types::event::summarize_params(action_name, params),\n });\n return PreflightResult::GatePaused(\n crate::runtime::messaging::ThreadOutcome::GatePaused {\n gate_name: \"approval\".into(),\n action_name: action_name.into(),\n call_id: call_id.into(),\n parameters: params.clone(),\n resume_kind: crate::gate::ResumeKind::Approval { allow_always: true },\n resume_output: None,\n },\n );\n }\n PolicyDecision::Allow => {}\n }\n }\n\n if let Err(e) = leases.consume_use(lease.id).await {\n return PreflightResult::Denied(ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"lease exhausted: {e}\")),\n )));\n }\n\n PreflightResult::Approved(lease)\n}\n\n// ── llm_query() — recursive subagent (RLM 3.5) ─────────────\n\n/// Handle `llm_query(prompt, context)` — single recursive sub-call.\nasync fn handle_llm_query(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n llm: &Arc,\n recursive_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n let prompt = extract_string_arg(args, kwargs, \"prompt\", 0);\n let context_arg = extract_string_arg(args, kwargs, \"context\", 1);\n // `model` must be parsed explicitly — `extract_string_arg` coerces via\n // `monty_to_string`, which turns `MontyObject::None` into the literal\n // string \"None\" and stringifies non-string values, both of which would\n // silently route the call to an invalid model ID. Accept only str or None.\n let model_arg = match extract_optional_string_kwarg(args, kwargs, \"model\", 2) {\n Ok(v) => v,\n Err(err) => return err,\n };\n\n let prompt = match prompt {\n Some(p) => p,\n None => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(\"llm_query() requires a 'prompt' argument\".into()),\n ));\n }\n };\n\n let mut messages = Vec::new();\n if let Some(ctx) = context_arg {\n messages.push(ThreadMessage::system(format!(\n \"You are a sub-agent. Answer concisely based on the context.\\n\\n{ctx}\"\n )));\n } else {\n // Some providers (e.g. OpenAI Codex Responses API) require a system\n // message / instructions field. Always include one.\n messages.push(ThreadMessage::system(\n \"You are a helpful sub-agent. Answer concisely.\",\n ));\n }\n messages.push(ThreadMessage::user(prompt));\n\n let config = LlmCallConfig {\n force_text: true,\n model: model_arg,\n ..LlmCallConfig::default()\n };\n\n match llm.complete(&messages, &[], &config).await {\n Ok(output) => {\n recursive_tokens.input_tokens += output.usage.input_tokens;\n recursive_tokens.output_tokens += output.usage.output_tokens;\n let text = match output.response {\n LlmResponse::Text(t) => t,\n LlmResponse::ActionCalls { content, .. } | LlmResponse::Code { content, .. } => {\n content.unwrap_or_default()\n }\n };\n ExtFunctionResult::Return(MontyObject::String(text))\n }\n Err(e) => ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"llm_query failed: {e}\")),\n )),\n }\n}\n\n/// Handle `llm_query_batched(prompts)` — parallel recursive sub-calls.\n///\n/// Takes a list of prompt strings and dispatches them concurrently.\n/// Returns a list of response strings in the same order.\nasync fn handle_llm_query_batched(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n llm: &Arc,\n recursive_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n // Extract prompts list (first arg or kwarg \"prompts\")\n let prompts_obj = args.first().or_else(|| {\n kwargs.iter().find_map(|(k, v)| {\n if let MontyObject::String(key) = k\n && key == \"prompts\"\n {\n return Some(v);\n }\n None\n })\n });\n\n let prompts: Vec = match prompts_obj {\n Some(MontyObject::List(items)) => items.iter().map(monty_to_string).collect(),\n Some(other) => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(format!(\n \"llm_query_batched() expects a list of prompts, got {other:?}\"\n )),\n ));\n }\n None => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(\"llm_query_batched() requires a 'prompts' argument\".into()),\n ));\n }\n };\n\n // Positional/keyword layout (matches the documented signature\n // `llm_query_batched(prompts, context=None, model=None, models=None)`):\n // arg 0 = prompts (already extracted above)\n // arg 1 = context\n // arg 2 = model\n // arg 3 = models\n // All three of context/model/models can also be passed by keyword.\n let context_arg = match extract_optional_string_kwarg(args, kwargs, \"context\", 1) {\n Ok(v) => v,\n Err(err) => return err,\n };\n\n // Optional model overrides:\n // - `model=\"...\"` applies the same model to every prompt\n // - `models=[...]` is a parallel array (must match prompts length); use\n // this to broadcast the same prompt across a council of models by\n // passing `prompts=[same]*N, models=[m1, m2, ...]`. Within `models`,\n // a `None` slot means \"no override for this prompt\" (the caller\n // opted out of routing for that slot); the singular `model=` kwarg\n // does NOT fill those slots, since mixing the two would be surprising.\n // See note in handle_llm_query: `model` must be parsed explicitly so that\n // `model=None` doesn't become the literal string \"None\".\n let single_model = match extract_optional_string_kwarg(args, kwargs, \"model\", 2) {\n Ok(v) => v,\n Err(err) => return err,\n };\n let models_kwarg = kwargs\n .iter()\n .find_map(|(k, v)| match k {\n MontyObject::String(key) if key == \"models\" => Some(v),\n _ => None,\n })\n .or_else(|| args.get(3));\n\n let models_list: Option>> = match models_kwarg {\n None | Some(MontyObject::None) => None,\n Some(MontyObject::List(items)) => {\n let mut out = Vec::with_capacity(items.len());\n for item in items {\n match item {\n MontyObject::String(s) => out.push(Some(s.clone())),\n MontyObject::None => out.push(None),\n other => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(format!(\n \"llm_query_batched(): models list entries must be str or None, got {other:?}\"\n )),\n ));\n }\n }\n }\n Some(out)\n }\n Some(other) => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(format!(\n \"llm_query_batched(): `models` must be a list of str or None, got {other:?}\"\n )),\n ));\n }\n };\n\n if let Some(ref ms) = models_list\n && ms.len() != prompts.len()\n {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::ValueError,\n Some(format!(\n \"llm_query_batched(): models list length ({}) must match prompts length ({})\",\n ms.len(),\n prompts.len()\n )),\n ));\n }\n\n let mut handles = Vec::with_capacity(prompts.len());\n for (i, prompt) in prompts.iter().enumerate() {\n let llm = Arc::clone(llm);\n let ctx = context_arg.clone();\n let prompt = prompt.clone();\n // If `models=` was provided, each slot is authoritative — `None` means\n // \"no override for this prompt\" and is NOT backfilled from `model=`.\n // Otherwise, fall back to the singular `model=` kwarg (or None).\n let model_override = match models_list.as_ref() {\n Some(ms) => ms[i].clone(),\n None => single_model.clone(),\n };\n let config = LlmCallConfig {\n force_text: true,\n model: model_override,\n ..LlmCallConfig::default()\n };\n handles.push(tokio::spawn(async move {\n let mut messages = Vec::new();\n if let Some(ctx) = ctx {\n messages.push(ThreadMessage::system(format!(\n \"You are a sub-agent. Answer concisely.\\n\\n{ctx}\"\n )));\n } else {\n messages.push(ThreadMessage::system(\n \"You are a helpful sub-agent. Answer concisely.\",\n ));\n }\n messages.push(ThreadMessage::user(prompt));\n llm.complete(&messages, &[], &config).await\n }));\n }\n\n // Collect results\n let mut results = Vec::with_capacity(prompts.len());\n let mut total_input = 0u64;\n let mut total_output = 0u64;\n\n for handle in handles {\n match handle.await {\n Ok(Ok(output)) => {\n total_input += output.usage.input_tokens;\n total_output += output.usage.output_tokens;\n let text = match output.response {\n LlmResponse::Text(t) => t,\n LlmResponse::ActionCalls { content, .. }\n | LlmResponse::Code { content, .. } => content.unwrap_or_default(),\n };\n results.push(MontyObject::String(text));\n }\n Ok(Err(e)) => {\n results.push(MontyObject::String(format!(\"Error: {e}\")));\n }\n Err(e) => {\n results.push(MontyObject::String(format!(\"Error: task failed: {e}\")));\n }\n }\n }\n\n recursive_tokens.input_tokens += total_input;\n recursive_tokens.output_tokens += total_output;\n\n ExtFunctionResult::Return(MontyObject::List(results))\n}\n\n// ── rlm_query() — full recursive sub-agent (RLM 3.5) ─────────\n\n/// Handle `rlm_query(prompt)` — spawn a child CodeAct thread with its own\n/// execution loop, tools, and iteration budget.\n///\n/// Unlike `llm_query()` (single-shot LLM call), `rlm_query()` creates a\n/// child thread with full CodeAct capabilities. The child inherits the\n/// parent's remaining budget and tool access.\n#[allow(clippy::too_many_arguments)]\nasync fn handle_rlm_query(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n parent_thread: &Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &LeaseManager,\n policy: &PolicyEngine,\n recursive_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n let prompt = extract_string_arg(args, kwargs, \"prompt\", 0);\n let prompt = match prompt {\n Some(p) => p,\n None => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(\"rlm_query() requires a 'prompt' argument\".into()),\n ));\n }\n };\n\n // Depth check — refuse if at max recursion depth\n let current_depth = parent_thread.config.depth;\n let max_depth = parent_thread.config.max_depth;\n if current_depth >= max_depth {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\n \"rlm_query() depth limit reached: depth {current_depth} >= max {max_depth}\"\n )),\n ));\n }\n\n // Build child thread with inherited budget\n let child_config = crate::types::thread::ThreadConfig {\n max_iterations: parent_thread.config.max_iterations.min(20), // cap child iterations\n enable_tool_intent_nudge: false,\n max_tokens_total: parent_thread\n .config\n .max_tokens_total\n .map(|max| max.saturating_sub(parent_thread.total_tokens_used)),\n max_budget_usd: parent_thread\n .config\n .max_budget_usd\n .map(|max| (max - parent_thread.total_cost_usd).max(0.0)),\n max_duration: parent_thread.config.max_duration,\n depth: current_depth + 1,\n max_depth,\n ..crate::types::thread::ThreadConfig::default()\n };\n\n let mut child_thread = crate::types::thread::Thread::new(\n &prompt,\n crate::types::thread::ThreadType::Research,\n parent_thread.project_id,\n &parent_thread.user_id,\n child_config,\n )\n .with_parent(parent_thread.id);\n\n // Add the prompt as a user message\n child_thread.add_message(ThreadMessage::user(&prompt));\n\n // Create signal channel and child's lease manager\n let (_tx, rx) = crate::runtime::messaging::signal_channel(8);\n let child_leases = Arc::new(LeaseManager::new());\n\n // Grant the child the same leases as the parent (in the child's manager)\n let parent_leases = leases.active_for_thread(parent_thread.id).await;\n let now = chrono::Utc::now();\n for parent_lease in &parent_leases {\n // Convert parent's expires_at to remaining duration\n let remaining_duration = parent_lease\n .expires_at\n .and_then(|exp| (exp - now).to_std().ok())\n .map(|d| chrono::Duration::from_std(d).unwrap_or(chrono::Duration::hours(1)));\n let lease = match child_leases\n .grant(\n child_thread.id,\n &parent_lease.capability_name,\n parent_lease.granted_actions.clone(),\n remaining_duration,\n parent_lease.max_uses,\n )\n .await\n {\n Ok(l) => l,\n Err(e) => {\n debug!(error = %e, \"rlm_query: skipping invalid lease for child thread\");\n continue;\n }\n };\n child_thread.capability_leases.push(lease.id);\n }\n let mut child_policy_engine = PolicyEngine::new();\n // Copy denied effects from parent policy\n for effect in &policy.denied_effects {\n child_policy_engine.deny_effect(*effect);\n }\n let child_policy = Arc::new(child_policy_engine);\n\n let mut child_loop = crate::executor::ExecutionLoop::new(\n child_thread,\n Arc::clone(llm),\n Arc::clone(effects),\n child_leases,\n child_policy,\n rx,\n \"rlm_child\".to_string(),\n );\n\n debug!(\n parent_thread = %parent_thread.id,\n depth = current_depth + 1,\n prompt_len = prompt.len(),\n \"rlm_query: spawning child CodeAct thread\"\n );\n\n // Run the child loop (Box::pin to avoid infinite future size from recursion)\n match Box::pin(child_loop.run()).await {\n Ok(outcome) => {\n // Track child's token usage\n recursive_tokens.input_tokens += child_loop.thread.total_tokens_used;\n recursive_tokens.cost_usd += child_loop.thread.total_cost_usd;\n\n let response = match outcome {\n crate::runtime::messaging::ThreadOutcome::Completed { response } => {\n response.unwrap_or_default()\n }\n crate::runtime::messaging::ThreadOutcome::Failed { error } => {\n format!(\"rlm_query child failed: {error}\")\n }\n crate::runtime::messaging::ThreadOutcome::MaxIterations => {\n \"rlm_query child reached max iterations\".to_string()\n }\n _ => String::new(),\n };\n\n ExtFunctionResult::Return(MontyObject::String(response))\n }\n Err(e) => ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"rlm_query failed: {e}\")),\n )),\n }\n}\n\n// ── Standalone async handlers (for tokio::spawn) ────────────\n\n/// `llm_query()` — standalone version that returns `(ExtFunctionResult, TokenUsage)`.\nasync fn handle_llm_query_standalone(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n llm: &Arc,\n) -> (ExtFunctionResult, TokenUsage) {\n let mut tokens = TokenUsage::default();\n let result = handle_llm_query(args, kwargs, llm, &mut tokens).await;\n (result, tokens)\n}\n\n/// `llm_query_batched()` — standalone version.\nasync fn handle_llm_query_batched_standalone(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n llm: &Arc,\n) -> (ExtFunctionResult, TokenUsage) {\n let mut tokens = TokenUsage::default();\n let result = handle_llm_query_batched(args, kwargs, llm, &mut tokens).await;\n (result, tokens)\n}\n\n// ── Future resolution helpers ───────────────────────────────\n\n/// Resolve a pending tool execution future.\n#[allow(clippy::too_many_arguments)]\nasync fn resolve_tool_future(\n handle: tokio::task::JoinHandle>,\n action_name: &str,\n call_id: &str,\n lease_id: crate::types::capability::LeaseId,\n parameters: serde_json::Value,\n params_summary: Option,\n leases: &LeaseManager,\n context: &ThreadExecutionContext,\n action_results: &mut Vec,\n events: &mut Vec,\n) -> ExtFunctionResult {\n match handle.await {\n Ok(Ok(result)) => {\n // If the effect adapter wrapped a tool error as an Ok(ActionResult)\n // with is_error=true (current convention in\n // `EffectBridgeAdapter::execute_action_internal`), surface it as\n // ActionFailed so traces, observers, and approval flows see the\n // failure correctly. Without this, every wrapped error looked like\n // a successful tool call to downstream consumers.\n if result.is_error {\n let error_msg = result\n .output\n .get(\"error\")\n .and_then(|v| v.as_str())\n .map(String::from)\n .unwrap_or_else(|| result.output.to_string());\n events.push(EventKind::ActionFailed {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n error: error_msg,\n params_summary,\n });\n } else {\n events.push(EventKind::ActionExecuted {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n duration_ms: result.duration.as_millis() as u64,\n params_summary,\n });\n }\n let monty_val = json_to_monty(&result.output);\n action_results.push(result);\n ExtFunctionResult::Return(monty_val)\n }\n Ok(Err(EngineError::GatePaused {\n gate_name,\n action_name,\n call_id,\n resume_kind,\n ..\n })) => {\n let _ = leases.refund_use(lease_id).await;\n events.push(EventKind::ApprovalRequested {\n action_name,\n call_id,\n parameters: Some(parameters),\n description: None,\n allow_always: match *resume_kind {\n crate::gate::ResumeKind::Approval { allow_always } => Some(allow_always),\n _ => None,\n },\n gate_name: Some(gate_name.clone()),\n params_summary,\n });\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"execution paused by gate '{gate_name}'\")),\n ))\n }\n Ok(Err(e)) => {\n events.push(EventKind::ActionFailed {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n error: e.to_string(),\n params_summary,\n });\n action_results.push(ActionResult {\n call_id: call_id.into(),\n action_name: action_name.into(),\n output: serde_json::json!({\"error\": e.to_string()}),\n is_error: true,\n duration: Duration::ZERO,\n });\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(e.to_string()),\n ))\n }\n Err(e) => {\n debug!(\"async tool task panicked: {e}\");\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"tool execution panicked: {e}\")),\n ))\n }\n }\n}\n\n/// Resolve a pending LLM call future, accumulating token usage.\nasync fn resolve_llm_future(\n handle: tokio::task::JoinHandle<(ExtFunctionResult, TokenUsage)>,\n recursive_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n match handle.await {\n Ok((result, tokens)) => {\n recursive_tokens.input_tokens += tokens.input_tokens;\n recursive_tokens.output_tokens += tokens.output_tokens;\n result\n }\n Err(e) => {\n debug!(\"async LLM task panicked: {e}\");\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"LLM call panicked: {e}\")),\n ))\n }\n }\n}\n\n// ── Helpers ─────────────────────────────────────────────────\n\nfn extract_string_arg(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n name: &str,\n position: usize,\n) -> Option {\n for (k, v) in kwargs {\n if let MontyObject::String(key) = k\n && key == name\n {\n return Some(monty_to_string(v));\n }\n }\n args.get(position).map(monty_to_string)\n}\n\n/// Strict optional-string extractor for arguments where silent coercion is\n/// dangerous (e.g. `model=` — passing the wrong type should NOT become an\n/// unintended model ID). Returns:\n/// - `Ok(None)` when the argument is missing or explicitly `None`\n/// - `Ok(Some(s))` when the argument is a string\n/// - `Err(TypeError)` for any other type\nfn extract_optional_string_kwarg(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n name: &str,\n position: usize,\n) -> Result, ExtFunctionResult> {\n let raw = kwargs\n .iter()\n .find_map(|(k, v)| match k {\n MontyObject::String(key) if key == name => Some(v),\n _ => None,\n })\n .or_else(|| args.get(position));\n\n match raw {\n None | Some(MontyObject::None) => Ok(None),\n Some(MontyObject::String(s)) => Ok(Some(s.clone())),\n Some(other) => Err(ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(format!(\"`{name}` must be a string or None, got {other:?}\")),\n ))),\n }\n}\n\npub(crate) fn monty_to_string(obj: &MontyObject) -> String {\n match obj {\n MontyObject::String(s) => s.clone(),\n MontyObject::None => \"None\".into(),\n MontyObject::Bool(b) => b.to_string(),\n MontyObject::Int(i) => i.to_string(),\n MontyObject::Float(f) => f.to_string(),\n other => {\n serde_json::to_string(&monty_to_json(other)).unwrap_or_else(|_| format!(\"{other:?}\"))\n }\n }\n}\n\n// Dispatch logic moved to orchestrator.rs (__execute_action__ handler).\n// GatePaused is handled via EngineError → JSON in orchestrator.rs.\n// ── MontyObject ↔ JSON ──────────────────────────────────────\n\npub(crate) fn monty_to_json(obj: &MontyObject) -> serde_json::Value {\n match obj {\n MontyObject::None => serde_json::Value::Null,\n MontyObject::Bool(b) => serde_json::Value::Bool(*b),\n MontyObject::Int(i) => serde_json::json!(i),\n MontyObject::BigInt(i) => serde_json::Value::String(i.to_string()),\n MontyObject::Float(f) => serde_json::json!(f),\n MontyObject::String(s) => serde_json::Value::String(s.clone()),\n MontyObject::List(items) | MontyObject::Tuple(items) => {\n serde_json::Value::Array(items.iter().map(monty_to_json).collect())\n }\n MontyObject::Dict(pairs) => {\n let map: serde_json::Map = pairs\n .into_iter()\n .map(|(k, v)| {\n let key = match k {\n MontyObject::String(s) => s.clone(),\n other => format!(\"{other:?}\"),\n };\n (key, monty_to_json(v))\n })\n .collect();\n serde_json::Value::Object(map)\n }\n MontyObject::Set(items) | MontyObject::FrozenSet(items) => {\n serde_json::Value::Array(items.iter().map(monty_to_json).collect())\n }\n MontyObject::Bytes(b) => {\n serde_json::Value::String(b.iter().map(|byte| format!(\"{byte:02x}\")).collect())\n }\n other => serde_json::Value::String(format!(\"{other:?}\")),\n }\n}\n\npub(crate) fn json_to_monty(val: &serde_json::Value) -> MontyObject {\n match val {\n serde_json::Value::Null => MontyObject::None,\n serde_json::Value::Bool(b) => MontyObject::Bool(*b),\n serde_json::Value::Number(n) => {\n if let Some(i) = n.as_i64() {\n MontyObject::Int(i)\n } else if let Some(f) = n.as_f64() {\n MontyObject::Float(f)\n } else {\n MontyObject::String(n.to_string())\n }\n }\n serde_json::Value::String(s) => MontyObject::String(s.clone()),\n serde_json::Value::Array(arr) => MontyObject::List(arr.iter().map(json_to_monty).collect()),\n serde_json::Value::Object(map) => MontyObject::dict(\n map.iter()\n .map(|(k, v)| (MontyObject::String(k.clone()), json_to_monty(v)))\n .collect::>(),\n ),\n }\n}\n\nfn monty_args_to_json(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n) -> serde_json::Value {\n let mut map = serde_json::Map::new();\n if !args.is_empty() {\n map.insert(\n \"_args\".into(),\n serde_json::Value::Array(args.iter().map(monty_to_json).collect()),\n );\n }\n for (k, v) in kwargs {\n let key = match k {\n MontyObject::String(s) => s.clone(),\n other => format!(\"{other:?}\"),\n };\n map.insert(key, monty_to_json(v));\n }\n serde_json::Value::Object(map)\n}\n\n#[cfg(test)]\nmod tests {\n use super::*;\n use crate::capability::lease::LeaseManager;\n use crate::capability::policy::PolicyEngine;\n use crate::traits::effect::ThreadExecutionContext;\n use crate::types::capability::{ActionDef, CapabilityLease, EffectType, GrantedActions};\n use crate::types::project::ProjectId;\n use crate::types::step::{ActionResult, StepId};\n use crate::types::thread::{Thread, ThreadConfig, ThreadType};\n use std::sync::Mutex;\n\n /// Truncate a string to at most `max_bytes`, snapping to a UTF-8 char\n /// boundary so assertion messages never panic on multibyte output.\n fn truncate_for_assert(s: &str, max_bytes: usize) -> &str {\n if s.len() <= max_bytes {\n return s;\n }\n let mut end = max_bytes;\n while end > 0 && !s.is_char_boundary(end) {\n end -= 1;\n }\n &s[..end] // safety: end is walked down to a valid char boundary above\n }\n\n struct MockEffects {\n results: Mutex>>,\n actions: Vec,\n }\n\n impl MockEffects {\n fn new(actions: Vec, results: Vec>) -> Self {\n Self {\n results: Mutex::new(results),\n actions,\n }\n }\n }\n\n #[async_trait::async_trait]\n impl EffectExecutor for MockEffects {\n async fn execute_action(\n &self,\n name: &str,\n _params: serde_json::Value,\n _lease: &CapabilityLease,\n _ctx: &ThreadExecutionContext,\n ) -> Result {\n let mut results = self.results.lock().unwrap();\n if results.is_empty() {\n Ok(ActionResult {\n call_id: String::new(),\n action_name: name.into(),\n output: serde_json::json!({\"result\": \"ok\"}),\n is_error: false,\n duration: Duration::from_millis(1),\n })\n } else {\n results.remove(0)\n }\n }\n\n async fn available_actions(\n &self,\n _leases: &[CapabilityLease],\n ) -> Result, EngineError> {\n Ok(self.actions.clone())\n }\n }\n\n fn test_action(name: &str) -> ActionDef {\n ActionDef {\n name: name.into(),\n description: \"Test tool\".into(),\n parameters_schema: serde_json::json!({\"type\": \"object\"}),\n effects: vec![EffectType::ReadLocal],\n requires_approval: false,\n }\n }\n\n fn make_test_thread() -> Thread {\n Thread::new(\n \"test goal\",\n ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n ThreadConfig::default(),\n )\n }\n\n fn make_exec_context(thread: &Thread) -> ThreadExecutionContext {\n ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: \"test\".into(),\n step_id: StepId::new(),\n current_call_id: None,\n source_channel: None,\n user_timezone: None,\n }\n }\n\n /// Stub LLM that always returns text \"stub\". Only used so execute_code\n /// doesn't need a real LLM — our tests exercise tool dispatch, not LLM calls.\n struct StubLlm;\n\n #[async_trait::async_trait]\n impl crate::traits::llm::LlmBackend for StubLlm {\n fn model_name(&self) -> &str {\n \"stub\"\n }\n\n async fn complete(\n &self,\n _messages: &[crate::types::message::ThreadMessage],\n _actions: &[ActionDef],\n _config: &crate::traits::llm::LlmCallConfig,\n ) -> Result {\n Ok(crate::traits::llm::LlmOutput {\n response: crate::types::step::LlmResponse::Text(\"stub\".into()),\n usage: crate::types::step::TokenUsage::default(),\n })\n }\n }\n\n async fn run_code(\n code: &str,\n effects: Arc,\n thread: &Thread,\n ) -> Result {\n let leases = LeaseManager::new();\n let policy = PolicyEngine::new();\n let ctx = make_exec_context(thread);\n\n // Grant a wildcard lease\n leases\n .grant(thread.id, \"tools\", GrantedActions::All, None, None)\n .await\n .unwrap();\n\n execute_code(\n code,\n thread,\n &(Arc::new(StubLlm) as Arc),\n &effects,\n &leases,\n &policy,\n &ctx,\n &[],\n &serde_json::json!({}),\n )\n .await\n }\n\n // ── Single await tool call ──────────────────────────────\n\n #[tokio::test]\n async fn single_await_tool_call() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"echo\")],\n vec![Ok(ActionResult {\n call_id: String::new(),\n action_name: \"echo\".into(),\n output: serde_json::json!(\"hello world\"),\n is_error: false,\n duration: Duration::from_millis(1),\n })],\n ));\n\n let code = r#\"\nresult = await echo(message=\"hello\")\nFINAL(str(result))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(\n result.final_answer.is_some(),\n \"should have final answer, stdout: {}\",\n result.stdout\n );\n assert!(\n result.failure.is_none(),\n \"should not error, stdout: {}\",\n result.stdout\n );\n assert_eq!(result.action_results.len(), 1);\n }\n\n // ── asyncio.gather parallel execution ───────────────────\n\n #[tokio::test]\n async fn asyncio_gather_two_tools() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"tool_a\"), test_action(\"tool_b\")],\n vec![\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"tool_a\".into(),\n output: serde_json::json!(10),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"tool_b\".into(),\n output: serde_json::json!(32),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n ],\n ));\n\n let code = r#\"\nimport asyncio\na, b = await asyncio.gather(tool_a(), tool_b())\nFINAL(str(a + b))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(\n result.final_answer.is_some(),\n \"should have final answer, stdout: {}\",\n result.stdout\n );\n assert_eq!(\n result.final_answer.as_deref(),\n Some(\"42\"),\n \"10 + 32 = 42, got: {:?}, stdout: {}\",\n result.final_answer,\n result.stdout\n );\n assert_eq!(result.action_results.len(), 2);\n assert!(result.failure.is_none());\n }\n\n // ── asyncio.gather three tools ──────────────────────────\n\n #[tokio::test]\n async fn asyncio_gather_three_tools() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![\n test_action(\"web_search\"),\n test_action(\"http\"),\n test_action(\"memory_search\"),\n ],\n vec![\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"web_search\".into(),\n output: serde_json::json!(\"search results\"),\n is_error: false,\n duration: Duration::from_millis(50),\n }),\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"http\".into(),\n output: serde_json::json!(\"page content\"),\n is_error: false,\n duration: Duration::from_millis(100),\n }),\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"memory_search\".into(),\n output: serde_json::json!(\"memories\"),\n is_error: false,\n duration: Duration::from_millis(25),\n }),\n ],\n ));\n\n let code = r#\"\nimport asyncio\ns, h, m = await asyncio.gather(\n web_search(query=\"test\"),\n http(url=\"https://example.com\"),\n memory_search(query=\"prior\"),\n)\nFINAL(str(s) + \"|\" + str(h) + \"|\" + str(m))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.action_results.len(), 3);\n let answer = result.final_answer.unwrap();\n assert!(answer.contains(\"search results\"), \"got: {answer}\");\n assert!(answer.contains(\"page content\"), \"got: {answer}\");\n assert!(answer.contains(\"memories\"), \"got: {answer}\");\n }\n\n // ── Data-dependent chain (sequential await) ─────────────\n\n #[tokio::test]\n async fn sequential_dependent_calls() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"step1\"), test_action(\"step2\")],\n vec![\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"step1\".into(),\n output: serde_json::json!(\"intermediate\"),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"step2\".into(),\n output: serde_json::json!(\"final\"),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n ],\n ));\n\n let code = r#\"\na = await step1()\nb = await step2(input=a)\nFINAL(str(b))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.action_results.len(), 2);\n assert_eq!(result.final_answer.as_deref(), Some(\"final\"));\n }\n\n // ── Error in one gathered tool ──────────────────────────\n\n #[tokio::test]\n async fn gather_with_error_propagates() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"good\"), test_action(\"bad\")],\n vec![\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"good\".into(),\n output: serde_json::json!(\"ok\"),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n Err(EngineError::Effect {\n reason: \"tool exploded\".into(),\n }),\n ],\n ));\n\n let code = r#\"\nimport asyncio\na, b = await asyncio.gather(good(), bad())\nFINAL(\"should not reach\")\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n // Error in gather propagates as exception — code should error\n assert!(\n result.failure.is_some(),\n \"should have error, stdout: {}\",\n result.stdout\n );\n assert!(\n result.final_answer.is_none()\n || result.final_answer.as_deref() != Some(\"should not reach\")\n );\n }\n\n // ── Tool with no lease (denied in preflight) ────────────\n\n #[tokio::test]\n async fn denied_tool_raises_exception() {\n let thread = make_test_thread();\n // No actions registered — tool has no lease\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n\n let code = r#\"\ntry:\n result = await unknown_tool()\n FINAL(\"should not reach\")\nexcept:\n FINAL(\"caught error\")\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n // Tool not found raises NameError before we even get to dispatch\n assert!(result.final_answer.is_some(), \"stdout: {}\", result.stdout);\n }\n\n // ── FINAL works without await ───────────────────────────\n\n #[tokio::test]\n async fn final_is_sync() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n\n let code = r#\"\nFINAL(\"hello from sync\")\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert_eq!(result.final_answer.as_deref(), Some(\"hello from sync\"));\n assert!(result.failure.is_none());\n }\n\n // ── globals() still works ───────────────────────────────\n\n #[tokio::test]\n async fn globals_returns_known_tools() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"web_search\"), test_action(\"http\")],\n vec![],\n ));\n\n let code = r#\"\ng = globals()\nhas_search = \"web_search\" in g\nhas_http = \"http\" in g\nFINAL(str(has_search) + \"|\" + str(has_http))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.final_answer.as_deref(), Some(\"True|True\"));\n }\n\n // ── Empty gather ────────────────────────────────────────\n\n #[tokio::test]\n async fn empty_gather() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n\n let code = r#\"\nimport asyncio\nresults = await asyncio.gather()\nFINAL(str(len(results)))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.final_answer.as_deref(), Some(\"0\"));\n }\n\n // ── Single-item gather ──────────────────────────────────\n\n #[tokio::test]\n async fn single_item_gather() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"echo\")],\n vec![Ok(ActionResult {\n call_id: String::new(),\n action_name: \"echo\".into(),\n output: serde_json::json!(\"gathered\"),\n is_error: false,\n duration: Duration::from_millis(1),\n })],\n ));\n\n let code = r#\"\nimport asyncio\nresults = await asyncio.gather(echo())\nFINAL(str(results[0]))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.final_answer.as_deref(), Some(\"gathered\"));\n assert_eq!(result.action_results.len(), 1);\n }\n\n // ── Sandbox security negative tests ────────────────────────\n\n /// OS-level operations must be denied or restricted by the Monty VM.\n #[tokio::test]\n async fn sandbox_denies_os_operations() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n // Try to import os and call os.system — should fail\n let code = r#\"\ntry:\n import os\n os.system(\"echo pwned\")\n FINAL(\"ESCAPED: os.system ran\")\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"os.system should be blocked, got: {answer}\",\n );\n }\n\n /// Resource limits must be enforced — infinite loops should be terminated.\n #[tokio::test]\n async fn sandbox_enforces_resource_limits() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n // Infinite allocation loop — should hit allocation or memory limit\n let code = r#\"\ndata = []\nwhile True:\n data.append(\"x\" * 10000)\n\"#;\n let result = run_code(code, effects, &thread).await;\n // Either returns an error or the stdout contains an error message —\n // the key assertion is that it DOES NOT run forever.\n if let Ok(r) = result {\n assert!(\n r.failure.is_some() || r.stdout.contains(\"Error\") || r.stdout.contains(\"limit\"),\n \"resource limit should terminate infinite loop, got stdout: {}\",\n truncate_for_assert(&r.stdout, 500),\n );\n }\n // Err(_) is also acceptable — means the VM was killed by resource limits\n }\n\n /// Python `import` of system modules must be restricted.\n #[tokio::test]\n async fn sandbox_restricts_imports() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n // Try to import subprocess — should fail\n let code = r#\"\ntry:\n import subprocess\n result = subprocess.run([\"echo\", \"escaped\"], capture_output=True, text=True)\n FINAL(\"ESCAPED: \" + result.stdout)\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"subprocess import should be blocked, got: {answer}\",\n );\n }\n\n /// File system access via open() must be blocked.\n #[tokio::test]\n async fn sandbox_denies_file_access() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n let code = r#\"\ntry:\n f = open(\"/etc/passwd\", \"r\")\n content = f.read()\n f.close()\n FINAL(\"ESCAPED: \" + content[:50])\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"open() should be blocked, got: {answer}\",\n );\n }\n\n /// Network access via socket must be blocked.\n #[tokio::test]\n async fn sandbox_denies_socket_access() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n let code = r#\"\ntry:\n import socket\n s = socket.socket(socket.AF_INET, socket.SOCK_STREAM)\n s.connect((\"127.0.0.1\", 80))\n FINAL(\"ESCAPED: connected\")\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"socket access should be blocked, got: {answer}\",\n );\n }\n\n /// Calls to tools not covered by the lease must be denied.\n #[tokio::test]\n async fn sandbox_unlicensed_tool_denied() {\n let effects: Arc =\n Arc::new(MockEffects::new(vec![test_action(\"allowed_tool\")], vec![]));\n let thread = make_test_thread();\n let leases = LeaseManager::new();\n let policy = PolicyEngine::new();\n let ctx = make_exec_context(&thread);\n\n // Grant a restricted lease — only \"allowed_tool\" is permitted.\n leases\n .grant(\n thread.id,\n \"tools\",\n GrantedActions::Specific(vec![\"allowed_tool\".into()]),\n None,\n None,\n )\n .await\n .unwrap();\n\n let code = r#\"\ntry:\n result = await secret_admin_tool(data=\"pwn\")\n FINAL(\"ESCAPED: \" + str(result))\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = execute_code(\n code,\n &thread,\n &(Arc::new(StubLlm) as Arc),\n &effects,\n &leases,\n &policy,\n &ctx,\n &[],\n &serde_json::json!({}),\n )\n .await\n .unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"unlicensed tool should be denied by preflight, got: {answer}\",\n );\n }\n\n /// CPU-bound infinite loops must be terminated by allocation/duration limits.\n #[tokio::test]\n async fn sandbox_enforces_cpu_limits() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n // Tight CPU-bound loop (no allocations to trip allocation limit)\n let code = r#\"\nx = 0\nwhile True:\n x += 1\n\"#;\n let result = run_code(code, effects, &thread).await;\n // Must terminate — either via error or resource limit\n if let Ok(r) = result {\n assert!(\n r.failure.is_some() || r.stdout.contains(\"Error\") || r.stdout.contains(\"limit\"),\n \"cpu-bound loop should be terminated, stdout: {}\",\n truncate_for_assert(&r.stdout, 500),\n );\n }\n // Err(_) is also acceptable — means the VM was killed by resource limits\n }\n\n /// FINAL() must capture the answer from the code.\n #[tokio::test]\n async fn sandbox_final_captures_answer() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n let code = r#\"\nx = 2 + 3\nFINAL(str(x))\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n assert_eq!(\n result.final_answer.as_deref(),\n Some(\"5\"),\n \"FINAL should capture the computed answer\"\n );\n }\n\n /// Syntax errors flow back as errors, not panics.\n #[tokio::test]\n async fn sandbox_handles_syntax_error() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n let code = \"def broken(\\nFINAL('nope')\";\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_some(), \"syntax error should set failure\");\n assert!(\n result.stdout.contains(\"SyntaxError\") || result.stdout.contains(\"Error\"),\n \"should contain SyntaxError, got: {}\",\n result.stdout,\n );\n }\n\n // ── llm_query model parameter plumbing ─────────────────────\n\n /// LLM backend that records every call's model + prompt for assertions.\n struct CapturingLlm {\n calls: tokio::sync::Mutex, String)>>,\n }\n\n impl CapturingLlm {\n fn new() -> Self {\n Self {\n calls: tokio::sync::Mutex::new(Vec::new()),\n }\n }\n }\n\n #[async_trait::async_trait]\n impl crate::traits::llm::LlmBackend for CapturingLlm {\n fn model_name(&self) -> &str {\n \"capturing\"\n }\n\n async fn complete(\n &self,\n messages: &[crate::types::message::ThreadMessage],\n _actions: &[ActionDef],\n config: &crate::traits::llm::LlmCallConfig,\n ) -> Result {\n let user_prompt = messages\n .iter()\n .rev()\n .find(|m| matches!(m.role, crate::types::message::MessageRole::User))\n .map(|m| m.content.clone())\n .unwrap_or_default();\n self.calls\n .lock()\n .await\n .push((config.model.clone(), user_prompt.clone()));\n Ok(crate::traits::llm::LlmOutput {\n response: crate::types::step::LlmResponse::Text(format!(\n \"ack:{}:{user_prompt}\",\n config.model.as_deref().unwrap_or(\"default\")\n )),\n usage: crate::types::step::TokenUsage::default(),\n })\n }\n }\n\n #[tokio::test]\n async fn llm_query_forwards_model_kwarg() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query(\n &[],\n &[\n (\n MontyObject::String(\"prompt\".into()),\n MontyObject::String(\"what is 2+2?\".into()),\n ),\n (\n MontyObject::String(\"model\".into()),\n MontyObject::String(\"gpt-4o\".into()),\n ),\n ],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n match result {\n ExtFunctionResult::Return(MontyObject::String(s)) => {\n assert!(s.contains(\"gpt-4o\"), \"got: {s}\");\n }\n other => panic!(\"expected string return, got {other:?}\"),\n }\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].0.as_deref(), Some(\"gpt-4o\"));\n assert_eq!(calls[0].1, \"what is 2+2?\");\n }\n\n #[tokio::test]\n async fn llm_query_without_model_passes_none() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let _ = handle_llm_query(\n &[MontyObject::String(\"hello\".into())],\n &[],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].0, None);\n }\n\n #[tokio::test]\n async fn llm_query_batched_broadcasts_with_models_list() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"Q\".into()),\n MontyObject::String(\"Q\".into()),\n MontyObject::String(\"Q\".into()),\n ]);\n let models = MontyObject::List(vec![\n MontyObject::String(\"gpt-4o\".into()),\n MontyObject::String(\"claude-sonnet-4-20250514\".into()),\n MontyObject::String(\"llama-3.1-70b-instruct\".into()),\n ]);\n let result = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"models\".into()), models)],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n match result {\n ExtFunctionResult::Return(MontyObject::List(items)) => {\n assert_eq!(items.len(), 3);\n }\n other => panic!(\"expected list return, got {other:?}\"),\n }\n\n let mut calls = llm.calls.lock().await;\n calls.sort_by(|a, b| a.0.cmp(&b.0));\n assert_eq!(calls.len(), 3);\n assert_eq!(calls[0].0.as_deref(), Some(\"claude-sonnet-4-20250514\"));\n assert_eq!(calls[1].0.as_deref(), Some(\"gpt-4o\"));\n assert_eq!(calls[2].0.as_deref(), Some(\"llama-3.1-70b-instruct\"));\n }\n\n #[tokio::test]\n async fn llm_query_batched_single_model_applies_to_all() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n let _ = handle_llm_query_batched(\n &[prompts],\n &[(\n MontyObject::String(\"model\".into()),\n MontyObject::String(\"gpt-4o\".into()),\n )],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 2);\n assert!(calls.iter().all(|(m, _)| m.as_deref() == Some(\"gpt-4o\")));\n }\n\n #[tokio::test]\n async fn llm_query_model_none_kwarg_is_no_override_not_literal_none_string() {\n // Regression: `extract_string_arg` would have coerced\n // MontyObject::None to the literal string \"None\", silently routing\n // every model=None call to an invalid model ID. Must stay None.\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let _ = handle_llm_query(\n &[],\n &[\n (\n MontyObject::String(\"prompt\".into()),\n MontyObject::String(\"hi\".into()),\n ),\n (MontyObject::String(\"model\".into()), MontyObject::None),\n ],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].0, None);\n }\n\n #[tokio::test]\n async fn llm_query_rejects_non_string_model_kwarg() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query(\n &[],\n &[\n (\n MontyObject::String(\"prompt\".into()),\n MontyObject::String(\"hi\".into()),\n ),\n (MontyObject::String(\"model\".into()), MontyObject::Int(42)),\n ],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Error(_)));\n assert!(llm.calls.lock().await.is_empty());\n }\n\n #[tokio::test]\n async fn llm_query_batched_single_model_none_kwarg_is_no_override() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n let _ = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"model\".into()), MontyObject::None)],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 2);\n assert!(calls.iter().all(|(m, _)| m.is_none()));\n }\n\n #[tokio::test]\n async fn llm_query_batched_honors_positional_context_and_model() {\n // Regression: `context`, `model`, and `models` used to be kwarg-only.\n // A positional call matching the documented signature\n // `llm_query_batched(prompts, context=None, model=None, models=None)`\n // silently dropped the model, violating the preamble.\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query_batched(\n &[\n MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]),\n MontyObject::String(\"shared context\".into()), // position 1: context\n MontyObject::String(\"gpt-4o\".into()), // position 2: model\n ],\n &[],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n match result {\n ExtFunctionResult::Return(MontyObject::List(items)) => assert_eq!(items.len(), 2),\n other => panic!(\"expected list return, got {other:?}\"),\n }\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 2);\n assert!(calls.iter().all(|(m, _)| m.as_deref() == Some(\"gpt-4o\")));\n }\n\n #[tokio::test]\n async fn llm_query_batched_honors_positional_models_list() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query_batched(\n &[\n MontyObject::List(vec![\n MontyObject::String(\"q\".into()),\n MontyObject::String(\"q\".into()),\n ]),\n MontyObject::None, // position 1: context = None\n MontyObject::None, // position 2: model = None\n MontyObject::List(vec![\n // position 3: models\n MontyObject::String(\"gpt-4o\".into()),\n MontyObject::String(\"claude-sonnet-4-6\".into()),\n ]),\n ],\n &[],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Return(_)));\n let mut calls = llm.calls.lock().await;\n calls.sort_by(|a, b| a.0.cmp(&b.0));\n assert_eq!(calls.len(), 2);\n assert_eq!(calls[0].0.as_deref(), Some(\"claude-sonnet-4-6\"));\n assert_eq!(calls[1].0.as_deref(), Some(\"gpt-4o\"));\n }\n\n #[tokio::test]\n async fn llm_query_batched_positional_none_for_models_is_no_override() {\n // `llm_query_batched(prompts, None, None, None)` should run with no\n // model overrides, not error on the positional None at slot 3.\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query_batched(\n &[\n MontyObject::List(vec![MontyObject::String(\"a\".into())]),\n MontyObject::None,\n MontyObject::None,\n MontyObject::None,\n ],\n &[],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Return(_)));\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].0, None);\n }\n\n #[tokio::test]\n async fn llm_query_batched_rejects_non_string_single_model() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![MontyObject::String(\"a\".into())]);\n let result = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"model\".into()), MontyObject::Int(7))],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Error(_)));\n assert!(llm.calls.lock().await.is_empty());\n }\n\n #[tokio::test]\n async fn llm_query_batched_rejects_non_string_models_entries() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n // Integers in the models list should fail loudly, not be coerced to \"1\"/\"2\".\n let models = MontyObject::List(vec![MontyObject::Int(1), MontyObject::Int(2)]);\n let result = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"models\".into()), models)],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Error(_)));\n assert!(llm.calls.lock().await.is_empty());\n }\n\n #[tokio::test]\n async fn llm_query_batched_none_in_models_list_does_not_backfill_from_model_kwarg() {\n // Regression: when `models=[None, \"gpt-4o\"]` and `model=\"claude-...\"`\n // are both passed, the None slot must NOT be backfilled by the\n // singular `model=` kwarg. Each slot is authoritative.\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n let models = MontyObject::List(vec![\n MontyObject::None,\n MontyObject::String(\"gpt-4o\".into()),\n ]);\n let _ = handle_llm_query_batched(\n &[prompts],\n &[\n (MontyObject::String(\"models\".into()), models),\n (\n MontyObject::String(\"model\".into()),\n MontyObject::String(\"claude-sonnet-4-20250514\".into()),\n ),\n ],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 2);\n // Slot 0 was None — must remain None, not become \"claude-sonnet-4-20250514\".\n let slot_a = calls.iter().find(|(_, p)| p == \"a\").expect(\"call for a\");\n let slot_b = calls.iter().find(|(_, p)| p == \"b\").expect(\"call for b\");\n assert_eq!(slot_a.0, None);\n assert_eq!(slot_b.0.as_deref(), Some(\"gpt-4o\"));\n }\n\n #[tokio::test]\n async fn llm_query_batched_models_length_mismatch_errors() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n let models = MontyObject::List(vec![MontyObject::String(\"only-one\".into())]);\n let result = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"models\".into()), models)],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Error(_)));\n assert!(llm.calls.lock().await.is_empty());\n }\n\n // ── Error classification tests ──────────────────────────────\n\n #[test]\n fn classify_syntax_error() {\n let cat = classify_runtime_error(\"SyntaxError: unexpected token\");\n assert_eq!(cat, CodeExecutionFailure::SyntaxError);\n }\n\n #[test]\n fn classify_timeout() {\n let cat = classify_runtime_error(\"execution timed out after 30s\");\n assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\n }\n\n #[test]\n fn classify_memory_limit() {\n let cat = classify_runtime_error(\"memory limit exceeded\");\n assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\n }\n\n #[test]\n fn classify_fuel_exhaustion() {\n let cat = classify_runtime_error(\"fuel exhausted during execution\");\n assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\n }\n\n #[test]\n fn classify_os_denied() {\n let cat = classify_runtime_error(\"OS operations are not permitted in CodeAct scripts\");\n assert_eq!(cat, CodeExecutionFailure::OsDenied);\n }\n\n #[test]\n fn classify_name_error_as_runtime() {\n // NameError from Monty (not NameLookup) is classified as RuntimeError\n let cat = classify_runtime_error(\"NameError: name 'foo' is not defined\");\n assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n }\n\n #[test]\n fn classify_type_error_as_runtime() {\n let cat = classify_runtime_error(\"TypeError: unsupported operand\");\n assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n }\n\n #[test]\n fn classify_module_not_found_as_runtime() {\n let cat = classify_runtime_error(\"ModuleNotFoundError: No module named 'csv'\");\n assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n }\n\n #[test]\n fn classify_syntax_word_is_not_syntaxerror() {\n // \"syntax\" alone should not trigger SyntaxError — only \"syntaxerror\" should.\n let cat = classify_runtime_error(\"unexpected syntax in expression\");\n assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n }\n\n #[test]\n fn vm_panic_variant_serializes_as_snake_case() {\n // VmPanic is set directly by catch_unwind paths, not by classify_runtime_error.\n // Verify it serializes consistently with Display (both snake_case).\n let failure = CodeExecutionFailure::VmPanic;\n assert_eq!(failure.to_string(), \"vm_panic\");\n let json = serde_json::to_value(&failure).unwrap();\n assert_eq!(json, serde_json::json!(\"vm_panic\"));\n }\n\n #[test]\n fn code_hash_deterministic() {\n let h1 = code_hash(\"print('hello')\");\n let h2 = code_hash(\"print('hello')\");\n assert_eq!(h1, h2);\n }\n\n #[test]\n fn code_hash_differs_for_different_code() {\n let h1 = code_hash(\"print('hello')\");\n let h2 = code_hash(\"print('world')\");\n assert_ne!(h1, h2);\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/executor/trace.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "x-xss-protection", + "0" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:11 GMT" + ], + [ + "x-ratelimit-used", + "23" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-ratelimit-remaining", + "4977" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "x-frame-options", + "deny" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "etag", + "\"563d5e8dd6835053c514ff45fcebbf5e94bf283f\"" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "content-length", + "26791" + ], + [ + "server", + "github.com" + ], + [ + "x-github-request-id", + "F588:8A635:41FB57:4D5555:69DFAEC7" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ] + ], + "body": "//! Execution trace analysis.\n//!\n//! Builds an in-memory `ExecutionTrace` from a completed `Thread` and runs a\n//! retrospective analyzer that flags common failure patterns. Used by the\n//! self-improvement mission and surfaced in debug logs.\n//!\n//! **There is no separate engine trace file.** Live trace recording for the\n//! whole system is handled by `RecordingLlm` in the host crate\n//! (`src/llm/recording.rs`), gated by `IRONCLAW_RECORD_TRACE`. Because the\n//! engine's `LlmBackend` is wired to the same provider chain, engine LLM\n//! interactions are captured by that single recorder — no engine-side env var\n//! and no second JSON file.\n\nuse chrono::Utc;\nuse serde::Serialize;\nuse tracing::debug;\n\nuse crate::types::event::ThreadEvent;\nuse crate::types::thread::{Thread, ThreadId, ThreadState};\n\n/// A complete execution trace for a single thread.\n#[derive(Debug, Serialize)]\npub struct ExecutionTrace {\n pub thread_id: ThreadId,\n pub goal: String,\n pub final_state: ThreadState,\n pub step_count: usize,\n pub total_tokens: u64,\n pub messages: Vec,\n pub events: Vec,\n pub issues: Vec,\n pub timestamp: chrono::DateTime,\n}\n\n/// A single doc record, for the trace.\n#[derive(Debug, Serialize)]\npub struct DocRecord {\n pub doc_type: String,\n pub title: String,\n pub content: String,\n}\n\n/// A message in the trace with role labeling.\n#[derive(Debug, Serialize)]\npub struct MessageRecord {\n pub role: String,\n pub content_length: usize,\n pub content_preview: String,\n pub full_content: String,\n pub action_name: Option,\n pub action_call_id: Option,\n}\n\n/// An issue detected by the retrospective analyzer.\n#[derive(Debug, Serialize)]\npub struct TraceIssue {\n pub severity: IssueSeverity,\n pub category: String,\n pub description: String,\n pub step: Option,\n}\n\n#[derive(Debug, PartialEq, Serialize)]\npub enum IssueSeverity {\n Error,\n Warning,\n Info,\n}\n\n/// Build a trace from a completed thread.\npub fn build_trace(thread: &Thread) -> ExecutionTrace {\n let messages: Vec = thread\n .messages\n .iter()\n .map(|m| {\n let preview: String = m.content.chars().take(300).collect();\n MessageRecord {\n role: format!(\"{:?}\", m.role),\n content_length: m.content.chars().count(),\n content_preview: if m.content.chars().count() > 300 {\n format!(\"{preview}...\")\n } else {\n preview\n },\n full_content: m.content.clone(),\n action_name: m.action_name.clone(),\n action_call_id: m.action_call_id.clone(),\n }\n })\n .collect();\n\n let issues = analyze_trace(thread);\n\n ExecutionTrace {\n thread_id: thread.id,\n goal: thread.goal.clone(),\n final_state: thread.state,\n step_count: thread.step_count,\n total_tokens: thread.total_tokens_used,\n messages,\n events: thread.events.clone(),\n issues,\n timestamp: Utc::now(),\n }\n}\n\n/// Print a summary of the trace to the log.\npub fn log_trace_summary(trace: &ExecutionTrace) {\n debug!(\n thread_id = %trace.thread_id,\n goal = %trace.goal,\n state = ?trace.final_state,\n steps = trace.step_count,\n tokens = trace.total_tokens,\n messages = trace.messages.len(),\n events = trace.events.len(),\n issues = trace.issues.len(),\n \"=== Engine V2 Trace Summary ===\"\n );\n\n for issue in &trace.issues {\n match issue.severity {\n IssueSeverity::Error => debug!(\n category = %issue.category,\n step = ?issue.step,\n \"ISSUE: {}\",\n issue.description\n ),\n IssueSeverity::Warning => debug!(\n category = %issue.category,\n step = ?issue.step,\n \"WARNING: {}\",\n issue.description\n ),\n IssueSeverity::Info => debug!(\n category = %issue.category,\n step = ?issue.step,\n \"NOTE: {}\",\n issue.description\n ),\n }\n }\n}\n\n// ── Retrospective analysis ──────────────────────────────────\n\n/// Analyze a completed thread for common issues.\nfn analyze_trace(thread: &Thread) -> Vec {\n let mut issues = Vec::new();\n\n // 1. Check if the thread failed\n if thread.state == ThreadState::Failed {\n issues.push(TraceIssue {\n severity: IssueSeverity::Error,\n category: \"thread_failure\".into(),\n description: \"Thread ended in Failed state\".into(),\n step: None,\n });\n }\n\n // 2. Check for empty response (no FINAL, no useful output)\n let has_assistant_response = thread\n .messages\n .iter()\n .any(|m| m.role == crate::types::message::MessageRole::Assistant && !m.content.is_empty());\n if !has_assistant_response {\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"no_response\".into(),\n description: \"No assistant message in thread — model may not have generated output\"\n .into(),\n step: None,\n });\n }\n\n // 3. Check for tool errors\n let tool_errors: Vec<&ThreadEvent> = thread\n .events\n .iter()\n .filter(|e| matches!(e.kind, crate::types::event::EventKind::ActionFailed { .. }))\n .collect();\n if !tool_errors.is_empty() {\n for event in &tool_errors {\n if let crate::types::event::EventKind::ActionFailed {\n action_name, error, ..\n } = &event.kind\n {\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"tool_error\".into(),\n description: format!(\"Tool '{action_name}' failed: {error}\"),\n step: None,\n });\n }\n }\n }\n\n // 4. Check for code execution errors via structured CodeExecutionFailed events.\n // These carry a classified failure category that tells us exactly what kind\n // of error occurred (syntax, runtime, name lookup, VM panic, resource limit,\n // tool error, OS denied, gate pause).\n let code_failures: Vec<&ThreadEvent> = thread\n .events\n .iter()\n .filter(|e| {\n matches!(\n e.kind,\n crate::types::event::EventKind::CodeExecutionFailed { .. }\n )\n })\n .collect();\n for event in &code_failures {\n if let crate::types::event::EventKind::CodeExecutionFailed {\n category, error, ..\n } = &event.kind\n {\n let preview: String = error.chars().take(200).collect();\n let severity = match category {\n crate::types::step::CodeExecutionFailure::VmPanic => IssueSeverity::Error,\n crate::types::step::CodeExecutionFailure::ResourceLimit => IssueSeverity::Error,\n _ => IssueSeverity::Warning,\n };\n issues.push(TraceIssue {\n severity,\n category: format!(\"code_{category}\"),\n description: format!(\"Code execution failed ({category}): {preview}\"),\n step: None,\n });\n }\n }\n\n // Fallback: also check message-level patterns for backward compatibility\n // with threads that ran before the CodeExecutionFailed instrumentation\n // was added (PR #2483). Note: threads from mixed eras (some steps\n // instrumented, some not) will only report structured events when any\n // exist, silently skipping message-level errors from uninstrumented steps.\n if code_failures.is_empty() {\n let error_patterns = [\n \"NameError\",\n \"SyntaxError\",\n \"TypeError\",\n \"NotImplementedError\",\n \"ValueError\",\n \"AttributeError\",\n \"IndexError\",\n \"KeyError\",\n \"ModuleNotFoundError\",\n \"RuntimeError\",\n ];\n for (i, msg) in thread.messages.iter().enumerate() {\n let is_code_output = msg.role == crate::types::message::MessageRole::User\n && (msg.content.starts_with(\"[stdout]\")\n || msg.content.starts_with(\"[stderr]\")\n || msg.content.starts_with(\"[code \")\n || msg.content.starts_with(\"Traceback\"));\n if is_code_output && error_patterns.iter().any(|p| msg.content.contains(p)) {\n let preview: String = msg.content.chars().take(200).collect();\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"code_error\".into(),\n description: format!(\"Code execution error in message {i}: {preview}\"),\n step: None,\n });\n }\n }\n }\n\n // 5. Check for empty call_id on ActionResult messages (causes LLM API rejection).\n for (i, msg) in thread.messages.iter().enumerate() {\n if msg.role == crate::types::message::MessageRole::ActionResult {\n let call_id_empty = msg.action_call_id.as_ref().is_none_or(|id| id.is_empty());\n if call_id_empty {\n let name = msg.action_name.as_deref().unwrap_or(\"unknown\");\n issues.push(TraceIssue {\n severity: IssueSeverity::Error,\n category: \"empty_call_id\".into(),\n description: format!(\n \"ActionResult message {i} (tool '{name}') has empty call_id — will cause LLM API rejection\"\n ),\n step: None,\n });\n }\n }\n }\n\n // 6. Check for model ignoring tool results (hallucination risk).\n // In Tier 0 (structured), results appear as ActionResult messages.\n // In Tier 1 (CodeAct), results appear as User messages with \"[tool result]\" prefixes.\n let has_tool_results = thread\n .messages\n .iter()\n .any(|m| m.role == crate::types::message::MessageRole::ActionResult);\n let has_tool_output_in_context = thread.messages.iter().any(|m| {\n m.role == crate::types::message::MessageRole::User\n && (m.content.contains(\" result]\") || m.content.contains(\" error]\"))\n });\n if has_tool_results && !has_tool_output_in_context {\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"missing_tool_output\".into(),\n description:\n \"Tool results exist but no tool output in messages — model may not see tool results\"\n .into(),\n step: None,\n });\n }\n\n // 7. Check for excessive iterations\n if thread.step_count > 10 {\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"excessive_steps\".into(),\n description: format!(\n \"Thread took {} steps — may be stuck in a loop\",\n thread.step_count\n ),\n step: None,\n });\n }\n\n // 8. Check for text response without FINAL (model answered from memory)\n let text_without_code = thread.events.iter().all(|e| {\n !matches!(\n e.kind,\n crate::types::event::EventKind::ActionExecuted { .. }\n )\n });\n if text_without_code && thread.step_count == 1 && has_assistant_response {\n issues.push(TraceIssue {\n severity: IssueSeverity::Info,\n category: \"no_tools_used\".into(),\n description: \"Model answered in one step without using any tools — may be answering from training data\".into(),\n step: Some(1),\n });\n }\n\n // 9. Check for LLM not producing code blocks\n let code_steps = thread\n .events\n .iter()\n .filter(|e| matches!(e.kind, crate::types::event::EventKind::StepStarted { .. }))\n .count();\n let text_responses_without_code = thread\n .messages\n .iter()\n .filter(|m| {\n m.role == crate::types::message::MessageRole::Assistant\n && !m.content.contains(\"```\")\n && !m.content.contains(\"FINAL(\")\n })\n .count();\n if text_responses_without_code > 0 && code_steps > 0 {\n issues.push(TraceIssue {\n severity: IssueSeverity::Info,\n category: \"mixed_mode\".into(),\n description: format!(\n \"{text_responses_without_code} text response(s) without code blocks — model may not be following CodeAct prompt\"\n ),\n step: None,\n });\n }\n\n // 10. Extract failure reason from StateChanged → Failed events\n for event in &thread.events {\n if let crate::types::event::EventKind::StateChanged {\n to: ThreadState::Failed,\n reason: Some(reason),\n ..\n } = &event.kind\n {\n if reason.contains(\"LLM\") || reason.contains(\"Provider\") {\n issues.push(TraceIssue {\n severity: IssueSeverity::Error,\n category: \"llm_error\".into(),\n description: format!(\"LLM provider error: {}\", truncate(reason, 300)),\n step: None,\n });\n } else if reason.contains(\"orchestrator\") {\n issues.push(TraceIssue {\n severity: IssueSeverity::Error,\n category: \"orchestrator_error\".into(),\n description: format!(\"Orchestrator error: {}\", truncate(reason, 300)),\n step: None,\n });\n }\n }\n }\n\n issues\n}\n\nfn truncate(s: &str, max_chars: usize) -> String {\n let chars: String = s.chars().take(max_chars).collect();\n if s.chars().count() > max_chars {\n format!(\"{chars}...\")\n } else {\n chars\n }\n}\n\n#[cfg(test)]\nmod tests {\n use super::*;\n use crate::types::event::EventKind;\n use crate::types::message::ThreadMessage;\n use crate::types::project::ProjectId;\n use crate::types::step::StepId;\n use crate::types::thread::{ThreadConfig, ThreadType};\n\n fn make_thread() -> Thread {\n Thread::new(\n \"test goal\",\n ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n ThreadConfig::default(),\n )\n }\n\n // ── empty_call_id detection (OpenAI / Codex rejection) ───\n\n /// OpenAI and Codex reject ActionResult messages with empty call_id.\n /// The trace analyzer must flag these as errors.\n #[test]\n fn detects_empty_call_id_on_action_result() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"calling tool\"));\n // Simulate the bug: empty call_id\n thread.add_message(ThreadMessage::action_result(\"\", \"web_search\", \"result\"));\n\n let issues = analyze_trace(&thread);\n let empty_id_issues: Vec<_> = issues\n .iter()\n .filter(|i| i.category == \"empty_call_id\")\n .collect();\n\n assert_eq!(empty_id_issues.len(), 1);\n assert_eq!(empty_id_issues[0].severity, IssueSeverity::Error);\n assert!(empty_id_issues[0].description.contains(\"web_search\"));\n }\n\n /// ActionResult with None call_id should also be flagged.\n #[test]\n fn detects_none_call_id_on_action_result() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"calling tool\"));\n // Manually construct a message with None call_id\n thread.add_message(ThreadMessage {\n role: crate::types::message::MessageRole::ActionResult,\n content: \"result\".into(),\n provenance: crate::types::provenance::Provenance::ToolOutput {\n action_name: \"shell\".into(),\n },\n action_call_id: None,\n action_name: Some(\"shell\".into()),\n action_calls: None,\n timestamp: chrono::Utc::now(),\n });\n\n let issues = analyze_trace(&thread);\n assert!(issues.iter().any(|i| i.category == \"empty_call_id\"));\n }\n\n /// No false positive: valid call_id should not be flagged.\n #[test]\n fn no_false_positive_for_valid_call_id() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"calling tool\"));\n thread.add_message(ThreadMessage::action_result(\n \"call_abc123\",\n \"web_search\",\n \"result\",\n ));\n\n let issues = analyze_trace(&thread);\n assert!(\n !issues.iter().any(|i| i.category == \"empty_call_id\"),\n \"valid call_id should not be flagged\"\n );\n }\n\n // ── tool_error detection ─────────────────────────────────\n\n /// ActionFailed events should produce tool_error warnings.\n #[test]\n fn detects_tool_failures_in_events() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"ok\"));\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::ActionFailed {\n step_id: StepId::new(),\n action_name: \"web_search\".into(),\n call_id: \"call_123\".into(),\n error: \"No lease for action 'web_search'\".into(),\n params_summary: None,\n },\n ));\n\n let issues = analyze_trace(&thread);\n let tool_errors: Vec<_> = issues\n .iter()\n .filter(|i| i.category == \"tool_error\")\n .collect();\n assert_eq!(tool_errors.len(), 1);\n assert!(tool_errors[0].description.contains(\"web_search\"));\n }\n\n // ── thread_failure detection ─────────────────────────────\n\n #[test]\n fn detects_failed_thread_state() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"trying\"));\n thread.state = ThreadState::Failed;\n\n let issues = analyze_trace(&thread);\n assert!(issues.iter().any(|i| i.category == \"thread_failure\"));\n }\n\n // ── LLM error detection from StateChanged events ─────────\n\n /// Reproduces the exact pattern from the trace: OpenAI rejects empty call_id.\n #[test]\n fn detects_llm_error_from_state_changed() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"ok\"));\n thread.state = ThreadState::Failed;\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::StateChanged {\n from: ThreadState::Running,\n to: ThreadState::Failed,\n reason: Some(\n \"LLM error: Provider openai_codex request failed: HTTP 400 Bad Request: \\\n Invalid 'input[5].call_id': empty string\"\n .into(),\n ),\n },\n ));\n\n let issues = analyze_trace(&thread);\n assert!(\n issues.iter().any(|i| i.category == \"llm_error\"),\n \"should detect LLM provider error in StateChanged reason\"\n );\n }\n\n // ── Multiple empty call_ids ──────────────────────────────\n\n /// Anthropic sends consecutive tool results merged into one User message.\n /// If multiple ActionResults have empty call_ids, each must be flagged.\n #[test]\n fn flags_each_empty_call_id_separately() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"parallel calls\"));\n thread.add_message(ThreadMessage::action_result(\"\", \"tool_a\", \"result_a\"));\n thread.add_message(ThreadMessage::action_result(\"\", \"tool_b\", \"result_b\"));\n thread.add_message(ThreadMessage::action_result(\n \"call_ok\", \"tool_c\", \"result_c\",\n ));\n\n let issues = analyze_trace(&thread);\n let empty_issues: Vec<_> = issues\n .iter()\n .filter(|i| i.category == \"empty_call_id\")\n .collect();\n assert_eq!(\n empty_issues.len(),\n 2,\n \"should flag exactly the 2 empty call_ids\"\n );\n }\n\n // ── CodeExecutionFailed event detection ────────────────────\n\n #[test]\n fn detects_code_execution_failure_from_event() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"```repl\\nimport csv\\n```\"));\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::CodeExecutionFailed {\n step_id: StepId::new(),\n category: crate::types::step::CodeExecutionFailure::RuntimeError,\n error: \"ModuleNotFoundError: No module named 'csv'\".into(),\n code_hash: Some(\"abc123\".into()),\n duration_ms: 42,\n },\n ));\n\n let issues = analyze_trace(&thread);\n let code_issues: Vec<_> = issues\n .iter()\n .filter(|i| i.category.starts_with(\"code_\"))\n .collect();\n assert_eq!(code_issues.len(), 1);\n assert_eq!(code_issues[0].category, \"code_runtime_error\");\n assert_eq!(code_issues[0].severity, IssueSeverity::Warning);\n assert!(code_issues[0].description.contains(\"ModuleNotFoundError\"));\n }\n\n #[test]\n fn vm_panic_is_error_severity() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"code\"));\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::CodeExecutionFailed {\n step_id: StepId::new(),\n category: crate::types::step::CodeExecutionFailure::VmPanic,\n error: \"Monty panicked: unreachable\".into(),\n code_hash: None,\n duration_ms: 0,\n },\n ));\n\n let issues = analyze_trace(&thread);\n let panic_issues: Vec<_> = issues\n .iter()\n .filter(|i| i.category == \"code_vm_panic\")\n .collect();\n assert_eq!(panic_issues.len(), 1);\n assert_eq!(panic_issues[0].severity, IssueSeverity::Error);\n }\n\n #[test]\n fn fallback_message_detection_when_no_events() {\n // Threads from before instrumentation should still be detected\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"code\"));\n thread.add_message(ThreadMessage::user(\n \"[stdout]\\nNameError: name 'foo' is not defined\",\n ));\n\n let issues = analyze_trace(&thread);\n assert!(\n issues.iter().any(|i| i.category == \"code_error\"),\n \"should detect code error from message when no CodeExecutionFailed events exist\"\n );\n }\n\n #[test]\n fn trace_serializes_approval_request_payload() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"installing notion\"));\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::ApprovalRequested {\n action_name: \"tool_install\".into(),\n call_id: \"call_install_1\".into(),\n parameters: Some(serde_json::json!({\"name\": \"notion\", \"kind\": \"mcp_server\"})),\n description: Some(\"Install an extension\".into()),\n allow_always: Some(true),\n gate_name: Some(\"approval\".into()),\n params_summary: Some(\"notion\".into()),\n },\n ));\n\n let trace = build_trace(&thread);\n // `Thread::add_message` records a `MessageAdded` event for each\n // message, so the `ApprovalRequested` event is no longer at index 0\n // — it's mixed in with the message events. Find it by kind.\n let approval = trace\n .events\n .iter()\n .find(|e| matches!(&e.kind, EventKind::ApprovalRequested { .. }))\n .expect(\"trace should contain an ApprovalRequested event\");\n match &approval.kind {\n EventKind::ApprovalRequested {\n action_name,\n call_id,\n parameters,\n description,\n allow_always,\n gate_name,\n params_summary,\n } => {\n assert_eq!(action_name, \"tool_install\");\n assert_eq!(call_id, \"call_install_1\");\n assert_eq!(\n parameters.as_ref().and_then(|p| p.get(\"name\")),\n Some(&serde_json::json!(\"notion\"))\n );\n assert_eq!(description.as_deref(), Some(\"Install an extension\"));\n assert_eq!(*allow_always, Some(true));\n assert_eq!(gate_name.as_deref(), Some(\"approval\"));\n assert_eq!(params_summary.as_deref(), Some(\"notion\"));\n }\n other => panic!(\"unexpected event kind: {other:?}\"),\n }\n\n let json = serde_json::to_string(&trace).expect(\"trace serializes\");\n assert!(json.contains(\"\\\"ApprovalRequested\\\"\"));\n assert!(json.contains(\"\\\"action_name\\\":\\\"tool_install\\\"\"));\n assert!(json.contains(\"\\\"call_id\\\":\\\"call_install_1\\\"\"));\n // Parameter map key order isn't stable across serde_json versions; check\n // both required keys are present rather than the exact serialized form.\n assert!(json.contains(\"\\\"name\\\":\\\"notion\\\"\"));\n assert!(json.contains(\"\\\"kind\\\":\\\"mcp_server\\\"\"));\n assert!(json.contains(\"\\\"description\\\":\\\"Install an extension\\\"\"));\n assert!(json.contains(\"\\\"allow_always\\\":true\"));\n assert!(json.contains(\"\\\"gate_name\\\":\\\"approval\\\"\"));\n assert!(json.contains(\"\\\"params_summary\\\":\\\"notion\\\"\"));\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/lib.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "x-ratelimit-resource", + "core" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "x-github-request-id", + "F599:77454:40288B:4B824F:69DFAEC7" + ], + [ + "x-ratelimit-remaining", + "4976" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "content-length", + "19797" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-frame-options", + "deny" + ], + [ + "x-xss-protection", + "0" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "etag", + "\"9ecaa6b575a369535e657b5f922187d3ce5149fa\"" + ], + [ + "server", + "github.com" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:11 GMT" + ], + [ + "x-ratelimit-used", + "24" + ] + ], + "body": "//! IronClaw Engine — unified thread-capability-CodeAct execution model.\n//!\n//! This crate provides the core execution engine for IronClaw, unifying\n//! ~10 separate abstractions (Session, Job, Routine, Channel, Tool, Skill,\n//! Hook, Observer, Extension, LoopDelegate) around 5 primitives:\n//!\n//! - **Thread** — unit of work (replaces Session + Job + Routine + Sub-agent)\n//! - **Step** — unit of execution (replaces agentic loop iteration + tool calls)\n//! - **Capability** — unit of effect (replaces Tool + Skill + Hook + Extension)\n//! - **MemoryDoc** — unit of durable knowledge (replaces workspace memory blobs)\n//! - **Project** — unit of context (replaces flat workspace namespace)\n//!\n//! The engine defines traits for external dependencies ([`LlmBackend`],\n//! [`Store`], [`EffectExecutor`]) that the host crate implements via bridge\n//! adapters over existing infrastructure.\n\n// Security: `__regex_match__` (in `executor/orchestrator.rs`) accepts\n// arbitrary patterns from the Python orchestrator and runs them on\n// user-supplied text. The default `regex` crate is linear-time. The\n// `fancy-regex` crate supports backreferences and is NOT linear-time, which\n// would turn that handler into a ReDoS vector. Cargo.toml depends on\n// `regex = \"1\"` with default features only — do NOT add `fancy-regex` to\n// this crate's dependency tree without first redesigning `__regex_match__`\n// to enforce a wall-clock matching budget.\n\npub mod capability;\npub mod executor;\npub mod gate;\npub mod memory;\npub mod reliability;\npub mod runtime;\npub mod traits;\npub mod types;\n\n// ── Re-exports: types ───────────────────────────────────────\n\npub use types::capability::{\n ActionDef, Capability, CapabilityLease, EffectType, GrantedActions, LeaseId, PolicyCondition,\n PolicyEffect, PolicyRule,\n};\npub use types::error::{CapabilityError, EngineError, StepError, ThreadError};\npub use types::event::{EventId, EventKind, ThreadEvent};\npub use types::memory::{DocId, DocType, MemoryDoc};\npub use types::message::{MessageRole, ThreadMessage};\npub use types::mission::{Mission, MissionCadence, MissionId, MissionStatus, ValidTimezone};\npub use types::project::{Project, ProjectId};\npub use types::provenance::Provenance;\npub use types::step::{\n ActionCall, ActionResult, CodeExecutionFailure, ExecutionTier, LlmResponse, Step, StepId,\n StepStatus, TokenUsage,\n};\npub use types::thread::{\n ActiveSkillProvenance, Thread, ThreadConfig, ThreadId, ThreadState, ThreadType,\n};\n\n// ── Re-exports: traits ──────────────────────────────────────\n\npub use traits::effect::{EffectExecutor, ThreadExecutionContext};\npub use traits::llm::{LlmBackend, LlmCallConfig, LlmOutput};\npub use traits::store::Store;\npub use traits::workspace::WorkspaceReader;\n\n// ── Re-exports: capability ────────────────────────────────────\n\npub use capability::lease::LeaseManager;\npub use capability::planner::{CapabilityGrantPlan, LeasePlanner};\npub use capability::policy::{PolicyDecision, PolicyEngine};\npub use capability::registry::CapabilityRegistry;\n\n// ── Re-exports: gate ─────────────────────────────────────────\n\npub use gate::lease::LeaseGate;\npub use gate::pipeline::GatePipeline;\npub use gate::tool_tier::{ToolTier, classify_tool_tier};\npub use gate::{\n ExecutionGate, ExecutionMode, GateContext, GateDecision, GateResolution, ResumeKind,\n};\n\n// ── Re-exports: runtime ───────────────────────────────────────\n\npub use executor::prompt::PlatformInfo;\npub use runtime::conversation::ConversationManager;\npub use runtime::manager::ThreadManager;\npub use runtime::messaging::ThreadOutcome;\npub use runtime::mission::{\n BudgetGate, FireRateLimit, MissionManager, MissionNotification, MissionUpdate,\n};\npub use runtime::tree::ThreadTree;\n\npub use types::conversation::{\n ConversationEntry, ConversationId, ConversationSurface, EntrySender,\n};\n\n// ── Re-exports: executor ──────────────────────────────────────\n\npub use executor::ExecutionLoop;\n\n// ── Re-exports: memory ────────────────────────────────────────\n\npub use memory::MemoryStore;\npub use memory::RetrievalEngine;\n\n// ── Re-exports: reliability ──────────────────────────────────\n\npub use reliability::ReliabilityTracker;\n\n// ── Test utilities ──────────────────────────────────────────\n\n#[cfg(test)]\npub(crate) mod tests {\n use tokio::sync::RwLock;\n\n use crate::traits::store::Store;\n use crate::types::capability::{CapabilityLease, LeaseId};\n use crate::types::conversation::{ConversationId, ConversationSurface};\n use crate::types::error::EngineError;\n use crate::types::event::ThreadEvent;\n use crate::types::memory::{DocId, MemoryDoc};\n use crate::types::mission::{Mission, MissionId, MissionStatus};\n use crate::types::project::{Project, ProjectId};\n use crate::types::step::Step;\n use crate::types::thread::{Thread, ThreadId, ThreadState};\n\n /// Shared in-memory Store implementation for tests.\n ///\n /// Stores all entity types with proper CRUD semantics and filtering by\n /// project_id / user_id. Use this instead of defining per-module mocks.\n pub struct InMemoryStore {\n threads: RwLock>,\n steps: RwLock>,\n events: RwLock>,\n projects: RwLock>,\n conversations: RwLock>,\n docs: RwLock>,\n leases: RwLock>,\n missions: RwLock>,\n }\n\n impl InMemoryStore {\n pub fn new() -> Self {\n Self {\n threads: RwLock::new(Vec::new()),\n steps: RwLock::new(Vec::new()),\n events: RwLock::new(Vec::new()),\n projects: RwLock::new(Vec::new()),\n conversations: RwLock::new(Vec::new()),\n docs: RwLock::new(Vec::new()),\n leases: RwLock::new(Vec::new()),\n missions: RwLock::new(Vec::new()),\n }\n }\n\n pub fn with_docs(docs: Vec) -> Self {\n Self {\n docs: RwLock::new(docs),\n ..Self::new()\n }\n }\n }\n\n #[async_trait::async_trait]\n impl Store for InMemoryStore {\n async fn save_thread(&self, thread: &Thread) -> Result<(), EngineError> {\n let mut threads = self.threads.write().await;\n threads.retain(|t| t.id != thread.id);\n threads.push(thread.clone());\n Ok(())\n }\n async fn load_thread(&self, id: ThreadId) -> Result, EngineError> {\n Ok(self\n .threads\n .read()\n .await\n .iter()\n .find(|t| t.id == id)\n .cloned())\n }\n async fn list_threads(\n &self,\n project_id: ProjectId,\n user_id: &str,\n ) -> Result, EngineError> {\n Ok(self\n .threads\n .read()\n .await\n .iter()\n .filter(|t| t.project_id == project_id && t.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn update_thread_state(\n &self,\n id: ThreadId,\n state: ThreadState,\n ) -> Result<(), EngineError> {\n let mut threads = self.threads.write().await;\n if let Some(t) = threads.iter_mut().find(|t| t.id == id) {\n t.state = state;\n }\n Ok(())\n }\n async fn save_step(&self, step: &Step) -> Result<(), EngineError> {\n let mut steps = self.steps.write().await;\n steps.retain(|s| s.id != step.id);\n steps.push(step.clone());\n Ok(())\n }\n async fn load_steps(&self, thread_id: ThreadId) -> Result, EngineError> {\n Ok(self\n .steps\n .read()\n .await\n .iter()\n .filter(|s| s.thread_id == thread_id)\n .cloned()\n .collect())\n }\n async fn append_events(&self, events: &[ThreadEvent]) -> Result<(), EngineError> {\n self.events.write().await.extend(events.iter().cloned());\n Ok(())\n }\n async fn load_events(&self, thread_id: ThreadId) -> Result, EngineError> {\n Ok(self\n .events\n .read()\n .await\n .iter()\n .filter(|e| e.thread_id == thread_id)\n .cloned()\n .collect())\n }\n async fn save_project(&self, project: &Project) -> Result<(), EngineError> {\n let mut projects = self.projects.write().await;\n projects.retain(|p| p.id != project.id);\n projects.push(project.clone());\n Ok(())\n }\n async fn load_project(&self, id: ProjectId) -> Result, EngineError> {\n Ok(self\n .projects\n .read()\n .await\n .iter()\n .find(|p| p.id == id)\n .cloned())\n }\n async fn list_projects(&self, user_id: &str) -> Result, EngineError> {\n Ok(self\n .projects\n .read()\n .await\n .iter()\n .filter(|p| p.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn list_all_projects(&self) -> Result, EngineError> {\n Ok(self.projects.read().await.iter().cloned().collect())\n }\n async fn save_conversation(\n &self,\n conversation: &ConversationSurface,\n ) -> Result<(), EngineError> {\n let mut conversations = self.conversations.write().await;\n conversations.retain(|c| c.id != conversation.id);\n conversations.push(conversation.clone());\n Ok(())\n }\n async fn load_conversation(\n &self,\n id: ConversationId,\n ) -> Result, EngineError> {\n Ok(self\n .conversations\n .read()\n .await\n .iter()\n .find(|c| c.id == id)\n .cloned())\n }\n async fn list_conversations(\n &self,\n user_id: &str,\n ) -> Result, EngineError> {\n Ok(self\n .conversations\n .read()\n .await\n .iter()\n .filter(|c| c.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn save_memory_doc(&self, doc: &MemoryDoc) -> Result<(), EngineError> {\n let mut docs = self.docs.write().await;\n docs.retain(|d| d.id != doc.id);\n docs.push(doc.clone());\n Ok(())\n }\n async fn load_memory_doc(&self, id: DocId) -> Result, EngineError> {\n Ok(self.docs.read().await.iter().find(|d| d.id == id).cloned())\n }\n async fn list_memory_docs(\n &self,\n project_id: ProjectId,\n user_id: &str,\n ) -> Result, EngineError> {\n Ok(self\n .docs\n .read()\n .await\n .iter()\n .filter(|d| d.project_id == project_id && d.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn save_lease(&self, lease: &CapabilityLease) -> Result<(), EngineError> {\n let mut leases = self.leases.write().await;\n leases.retain(|l| l.id != lease.id);\n leases.push(lease.clone());\n Ok(())\n }\n async fn load_active_leases(\n &self,\n thread_id: ThreadId,\n ) -> Result, EngineError> {\n Ok(self\n .leases\n .read()\n .await\n .iter()\n .filter(|l| l.thread_id == thread_id && !l.revoked)\n .cloned()\n .collect())\n }\n async fn revoke_lease(&self, lease_id: LeaseId, _reason: &str) -> Result<(), EngineError> {\n let mut leases = self.leases.write().await;\n if let Some(l) = leases.iter_mut().find(|l| l.id == lease_id) {\n l.revoked = true;\n }\n Ok(())\n }\n async fn save_mission(&self, mission: &Mission) -> Result<(), EngineError> {\n let mut missions = self.missions.write().await;\n missions.retain(|m| m.id != mission.id);\n missions.push(mission.clone());\n Ok(())\n }\n async fn load_mission(&self, id: MissionId) -> Result, EngineError> {\n Ok(self\n .missions\n .read()\n .await\n .iter()\n .find(|m| m.id == id)\n .cloned())\n }\n async fn list_missions(\n &self,\n project_id: ProjectId,\n user_id: &str,\n ) -> Result, EngineError> {\n Ok(self\n .missions\n .read()\n .await\n .iter()\n .filter(|m| m.project_id == project_id && m.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn update_mission_status(\n &self,\n id: MissionId,\n status: MissionStatus,\n ) -> Result<(), EngineError> {\n let mut missions = self.missions.write().await;\n if let Some(m) = missions.iter_mut().find(|m| m.id == id) {\n m.status = status;\n }\n Ok(())\n }\n async fn list_all_threads(\n &self,\n project_id: ProjectId,\n ) -> Result, EngineError> {\n Ok(self\n .threads\n .read()\n .await\n .iter()\n .filter(|t| t.project_id == project_id)\n .cloned()\n .collect())\n }\n async fn list_all_missions(\n &self,\n project_id: ProjectId,\n ) -> Result, EngineError> {\n Ok(self\n .missions\n .read()\n .await\n .iter()\n .filter(|m| m.project_id == project_id)\n .cloned()\n .collect())\n }\n }\n\n struct MinimalStore;\n\n #[async_trait::async_trait]\n impl Store for MinimalStore {\n async fn save_thread(&self, _thread: &Thread) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_thread(&self, _id: ThreadId) -> Result, EngineError> {\n Ok(None)\n }\n async fn list_threads(\n &self,\n _project_id: ProjectId,\n _user_id: &str,\n ) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn update_thread_state(\n &self,\n _id: ThreadId,\n _state: ThreadState,\n ) -> Result<(), EngineError> {\n Ok(())\n }\n async fn save_step(&self, _step: &Step) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_steps(&self, _thread_id: ThreadId) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn append_events(&self, _events: &[ThreadEvent]) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_events(&self, _thread_id: ThreadId) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn save_project(&self, _project: &Project) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_project(&self, _id: ProjectId) -> Result, EngineError> {\n Ok(None)\n }\n async fn save_memory_doc(&self, _doc: &MemoryDoc) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_memory_doc(&self, _id: DocId) -> Result, EngineError> {\n Ok(None)\n }\n async fn list_memory_docs(\n &self,\n _project_id: ProjectId,\n _user_id: &str,\n ) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn save_lease(&self, _lease: &CapabilityLease) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_active_leases(\n &self,\n _thread_id: ThreadId,\n ) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn revoke_lease(&self, _lease_id: LeaseId, _reason: &str) -> Result<(), EngineError> {\n Ok(())\n }\n async fn save_mission(&self, _mission: &Mission) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_mission(&self, _id: MissionId) -> Result, EngineError> {\n Ok(None)\n }\n async fn list_missions(\n &self,\n _project_id: ProjectId,\n _user_id: &str,\n ) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn update_mission_status(\n &self,\n _id: MissionId,\n _status: MissionStatus,\n ) -> Result<(), EngineError> {\n Ok(())\n }\n }\n\n #[tokio::test]\n async fn store_defaults_fail_closed() {\n let store = MinimalStore;\n assert!(matches!(\n store.list_projects(\"alice\").await,\n Err(EngineError::Store { .. })\n ));\n assert!(matches!(\n store.list_all_projects().await,\n Err(EngineError::Store { .. })\n ));\n assert!(matches!(\n store.load_conversation(ConversationId::new()).await,\n Err(EngineError::Store { .. })\n ));\n assert!(matches!(\n store.list_all_threads(ProjectId::new()).await,\n Err(EngineError::Store { .. })\n ));\n }\n\n #[tokio::test]\n async fn shared_queries_include_legacy_and_current_shared_owner() {\n use crate::types::memory::DocType;\n use crate::types::{LEGACY_SHARED_OWNER_ID, shared_owner_id};\n\n let project_id = ProjectId::new();\n let mut legacy = MemoryDoc::new(\n project_id,\n LEGACY_SHARED_OWNER_ID,\n DocType::Note,\n \"legacy\",\n \"a\",\n );\n let current = MemoryDoc::new(project_id, shared_owner_id(), DocType::Note, \"current\", \"b\");\n legacy.id = DocId::new();\n let store = InMemoryStore::with_docs(vec![legacy, current]);\n\n let docs = store\n .list_memory_docs_with_shared(project_id, \"alice\")\n .await\n .unwrap();\n assert_eq!(docs.len(), 2);\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/types/event.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "access-control-allow-origin", + "*" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:12 GMT" + ], + [ + "x-xss-protection", + "0" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "server", + "github.com" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-ratelimit-remaining", + "4975" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "x-ratelimit-used", + "25" + ], + [ + "x-github-request-id", + "F5B1:A381C:403C5C:4B95A1:69DFAEC8" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-frame-options", + "deny" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "etag", + "\"b61185f07131d737272f1068c899f7f6b5821960\"" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "content-length", + "8364" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ] + ], + "body": "//! Event sourcing types.\n//!\n//! Every significant action within a thread is recorded as an event.\n//! This enables replay, debugging, reflection, and trace-based testing.\n\nuse chrono::{DateTime, Utc};\nuse serde::{Deserialize, Serialize};\nuse uuid::Uuid;\n\nuse crate::types::capability::LeaseId;\n\n/// Generate a short human-readable summary of tool parameters for display.\n///\n/// For `http`: shows the URL. For `web_search`: shows the query.\n/// For other tools: shows the first string argument, truncated.\n/// Returns `None` for empty or unrecognizable params.\npub fn summarize_params(action_name: &str, params: &serde_json::Value) -> Option {\n let summary = match action_name {\n \"http\" | \"web_fetch\" => params\n .get(\"url\")\n .and_then(|v| v.as_str())\n .map(|u| truncate(u, 80)),\n \"web_search\" | \"llm_context\" => params\n .get(\"query\")\n .and_then(|v| v.as_str())\n .map(|q| truncate(q, 60)),\n \"memory_search\" => params\n .get(\"query\")\n .and_then(|v| v.as_str())\n .map(|q| truncate(q, 60)),\n \"memory_write\" => params\n .get(\"target\")\n .and_then(|v| v.as_str())\n .map(|t| t.to_string()),\n \"memory_read\" => params\n .get(\"path\")\n .and_then(|v| v.as_str())\n .map(|p| p.to_string()),\n \"shell\" => params\n .get(\"command\")\n .and_then(|v| v.as_str())\n .map(|c| truncate(c, 60)),\n \"message\" => params\n .get(\"content\")\n .and_then(|v| v.as_str())\n .map(|c| truncate(c, 40)),\n _ => {\n // Generic: show first string value\n if let Some(obj) = params.as_object() {\n obj.values()\n .find_map(|v| v.as_str())\n .map(|s| truncate(s, 50))\n } else {\n None\n }\n }\n };\n summary.filter(|s| !s.is_empty())\n}\n\nfn truncate(s: &str, max: usize) -> String {\n if s.len() <= max {\n s.to_string()\n } else {\n // Find a safe UTF-8 boundary\n let mut end = max.min(s.len());\n while end > 0 && !s.is_char_boundary(end) {\n end -= 1;\n }\n format!(\"{}...\", &s[..end]) // safety: end is validated by is_char_boundary loop above\n }\n}\nuse crate::types::step::{StepId, TokenUsage};\nuse crate::types::thread::{ThreadId, ThreadState};\n\n/// Strongly-typed event identifier.\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]\npub struct EventId(pub Uuid);\n\nimpl EventId {\n pub fn new() -> Self {\n Self(Uuid::new_v4())\n }\n}\n\nimpl Default for EventId {\n fn default() -> Self {\n Self::new()\n }\n}\n\n/// A recorded event in a thread's execution history.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct ThreadEvent {\n pub id: EventId,\n pub thread_id: ThreadId,\n pub timestamp: DateTime,\n pub kind: EventKind,\n}\n\nimpl ThreadEvent {\n pub fn new(thread_id: ThreadId, kind: EventKind) -> Self {\n Self {\n id: EventId::new(),\n thread_id,\n timestamp: Utc::now(),\n kind,\n }\n }\n}\n\n/// The specific kind of event that occurred.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub enum EventKind {\n // ── Thread lifecycle ────────────────────────────────────\n StateChanged {\n from: ThreadState,\n to: ThreadState,\n reason: Option,\n },\n\n // ── Step lifecycle ──────────────────────────────────────\n StepStarted {\n step_id: StepId,\n },\n StepCompleted {\n step_id: StepId,\n tokens: TokenUsage,\n },\n StepFailed {\n step_id: StepId,\n error: String,\n },\n\n // ── Action execution ────────────────────────────────────\n ActionExecuted {\n step_id: StepId,\n action_name: String,\n call_id: String,\n duration_ms: u64,\n /// Short human-readable summary of parameters (e.g., URL for http tool).\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n params_summary: Option,\n },\n ActionFailed {\n step_id: StepId,\n action_name: String,\n call_id: String,\n error: String,\n /// Short human-readable summary of parameters.\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n params_summary: Option,\n },\n\n // ── Capability leases ───────────────────────────────────\n LeaseGranted {\n lease_id: LeaseId,\n capability_name: String,\n },\n LeaseRevoked {\n lease_id: LeaseId,\n reason: String,\n },\n LeaseExpired {\n lease_id: LeaseId,\n },\n\n // ── Messages ────────────────────────────────────────────\n MessageAdded {\n role: String,\n content_preview: String,\n },\n\n // ── Thread tree ─────────────────────────────────────────\n ChildSpawned {\n child_id: ThreadId,\n goal: String,\n },\n ChildCompleted {\n child_id: ThreadId,\n },\n\n // ── Approval flow ───────────────────────────────────────\n ApprovalRequested {\n action_name: String,\n call_id: String,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n parameters: Option,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n description: Option,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n allow_always: Option,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n gate_name: Option,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n params_summary: Option,\n },\n ApprovalReceived {\n call_id: String,\n approved: bool,\n },\n\n // ── Self-improvement ──────────────────────────────────────\n SelfImprovementStarted,\n SelfImprovementComplete {\n prompt_updated: bool,\n patterns_added: usize,\n },\n SelfImprovementFailed {\n error: String,\n },\n\n // ── Skill activation ───────────────────────────────────────\n SkillActivated {\n skill_names: Vec,\n },\n\n // ── Code execution instrumentation ────────────────────────\n /// Emitted when a code (REPL) execution attempt fails. Enables aggregate\n /// analysis of code execution failure modes to determine whether the\n /// runtime (Monty), the LLM, or tool dispatch is the primary source of\n /// failures.\n CodeExecutionFailed {\n step_id: StepId,\n /// Classified failure category.\n category: crate::types::step::CodeExecutionFailure,\n /// The error message text (truncated to 500 chars).\n error: String,\n /// Hash of the Python code that was executed, for dedup/correlation.\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n code_hash: Option,\n /// Duration of the code execution attempt in milliseconds.\n #[serde(default)]\n duration_ms: u64,\n },\n\n // ── Orchestrator versioning ───────────────────────────────\n OrchestratorRollback {\n from_version: u64,\n to_version: u64,\n reason: String,\n },\n\n /// Unknown event kind — catch-all for forward compatibility during\n /// rolling deploys. Older binaries deserializing events written by\n /// newer binaries will produce this variant instead of failing.\n #[serde(other)]\n Unknown,\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/types/step.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:13 GMT" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "x-ratelimit-remaining", + "4974" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "x-xss-protection", + "0" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "x-github-request-id", + "F5C3:365517:3FF698:4B5015:69DFAEC8" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "content-length", + "6490" + ], + [ + "x-ratelimit-used", + "26" + ], + [ + "x-frame-options", + "deny" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "etag", + "\"161096e9b02238d78d86f611befa72d32fbe0a76\"" + ], + [ + "server", + "github.com" + ] + ], + "body": "//! Step — the unit of execution within a thread.\n//!\n//! Each step corresponds to one LLM call plus its subsequent action\n//! executions. This replaces the implicit \"iteration\" counter in the\n//! existing `run_agentic_loop`.\n\nuse std::time::Duration;\n\nuse chrono::{DateTime, Utc};\nuse serde::{Deserialize, Serialize};\nuse uuid::Uuid;\n\nuse crate::types::thread::ThreadId;\n\n/// Strongly-typed step identifier.\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]\npub struct StepId(pub Uuid);\n\nimpl StepId {\n pub fn new() -> Self {\n Self(Uuid::new_v4())\n }\n}\n\nimpl Default for StepId {\n fn default() -> Self {\n Self::new()\n }\n}\n\n/// Status of a step within its lifecycle.\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]\npub enum StepStatus {\n Pending,\n LlmCalling,\n Executing,\n Completed,\n Failed,\n}\n\n/// Which execution tier handles the step's code/actions.\n///\n/// Monty is the sole CodeAct/RLM executor. WASM and Docker are used for\n/// third-party tool isolation and thread sandboxing (Phase 8), not for\n/// running LLM-generated Python.\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]\npub enum ExecutionTier {\n /// Structured tool calls (JSON action calls from LLM).\n Structured,\n /// Embedded Python via Monty (CodeAct/RLM pattern).\n Scripting,\n}\n\n/// A single execution step within a thread.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct Step {\n pub id: StepId,\n pub thread_id: ThreadId,\n /// 1-indexed sequence within the thread.\n pub sequence: usize,\n pub status: StepStatus,\n pub tier: ExecutionTier,\n pub llm_response: Option,\n pub action_results: Vec,\n pub tokens_used: TokenUsage,\n pub started_at: DateTime,\n pub completed_at: Option>,\n}\n\nimpl Step {\n pub fn new(thread_id: ThreadId, sequence: usize) -> Self {\n Self {\n id: StepId::new(),\n thread_id,\n sequence,\n status: StepStatus::Pending,\n tier: ExecutionTier::Structured,\n llm_response: None,\n action_results: Vec::new(),\n tokens_used: TokenUsage::default(),\n started_at: Utc::now(),\n completed_at: None,\n }\n }\n}\n\n// ── LLM response types ─────────────────────────────────────\n\n/// Response from the LLM: text, action calls, or executable code.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub enum LlmResponse {\n /// Final text response.\n Text(String),\n /// One or more action calls (with optional reasoning text).\n ActionCalls {\n calls: Vec,\n content: Option,\n },\n /// Executable Python code (CodeAct). Tool calls happen as function\n /// calls within the code; the runtime suspends at each one and\n /// delegates to the EffectExecutor.\n Code {\n code: String,\n content: Option,\n },\n}\n\n/// A request from the LLM to execute a capability action.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct ActionCall {\n /// Unique call identifier (echoed in the result).\n pub id: String,\n /// Action name (e.g. \"web_fetch\", \"create_issue\").\n pub action_name: String,\n /// Action parameters as JSON.\n pub parameters: serde_json::Value,\n}\n\n/// Result of executing a capability action.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct ActionResult {\n /// The call ID this result corresponds to.\n pub call_id: String,\n /// The action that was executed.\n pub action_name: String,\n /// Output value.\n pub output: serde_json::Value,\n /// Whether this result represents an error.\n pub is_error: bool,\n /// How long the action took.\n #[serde(with = \"duration_millis\")]\n pub duration: Duration,\n}\n\n/// Classification of code execution failures.\n///\n/// Used by the instrumentation layer to distinguish Monty VM limitations\n/// from LLM logic errors, tool dispatch failures, and resource exhaustion.\n/// This data enables informed decisions about runtime alternatives.\n#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]\n#[serde(rename_all = \"snake_case\")]\npub enum CodeExecutionFailure {\n /// Python parse error — LLM generated invalid syntax.\n SyntaxError,\n /// Python runtime error (NameError, TypeError, ValueError, etc.) —\n /// LLM logic bug or use of unsupported feature.\n RuntimeError,\n /// Name lookup failed — function/variable not in scope and not a known tool.\n NameLookup,\n /// Monty VM panicked (catch_unwind caught it). Indicates a Monty bug,\n /// not a user code issue.\n VmPanic,\n /// Resource limit hit (timeout, memory, or allocation cap).\n ResourceLimit,\n /// A tool call inside code returned an error.\n ToolError,\n /// OS operation attempted (blocked by sandbox).\n OsDenied,\n}\n\nimpl std::fmt::Display for CodeExecutionFailure {\n fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {\n match self {\n Self::SyntaxError => write!(f, \"syntax_error\"),\n Self::RuntimeError => write!(f, \"runtime_error\"),\n Self::NameLookup => write!(f, \"name_lookup\"),\n Self::VmPanic => write!(f, \"vm_panic\"),\n Self::ResourceLimit => write!(f, \"resource_limit\"),\n Self::ToolError => write!(f, \"tool_error\"),\n Self::OsDenied => write!(f, \"os_denied\"),\n }\n }\n}\n\n/// Token usage for a single LLM call.\n#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize)]\npub struct TokenUsage {\n pub input_tokens: u64,\n pub output_tokens: u64,\n pub cache_read_tokens: u64,\n pub cache_write_tokens: u64,\n /// USD cost for this call (populated by LlmBackend if cost data is available).\n pub cost_usd: f64,\n}\n\nimpl TokenUsage {\n pub fn total(&self) -> u64 {\n self.input_tokens + self.output_tokens\n }\n}\n\n/// Serde helper for Duration as milliseconds.\nmod duration_millis {\n use std::time::Duration;\n\n use serde::{Deserialize, Deserializer, Serializer};\n\n pub fn serialize(d: &Duration, s: S) -> Result {\n s.serialize_u64(d.as_millis() as u64)\n }\n\n pub fn deserialize<'de, D: Deserializer<'de>>(d: D) -> Result {\n let millis = u64::deserialize(d)?;\n Ok(Duration::from_millis(millis))\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/pulls/2483" + }, + "response": { + "status": 200, + "headers": [ + [ + "x-content-type-options", + "nosniff" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-github-media-type", + "github.v3; format=json" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP,Accept-Encoding, Accept, X-Requested-With" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-github-request-id", + "FA7C:3B0091:423A54:4D98B1:69DFAEEA" + ], + [ + "content-type", + "application/json; charset=utf-8" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-xss-protection", + "0" + ], + [ + "etag", + "W/\"9e16e6116f0660a9e6887ab1801d96ff4f14bb9a0f40491f309b0afb108e8b95\"" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "server", + "github.com" + ], + [ + "x-ratelimit-used", + "27" + ], + [ + "x-frame-options", + "deny" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-ratelimit-remaining", + "4973" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:47 GMT" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:23:08 GMT" + ] + ], + "body": "{\"url\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483\",\"id\":3532069864,\"node_id\":\"PR_kwDORHZ7Z87Shxvo\",\"html_url\":\"https://github.com/nearai/ironclaw/pull/2483\",\"diff_url\":\"https://github.com/nearai/ironclaw/pull/2483.diff\",\"patch_url\":\"https://github.com/nearai/ironclaw/pull/2483.patch\",\"issue_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/2483\",\"number\":2483,\"state\":\"open\",\"locked\":false,\"title\":\"feat(engine): add code execution failure categorization instrumentation\",\"user\":{\"login\":\"serrrfirat\",\"id\":5748809,\"node_id\":\"MDQ6VXNlcjU3NDg4MDk=\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/5748809?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/serrrfirat\",\"html_url\":\"https://github.com/serrrfirat\",\"followers_url\":\"https://api.github.com/users/serrrfirat/followers\",\"following_url\":\"https://api.github.com/users/serrrfirat/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/serrrfirat/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/serrrfirat/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/serrrfirat/subscriptions\",\"organizations_url\":\"https://api.github.com/users/serrrfirat/orgs\",\"repos_url\":\"https://api.github.com/users/serrrfirat/repos\",\"events_url\":\"https://api.github.com/users/serrrfirat/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/serrrfirat/received_events\",\"type\":\"User\",\"user_view_type\":\"public\",\"site_admin\":false},\"body\":\"## Summary\\n- Adds `CodeExecutionFailure` enum (7 variants: SyntaxError, RuntimeError, NameLookup, VmPanic, ResourceLimit, ToolError, OsDenied) to classify why code execution fails\\n- Threads `failure` through `CodeExecutionResult` (replacing `had_error: bool`) and emits structured `CodeExecutionFailed` events for aggregate analysis of failure modes (Monty limitation vs LLM logic error vs tool dispatch failure)\\n- Upgrades trace analysis to use structured events with severity escalation (VmPanic/ResourceLimit → Error), with backward-compatible fallback for pre-instrumentation threads\\n\\n### Caller audit (Err → Ok(failure=VmPanic) shift)\\n`execute_code` / `execute_code_with_skills` previously returned `Err(EngineError::Effect)` on VM panic; now returns `Ok(CodeExecutionResult { failure: Some(VmPanic) })`. The only production call site is `orchestrator.rs:handle_execute_code_step` which correctly dispatches on `result.failure`. Test-only callers in `scripting.rs` were also updated. No other callers exist.\\n\\n### Known limitation: mixed-era fallback\\nThe trace analyzer's backward-compatible fallback (message-scraping for pre-instrumentation threads) is all-or-nothing: if a thread has *any* `CodeExecutionFailed` event, message-scraping is skipped entirely. Errors from pre-instrumentation steps in a mixed-era thread will go unreported. This is acceptable during the transition period since all new threads will have full instrumentation.\\n\\n## Test plan\\n- [x] 11 new unit tests for error classifier, code hash, and trace detection\\n- [x] `cargo test -p ironclaw_engine --lib` — 395 pass\\n- [x] `cargo clippy --all --all-features` — zero engine warnings\\n\\n🤖 Generated with [Claude Code](https://claude.com/claude-code)\",\"created_at\":\"2026-04-15T05:50:13Z\",\"updated_at\":\"2026-04-15T14:23:08Z\",\"closed_at\":null,\"merged_at\":null,\"merge_commit_sha\":\"35b0564ebf9bd9aef17e24c821ea98158844030d\",\"assignees\":[],\"requested_reviewers\":[{\"login\":\"ilblackdragon\",\"id\":175486,\"node_id\":\"MDQ6VXNlcjE3NTQ4Ng==\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/175486?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/ilblackdragon\",\"html_url\":\"https://github.com/ilblackdragon\",\"followers_url\":\"https://api.github.com/users/ilblackdragon/followers\",\"following_url\":\"https://api.github.com/users/ilblackdragon/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/ilblackdragon/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/ilblackdragon/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/ilblackdragon/subscriptions\",\"organizations_url\":\"https://api.github.com/users/ilblackdragon/orgs\",\"repos_url\":\"https://api.github.com/users/ilblackdragon/repos\",\"events_url\":\"https://api.github.com/users/ilblackdragon/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/ilblackdragon/received_events\",\"type\":\"User\",\"user_view_type\":\"public\",\"site_admin\":false}],\"requested_teams\":[],\"labels\":[{\"id\":10248775942,\"node_id\":\"LA_kwDORHZ7Z88AAAACYt_pBg\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/size:%20XL\",\"name\":\"size: XL\",\"color\":\"B71C1C\",\"default\":false,\"description\":\"500+ changed lines\"},{\"id\":10248776001,\"node_id\":\"LA_kwDORHZ7Z88AAAACYt_pQQ\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/risk:%20low\",\"name\":\"risk: low\",\"color\":\"4CAF50\",\"default\":false,\"description\":\"Changes to docs, tests, or low-risk modules\"},{\"id\":10248776633,\"node_id\":\"LA_kwDORHZ7Z88AAAACYt_ruQ\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/scope:%20db/postgres\",\"name\":\"scope: db/postgres\",\"color\":\"6A1B9A\",\"default\":false,\"description\":\"PostgreSQL backend\"},{\"id\":10248777536,\"node_id\":\"LA_kwDORHZ7Z88AAAACYt_vQA\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/contributor:%20core\",\"name\":\"contributor: core\",\"color\":\"FF8A65\",\"default\":false,\"description\":\"20+ merged PRs\"},{\"id\":10665130192,\"node_id\":\"LA_kwDORHZ7Z88AAAACe7D40A\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/DB%20MIGRATION\",\"name\":\"DB MIGRATION\",\"color\":\"C62828\",\"default\":false,\"description\":\"PR adds or modifies PostgreSQL or libSQL migration definitions\"}],\"milestone\":null,\"draft\":false,\"commits_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483/commits\",\"review_comments_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483/comments\",\"review_comment_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls/comments{/number}\",\"comments_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/2483/comments\",\"statuses_url\":\"https://api.github.com/repos/nearai/ironclaw/statuses/5496341c49b2abaa9842e2f939ea349c75fd7340\",\"head\":{\"label\":\"nearai:claude/audit-v2-engine-usage-I7dIz\",\"ref\":\"claude/audit-v2-engine-usage-I7dIz\",\"sha\":\"5496341c49b2abaa9842e2f939ea349c75fd7340\",\"user\":{\"login\":\"nearai\",\"id\":29134221,\"node_id\":\"MDEyOk9yZ2FuaXphdGlvbjI5MTM0MjIx\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/29134221?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/nearai\",\"html_url\":\"https://github.com/nearai\",\"followers_url\":\"https://api.github.com/users/nearai/followers\",\"following_url\":\"https://api.github.com/users/nearai/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/nearai/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/nearai/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/nearai/subscriptions\",\"organizations_url\":\"https://api.github.com/users/nearai/orgs\",\"repos_url\":\"https://api.github.com/users/nearai/repos\",\"events_url\":\"https://api.github.com/users/nearai/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/nearai/received_events\",\"type\":\"Organization\",\"user_view_type\":\"public\",\"site_admin\":false},\"repo\":{\"id\":1148615527,\"node_id\":\"R_kgDORHZ7Zw\",\"name\":\"ironclaw\",\"full_name\":\"nearai/ironclaw\",\"private\":false,\"owner\":{\"login\":\"nearai\",\"id\":29134221,\"node_id\":\"MDEyOk9yZ2FuaXphdGlvbjI5MTM0MjIx\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/29134221?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/nearai\",\"html_url\":\"https://github.com/nearai\",\"followers_url\":\"https://api.github.com/users/nearai/followers\",\"following_url\":\"https://api.github.com/users/nearai/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/nearai/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/nearai/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/nearai/subscriptions\",\"organizations_url\":\"https://api.github.com/users/nearai/orgs\",\"repos_url\":\"https://api.github.com/users/nearai/repos\",\"events_url\":\"https://api.github.com/users/nearai/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/nearai/received_events\",\"type\":\"Organization\",\"user_view_type\":\"public\",\"site_admin\":false},\"html_url\":\"https://github.com/nearai/ironclaw\",\"description\":\"IronClaw is OpenClaw inspired implementation in Rust focused on privacy and security\",\"fork\":false,\"url\":\"https://api.github.com/repos/nearai/ironclaw\",\"forks_url\":\"https://api.github.com/repos/nearai/ironclaw/forks\",\"keys_url\":\"https://api.github.com/repos/nearai/ironclaw/keys{/key_id}\",\"collaborators_url\":\"https://api.github.com/repos/nearai/ironclaw/collaborators{/collaborator}\",\"teams_url\":\"https://api.github.com/repos/nearai/ironclaw/teams\",\"hooks_url\":\"https://api.github.com/repos/nearai/ironclaw/hooks\",\"issue_events_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/events{/number}\",\"events_url\":\"https://api.github.com/repos/nearai/ironclaw/events\",\"assignees_url\":\"https://api.github.com/repos/nearai/ironclaw/assignees{/user}\",\"branches_url\":\"https://api.github.com/repos/nearai/ironclaw/branches{/branch}\",\"tags_url\":\"https://api.github.com/repos/nearai/ironclaw/tags\",\"blobs_url\":\"https://api.github.com/repos/nearai/ironclaw/git/blobs{/sha}\",\"git_tags_url\":\"https://api.github.com/repos/nearai/ironclaw/git/tags{/sha}\",\"git_refs_url\":\"https://api.github.com/repos/nearai/ironclaw/git/refs{/sha}\",\"trees_url\":\"https://api.github.com/repos/nearai/ironclaw/git/trees{/sha}\",\"statuses_url\":\"https://api.github.com/repos/nearai/ironclaw/statuses/{sha}\",\"languages_url\":\"https://api.github.com/repos/nearai/ironclaw/languages\",\"stargazers_url\":\"https://api.github.com/repos/nearai/ironclaw/stargazers\",\"contributors_url\":\"https://api.github.com/repos/nearai/ironclaw/contributors\",\"subscribers_url\":\"https://api.github.com/repos/nearai/ironclaw/subscribers\",\"subscription_url\":\"https://api.github.com/repos/nearai/ironclaw/subscription\",\"commits_url\":\"https://api.github.com/repos/nearai/ironclaw/commits{/sha}\",\"git_commits_url\":\"https://api.github.com/repos/nearai/ironclaw/git/commits{/sha}\",\"comments_url\":\"https://api.github.com/repos/nearai/ironclaw/comments{/number}\",\"issue_comment_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/comments{/number}\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/{+path}\",\"compare_url\":\"https://api.github.com/repos/nearai/ironclaw/compare/{base}...{head}\",\"merges_url\":\"https://api.github.com/repos/nearai/ironclaw/merges\",\"archive_url\":\"https://api.github.com/repos/nearai/ironclaw/{archive_format}{/ref}\",\"downloads_url\":\"https://api.github.com/repos/nearai/ironclaw/downloads\",\"issues_url\":\"https://api.github.com/repos/nearai/ironclaw/issues{/number}\",\"pulls_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls{/number}\",\"milestones_url\":\"https://api.github.com/repos/nearai/ironclaw/milestones{/number}\",\"notifications_url\":\"https://api.github.com/repos/nearai/ironclaw/notifications{?since,all,participating}\",\"labels_url\":\"https://api.github.com/repos/nearai/ironclaw/labels{/name}\",\"releases_url\":\"https://api.github.com/repos/nearai/ironclaw/releases{/id}\",\"deployments_url\":\"https://api.github.com/repos/nearai/ironclaw/deployments\",\"created_at\":\"2026-02-03T06:57:10Z\",\"updated_at\":\"2026-04-15T15:21:34Z\",\"pushed_at\":\"2026-04-15T15:27:33Z\",\"git_url\":\"git://github.com/nearai/ironclaw.git\",\"ssh_url\":\"git@github.com:nearai/ironclaw.git\",\"clone_url\":\"https://github.com/nearai/ironclaw.git\",\"svn_url\":\"https://github.com/nearai/ironclaw\",\"homepage\":\"https://www.ironclaw.com\",\"size\":30129,\"stargazers_count\":11789,\"watchers_count\":11789,\"language\":\"Rust\",\"has_issues\":true,\"has_projects\":false,\"has_downloads\":true,\"has_wiki\":false,\"has_pages\":false,\"has_discussions\":false,\"forks_count\":1349,\"mirror_url\":null,\"archived\":false,\"disabled\":false,\"open_issues_count\":640,\"license\":{\"key\":\"apache-2.0\",\"name\":\"Apache License 2.0\",\"spdx_id\":\"Apache-2.0\",\"url\":\"https://api.github.com/licenses/apache-2.0\",\"node_id\":\"MDc6TGljZW5zZTI=\"},\"allow_forking\":true,\"is_template\":false,\"web_commit_signoff_required\":false,\"has_pull_requests\":true,\"pull_request_creation_policy\":\"all\",\"topics\":[],\"visibility\":\"public\",\"forks\":1349,\"open_issues\":640,\"watchers\":11789,\"default_branch\":\"staging\"}},\"base\":{\"label\":\"nearai:staging\",\"ref\":\"staging\",\"sha\":\"16a07316d03430066f420e0d19e98a7506fec032\",\"user\":{\"login\":\"nearai\",\"id\":29134221,\"node_id\":\"MDEyOk9yZ2FuaXphdGlvbjI5MTM0MjIx\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/29134221?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/nearai\",\"html_url\":\"https://github.com/nearai\",\"followers_url\":\"https://api.github.com/users/nearai/followers\",\"following_url\":\"https://api.github.com/users/nearai/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/nearai/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/nearai/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/nearai/subscriptions\",\"organizations_url\":\"https://api.github.com/users/nearai/orgs\",\"repos_url\":\"https://api.github.com/users/nearai/repos\",\"events_url\":\"https://api.github.com/users/nearai/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/nearai/received_events\",\"type\":\"Organization\",\"user_view_type\":\"public\",\"site_admin\":false},\"repo\":{\"id\":1148615527,\"node_id\":\"R_kgDORHZ7Zw\",\"name\":\"ironclaw\",\"full_name\":\"nearai/ironclaw\",\"private\":false,\"owner\":{\"login\":\"nearai\",\"id\":29134221,\"node_id\":\"MDEyOk9yZ2FuaXphdGlvbjI5MTM0MjIx\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/29134221?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/nearai\",\"html_url\":\"https://github.com/nearai\",\"followers_url\":\"https://api.github.com/users/nearai/followers\",\"following_url\":\"https://api.github.com/users/nearai/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/nearai/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/nearai/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/nearai/subscriptions\",\"organizations_url\":\"https://api.github.com/users/nearai/orgs\",\"repos_url\":\"https://api.github.com/users/nearai/repos\",\"events_url\":\"https://api.github.com/users/nearai/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/nearai/received_events\",\"type\":\"Organization\",\"user_view_type\":\"public\",\"site_admin\":false},\"html_url\":\"https://github.com/nearai/ironclaw\",\"description\":\"IronClaw is OpenClaw inspired implementation in Rust focused on privacy and security\",\"fork\":false,\"url\":\"https://api.github.com/repos/nearai/ironclaw\",\"forks_url\":\"https://api.github.com/repos/nearai/ironclaw/forks\",\"keys_url\":\"https://api.github.com/repos/nearai/ironclaw/keys{/key_id}\",\"collaborators_url\":\"https://api.github.com/repos/nearai/ironclaw/collaborators{/collaborator}\",\"teams_url\":\"https://api.github.com/repos/nearai/ironclaw/teams\",\"hooks_url\":\"https://api.github.com/repos/nearai/ironclaw/hooks\",\"issue_events_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/events{/number}\",\"events_url\":\"https://api.github.com/repos/nearai/ironclaw/events\",\"assignees_url\":\"https://api.github.com/repos/nearai/ironclaw/assignees{/user}\",\"branches_url\":\"https://api.github.com/repos/nearai/ironclaw/branches{/branch}\",\"tags_url\":\"https://api.github.com/repos/nearai/ironclaw/tags\",\"blobs_url\":\"https://api.github.com/repos/nearai/ironclaw/git/blobs{/sha}\",\"git_tags_url\":\"https://api.github.com/repos/nearai/ironclaw/git/tags{/sha}\",\"git_refs_url\":\"https://api.github.com/repos/nearai/ironclaw/git/refs{/sha}\",\"trees_url\":\"https://api.github.com/repos/nearai/ironclaw/git/trees{/sha}\",\"statuses_url\":\"https://api.github.com/repos/nearai/ironclaw/statuses/{sha}\",\"languages_url\":\"https://api.github.com/repos/nearai/ironclaw/languages\",\"stargazers_url\":\"https://api.github.com/repos/nearai/ironclaw/stargazers\",\"contributors_url\":\"https://api.github.com/repos/nearai/ironclaw/contributors\",\"subscribers_url\":\"https://api.github.com/repos/nearai/ironclaw/subscribers\",\"subscription_url\":\"https://api.github.com/repos/nearai/ironclaw/subscription\",\"commits_url\":\"https://api.github.com/repos/nearai/ironclaw/commits{/sha}\",\"git_commits_url\":\"https://api.github.com/repos/nearai/ironclaw/git/commits{/sha}\",\"comments_url\":\"https://api.github.com/repos/nearai/ironclaw/comments{/number}\",\"issue_comment_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/comments{/number}\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/{+path}\",\"compare_url\":\"https://api.github.com/repos/nearai/ironclaw/compare/{base}...{head}\",\"merges_url\":\"https://api.github.com/repos/nearai/ironclaw/merges\",\"archive_url\":\"https://api.github.com/repos/nearai/ironclaw/{archive_format}{/ref}\",\"downloads_url\":\"https://api.github.com/repos/nearai/ironclaw/downloads\",\"issues_url\":\"https://api.github.com/repos/nearai/ironclaw/issues{/number}\",\"pulls_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls{/number}\",\"milestones_url\":\"https://api.github.com/repos/nearai/ironclaw/milestones{/number}\",\"notifications_url\":\"https://api.github.com/repos/nearai/ironclaw/notifications{?since,all,participating}\",\"labels_url\":\"https://api.github.com/repos/nearai/ironclaw/labels{/name}\",\"releases_url\":\"https://api.github.com/repos/nearai/ironclaw/releases{/id}\",\"deployments_url\":\"https://api.github.com/repos/nearai/ironclaw/deployments\",\"created_at\":\"2026-02-03T06:57:10Z\",\"updated_at\":\"2026-04-15T15:21:34Z\",\"pushed_at\":\"2026-04-15T15:27:33Z\",\"git_url\":\"git://github.com/nearai/ironclaw.git\",\"ssh_url\":\"git@github.com:nearai/ironclaw.git\",\"clone_url\":\"https://github.com/nearai/ironclaw.git\",\"svn_url\":\"https://github.com/nearai/ironclaw\",\"homepage\":\"https://www.ironclaw.com\",\"size\":30129,\"stargazers_count\":11789,\"watchers_count\":11789,\"language\":\"Rust\",\"has_issues\":true,\"has_projects\":false,\"has_downloads\":true,\"has_wiki\":false,\"has_pages\":false,\"has_discussions\":false,\"forks_count\":1349,\"mirror_url\":null,\"archived\":false,\"disabled\":false,\"open_issues_count\":640,\"license\":{\"key\":\"apache-2.0\",\"name\":\"Apache License 2.0\",\"spdx_id\":\"Apache-2.0\",\"url\":\"https://api.github.com/licenses/apache-2.0\",\"node_id\":\"MDc6TGljZW5zZTI=\"},\"allow_forking\":true,\"is_template\":false,\"web_commit_signoff_required\":false,\"has_pull_requests\":true,\"pull_request_creation_policy\":\"all\",\"topics\":[],\"visibility\":\"public\",\"forks\":1349,\"open_issues\":640,\"watchers\":11789,\"default_branch\":\"staging\"}},\"_links\":{\"self\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483\"},\"html\":{\"href\":\"https://github.com/nearai/ironclaw/pull/2483\"},\"issue\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/issues/2483\"},\"comments\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/issues/2483/comments\"},\"review_comments\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483/comments\"},\"review_comment\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/pulls/comments{/number}\"},\"commits\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483/commits\"},\"statuses\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/statuses/5496341c49b2abaa9842e2f939ea349c75fd7340\"}},\"author_association\":\"MEMBER\",\"auto_merge\":null,\"assignee\":null,\"active_lock_reason\":null,\"merged\":false,\"mergeable\":true,\"rebaseable\":false,\"mergeable_state\":\"blocked\",\"merged_by\":null,\"comments\":5,\"review_comments\":4,\"maintainer_can_modify\":false,\"commits\":5,\"additions\":570,\"deletions\":78,\"changed_files\":6}" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/pulls/2483/files?per_page=100" + }, + "response": { + "status": 200, + "headers": [ + [ + "x-accepted-oauth-scopes", + "" + ], + [ + "server", + "github.com" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP,Accept-Encoding, Accept, X-Requested-With" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "x-frame-options", + "deny" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-ratelimit-remaining", + "4972" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:48 GMT" + ], + [ + "x-github-request-id", + "FAC4:1FD39D:402A06:4B8676:69DFAEEC" + ], + [ + "etag", + "\"684a195e98a258be2a092400e150c420a630966f3f2d516f4f9ff7a9ce4d8fa9\"" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:23:08 GMT" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "strict-transport-security", + "max-age=31536000; includeSubdomains; preload" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-xss-protection", + "0" + ], + [ + "x-ratelimit-used", + "28" + ], + [ + "x-github-media-type", + "github.v3; format=json" + ], + [ + "content-length", + "49621" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "content-type", + "application/json; charset=utf-8" + ] + ], + "body": "[{\"sha\":\"76bf321deece47e13ba3044a33ee11da44d6130c\",\"filename\":\"crates/ironclaw_engine/src/executor/orchestrator.rs\",\"status\":\"modified\",\"additions\":118,\"deletions\":4,\"changes\":122,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Forchestrator.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Forchestrator.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Forchestrator.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -720,6 +720,7 @@ async fn handle_execute_code_step(\\n };\\n \\n // Run user code in a nested Monty VM (same pattern as rlm_query)\\n+ let code_start = std::time::Instant::now();\\n match Box::pin(execute_code(\\n &code,\\n thread,\\n@@ -748,10 +749,12 @@ async fn handle_execute_code_step(\\n // error, etc.), surface it as an ActionFailed event so traces and\\n // observers see the failure. Without this, parse errors silently\\n // fall back to the LLM via the result dict and never warn callers.\\n- if result.had_error {\\n+ if let Some(ref category) = result.failure {\\n let error_msg = if !result.stdout.is_empty() {\\n- let snippet: String = result.stdout.chars().take(500).collect();\\n- format!(\\\"CodeAct execution failed: {snippet}\\\")\\n+ format!(\\n+ \\\"CodeAct execution failed: {}\\\",\\n+ tail_chars(&result.stdout, 500)\\n+ )\\n } else {\\n \\\"CodeAct execution failed (no stdout)\\\".to_string()\\n };\\n@@ -774,6 +777,25 @@ async fn handle_execute_code_step(\\n let _ = tx.send(failed_event.clone());\\n }\\n thread.events.push(failed_event);\\n+\\n+ // Emit structured CodeExecutionFailed event for instrumentation.\\n+ // This enables aggregate analysis of WHY code execution fails\\n+ // (Monty limitation vs LLM logic error vs tool dispatch failure).\\n+ let error_text = tail_chars(&result.stdout, 500);\\n+ let instrumentation_event = ThreadEvent::new(\\n+ thread.id,\\n+ EventKind::CodeExecutionFailed {\\n+ step_id: exec_ctx.step_id,\\n+ category: category.clone(),\\n+ error: error_text,\\n+ code_hash: Some(crate::executor::scripting::code_hash(&code)),\\n+ duration_ms: code_start.elapsed().as_millis() as u64,\\n+ },\\n+ );\\n+ if let Some(tx) = event_tx {\\n+ let _ = tx.send(instrumentation_event.clone());\\n+ }\\n+ thread.events.push(instrumentation_event);\\n }\\n thread.updated_at = chrono::Utc::now();\\n \\n@@ -795,7 +817,7 @@ async fn handle_execute_code_step(\\n \\\"stdout\\\": result.stdout,\\n \\\"action_results\\\": action_results,\\n \\\"final_answer\\\": result.final_answer,\\n- \\\"had_error\\\": result.had_error,\\n+ \\\"had_error\\\": result.failure.is_some(),\\n \\\"pending_gate\\\": result.need_approval.as_ref().map(|na| {\\n match na {\\n ThreadOutcome::GatePaused { gate_name, action_name, call_id, parameters, resume_kind, resume_output } => serde_json::json!({\\n@@ -2196,6 +2218,19 @@ fn action_calls_to_python_json(calls: &[ActionCall]) -> Vec {\\n .collect()\\n }\\n \\n+/// Extract the last `n` characters from `s`.\\n+///\\n+/// Error tracebacks appear at the end of stdout, after any `print()` output.\\n+/// Using the head would capture the print statements instead of the error.\\n+fn tail_chars(s: &str, n: usize) -> String {\\n+ let char_count = s.chars().count();\\n+ if char_count > n {\\n+ s.chars().skip(char_count - n).collect()\\n+ } else {\\n+ s.to_owned()\\n+ }\\n+}\\n+\\n /// Build a PII-safe summary of an `action_calls` JSON value for log output.\\n ///\\n /// The action_calls payload contains tool parameters, which can carry user\\n@@ -3860,4 +3895,83 @@ FINAL(batch_error_count)\\n );\\n }\\n }\\n+\\n+ // ── CodeExecutionFailed event emission (caller test) ────────\\n+\\n+ #[tokio::test]\\n+ async fn execute_code_step_emits_code_execution_failed_event() {\\n+ let llm: Arc = Arc::new(ModelCapturingLlm {\\n+ captured: tokio::sync::Mutex::new(Vec::new()),\\n+ });\\n+ let effects: Arc = Arc::new(NoopEffects);\\n+ let leases = Arc::new(LeaseManager::new());\\n+ let policy = Arc::new(PolicyEngine::new());\\n+\\n+ let mut thread = Thread::new(\\n+ \\\"test code execution failure instrumentation\\\",\\n+ crate::types::thread::ThreadType::Foreground,\\n+ ProjectId::new(),\\n+ \\\"test-user\\\",\\n+ crate::types::thread::ThreadConfig::default(),\\n+ );\\n+ thread.transition_to(ThreadState::Running, None).unwrap();\\n+\\n+ // Pass intentionally broken Python code (syntax error)\\n+ let args = &[\\n+ json_to_monty(&serde_json::json!(\\\"def ==\\\")),\\n+ json_to_monty(&serde_json::json!({})),\\n+ ];\\n+\\n+ let (tx, _rx) = tokio::sync::broadcast::channel(16);\\n+ let _result = handle_execute_code_step(\\n+ args,\\n+ &[],\\n+ &mut thread,\\n+ &llm,\\n+ &effects,\\n+ &leases,\\n+ &policy,\\n+ Some(&tx),\\n+ )\\n+ .await;\\n+\\n+ // Verify CodeExecutionFailed event was emitted on thread.events\\n+ let code_failed_events: Vec<_> = thread\\n+ .events\\n+ .iter()\\n+ .filter(|e| matches!(&e.kind, EventKind::CodeExecutionFailed { .. }))\\n+ .collect();\\n+\\n+ assert_eq!(\\n+ code_failed_events.len(),\\n+ 1,\\n+ \\\"expected exactly one CodeExecutionFailed event, got {}\\\",\\n+ code_failed_events.len()\\n+ );\\n+\\n+ if let EventKind::CodeExecutionFailed {\\n+ category,\\n+ code_hash,\\n+ ..\\n+ } = &code_failed_events[0].kind\\n+ {\\n+ assert_eq!(\\n+ *category,\\n+ crate::types::step::CodeExecutionFailure::SyntaxError\\n+ );\\n+ assert!(code_hash.is_some());\\n+ } else {\\n+ panic!(\\\"expected CodeExecutionFailed event kind\\\");\\n+ }\\n+\\n+ // Also verify ActionFailed was emitted (existing behavior)\\n+ let action_failed = thread\\n+ .events\\n+ .iter()\\n+ .any(|e| matches!(&e.kind, EventKind::ActionFailed { .. }));\\n+ assert!(\\n+ action_failed,\\n+ \\\"expected ActionFailed event alongside CodeExecutionFailed\\\"\\n+ );\\n+ }\\n }\"},{\"sha\":\"674fca2750840d6f245ff8676d54df8b83ca74f1\",\"filename\":\"crates/ironclaw_engine/src/executor/scripting.rs\",\"status\":\"modified\",\"additions\":250,\"deletions\":52,\"changes\":302,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Fscripting.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Fscripting.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Fscripting.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -33,7 +33,7 @@ use crate::traits::llm::{LlmBackend, LlmCallConfig};\\n use crate::types::error::EngineError;\\n use crate::types::event::EventKind;\\n use crate::types::message::{MessageRole, ThreadMessage};\\n-use crate::types::step::{ActionResult, LlmResponse, TokenUsage};\\n+use crate::types::step::{ActionResult, CodeExecutionFailure, LlmResponse, TokenUsage};\\n use crate::types::thread::Thread;\\n use ironclaw_common::ValidTimezone;\\n \\n@@ -72,8 +72,10 @@ pub struct CodeExecutionResult {\\n pub recursive_tokens: TokenUsage,\\n /// If set, the code called FINAL() or FINAL_VAR() with this answer.\\n pub final_answer: Option,\\n- /// Whether the code execution hit an error (traceback included in stdout).\\n- pub had_error: bool,\\n+ /// Classified failure category. `None` when execution succeeded or was\\n+ /// paused by a gate. `Some(category)` when code execution failed —\\n+ /// `failure.is_some()` replaces the former `had_error: bool` field.\\n+ pub failure: Option,\\n }\\n \\n /// Build a compact output summary for inclusion in LLM context between steps.\\n@@ -297,7 +299,6 @@ pub async fn execute_code_with_skills(\\n let mut events = Vec::new();\\n let mut recursive_tokens = TokenUsage::default();\\n let mut final_answer: Option = None;\\n- let mut had_error = false;\\n \\n // Build context variables including persisted state from prior steps\\n let (input_names, input_values) = build_context_inputs(thread, persisted_state);\\n@@ -335,12 +336,19 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer: None,\\n- had_error: true,\\n+ failure: Some(CodeExecutionFailure::SyntaxError),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during code parsing\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\\"{stdout}\\\\nVmPanic: Monty VM panicked during code parsing\\\"),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer: None,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n };\\n@@ -356,6 +364,7 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => p,\\n Ok(Err(e)) => {\\n // Runtime error flows back to LLM\\n+ let category = classify_runtime_error(&e.to_string());\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout: format!(\\\"{stdout}\\\\nError: {e}\\\"),\\n@@ -364,12 +373,19 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer: None,\\n- had_error: true,\\n+ failure: Some(category),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during execution start\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\\"{stdout}\\\\nVmPanic: Monty VM panicked during execution start\\\"),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer: None,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n };\\n@@ -392,7 +408,7 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: None,\\n });\\n }\\n \\n@@ -479,7 +495,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -488,12 +503,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(classify_runtime_error(&e.to_string())),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during resume\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during resume\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -509,7 +533,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -518,12 +541,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(classify_runtime_error(&e.to_string())),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during resume_pending\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during resume_pending\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -585,7 +617,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -594,12 +625,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(CodeExecutionFailure::ToolError),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during resume_pending\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during resume_pending\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -612,7 +652,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -621,12 +660,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(CodeExecutionFailure::ToolError),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during resume\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during resume\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -640,7 +688,7 @@ pub async fn execute_code_with_skills(\\n need_approval: Some(outcome),\\n recursive_tokens,\\n final_answer: None,\\n- had_error,\\n+ failure: None,\\n });\\n }\\n }\\n@@ -702,7 +750,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -711,12 +758,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(classify_runtime_error(&e.to_string())),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during ResolveFutures resume\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during ResolveFutures resume\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -747,7 +803,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nNameError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -756,12 +811,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(CodeExecutionFailure::NameLookup),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during name lookup\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during name lookup\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -779,7 +843,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nOSError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -788,12 +851,19 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(CodeExecutionFailure::OsDenied),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during OS call\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\\"{stdout}\\\\nVmPanic: Monty VM panicked during OS call\\\"),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -802,6 +872,52 @@ pub async fn execute_code_with_skills(\\n }\\n }\\n \\n+// ── Error classification ────────────────────────────────────\\n+\\n+/// Classify a runtime error message into a failure category.\\n+///\\n+/// Parses the error text from Monty to distinguish between LLM logic bugs\\n+/// (NameError, TypeError, etc.), resource limit hits, and Monty VM issues.\\n+fn classify_runtime_error(error_msg: &str) -> CodeExecutionFailure {\\n+ let lower = error_msg.to_ascii_lowercase();\\n+\\n+ // Most specific checks first to avoid substring false positives.\\n+ if lower.contains(\\\"timed out\\\")\\n+ || lower.contains(\\\"timeout\\\")\\n+ || lower.contains(\\\"memory limit\\\")\\n+ || lower.contains(\\\"allocation limit\\\")\\n+ || lower.contains(\\\"out of fuel\\\")\\n+ || lower.contains(\\\"fuel exhausted\\\")\\n+ || lower.contains(\\\"resource limit\\\")\\n+ {\\n+ CodeExecutionFailure::ResourceLimit\\n+ } else if lower.contains(\\\"os operations are not permitted\\\") || lower.contains(\\\"oserror\\\") {\\n+ CodeExecutionFailure::OsDenied\\n+ } else if lower.contains(\\\"syntaxerror\\\") {\\n+ CodeExecutionFailure::SyntaxError\\n+ } else {\\n+ // NameError, TypeError, ValueError, AttributeError, IndexError,\\n+ // KeyError, ModuleNotFoundError, NotImplementedError, etc.\\n+ CodeExecutionFailure::RuntimeError\\n+ }\\n+}\\n+\\n+/// Compute a short hash of Python code for dedup/correlation in events.\\n+///\\n+/// Uses FNV-1a (64-bit) which is stable across Rust versions, unlike\\n+/// `DefaultHasher`. Not cryptographic — collision probability is ~2^-32\\n+/// at typical usage levels, sufficient for dedup but not for security.\\n+pub fn code_hash(code: &str) -> String {\\n+ const FNV_OFFSET: u64 = 0xcbf29ce484222325;\\n+ const FNV_PRIME: u64 = 0x00000100000001B3;\\n+ let mut hash = FNV_OFFSET;\\n+ for byte in code.as_bytes() {\\n+ hash ^= *byte as u64;\\n+ hash = hash.wrapping_mul(FNV_PRIME);\\n+ }\\n+ format!(\\\"{hash:016x}\\\")\\n+}\\n+\\n // ── Pending future tracking ─────────────────────────────────\\n \\n /// A deferred computation spawned as a tokio task, pending resolution\\n@@ -1803,7 +1919,7 @@ FINAL(str(result))\\n result.stdout\\n );\\n assert!(\\n- !result.had_error,\\n+ result.failure.is_none(),\\n \\\"should not error, stdout: {}\\\",\\n result.stdout\\n );\\n@@ -1855,7 +1971,7 @@ FINAL(str(a + b))\\n result.stdout\\n );\\n assert_eq!(result.action_results.len(), 2);\\n- assert!(!result.had_error);\\n+ assert!(result.failure.is_none());\\n }\\n \\n // ── asyncio.gather three tools ──────────────────────────\\n@@ -1905,7 +2021,7 @@ FINAL(str(s) + \\\"|\\\" + str(h) + \\\"|\\\" + str(m))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.action_results.len(), 3);\\n let answer = result.final_answer.unwrap();\\n assert!(answer.contains(\\\"search results\\\"), \\\"got: {answer}\\\");\\n@@ -1945,7 +2061,7 @@ FINAL(str(b))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.action_results.len(), 2);\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"final\\\"));\\n }\\n@@ -1980,7 +2096,7 @@ FINAL(\\\"should not reach\\\")\\n let result = run_code(code, effects, &thread).await.unwrap();\\n // Error in gather propagates as exception — code should error\\n assert!(\\n- result.had_error,\\n+ result.failure.is_some(),\\n \\\"should have error, stdout: {}\\\",\\n result.stdout\\n );\\n@@ -2024,7 +2140,7 @@ FINAL(\\\"hello from sync\\\")\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"hello from sync\\\"));\\n- assert!(!result.had_error);\\n+ assert!(result.failure.is_none());\\n }\\n \\n // ── globals() still works ───────────────────────────────\\n@@ -2045,7 +2161,7 @@ FINAL(str(has_search) + \\\"|\\\" + str(has_http))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"True|True\\\"));\\n }\\n \\n@@ -2063,7 +2179,7 @@ FINAL(str(len(results)))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"0\\\"));\\n }\\n \\n@@ -2090,7 +2206,7 @@ FINAL(str(results[0]))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"gathered\\\"));\\n assert_eq!(result.action_results.len(), 1);\\n }\\n@@ -2137,7 +2253,7 @@ while True:\\n // the key assertion is that it DOES NOT run forever.\\n if let Ok(r) = result {\\n assert!(\\n- r.had_error || r.stdout.contains(\\\"Error\\\") || r.stdout.contains(\\\"limit\\\"),\\n+ r.failure.is_some() || r.stdout.contains(\\\"Error\\\") || r.stdout.contains(\\\"limit\\\"),\\n \\\"resource limit should terminate infinite loop, got stdout: {}\\\",\\n truncate_for_assert(&r.stdout, 500),\\n );\\n@@ -2279,7 +2395,7 @@ while True:\\n // Must terminate — either via error or resource limit\\n if let Ok(r) = result {\\n assert!(\\n- r.had_error || r.stdout.contains(\\\"Error\\\") || r.stdout.contains(\\\"limit\\\"),\\n+ r.failure.is_some() || r.stdout.contains(\\\"Error\\\") || r.stdout.contains(\\\"limit\\\"),\\n \\\"cpu-bound loop should be terminated, stdout: {}\\\",\\n truncate_for_assert(&r.stdout, 500),\\n );\\n@@ -2313,7 +2429,7 @@ FINAL(str(x))\\n \\n let code = \\\"def broken(\\\\nFINAL('nope')\\\";\\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(result.had_error, \\\"syntax error should set had_error\\\");\\n+ assert!(result.failure.is_some(), \\\"syntax error should set failure\\\");\\n assert!(\\n result.stdout.contains(\\\"SyntaxError\\\") || result.stdout.contains(\\\"Error\\\"),\\n \\\"should contain SyntaxError, got: {}\\\",\\n@@ -2736,4 +2852,86 @@ FINAL(str(x))\\n assert!(matches!(result, ExtFunctionResult::Error(_)));\\n assert!(llm.calls.lock().await.is_empty());\\n }\\n+\\n+ // ── Error classification tests ──────────────────────────────\\n+\\n+ #[test]\\n+ fn classify_syntax_error() {\\n+ let cat = classify_runtime_error(\\\"SyntaxError: unexpected token\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::SyntaxError);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_timeout() {\\n+ let cat = classify_runtime_error(\\\"execution timed out after 30s\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_memory_limit() {\\n+ let cat = classify_runtime_error(\\\"memory limit exceeded\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_fuel_exhaustion() {\\n+ let cat = classify_runtime_error(\\\"fuel exhausted during execution\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_os_denied() {\\n+ let cat = classify_runtime_error(\\\"OS operations are not permitted in CodeAct scripts\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::OsDenied);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_name_error_as_runtime() {\\n+ // NameError from Monty (not NameLookup) is classified as RuntimeError\\n+ let cat = classify_runtime_error(\\\"NameError: name 'foo' is not defined\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_type_error_as_runtime() {\\n+ let cat = classify_runtime_error(\\\"TypeError: unsupported operand\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_module_not_found_as_runtime() {\\n+ let cat = classify_runtime_error(\\\"ModuleNotFoundError: No module named 'csv'\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_syntax_word_is_not_syntaxerror() {\\n+ // \\\"syntax\\\" alone should not trigger SyntaxError — only \\\"syntaxerror\\\" should.\\n+ let cat = classify_runtime_error(\\\"unexpected syntax in expression\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\\n+ }\\n+\\n+ #[test]\\n+ fn vm_panic_variant_serializes_as_snake_case() {\\n+ // VmPanic is set directly by catch_unwind paths, not by classify_runtime_error.\\n+ // Verify it serializes consistently with Display (both snake_case).\\n+ let failure = CodeExecutionFailure::VmPanic;\\n+ assert_eq!(failure.to_string(), \\\"vm_panic\\\");\\n+ let json = serde_json::to_value(&failure).unwrap();\\n+ assert_eq!(json, serde_json::json!(\\\"vm_panic\\\"));\\n+ }\\n+\\n+ #[test]\\n+ fn code_hash_deterministic() {\\n+ let h1 = code_hash(\\\"print('hello')\\\");\\n+ let h2 = code_hash(\\\"print('hello')\\\");\\n+ assert_eq!(h1, h2);\\n+ }\\n+\\n+ #[test]\\n+ fn code_hash_differs_for_different_code() {\\n+ let h1 = code_hash(\\\"print('hello')\\\");\\n+ let h2 = code_hash(\\\"print('world')\\\");\\n+ assert_ne!(h1, h2);\\n+ }\\n }\"},{\"sha\":\"563d5e8dd6835053c514ff45fcebbf5e94bf283f\",\"filename\":\"crates/ironclaw_engine/src/executor/trace.rs\",\"status\":\"modified\",\"additions\":135,\"deletions\":21,\"changes\":156,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Ftrace.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Ftrace.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Ftrace.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -195,33 +195,76 @@ fn analyze_trace(thread: &Thread) -> Vec {\\n }\\n }\\n \\n- // 4. Check for code execution errors in output messages.\\n- // Code output appears as User-role messages (Monty stdout/stderr) with\\n- // prefixes like \\\"[stdout]\\\" or \\\"[stderr]\\\". Skip the System prompt (index 0)\\n- // and Assistant messages to avoid false positives from example text.\\n- let error_patterns = [\\n- \\\"NameError\\\",\\n- \\\"SyntaxError\\\",\\n- \\\"TypeError\\\",\\n- \\\"NotImplementedError\\\",\\n- ];\\n- for (i, msg) in thread.messages.iter().enumerate() {\\n- let is_code_output = msg.role == crate::types::message::MessageRole::User\\n- && (msg.content.starts_with(\\\"[stdout]\\\")\\n- || msg.content.starts_with(\\\"[stderr]\\\")\\n- || msg.content.starts_with(\\\"[code \\\")\\n- || msg.content.starts_with(\\\"Traceback\\\"));\\n- if is_code_output && error_patterns.iter().any(|p| msg.content.contains(p)) {\\n- let preview: String = msg.content.chars().take(200).collect();\\n+ // 4. Check for code execution errors via structured CodeExecutionFailed events.\\n+ // These carry a classified failure category that tells us exactly what kind\\n+ // of error occurred (syntax, runtime, name lookup, VM panic, resource limit,\\n+ // tool error, OS denied, gate pause).\\n+ let code_failures: Vec<&ThreadEvent> = thread\\n+ .events\\n+ .iter()\\n+ .filter(|e| {\\n+ matches!(\\n+ e.kind,\\n+ crate::types::event::EventKind::CodeExecutionFailed { .. }\\n+ )\\n+ })\\n+ .collect();\\n+ for event in &code_failures {\\n+ if let crate::types::event::EventKind::CodeExecutionFailed {\\n+ category, error, ..\\n+ } = &event.kind\\n+ {\\n+ let preview: String = error.chars().take(200).collect();\\n+ let severity = match category {\\n+ crate::types::step::CodeExecutionFailure::VmPanic => IssueSeverity::Error,\\n+ crate::types::step::CodeExecutionFailure::ResourceLimit => IssueSeverity::Error,\\n+ _ => IssueSeverity::Warning,\\n+ };\\n issues.push(TraceIssue {\\n- severity: IssueSeverity::Warning,\\n- category: \\\"code_error\\\".into(),\\n- description: format!(\\\"Code execution error in message {i}: {preview}\\\"),\\n+ severity,\\n+ category: format!(\\\"code_{category}\\\"),\\n+ description: format!(\\\"Code execution failed ({category}): {preview}\\\"),\\n step: None,\\n });\\n }\\n }\\n \\n+ // Fallback: also check message-level patterns for backward compatibility\\n+ // with threads that ran before the CodeExecutionFailed instrumentation\\n+ // was added (PR #2483). Note: threads from mixed eras (some steps\\n+ // instrumented, some not) will only report structured events when any\\n+ // exist, silently skipping message-level errors from uninstrumented steps.\\n+ if code_failures.is_empty() {\\n+ let error_patterns = [\\n+ \\\"NameError\\\",\\n+ \\\"SyntaxError\\\",\\n+ \\\"TypeError\\\",\\n+ \\\"NotImplementedError\\\",\\n+ \\\"ValueError\\\",\\n+ \\\"AttributeError\\\",\\n+ \\\"IndexError\\\",\\n+ \\\"KeyError\\\",\\n+ \\\"ModuleNotFoundError\\\",\\n+ \\\"RuntimeError\\\",\\n+ ];\\n+ for (i, msg) in thread.messages.iter().enumerate() {\\n+ let is_code_output = msg.role == crate::types::message::MessageRole::User\\n+ && (msg.content.starts_with(\\\"[stdout]\\\")\\n+ || msg.content.starts_with(\\\"[stderr]\\\")\\n+ || msg.content.starts_with(\\\"[code \\\")\\n+ || msg.content.starts_with(\\\"Traceback\\\"));\\n+ if is_code_output && error_patterns.iter().any(|p| msg.content.contains(p)) {\\n+ let preview: String = msg.content.chars().take(200).collect();\\n+ issues.push(TraceIssue {\\n+ severity: IssueSeverity::Warning,\\n+ category: \\\"code_error\\\".into(),\\n+ description: format!(\\\"Code execution error in message {i}: {preview}\\\"),\\n+ step: None,\\n+ });\\n+ }\\n+ }\\n+ }\\n+\\n // 5. Check for empty call_id on ActionResult messages (causes LLM API rejection).\\n for (i, msg) in thread.messages.iter().enumerate() {\\n if msg.role == crate::types::message::MessageRole::ActionResult {\\n@@ -536,6 +579,77 @@ mod tests {\\n );\\n }\\n \\n+ // ── CodeExecutionFailed event detection ────────────────────\\n+\\n+ #[test]\\n+ fn detects_code_execution_failure_from_event() {\\n+ let mut thread = make_thread();\\n+ thread.add_message(ThreadMessage::system(\\\"sys\\\"));\\n+ thread.add_message(ThreadMessage::assistant(\\\"```repl\\\\nimport csv\\\\n```\\\"));\\n+ thread.events.push(ThreadEvent::new(\\n+ thread.id,\\n+ EventKind::CodeExecutionFailed {\\n+ step_id: StepId::new(),\\n+ category: crate::types::step::CodeExecutionFailure::RuntimeError,\\n+ error: \\\"ModuleNotFoundError: No module named 'csv'\\\".into(),\\n+ code_hash: Some(\\\"abc123\\\".into()),\\n+ duration_ms: 42,\\n+ },\\n+ ));\\n+\\n+ let issues = analyze_trace(&thread);\\n+ let code_issues: Vec<_> = issues\\n+ .iter()\\n+ .filter(|i| i.category.starts_with(\\\"code_\\\"))\\n+ .collect();\\n+ assert_eq!(code_issues.len(), 1);\\n+ assert_eq!(code_issues[0].category, \\\"code_runtime_error\\\");\\n+ assert_eq!(code_issues[0].severity, IssueSeverity::Warning);\\n+ assert!(code_issues[0].description.contains(\\\"ModuleNotFoundError\\\"));\\n+ }\\n+\\n+ #[test]\\n+ fn vm_panic_is_error_severity() {\\n+ let mut thread = make_thread();\\n+ thread.add_message(ThreadMessage::system(\\\"sys\\\"));\\n+ thread.add_message(ThreadMessage::assistant(\\\"code\\\"));\\n+ thread.events.push(ThreadEvent::new(\\n+ thread.id,\\n+ EventKind::CodeExecutionFailed {\\n+ step_id: StepId::new(),\\n+ category: crate::types::step::CodeExecutionFailure::VmPanic,\\n+ error: \\\"Monty panicked: unreachable\\\".into(),\\n+ code_hash: None,\\n+ duration_ms: 0,\\n+ },\\n+ ));\\n+\\n+ let issues = analyze_trace(&thread);\\n+ let panic_issues: Vec<_> = issues\\n+ .iter()\\n+ .filter(|i| i.category == \\\"code_vm_panic\\\")\\n+ .collect();\\n+ assert_eq!(panic_issues.len(), 1);\\n+ assert_eq!(panic_issues[0].severity, IssueSeverity::Error);\\n+ }\\n+\\n+ #[test]\\n+ fn fallback_message_detection_when_no_events() {\\n+ // Threads from before instrumentation should still be detected\\n+ let mut thread = make_thread();\\n+ thread.add_message(ThreadMessage::system(\\\"sys\\\"));\\n+ thread.add_message(ThreadMessage::assistant(\\\"code\\\"));\\n+ thread.add_message(ThreadMessage::user(\\n+ \\\"[stdout]\\\\nNameError: name 'foo' is not defined\\\",\\n+ ));\\n+\\n+ let issues = analyze_trace(&thread);\\n+ assert!(\\n+ issues.iter().any(|i| i.category == \\\"code_error\\\"),\\n+ \\\"should detect code error from message when no CodeExecutionFailed events exist\\\"\\n+ );\\n+ }\\n+\\n #[test]\\n fn trace_serializes_approval_request_payload() {\\n let mut thread = make_thread();\"},{\"sha\":\"9ecaa6b575a369535e657b5f922187d3ce5149fa\",\"filename\":\"crates/ironclaw_engine/src/lib.rs\",\"status\":\"modified\",\"additions\":2,\"deletions\":1,\"changes\":3,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Flib.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Flib.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Flib.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -46,7 +46,8 @@ pub use types::mission::{Mission, MissionCadence, MissionId, MissionStatus, Vali\\n pub use types::project::{Project, ProjectId};\\n pub use types::provenance::Provenance;\\n pub use types::step::{\\n- ActionCall, ActionResult, ExecutionTier, LlmResponse, Step, StepId, StepStatus, TokenUsage,\\n+ ActionCall, ActionResult, CodeExecutionFailure, ExecutionTier, LlmResponse, Step, StepId,\\n+ StepStatus, TokenUsage,\\n };\\n pub use types::thread::{\\n ActiveSkillProvenance, Thread, ThreadConfig, ThreadId, ThreadState, ThreadType,\"},{\"sha\":\"b61185f07131d737272f1068c899f7f6b5821960\",\"filename\":\"crates/ironclaw_engine/src/types/event.rs\",\"status\":\"modified\",\"additions\":25,\"deletions\":0,\"changes\":25,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fevent.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fevent.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fevent.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -215,10 +215,35 @@ pub enum EventKind {\\n skill_names: Vec,\\n },\\n \\n+ // ── Code execution instrumentation ────────────────────────\\n+ /// Emitted when a code (REPL) execution attempt fails. Enables aggregate\\n+ /// analysis of code execution failure modes to determine whether the\\n+ /// runtime (Monty), the LLM, or tool dispatch is the primary source of\\n+ /// failures.\\n+ CodeExecutionFailed {\\n+ step_id: StepId,\\n+ /// Classified failure category.\\n+ category: crate::types::step::CodeExecutionFailure,\\n+ /// The error message text (truncated to 500 chars).\\n+ error: String,\\n+ /// Hash of the Python code that was executed, for dedup/correlation.\\n+ #[serde(default, skip_serializing_if = \\\"Option::is_none\\\")]\\n+ code_hash: Option,\\n+ /// Duration of the code execution attempt in milliseconds.\\n+ #[serde(default)]\\n+ duration_ms: u64,\\n+ },\\n+\\n // ── Orchestrator versioning ───────────────────────────────\\n OrchestratorRollback {\\n from_version: u64,\\n to_version: u64,\\n reason: String,\\n },\\n+\\n+ /// Unknown event kind — catch-all for forward compatibility during\\n+ /// rolling deploys. Older binaries deserializing events written by\\n+ /// newer binaries will produce this variant instead of failing.\\n+ #[serde(other)]\\n+ Unknown,\\n }\"},{\"sha\":\"161096e9b02238d78d86f611befa72d32fbe0a76\",\"filename\":\"crates/ironclaw_engine/src/types/step.rs\",\"status\":\"modified\",\"additions\":40,\"deletions\":0,\"changes\":40,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fstep.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fstep.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fstep.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -132,6 +132,46 @@ pub struct ActionResult {\\n pub duration: Duration,\\n }\\n \\n+/// Classification of code execution failures.\\n+///\\n+/// Used by the instrumentation layer to distinguish Monty VM limitations\\n+/// from LLM logic errors, tool dispatch failures, and resource exhaustion.\\n+/// This data enables informed decisions about runtime alternatives.\\n+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]\\n+#[serde(rename_all = \\\"snake_case\\\")]\\n+pub enum CodeExecutionFailure {\\n+ /// Python parse error — LLM generated invalid syntax.\\n+ SyntaxError,\\n+ /// Python runtime error (NameError, TypeError, ValueError, etc.) —\\n+ /// LLM logic bug or use of unsupported feature.\\n+ RuntimeError,\\n+ /// Name lookup failed — function/variable not in scope and not a known tool.\\n+ NameLookup,\\n+ /// Monty VM panicked (catch_unwind caught it). Indicates a Monty bug,\\n+ /// not a user code issue.\\n+ VmPanic,\\n+ /// Resource limit hit (timeout, memory, or allocation cap).\\n+ ResourceLimit,\\n+ /// A tool call inside code returned an error.\\n+ ToolError,\\n+ /// OS operation attempted (blocked by sandbox).\\n+ OsDenied,\\n+}\\n+\\n+impl std::fmt::Display for CodeExecutionFailure {\\n+ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {\\n+ match self {\\n+ Self::SyntaxError => write!(f, \\\"syntax_error\\\"),\\n+ Self::RuntimeError => write!(f, \\\"runtime_error\\\"),\\n+ Self::NameLookup => write!(f, \\\"name_lookup\\\"),\\n+ Self::VmPanic => write!(f, \\\"vm_panic\\\"),\\n+ Self::ResourceLimit => write!(f, \\\"resource_limit\\\"),\\n+ Self::ToolError => write!(f, \\\"tool_error\\\"),\\n+ Self::OsDenied => write!(f, \\\"os_denied\\\"),\\n+ }\\n+ }\\n+}\\n+\\n /// Token usage for a single LLM call.\\n #[derive(Debug, Clone, Copy, Default, Serialize, Deserialize)]\\n pub struct TokenUsage {\"}]" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/executor/orchestrator.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:49 GMT" + ], + [ + "x-xss-protection", + "0" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "etag", + "\"76bf321deece47e13ba3044a33ee11da44d6130c\"" + ], + [ + "content-length", + "152011" + ], + [ + "x-ratelimit-remaining", + "4971" + ], + [ + "x-ratelimit-used", + "29" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-github-request-id", + "FAE9:17B00F:473F7A:52A24F:69DFAEED" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-frame-options", + "deny" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "server", + "github.com" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ] + ], + "body": "//! Python orchestrator — the self-modifiable execution loop.\n//!\n//! Replaces the Rust `ExecutionLoop::run()` with versioned Python code\n//! executed via Monty. The orchestrator is the \"glue layer\" between the\n//! LLM and tools — tool dispatch, output formatting, state management,\n//! truncation — all in Python, patchable by the self-improvement Mission.\n//!\n//! Host functions exposed to the orchestrator Python:\n//! - `__llm_complete__` — make an LLM call\n//! - `__execute_code_step__` — run user CodeAct code in a nested Monty VM\n//! - `__execute_action__` — execute a single tool action\n//! - `__execute_actions_parallel__` — execute multiple tool actions concurrently\n//! - `__check_signals__` — poll for stop/inject signals\n//! - `__emit_event__` — broadcast a ThreadEvent\n//! - `__save_checkpoint__` — persist thread state\n//! - `__transition_to__` — change thread state (validated)\n//! - `__retrieve_docs__` — query memory docs\n//! - `__check_budget__` — remaining tokens/time/USD\n//! - `__get_actions__` — available tool definitions\n\nuse std::sync::Arc;\nuse std::sync::atomic::{AtomicU64, Ordering};\n\nuse std::collections::HashMap;\n\nuse monty::{\n ExtFunctionResult, LimitedTracker, MontyObject, MontyRun, NameLookupResult, PrintWriter,\n ResourceLimits, RunProgress,\n};\nuse tracing::{debug, warn};\n\nuse crate::capability::lease::LeaseManager;\nuse crate::capability::policy::PolicyEngine;\nuse crate::memory::RetrievalEngine;\nuse crate::runtime::lease_refresh::reconcile_dynamic_tool_lease;\nuse crate::runtime::messaging::{SignalReceiver, ThreadOutcome, ThreadSignal};\nuse crate::traits::effect::{EffectExecutor, ThreadExecutionContext};\nuse crate::traits::llm::{LlmBackend, LlmCallConfig};\nuse crate::traits::store::Store;\nuse crate::types::error::EngineError;\nuse crate::types::event::{EventKind, ThreadEvent, summarize_params};\nuse crate::types::message::ThreadMessage;\nuse crate::types::project::ProjectId;\nuse crate::types::shared_owner_id;\nuse crate::types::step::{ActionCall, StepId, TokenUsage};\nuse crate::types::thread::{ActiveSkillProvenance, Thread, ThreadState};\nuse ironclaw_common::ValidTimezone;\n\nuse super::scripting::{execute_code, json_to_monty, monty_to_json, monty_to_string};\n\n/// The compiled-in default orchestrator (v0).\npub(crate) const DEFAULT_ORCHESTRATOR: &str = include_str!(\"../../orchestrator/default.py\");\n\n/// Well-known title for orchestrator code in the Store.\npub const ORCHESTRATOR_TITLE: &str = \"orchestrator:main\";\n\n/// Well-known tag for orchestrator code docs.\npub const ORCHESTRATOR_TAG: &str = \"orchestrator_code\";\n\n/// Result of running the orchestrator.\npub struct OrchestratorResult {\n /// The thread outcome parsed from the orchestrator's return value.\n pub outcome: ThreadOutcome,\n /// Total tokens used by LLM calls within the orchestrator.\n pub tokens_used: TokenUsage,\n}\n\n/// Extract source_channel from thread metadata (set by ConversationManager).\nfn thread_source_channel(thread: &Thread) -> Option {\n thread\n .metadata\n .get(\"source_channel\")\n .and_then(|v| v.as_str())\n .map(String::from)\n}\n\n/// Extract and validate user_timezone from thread metadata (set by bridge router).\nfn thread_user_timezone(thread: &Thread) -> Option {\n thread\n .metadata\n .get(\"user_timezone\")\n .and_then(|v| v.as_str())\n .and_then(ValidTimezone::parse)\n}\n\nfn normalize_pause_outcome(\n thread: &mut Thread,\n outcome: &ThreadOutcome,\n) -> Result<(), EngineError> {\n if matches!(outcome, ThreadOutcome::GatePaused { .. }) && thread.state != ThreadState::Waiting {\n thread.transition_to(\n ThreadState::Waiting,\n Some(\"waiting on external gate resolution\".into()),\n )?;\n }\n Ok(())\n}\n\n/// Resource limits for the orchestrator VM.\nfn orchestrator_limits() -> ResourceLimits {\n ResourceLimits::new()\n .max_duration(std::time::Duration::from_secs(300)) // 5 min (longer than user code)\n .max_allocations(5_000_000)\n .max_memory(128 * 1024 * 1024) // 128 MB\n}\n\n/// Maximum consecutive failures before auto-rollback.\nconst MAX_FAILURES_BEFORE_ROLLBACK: u64 = 3;\n\n/// Well-known title for orchestrator failure tracking.\nconst FAILURE_TRACKER_TITLE: &str = \"orchestrator:failures\";\nconst LEASE_REFRESH_WARN_INTERVAL_SECS: u64 = 60;\n\nfn warn_on_lease_refresh_failure(context: &'static str, error: &crate::types::error::EngineError) {\n static LAST_WARN_TS: AtomicU64 = AtomicU64::new(0);\n\n let now = chrono::Utc::now().timestamp().max(0) as u64;\n let last = LAST_WARN_TS.load(Ordering::Relaxed);\n if now.saturating_sub(last) >= LEASE_REFRESH_WARN_INTERVAL_SECS\n && LAST_WARN_TS\n .compare_exchange(last, now, Ordering::Relaxed, Ordering::Relaxed)\n .is_ok()\n {\n warn!(context, error = %error, \"dynamic lease refresh failed\");\n } else {\n debug!(context, error = %error, \"dynamic lease refresh failed\");\n }\n}\n\n/// Load orchestrator code: runtime version from Store, or compiled-in default.\n///\n/// When `allow_self_modify` is false, always uses the compiled-in default\n/// regardless of any runtime versions in the Store. This is the safe default\n/// for production — runtime orchestrator patching is opt-in.\n///\n/// Checks the failure tracker — if the latest version has >= 3 consecutive\n/// failures, falls back to the previous version (or compiled-in default).\npub async fn load_orchestrator(\n store: Option<&Arc>,\n project_id: ProjectId,\n allow_self_modify: bool,\n) -> (String, u64) {\n if !allow_self_modify {\n debug!(\"orchestrator self-modification disabled, using compiled-in default (v0)\");\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n }\n\n let Some(store) = store else {\n debug!(\"using compiled-in default orchestrator (v0, no store)\");\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n };\n\n let docs = match store.list_shared_memory_docs(project_id).await {\n Ok(d) => d,\n Err(_) => {\n debug!(\"using compiled-in default orchestrator (v0, store error)\");\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n }\n };\n\n load_orchestrator_from_docs(&docs, allow_self_modify)\n}\n\n/// Load orchestrator from pre-fetched system memory docs.\n///\n/// When the caller already has the `list_memory_docs` result, use this to\n/// avoid a duplicate Store query. Returns `(code, version)`.\n///\n/// Respects `allow_self_modify` — when false, always returns the compiled-in\n/// default. The caller in `loop_engine.rs` passes this from engine config.\npub fn load_orchestrator_from_docs(\n docs: &[crate::types::memory::MemoryDoc],\n allow_self_modify: bool,\n) -> (String, u64) {\n if !allow_self_modify {\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n }\n\n // Find all orchestrator versions, sorted by version number descending\n let mut versions: Vec<_> = docs\n .iter()\n .filter(|d| d.title == ORCHESTRATOR_TITLE && d.tags.contains(&ORCHESTRATOR_TAG.to_string()))\n .collect();\n versions.sort_by(|a, b| {\n let va = a\n .metadata\n .get(\"version\")\n .and_then(|v| v.as_u64())\n .unwrap_or(0);\n let vb = b\n .metadata\n .get(\"version\")\n .and_then(|v| v.as_u64())\n .unwrap_or(0);\n vb.cmp(&va) // descending\n });\n\n if versions.is_empty() {\n debug!(\"using compiled-in default orchestrator (v0)\");\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n }\n\n // Check failure count for the latest version\n let failures = load_failure_count(docs);\n\n for doc in &versions {\n let version = doc\n .metadata\n .get(\"version\")\n .and_then(|v| v.as_u64())\n .unwrap_or(1);\n\n // Skip versions with too many failures (only check the latest)\n if version\n == versions[0]\n .metadata\n .get(\"version\")\n .and_then(|v| v.as_u64())\n .unwrap_or(1)\n && failures >= MAX_FAILURES_BEFORE_ROLLBACK\n {\n debug!(\n version,\n failures, \"orchestrator version has too many failures, skipping\"\n );\n continue;\n }\n\n debug!(version, \"loaded runtime orchestrator\");\n return (doc.content.clone(), version);\n }\n\n // All versions failed — fall back to compiled-in default\n debug!(\"all orchestrator versions failed, using compiled-in default (v0)\");\n (DEFAULT_ORCHESTRATOR.to_string(), 0)\n}\n\n/// Record a failure for the current orchestrator version.\npub async fn record_orchestrator_failure(\n store: &Arc,\n project_id: ProjectId,\n version: u64,\n) {\n use crate::types::memory::{DocType, MemoryDoc};\n\n let docs = match store.list_shared_memory_docs(project_id).await {\n Ok(docs) => docs,\n Err(e) => {\n debug!(\"failed to list memory docs for failure tracker: {e}\");\n return;\n }\n };\n let existing = docs.iter().find(|d| d.title == FAILURE_TRACKER_TITLE);\n\n let mut tracker = if let Some(doc) = existing {\n doc.clone()\n } else {\n MemoryDoc::new(\n project_id,\n shared_owner_id(),\n DocType::Note,\n FAILURE_TRACKER_TITLE,\n \"\",\n )\n .with_tags(vec![\"orchestrator_meta\".to_string()])\n };\n\n // Store failure count as JSON in content: {\"version\": N, \"count\": M}\n let current: serde_json::Value =\n serde_json::from_str(&tracker.content).unwrap_or(serde_json::json!({}));\n let current_version = current.get(\"version\").and_then(|v| v.as_u64()).unwrap_or(0);\n let current_count = current.get(\"count\").and_then(|v| v.as_u64()).unwrap_or(0);\n\n let new_count = if current_version == version {\n current_count + 1\n } else {\n 1 // new version, reset count\n };\n\n tracker.content = serde_json::json!({\n \"version\": version,\n \"count\": new_count,\n })\n .to_string();\n tracker.updated_at = chrono::Utc::now();\n\n if let Err(e) = store.save_memory_doc(&tracker).await {\n debug!(\"failed to save orchestrator failure tracker: {e}\");\n }\n\n debug!(version, count = new_count, \"recorded orchestrator failure\");\n}\n\n/// Reset the failure counter (called after successful execution).\npub async fn reset_orchestrator_failures(store: &Arc, project_id: ProjectId) {\n let docs = store\n .list_shared_memory_docs(project_id)\n .await\n .unwrap_or_default();\n let existing = docs.iter().find(|d| d.title == FAILURE_TRACKER_TITLE);\n\n if let Some(doc) = existing {\n let mut tracker = doc.clone();\n tracker.content = serde_json::json!({\"version\": 0, \"count\": 0}).to_string();\n tracker.updated_at = chrono::Utc::now();\n let _ = store.save_memory_doc(&tracker).await;\n }\n}\n\n/// Load failure count for the latest orchestrator version.\nfn load_failure_count(docs: &[crate::types::memory::MemoryDoc]) -> u64 {\n docs.iter()\n .find(|d| d.title == FAILURE_TRACKER_TITLE)\n .and_then(|d| serde_json::from_str::(&d.content).ok())\n .and_then(|v| v.get(\"count\").and_then(|c| c.as_u64()))\n .unwrap_or(0)\n}\n\n/// Execute the orchestrator Python code with host function dispatch.\n///\n/// This is the core function that replaces `ExecutionLoop::run()`'s inner loop.\n/// The orchestrator Python calls host functions via Monty's suspension mechanism,\n/// and this function handles each suspension by delegating to the appropriate\n/// Rust implementation.\n#[allow(clippy::too_many_arguments)]\npub async fn execute_orchestrator(\n code: &str,\n thread: &mut Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &Arc,\n policy: &Arc,\n signal_rx: &mut SignalReceiver,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n retrieval: Option<&RetrievalEngine>,\n store: Option<&Arc>,\n persisted_state: &serde_json::Value,\n) -> Result {\n let mut total_tokens = TokenUsage::default();\n\n // Build context variables for the orchestrator\n let (input_names, input_values) = build_orchestrator_inputs(thread, persisted_state);\n\n // Parse and compile\n let runner = match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n MontyRun::new(code.to_string(), \"orchestrator.py\", input_names)\n })) {\n Ok(Ok(runner)) => runner,\n Ok(Err(e)) => {\n return Err(EngineError::Effect {\n reason: format!(\"Orchestrator parse error: {e}\"),\n });\n }\n Err(_) => {\n return Err(EngineError::Effect {\n reason: \"Monty VM panicked during orchestrator parsing\".into(),\n });\n }\n };\n\n // Start execution\n let mut stdout = String::new();\n let tracker = LimitedTracker::new(orchestrator_limits());\n\n let run_result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n runner.start(input_values, tracker, PrintWriter::Collect(&mut stdout))\n }));\n\n let mut progress = match run_result {\n Ok(Ok(p)) => p,\n Ok(Err(e)) => {\n return Err(EngineError::Effect {\n reason: format!(\"Orchestrator runtime error: {e}\"),\n });\n }\n Err(_) => {\n return Err(EngineError::Effect {\n reason: \"Monty VM panicked during orchestrator start\".into(),\n });\n }\n };\n\n // Drive the orchestrator dispatch loop\n let mut final_result: Option = None;\n\n loop {\n match progress {\n RunProgress::Complete(obj) => {\n // Use FINAL result if set, otherwise fall back to VM return value\n let result = if let Some(ref fr) = final_result {\n fr.clone()\n } else {\n monty_to_json(&obj)\n };\n sync_runtime_state(thread, result.get(\"state\"));\n let outcome = parse_outcome(&result);\n sync_visible_outcome(thread, &outcome);\n normalize_pause_outcome(thread, &outcome)?;\n return Ok(OrchestratorResult {\n outcome,\n tokens_used: total_tokens,\n });\n }\n\n RunProgress::FunctionCall(call) => {\n let action_name = call.function_name.clone();\n let args = &call.args;\n let kwargs = &call.kwargs;\n\n debug!(action = %action_name, \"orchestrator: host function call\");\n\n let ext_result = match action_name.as_str() {\n // FINAL(result) — orchestrator returns its outcome\n \"FINAL\" => {\n let val = args.first().map(monty_to_json).unwrap_or_default();\n final_result = Some(val);\n ExtFunctionResult::Return(MontyObject::None)\n }\n\n // __llm_complete__(messages, actions, config)\n \"__llm_complete__\" => {\n handle_llm_complete(\n args,\n kwargs,\n thread,\n LlmCompleteDeps {\n llm,\n effects,\n leases,\n store,\n },\n &mut total_tokens,\n )\n .await\n }\n\n // __execute_code_step__(code, state)\n \"__execute_code_step__\" => {\n handle_execute_code_step(\n args, kwargs, thread, llm, effects, leases, policy, event_tx,\n )\n .await\n }\n\n // __execute_action__(name, params, call_id=...)\n \"__execute_action__\" => {\n handle_execute_action(\n args, kwargs, thread, effects, leases, policy, event_tx,\n )\n .await\n }\n\n // __execute_actions_parallel__(calls)\n \"__execute_actions_parallel__\" => {\n handle_execute_actions_parallel(\n args, thread, effects, leases, policy, event_tx,\n )\n .await\n }\n\n // __check_signals__()\n \"__check_signals__\" => handle_check_signals(signal_rx, thread),\n\n // __emit_event__(kind, **data)\n \"__emit_event__\" => handle_emit_event(args, kwargs, thread, event_tx),\n\n // __save_checkpoint__(state, counters)\n \"__save_checkpoint__\" => handle_save_checkpoint(args, kwargs, thread),\n\n // __transition_to__(state, reason)\n \"__transition_to__\" => handle_transition_to(args, kwargs, thread),\n\n // __retrieve_docs__(goal, max_docs)\n \"__retrieve_docs__\" => {\n handle_retrieve_docs(args, kwargs, thread, retrieval).await\n }\n\n // __check_budget__()\"\n \"__check_budget__\" => handle_check_budget(thread),\n\n // __get_actions__()\n \"__get_actions__\" => handle_get_actions(thread, effects, leases, store).await,\n\n // __list_skills__(max_candidates, max_tokens)\n \"__list_skills__\" => handle_list_skills(args, thread, store).await,\n\n // __record_skill_usage__(doc_id, success)\n \"__record_skill_usage__\" => handle_record_skill_usage(args, store).await,\n\n // __regex_match__(pattern, text) -> bool\n // Evaluates a regex against text using Rust's regex crate.\n // Invalid patterns return False silently. Monty has no `re`\n // module, so this host function bridges the gap for the\n // skill selector's pattern-based scoring.\n \"__regex_match__\" => handle_regex_match(args),\n\n // __set_active_skills__(skills)\n \"__set_active_skills__\" => handle_set_active_skills(args, thread),\n\n // Unknown — let Monty resolve it (user-defined functions, builtins)\n other => ExtFunctionResult::NotFound(other.to_string()),\n };\n\n // Resume the orchestrator VM\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume(ext_result, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n return Err(EngineError::Effect {\n reason: format!(\"Orchestrator error after resume: {e}\"),\n });\n }\n Err(_) => {\n return Err(EngineError::Effect {\n reason: \"Monty VM panicked during orchestrator resume\".into(),\n });\n }\n }\n\n // If FINAL was called, the VM should complete on next iteration\n if final_result.is_some() {\n continue;\n }\n }\n\n RunProgress::NameLookup(lookup) => {\n // Undefined variable — resume with NameError\n let name = lookup.name.clone();\n debug!(name = %name, \"orchestrator: unresolved name\");\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n lookup.resume(\n NameLookupResult::Undefined,\n PrintWriter::Collect(&mut stdout),\n )\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n return Err(EngineError::Effect {\n reason: format!(\"Orchestrator NameError '{name}': {e}\"),\n });\n }\n Err(_) => {\n return Err(EngineError::Effect {\n reason: format!(\"Monty panic on NameLookup '{name}'\"),\n });\n }\n }\n }\n\n RunProgress::OsCall(_) => {\n return Err(EngineError::Effect {\n reason: \"Orchestrator attempted OS call (blocked)\".into(),\n });\n }\n\n RunProgress::ResolveFutures(_) => {\n return Err(EngineError::Effect {\n reason: \"Orchestrator attempted async (not supported)\".into(),\n });\n }\n }\n }\n}\n\n// ── Host function handlers ──────────────────────────────────\n\nstruct LlmCompleteDeps<'a> {\n llm: &'a Arc,\n effects: &'a Arc,\n leases: &'a Arc,\n store: Option<&'a Arc>,\n}\n\n/// Handle `__llm_complete__(messages, actions, config)`.\n///\n/// Calls the LLM and returns the response as a dict:\n/// `{type: \"text\"|\"code\"|\"actions\", content/code/calls: ..., usage: {...}}`\n///\nasync fn handle_llm_complete(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n deps: LlmCompleteDeps<'_>,\n total_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n use crate::types::step::LlmResponse;\n\n let explicit_messages = args.first().map(monty_to_json).filter(|v| !v.is_null());\n let explicit_config = args.get(2).map(monty_to_json).filter(|v| !v.is_null());\n let messages = explicit_messages\n .as_ref()\n .and_then(json_to_thread_messages)\n .unwrap_or_else(|| thread.messages.clone());\n\n if let Err(e) = reconcile_dynamic_tool_lease(\n thread,\n deps.effects,\n deps.leases,\n deps.store,\n &crate::LeasePlanner::new(),\n )\n .await\n {\n warn_on_lease_refresh_failure(\"llm_complete\", &e);\n }\n\n let active_leases = deps.leases.active_for_thread(thread.id).await;\n let actions = deps\n .effects\n .available_actions(&active_leases)\n .await\n .unwrap_or_default();\n\n let config = LlmCallConfig {\n max_tokens: explicit_config\n .as_ref()\n .and_then(|cfg| cfg.get(\"max_tokens\"))\n .and_then(|v| v.as_u64())\n .and_then(|v| u32::try_from(v).ok()),\n temperature: explicit_config\n .as_ref()\n .and_then(|cfg| cfg.get(\"temperature\"))\n .and_then(|v| v.as_f64())\n .map(|v| v as f32),\n force_text: explicit_config\n .as_ref()\n .and_then(|cfg| cfg.get(\"force_text\"))\n .and_then(|v| v.as_bool())\n .unwrap_or(false),\n depth: thread.config.depth,\n model: explicit_config\n .as_ref()\n .and_then(|cfg| cfg.get(\"model\"))\n .and_then(|v| v.as_str())\n .map(String::from),\n metadata: HashMap::new(),\n };\n\n match deps.llm.complete(&messages, &actions, &config).await {\n Ok(output) => {\n total_tokens.input_tokens += output.usage.input_tokens;\n total_tokens.output_tokens += output.usage.output_tokens;\n total_tokens.cost_usd += output.usage.cost_usd;\n\n let usage = serde_json::json!({\n \"input_tokens\": output.usage.input_tokens,\n \"output_tokens\": output.usage.output_tokens,\n \"cost_usd\": output.usage.cost_usd,\n });\n\n let result = match output.response {\n LlmResponse::Text(text) => {\n serde_json::json!({\"type\": \"text\", \"content\": text, \"usage\": usage})\n }\n LlmResponse::Code { code, .. } => {\n serde_json::json!({\"type\": \"code\", \"code\": code, \"usage\": usage})\n }\n LlmResponse::ActionCalls { calls, content } => {\n // Single source of truth for the Python interchange\n // shape — must round-trip via `python_json_to_action_calls`.\n let calls_json = action_calls_to_python_json(&calls);\n serde_json::json!({\n \"type\": \"actions\",\n \"content\": content,\n \"calls\": calls_json,\n \"usage\": usage\n })\n }\n };\n\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n Err(e) => ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::RuntimeError,\n Some(format!(\"LLM call failed: {e}\")),\n )),\n }\n}\n\n/// Handle `__execute_code_step__(code, state)`.\n///\n/// Runs user CodeAct code in a nested Monty VM with full tool dispatch.\n/// Returns a dict with stdout, return_value, action_results, etc.\n#[allow(clippy::too_many_arguments)]\nasync fn handle_execute_code_step(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &Arc,\n policy: &Arc,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n) -> ExtFunctionResult {\n let code = match args.first() {\n Some(obj) => monty_to_string(obj),\n None => {\n return ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::TypeError,\n Some(\"__execute_code_step__ requires a code string\".into()),\n ));\n }\n };\n\n let state = args\n .get(1)\n .map(monty_to_json)\n .unwrap_or(serde_json::json!({}));\n\n let exec_ctx = ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: thread.user_id.clone(),\n step_id: StepId::new(),\n current_call_id: None,\n source_channel: thread_source_channel(thread),\n user_timezone: thread_user_timezone(thread),\n };\n\n // Run user code in a nested Monty VM (same pattern as rlm_query)\n let code_start = std::time::Instant::now();\n match Box::pin(execute_code(\n &code,\n thread,\n llm,\n effects,\n leases,\n policy,\n &exec_ctx,\n &[],\n &state,\n ))\n .await\n {\n Ok(result) => {\n // Broadcast events from code execution to the thread and event channel.\n // Without this, ActionExecuted events from CodeAct tool calls are lost\n // and never appear in traces.\n for event_kind in &result.events {\n let event = ThreadEvent::new(thread.id, event_kind.clone());\n if let Some(tx) = event_tx {\n let _ = tx.send(event.clone());\n }\n thread.events.push(event);\n }\n // If the CodeAct snippet itself failed (Python SyntaxError, runtime\n // error, etc.), surface it as an ActionFailed event so traces and\n // observers see the failure. Without this, parse errors silently\n // fall back to the LLM via the result dict and never warn callers.\n if let Some(ref category) = result.failure {\n let error_msg = if !result.stdout.is_empty() {\n format!(\n \"CodeAct execution failed: {}\",\n tail_chars(&result.stdout, 500)\n )\n } else {\n \"CodeAct execution failed (no stdout)\".to_string()\n };\n let failed_event = ThreadEvent::new(\n thread.id,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: \"__codeact__\".to_string(),\n // Synthetic call_id derived from the step id —\n // CodeAct snippet failures don't have an LLM-provided\n // call_id, but `loop_engine.rs:1277` asserts that\n // ActionFailed events carry a non-empty call_id for\n // trace correlation.\n call_id: format!(\"codeact-step-{}\", exec_ctx.step_id.0),\n error: error_msg,\n params_summary: None,\n },\n );\n if let Some(tx) = event_tx {\n let _ = tx.send(failed_event.clone());\n }\n thread.events.push(failed_event);\n\n // Emit structured CodeExecutionFailed event for instrumentation.\n // This enables aggregate analysis of WHY code execution fails\n // (Monty limitation vs LLM logic error vs tool dispatch failure).\n let error_text = tail_chars(&result.stdout, 500);\n let instrumentation_event = ThreadEvent::new(\n thread.id,\n EventKind::CodeExecutionFailed {\n step_id: exec_ctx.step_id,\n category: category.clone(),\n error: error_text,\n code_hash: Some(crate::executor::scripting::code_hash(&code)),\n duration_ms: code_start.elapsed().as_millis() as u64,\n },\n );\n if let Some(tx) = event_tx {\n let _ = tx.send(instrumentation_event.clone());\n }\n thread.events.push(instrumentation_event);\n }\n thread.updated_at = chrono::Utc::now();\n\n let action_results: Vec = result\n .action_results\n .iter()\n .map(|r| {\n serde_json::json!({\n \"action_name\": r.action_name,\n \"output\": r.output,\n \"is_error\": r.is_error,\n \"duration_ms\": r.duration.as_millis(),\n })\n })\n .collect();\n\n let result_json = serde_json::json!({\n \"return_value\": result.return_value,\n \"stdout\": result.stdout,\n \"action_results\": action_results,\n \"final_answer\": result.final_answer,\n \"had_error\": result.failure.is_some(),\n \"pending_gate\": result.need_approval.as_ref().map(|na| {\n match na {\n ThreadOutcome::GatePaused { gate_name, action_name, call_id, parameters, resume_kind, resume_output } => serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": gate_name,\n \"action_name\": action_name,\n \"call_id\": call_id,\n \"parameters\": parameters,\n \"resume_kind\": serde_json::to_value(resume_kind).unwrap_or_default(),\n \"resume_output\": resume_output,\n }),\n _ => serde_json::Value::Null,\n }\n }),\n });\n\n ExtFunctionResult::Return(json_to_monty(&result_json))\n }\n Err(e) => ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::RuntimeError,\n Some(format!(\"Code execution failed: {e}\")),\n )),\n }\n}\n\n/// Handle `__execute_action__(name, params, call_id=...)`.\n///\n/// Single source of truth for action execution. Performs:\n/// 1. Lease lookup\n/// 2. Policy check\n/// 3. Lease consumption\n/// 4. Action execution via EffectExecutor\n/// 5. Event emission (ActionExecuted/ActionFailed)\n///\n/// Python owns the working transcript and decides how tool outputs are\n/// represented in internal message history.\nasync fn handle_execute_action(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n effects: &Arc,\n leases: &Arc,\n policy: &Arc,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n) -> ExtFunctionResult {\n let name = match extract_string_arg(args, kwargs, \"name\", 0) {\n Some(n) => n,\n None => {\n return ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::TypeError,\n Some(\"__execute_action__ requires a name argument\".into()),\n ));\n }\n };\n\n let params = args\n .get(1)\n .map(monty_to_json)\n .unwrap_or(serde_json::json!({}));\n\n let call_id = extract_string_kwarg(kwargs, \"call_id\").unwrap_or_default();\n\n let exec_ctx = ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: thread.user_id.clone(),\n step_id: StepId::new(),\n current_call_id: Some(call_id.clone()),\n source_channel: thread_source_channel(thread),\n user_timezone: thread_user_timezone(thread),\n };\n\n // Helper: emit event only. The orchestrator owns transcript recording.\n let emit_and_record = |thread: &mut Thread,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n event_kind: EventKind,\n _call_id: &str,\n _action_name: &str,\n _output: &serde_json::Value| {\n let event = ThreadEvent::new(thread.id, event_kind);\n if let Some(tx) = event_tx {\n let _ = tx.send(event.clone());\n }\n thread.events.push(event);\n thread.updated_at = chrono::Utc::now();\n };\n\n // 1. Find lease for this action\n let lease = match leases.find_lease_for_action(thread.id, &name).await {\n Some(l) => l,\n None => {\n let error = format!(\"No lease for action '{name}'\");\n let output = serde_json::json!({\"error\": &error});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error,\n params_summary: None,\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"output\": output,\n \"is_error\": true,\n });\n return ExtFunctionResult::Return(json_to_monty(&result));\n }\n };\n\n // 2. Check policy\n let action_def = effects\n .available_actions(std::slice::from_ref(&lease))\n .await\n .ok()\n .and_then(|actions| actions.into_iter().find(|a| a.name == name));\n\n if let Some(ref ad) = action_def {\n match policy.evaluate(ad, &lease, &[]) {\n crate::capability::policy::PolicyDecision::Deny { reason } => {\n let output = serde_json::json!({\"error\": format!(\"Denied: {reason}\")});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error: reason,\n params_summary: None,\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"output\": output,\n \"is_error\": true,\n });\n return ExtFunctionResult::Return(json_to_monty(&result));\n }\n crate::capability::policy::PolicyDecision::RequireApproval { .. } => {\n let output = serde_json::json!({\"status\": \"gate_paused\", \"gate_name\": \"approval\"});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ApprovalRequested {\n action_name: name.clone(),\n call_id: call_id.clone(),\n parameters: Some(params.clone()),\n description: None,\n allow_always: None,\n gate_name: None,\n params_summary: summarize_params(&name, ¶ms),\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": \"approval\",\n \"action_name\": name,\n \"call_id\": call_id,\n \"parameters\": params,\n \"resume_kind\": serde_json::to_value(crate::gate::ResumeKind::Approval {\n allow_always: true,\n })\n .unwrap_or_default(),\n });\n return ExtFunctionResult::Return(json_to_monty(&result));\n }\n crate::capability::policy::PolicyDecision::Allow => {}\n }\n }\n\n // 3. Atomically re-find + consume a lease use under a single write\n // lock. This closes the TOCTOU window between the read-only\n // `find_lease_for_action` (used above for the policy check) and the\n // consume — without it, two concurrent calls could both observe a\n // lease with one remaining use and both proceed to execute. Mirrors\n // `structured.rs::execute_action_batch_with_results`.\n let lease = match leases.find_and_consume(thread.id, &name).await {\n Ok(l) => l,\n Err(e) => {\n debug!(error = %e, \"atomic lease find_and_consume failed\");\n let error = format!(\"lease consumption failed for action '{name}': {e}\");\n let output = serde_json::json!({\"error\": &error});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error,\n params_summary: None,\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"output\": output,\n \"is_error\": true,\n });\n return ExtFunctionResult::Return(json_to_monty(&result));\n }\n };\n\n // 4. Execute\n let ps = summarize_params(&name, ¶ms);\n match effects\n .execute_action(&name, params, &lease, &exec_ctx)\n .await\n {\n Ok(r) => {\n // Effect adapters wrap tool errors as `Ok(ActionResult { is_error: true })`\n // — surface them as `ActionFailed` so traces and observers see the\n // failure. See `resolve_tool_future` in `scripting.rs` for the same\n // pattern on the structured-tool path.\n if r.is_error {\n let error_msg = r\n .output\n .get(\"error\")\n .and_then(|v| v.as_str())\n .map(String::from)\n .unwrap_or_else(|| r.output.to_string());\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error: error_msg,\n params_summary: ps.clone(),\n },\n &call_id,\n &name,\n &r.output,\n );\n } else {\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionExecuted {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n duration_ms: r.duration.as_millis() as u64,\n params_summary: ps.clone(),\n },\n &call_id,\n &name,\n &r.output,\n );\n }\n let result = serde_json::json!({\n \"action_name\": r.action_name,\n \"output\": r.output,\n \"is_error\": r.is_error,\n \"duration_ms\": r.duration.as_millis(),\n });\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n Err(EngineError::GatePaused {\n gate_name,\n action_name: _,\n call_id: _,\n parameters,\n resume_kind,\n resume_output,\n }) => {\n let _ = leases.refund_use(lease.id).await;\n let output = serde_json::json!({\"status\": \"gate_paused\", \"gate_name\": gate_name});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ApprovalRequested {\n action_name: name.clone(),\n call_id: call_id.clone(),\n parameters: Some((*parameters).clone()),\n description: None,\n allow_always: match resume_kind.as_ref() {\n crate::gate::ResumeKind::Approval { allow_always } => Some(*allow_always),\n _ => None,\n },\n gate_name: Some(gate_name.clone()),\n params_summary: summarize_params(&name, ¶meters),\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": gate_name,\n \"action_name\": name,\n \"call_id\": call_id,\n \"parameters\": parameters,\n \"resume_kind\": serde_json::to_value(&*resume_kind).unwrap_or_default(),\n \"resume_output\": resume_output,\n });\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n Err(e) => {\n let output = serde_json::json!({\"error\": e.to_string()});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error: e.to_string(),\n params_summary: ps,\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"output\": output,\n \"is_error\": true,\n });\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n }\n}\n\n/// Handle `__execute_actions_parallel__(calls)`.\n///\n/// Batch host function that receives a list of action calls and executes them\n/// concurrently. Each call is a dict with `name`, `params`, and optionally `call_id`.\n///\n/// Returns a list of result dicts (one per call, in order). Each result has the\n/// same shape as `__execute_action__` output, plus an optional gate pause payload.\n///\n/// Events are emitted in original call order after all parallel executions complete.\nasync fn handle_execute_actions_parallel(\n args: &[MontyObject],\n thread: &mut Thread,\n effects: &Arc,\n leases: &Arc,\n policy: &Arc,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n) -> ExtFunctionResult {\n // Parse the calls list from the first argument (list of dicts)\n let calls_json = args\n .first()\n .map(monty_to_json)\n .unwrap_or(serde_json::json!([]));\n let calls_array = match calls_json.as_array() {\n Some(arr) => arr.clone(),\n None => {\n return ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::TypeError,\n Some(\"__execute_actions_parallel__ requires a list of call dicts\".into()),\n ));\n }\n };\n\n if calls_array.is_empty() {\n return ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])));\n }\n\n // Parse each call dict into (name, params, call_id)\n struct ParsedCall {\n name: String,\n params: serde_json::Value,\n call_id: String,\n }\n\n let mut parsed: Vec = Vec::with_capacity(calls_array.len());\n for c in &calls_array {\n let name = c\n .get(\"name\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"\")\n .to_string();\n let params = c.get(\"params\").cloned().unwrap_or(serde_json::json!({}));\n let call_id = c\n .get(\"call_id\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"\")\n .to_string();\n parsed.push(ParsedCall {\n name,\n params,\n call_id,\n });\n }\n\n let step_id = StepId::new();\n\n // ── Phase 1: Preflight (sequential) ─────────────────────────\n // Check leases and policies. Denied → error result. Approval → interrupt.\n\n enum PfOutcome {\n Runnable {\n lease: crate::types::capability::CapabilityLease,\n },\n Error {\n result_json: serde_json::Value,\n event: EventKind,\n output: serde_json::Value,\n },\n }\n\n let mut preflight: Vec> = Vec::with_capacity(parsed.len());\n\n for pc in &parsed {\n // Find lease\n let lease = match leases.find_lease_for_action(thread.id, &pc.name).await {\n Some(l) => l,\n None => {\n let error = format!(\"No lease for action '{}'\", pc.name);\n let output = serde_json::json!({\"error\": &error});\n let result_json = serde_json::json!({\n \"output\": &output,\n \"is_error\": true,\n });\n let event = EventKind::ActionFailed {\n step_id,\n action_name: pc.name.clone(),\n call_id: pc.call_id.clone(),\n error,\n params_summary: None,\n };\n preflight.push(Some(PfOutcome::Error {\n result_json,\n event,\n output,\n }));\n continue;\n }\n };\n\n // Check policy\n let action_def = effects\n .available_actions(std::slice::from_ref(&lease))\n .await\n .ok()\n .and_then(|actions| actions.into_iter().find(|a| a.name == pc.name));\n\n if let Some(ref ad) = action_def {\n match policy.evaluate(ad, &lease, &[]) {\n crate::capability::policy::PolicyDecision::Deny { reason } => {\n let output = serde_json::json!({\"error\": format!(\"Denied: {reason}\")});\n let result_json = serde_json::json!({\n \"output\": &output,\n \"is_error\": true,\n });\n let event = EventKind::ActionFailed {\n step_id,\n action_name: pc.name.clone(),\n call_id: pc.call_id.clone(),\n error: reason,\n params_summary: None,\n };\n preflight.push(Some(PfOutcome::Error {\n result_json,\n event,\n output,\n }));\n continue;\n }\n crate::capability::policy::PolicyDecision::RequireApproval { .. } => {\n // Emit events for earlier errors, then interrupt\n let mut results_json = Vec::with_capacity(preflight.len() + 1);\n for pf in preflight {\n match pf {\n Some(PfOutcome::Error {\n result_json,\n event,\n output: _,\n }) => {\n let ev = ThreadEvent::new(thread.id, event);\n if let Some(tx) = event_tx {\n let _ = tx.send(ev.clone());\n }\n thread.events.push(ev);\n results_json.push(result_json);\n }\n Some(PfOutcome::Runnable { .. }) | None => {\n results_json.push(serde_json::json!(null));\n }\n }\n }\n // Add the approval entry\n let ev = ThreadEvent::new(\n thread.id,\n EventKind::ApprovalRequested {\n action_name: pc.name.clone(),\n call_id: pc.call_id.clone(),\n parameters: Some(pc.params.clone()),\n description: None,\n allow_always: None,\n gate_name: None,\n params_summary: summarize_params(&pc.name, &pc.params),\n },\n );\n if let Some(tx) = event_tx {\n let _ = tx.send(ev.clone());\n }\n thread.events.push(ev);\n thread.updated_at = chrono::Utc::now();\n\n results_json.push(serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": \"approval\",\n \"action_name\": &pc.name,\n \"call_id\": &pc.call_id,\n \"parameters\": &pc.params,\n \"resume_kind\": serde_json::to_value(crate::gate::ResumeKind::Approval {\n allow_always: true,\n })\n .unwrap_or_default(),\n }));\n // Pad with nulls for calls that weren't reached so the\n // Python-side loop can emit ActionResult placeholders for\n // every tool call in the assistant message.\n while results_json.len() < parsed.len() {\n results_json.push(serde_json::json!(null));\n }\n return ExtFunctionResult::Return(json_to_monty(&serde_json::json!(\n results_json\n )));\n }\n crate::capability::policy::PolicyDecision::Allow => {}\n }\n }\n\n // Atomically re-find + consume a lease use under a single write\n // lock, closing the TOCTOU window between the read-only\n // `find_lease_for_action` above and the consume. Mirrors\n // `structured.rs::execute_action_batch_with_results`.\n let lease = match leases.find_and_consume(thread.id, &pc.name).await {\n Ok(l) => l,\n Err(e) => {\n debug!(error = %e, \"atomic lease find_and_consume failed\");\n let error = format!(\"lease consumption failed for action '{}': {e}\", pc.name);\n let output = serde_json::json!({\"error\": &error});\n let result_json = serde_json::json!({\n \"output\": &output,\n \"is_error\": true,\n });\n let event = EventKind::ActionFailed {\n step_id,\n action_name: pc.name.clone(),\n call_id: pc.call_id.clone(),\n error,\n params_summary: None,\n };\n preflight.push(Some(PfOutcome::Error {\n result_json,\n event,\n output,\n }));\n continue;\n }\n };\n\n preflight.push(Some(PfOutcome::Runnable { lease }));\n }\n\n // ── Phase 2: Execute in parallel ────────────────────────────\n\n // Slot array: index → execution result\n let mut slot_results: Vec> = vec![None; parsed.len()];\n let mut slot_events: Vec> = vec![None; parsed.len()];\n let mut slot_outputs: Vec> = vec![None; parsed.len()];\n\n // Separate runnable from errors\n let mut runnable: Vec<(usize, crate::types::capability::CapabilityLease)> = Vec::new();\n for (idx, pf) in preflight.into_iter().enumerate() {\n match pf {\n Some(PfOutcome::Error {\n result_json,\n event,\n output,\n }) => {\n slot_results[idx] = Some(result_json);\n slot_events[idx] = Some(event);\n slot_outputs[idx] = Some(output);\n }\n Some(PfOutcome::Runnable { lease }) => {\n runnable.push((idx, lease));\n }\n None => {}\n }\n }\n\n if runnable.len() == 1 {\n // Single call: execute directly\n let (idx, lease) = runnable.into_iter().next().unwrap(); // safety: len()==1 checked above\n let pc = &parsed[idx];\n let exec_ctx = ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: thread.user_id.clone(),\n step_id,\n current_call_id: Some(pc.call_id.clone()),\n // Read source_channel from thread metadata so downstream tools\n // (e.g. mission_create) can default notify_channels to the\n // originating channel. Hardcoding `None` here was a bug — it\n // silently dropped the gateway routing for any tool dispatched\n // through the parallel batch path.\n source_channel: thread_source_channel(thread),\n user_timezone: thread_user_timezone(thread),\n };\n let ps = summarize_params(&pc.name, &pc.params);\n let (result_json, event, output) = execute_single_action(\n effects,\n &pc.name,\n pc.params.clone(),\n &pc.call_id,\n &lease,\n &exec_ctx,\n ps,\n )\n .await;\n if interrupted_result_needs_refund(&result_json) {\n let _ = leases.refund_use(lease.id).await;\n }\n slot_results[idx] = Some(result_json);\n slot_events[idx] = Some(event);\n slot_outputs[idx] = Some(output);\n } else if runnable.len() > 1 {\n // Multiple calls: execute in parallel via JoinSet\n let mut join_set = tokio::task::JoinSet::new();\n let effects = effects.clone();\n // Capture once outside the loop — the thread's metadata is stable\n // for the duration of the parallel batch.\n let parallel_source_channel = thread_source_channel(thread);\n let parallel_user_timezone = thread_user_timezone(thread);\n\n for (idx, lease) in runnable {\n let pc_name = parsed[idx].name.clone();\n let pc_params = parsed[idx].params.clone();\n let pc_call_id = parsed[idx].call_id.clone();\n let effects = effects.clone();\n let lease = lease.clone();\n let exec_ctx = ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: thread.user_id.clone(),\n step_id,\n current_call_id: Some(pc_call_id.clone()),\n // See comment above — read from thread metadata, not None.\n source_channel: parallel_source_channel.clone(),\n user_timezone: parallel_user_timezone,\n };\n let ps = summarize_params(&pc_name, &pc_params);\n\n join_set.spawn(async move {\n let (result_json, event, output) = execute_single_action(\n &effects,\n &pc_name,\n pc_params,\n &pc_call_id,\n &lease,\n &exec_ctx,\n ps,\n )\n .await;\n (idx, lease.id, result_json, event, output)\n });\n }\n\n while let Some(join_result) = join_set.join_next().await {\n match join_result {\n Ok((idx, lease_id, result_json, event, output)) => {\n if interrupted_result_needs_refund(&result_json) {\n let _ = leases.refund_use(lease_id).await;\n }\n slot_results[idx] = Some(result_json);\n slot_events[idx] = Some(event);\n slot_outputs[idx] = Some(output);\n }\n Err(e) => {\n debug!(\"parallel action execution task panicked: {e}\");\n }\n }\n }\n }\n\n // ── Phase 3: Emit events in order ───────────────────────────\n\n let mut results_json = Vec::with_capacity(parsed.len());\n for idx in 0..parsed.len() {\n let result_json = slot_results[idx].take().unwrap_or(\n serde_json::json!({\"is_error\": true, \"output\": {\"error\": \"execution slot empty\"}}),\n );\n let _output = slot_outputs[idx]\n .take()\n .unwrap_or(serde_json::json!({\"error\": \"no output\"}));\n\n if let Some(event) = slot_events[idx].take() {\n let ev = ThreadEvent::new(thread.id, event);\n if let Some(tx) = event_tx {\n let _ = tx.send(ev.clone());\n }\n thread.events.push(ev);\n }\n\n results_json.push(result_json.clone());\n }\n\n thread.updated_at = chrono::Utc::now();\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!(results_json)))\n}\n\n/// Execute a single action and return (result_json, event, output) for the\n/// batch handler to record. Shared by both single-call and parallel paths.\nasync fn execute_single_action(\n effects: &Arc,\n name: &str,\n params: serde_json::Value,\n call_id: &str,\n lease: &crate::types::capability::CapabilityLease,\n exec_ctx: &ThreadExecutionContext,\n params_summary: Option,\n) -> (serde_json::Value, EventKind, serde_json::Value) {\n match effects.execute_action(name, params, lease, exec_ctx).await {\n Ok(r) => {\n // Surface wrapped errors as ActionFailed (see resolve_tool_future\n // and the parallel execute path for the same pattern).\n let event = if r.is_error {\n let error_msg = r\n .output\n .get(\"error\")\n .and_then(|v| v.as_str())\n .map(String::from)\n .unwrap_or_else(|| r.output.to_string());\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.to_string(),\n call_id: call_id.to_string(),\n error: error_msg,\n params_summary: params_summary.clone(),\n }\n } else {\n EventKind::ActionExecuted {\n step_id: exec_ctx.step_id,\n action_name: name.to_string(),\n call_id: call_id.to_string(),\n duration_ms: r.duration.as_millis() as u64,\n params_summary: params_summary.clone(),\n }\n };\n let result_json = serde_json::json!({\n \"action_name\": r.action_name,\n \"output\": r.output,\n \"is_error\": r.is_error,\n \"duration_ms\": r.duration.as_millis(),\n });\n (result_json, event, r.output)\n }\n Err(EngineError::GatePaused {\n gate_name,\n action_name: _,\n call_id: _,\n parameters,\n resume_kind,\n resume_output,\n }) => {\n let output = serde_json::json!({\"status\": \"gate_paused\", \"gate_name\": &gate_name});\n let event = EventKind::ApprovalRequested {\n action_name: name.to_string(),\n call_id: call_id.to_string(),\n parameters: Some((*parameters).clone()),\n description: None,\n allow_always: match resume_kind.as_ref() {\n crate::gate::ResumeKind::Approval { allow_always } => Some(*allow_always),\n _ => None,\n },\n gate_name: Some(gate_name.clone()),\n params_summary: summarize_params(name, ¶meters),\n };\n let result_json = serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": gate_name,\n \"action_name\": name,\n \"call_id\": call_id,\n \"parameters\": parameters,\n \"resume_kind\": serde_json::to_value(&*resume_kind).unwrap_or_default(),\n \"resume_output\": resume_output,\n });\n (result_json, event, output)\n }\n Err(e) => {\n let output = serde_json::json!({\"error\": e.to_string()});\n let event = EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.to_string(),\n call_id: call_id.to_string(),\n error: e.to_string(),\n params_summary,\n };\n let result_json = serde_json::json!({\n \"output\": &output,\n \"is_error\": true,\n });\n (result_json, event, output)\n }\n }\n}\n\nfn interrupted_result_needs_refund(result: &serde_json::Value) -> bool {\n result.get(\"gate_paused\").and_then(|v| v.as_bool()) == Some(true)\n}\n\n/// Handle `__check_signals__()`.\nfn handle_check_signals(signal_rx: &mut SignalReceiver, thread: &mut Thread) -> ExtFunctionResult {\n match signal_rx.try_recv() {\n Ok(ThreadSignal::Stop) | Ok(ThreadSignal::Suspend) => {\n ExtFunctionResult::Return(MontyObject::String(\"stop\".into()))\n }\n Ok(ThreadSignal::InjectMessage(msg)) => {\n thread.add_message(msg.clone());\n let result = serde_json::json!({\"inject\": msg.content});\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n Ok(ThreadSignal::Resume) | Ok(ThreadSignal::ChildCompleted { .. }) => {\n ExtFunctionResult::Return(MontyObject::None)\n }\n Err(_) => ExtFunctionResult::Return(MontyObject::None),\n }\n}\n\n/// Handle `__emit_event__(kind, **data)`.\nfn handle_emit_event(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n) -> ExtFunctionResult {\n let kind_str = args.first().map(monty_to_string).unwrap_or_default();\n\n let kind = match kind_str.as_str() {\n \"step_started\" => {\n let _step = extract_u64_kwarg(kwargs, \"step\").unwrap_or(0);\n EventKind::StepStarted {\n step_id: StepId::new(),\n }\n }\n \"step_completed\" => {\n let input = extract_u64_kwarg(kwargs, \"input_tokens\").unwrap_or(0);\n let output = extract_u64_kwarg(kwargs, \"output_tokens\").unwrap_or(0);\n // Increment step count (mirrors the old Rust loop's step_count += 1)\n thread.step_count += 1;\n // Track token usage\n thread.total_tokens_used += input + output;\n EventKind::StepCompleted {\n step_id: StepId::new(),\n tokens: TokenUsage {\n input_tokens: input,\n output_tokens: output,\n ..Default::default()\n },\n }\n }\n \"action_executed\" => {\n let action_name = extract_string_kwarg(kwargs, \"action_name\").unwrap_or_default();\n let call_id = extract_string_kwarg(kwargs, \"call_id\").unwrap_or_default();\n EventKind::ActionExecuted {\n step_id: StepId::new(),\n action_name,\n call_id,\n duration_ms: 0,\n params_summary: None,\n }\n }\n \"action_failed\" => {\n let action_name = extract_string_kwarg(kwargs, \"action_name\").unwrap_or_default();\n let call_id = extract_string_kwarg(kwargs, \"call_id\").unwrap_or_default();\n let error = extract_string_kwarg(kwargs, \"error\").unwrap_or_default();\n EventKind::ActionFailed {\n step_id: StepId::new(),\n action_name,\n call_id,\n error,\n params_summary: None,\n }\n }\n \"skill_activated\" => {\n let names_str = extract_string_kwarg(kwargs, \"skill_names\").unwrap_or_default();\n let skill_names: Vec = names_str\n .split(',')\n .map(|s| s.trim().to_string())\n .filter(|s| !s.is_empty())\n .collect();\n EventKind::SkillActivated { skill_names }\n }\n _ => {\n debug!(kind = %kind_str, \"orchestrator: unknown event kind, skipping\");\n return ExtFunctionResult::Return(MontyObject::None);\n }\n };\n\n let event = ThreadEvent::new(thread.id, kind);\n if let Some(tx) = event_tx {\n let _ = tx.send(event.clone());\n }\n thread.events.push(event);\n thread.updated_at = chrono::Utc::now();\n\n ExtFunctionResult::Return(MontyObject::None)\n}\n\n/// Handle `__save_checkpoint__(state, counters)`.\nfn handle_save_checkpoint(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n) -> ExtFunctionResult {\n let state = args\n .first()\n .map(monty_to_json)\n .unwrap_or(serde_json::json!({}));\n let counters = args\n .get(1)\n .map(monty_to_json)\n .unwrap_or(serde_json::json!({}));\n\n sync_runtime_state(thread, Some(&state));\n\n if let Some(metadata) = thread.metadata.as_object_mut() {\n metadata.insert(\n \"runtime_checkpoint\".into(),\n serde_json::json!({\n \"persisted_state\": state,\n \"nudge_count\": counters.get(\"nudge_count\").and_then(|v| v.as_u64()).unwrap_or(0),\n \"consecutive_errors\": counters.get(\"consecutive_errors\").and_then(|v| v.as_u64()).unwrap_or(0),\n \"consecutive_action_errors\": counters.get(\"consecutive_action_errors\").and_then(|v| v.as_u64()).unwrap_or(0),\n \"compaction_count\": counters.get(\"compaction_count\").and_then(|v| v.as_u64()).unwrap_or(0),\n }),\n );\n }\n thread.updated_at = chrono::Utc::now();\n\n ExtFunctionResult::Return(MontyObject::None)\n}\n\n/// Handle `__transition_to__(state, reason)`.\nfn handle_transition_to(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n) -> ExtFunctionResult {\n let state_str = args.first().map(monty_to_string).unwrap_or_default();\n let reason = args.get(1).map(monty_to_string);\n\n let target = match state_str.as_str() {\n \"running\" => crate::types::thread::ThreadState::Running,\n \"completed\" => crate::types::thread::ThreadState::Completed,\n \"failed\" => crate::types::thread::ThreadState::Failed,\n \"waiting\" => crate::types::thread::ThreadState::Waiting,\n \"suspended\" => crate::types::thread::ThreadState::Suspended,\n other => {\n return ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::ValueError,\n Some(format!(\"Unknown thread state: {other}\")),\n ));\n }\n };\n\n match thread.transition_to(target, reason) {\n Ok(()) => ExtFunctionResult::Return(MontyObject::None),\n Err(e) => ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::RuntimeError,\n Some(format!(\"State transition failed: {e}\")),\n )),\n }\n}\n\n/// Handle `__retrieve_docs__(goal, max_docs)`.\nasync fn handle_retrieve_docs(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &Thread,\n retrieval: Option<&RetrievalEngine>,\n) -> ExtFunctionResult {\n let retrieval = match retrieval {\n Some(r) => r,\n None => return ExtFunctionResult::Return(json_to_monty(&serde_json::json!([]))),\n };\n\n let goal = args.first().map(monty_to_string).unwrap_or_default();\n let max_docs = args\n .get(1)\n .and_then(|v| match v {\n MontyObject::Int(i) => Some(*i as usize),\n _ => None,\n })\n .unwrap_or(5);\n\n match retrieval\n .retrieve_context(thread.project_id, &thread.user_id, &goal, max_docs)\n .await\n {\n Ok(docs) => {\n let docs_json: Vec = docs\n .iter()\n .map(|d| {\n serde_json::json!({\n \"type\": format!(\"{:?}\", d.doc_type),\n \"title\": d.title,\n \"content\": d.content,\n })\n })\n .collect();\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!(docs_json)))\n }\n Err(e) => {\n debug!(\"retrieve_docs failed: {e}\");\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])))\n }\n }\n}\n\n/// Handle `__check_budget__()`.\nfn handle_check_budget(thread: &Thread) -> ExtFunctionResult {\n let tokens_remaining = thread\n .config\n .max_tokens_total\n .map(|max| max.saturating_sub(thread.total_tokens_used))\n .unwrap_or(u64::MAX);\n\n let time_remaining_ms = thread\n .config\n .max_duration\n .map(|dur| {\n let elapsed = chrono::Utc::now()\n .signed_duration_since(thread.created_at)\n .num_milliseconds()\n .max(0) as u64;\n dur.as_millis() as u64 - elapsed.min(dur.as_millis() as u64)\n })\n .unwrap_or(u64::MAX);\n\n let usd_remaining = thread\n .config\n .max_budget_usd\n .map(|max| (max - thread.total_cost_usd).max(0.0));\n\n let result = serde_json::json!({\n \"tokens_remaining\": tokens_remaining,\n \"time_remaining_ms\": time_remaining_ms,\n \"usd_remaining\": usd_remaining,\n });\n\n ExtFunctionResult::Return(json_to_monty(&result))\n}\n\n/// Handle `__get_actions__()`.\nasync fn handle_get_actions(\n thread: &mut Thread,\n effects: &Arc,\n leases: &Arc,\n store: Option<&Arc>,\n) -> ExtFunctionResult {\n if let Err(e) =\n reconcile_dynamic_tool_lease(thread, effects, leases, store, &crate::LeasePlanner::new())\n .await\n {\n warn_on_lease_refresh_failure(\"get_actions\", &e);\n }\n\n let active_leases = leases.active_for_thread(thread.id).await;\n match effects.available_actions(&active_leases).await {\n Ok(actions) => {\n let actions_json: Vec = actions\n .iter()\n .map(|a| {\n serde_json::json!({\n \"name\": a.name,\n \"description\": a.description,\n \"params\": a.parameters_schema,\n })\n })\n .collect();\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!(actions_json)))\n }\n Err(e) => {\n debug!(\"get_actions failed: {e}\");\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])))\n }\n }\n}\n\n/// Handle `__list_skills__()`.\n///\n/// Loads all `DocType::Skill` MemoryDocs from the project and returns them\n/// as a list of Python dicts. The Python orchestrator handles scoring,\n/// selection, and injection — Rust just provides data access.\nasync fn handle_list_skills(\n _args: &[MontyObject],\n thread: &Thread,\n store: Option<&Arc>,\n) -> ExtFunctionResult {\n let Some(store) = store else {\n return ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])));\n };\n\n // Use shared listing: user's own skills + system/admin-installed skills.\n let docs = match store\n .list_memory_docs_with_shared(thread.project_id, &thread.user_id)\n .await\n {\n Ok(docs) => docs,\n Err(e) => {\n debug!(\"__list_skills__: failed to load docs: {e}\");\n return ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])));\n }\n };\n\n let skills: Vec = docs\n .into_iter()\n .filter(|d| d.doc_type == crate::types::memory::DocType::Skill)\n .map(|d| {\n serde_json::json!({\n \"doc_id\": d.id.0.to_string(),\n \"title\": d.title,\n \"content\": d.content,\n \"metadata\": d.metadata,\n })\n })\n .collect();\n\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!(skills)))\n}\n\n/// Handle `__record_skill_usage__(doc_id, success)`.\n///\n/// Records that a skill was used in this thread. Called by the Python\n/// orchestrator after skill-assisted execution completes.\nasync fn handle_record_skill_usage(\n args: &[MontyObject],\n store: Option<&Arc>,\n) -> ExtFunctionResult {\n let Some(store) = store else {\n return ExtFunctionResult::Return(MontyObject::None);\n };\n\n let doc_id_str = args.first().map(monty_to_string).unwrap_or_default();\n let success = args\n .get(1)\n .map(|o| matches!(o, MontyObject::Bool(true)))\n .unwrap_or(false);\n\n let Ok(uuid) = uuid::Uuid::parse_str(&doc_id_str) else {\n debug!(\"__record_skill_usage__: invalid doc_id: {doc_id_str}\");\n return ExtFunctionResult::Return(MontyObject::None);\n };\n\n let tracker = crate::memory::SkillTracker::new(Arc::clone(store));\n if let Err(e) = tracker\n .record_usage(crate::types::memory::DocId(uuid), success)\n .await\n {\n debug!(\"__record_skill_usage__: failed: {e}\");\n }\n\n ExtFunctionResult::Return(MontyObject::None)\n}\n\n/// Handle `__regex_match__(pattern, text) -> bool`.\n///\n/// Compiles `pattern` with a bounded size limit and returns whether it\n/// matches anywhere in `text`. Invalid regex or a size-limit violation\n/// returns `False` silently. Used by the Python skill selector for regex\n/// pattern scoring (Monty has no `re` module).\n///\n/// **Security: ReDoS safety.** This handler accepts arbitrary patterns from\n/// the Python orchestrator (which itself receives them from skill manifests)\n/// and runs them on user-supplied text. Safety relies on the `regex` crate's\n/// linear-time matching guarantee (no backreferences, no lookaround) plus the\n/// 64 KiB compiled-size cap and DFA-size cap below. If the `regex` crate is\n/// ever swapped for `fancy-regex` (which supports backreferences and is NOT\n/// linear-time), this becomes a real ReDoS vector. This is enforced by\n/// convention and documentation only — see the top-of-crate comment in\n/// `crates/ironclaw_engine/src/lib.rs`. (A `#[cfg(feature = \"fancy-regex\")]\n/// compile_error!` tripwire was evaluated but conflicts with\n/// `cargo clippy --all-features` which is the standard CI command.)\nfn handle_regex_match(args: &[MontyObject]) -> ExtFunctionResult {\n let pattern = args.first().map(monty_to_string).unwrap_or_default();\n let text = args.get(1).map(monty_to_string).unwrap_or_default();\n if pattern.is_empty() {\n return ExtFunctionResult::Return(MontyObject::Bool(false));\n }\n // Cap compiled regex size to prevent ReDoS (matches the 64 KiB limit used\n // by `LoadedSkill::compile_patterns` in `ironclaw_skills`). Also cap the\n // lazy-DFA cache: the `regex` crate's DFA can grow beyond `size_limit`\n // during matching, so `dfa_size_limit` is a separate defensive cap on\n // memory allocation from a crafted pattern over untrusted skill manifests.\n const MAX_REGEX_SIZE: usize = 1 << 16;\n let matched = match regex::RegexBuilder::new(&pattern)\n .size_limit(MAX_REGEX_SIZE)\n .dfa_size_limit(MAX_REGEX_SIZE)\n .build()\n {\n Ok(re) => re.is_match(&text),\n Err(e) => {\n debug!(\"__regex_match__: invalid pattern '{pattern}': {e}\");\n false\n }\n };\n ExtFunctionResult::Return(MontyObject::Bool(matched))\n}\n\n/// Handle `__set_active_skills__(skills)`.\n///\n/// Persists the selected skill provenance onto the thread so post-run learning\n/// flows can reason about the exact skill versions and snippets that were active.\nfn handle_set_active_skills(args: &[MontyObject], thread: &mut Thread) -> ExtFunctionResult {\n let skills_json = args\n .first()\n .map(monty_to_json)\n .unwrap_or_else(|| serde_json::json!([]));\n\n let skills = match serde_json::from_value::>(skills_json) {\n Ok(skills) => skills,\n Err(e) => {\n debug!(\"__set_active_skills__: invalid payload: {e}\");\n return ExtFunctionResult::Return(MontyObject::None);\n }\n };\n\n if let Err(e) = thread.set_active_skills(&skills) {\n debug!(\"__set_active_skills__: failed to persist active skills: {e}\");\n }\n\n ExtFunctionResult::Return(MontyObject::None)\n}\n\n// ── Helpers ─────────────────────────────────────────────────\n\n/// Build the context variables injected into the orchestrator Python.\nfn build_orchestrator_inputs(\n thread: &Thread,\n persisted_state: &serde_json::Value,\n) -> (Vec, Vec) {\n let names = vec![\n \"context\".into(),\n \"goal\".into(),\n \"actions\".into(),\n \"state\".into(),\n \"config\".into(),\n ];\n\n // Build orchestrator bootstrap context. Prefer the internal execution\n // transcript when present, otherwise fall back to the user-visible transcript.\n let bootstrap_messages = if thread.internal_messages.is_empty() {\n &thread.messages\n } else {\n &thread.internal_messages\n };\n let context: Vec = bootstrap_messages\n .iter()\n .map(|m| {\n // Serialize action_calls through the Python interchange shape\n // (`{name, call_id, params}`) so the bootstrap context is\n // round-trip compatible with `python_json_to_action_calls`.\n // Using bare `m.action_calls` here produces the canonical Rust\n // serde format (`{action_name, id, parameters}`), which the\n // Python orchestrator passes back verbatim on the next\n // `__llm_complete__` call — and `python_json_to_action_calls`\n // then fails with \"missing field `name`\", orphaning every\n // subsequent tool result. This is the SECOND code path (after\n // `handle_llm_complete`) that feeds action_calls into the\n // Python working transcript; both must use the same shape.\n let calls_json = m\n .action_calls\n .as_ref()\n .map(|calls| serde_json::Value::Array(action_calls_to_python_json(calls)));\n serde_json::json!({\n \"role\": format!(\"{:?}\", m.role),\n \"content\": m.content,\n \"action_name\": m.action_name,\n \"action_call_id\": m.action_call_id,\n \"action_calls\": calls_json,\n })\n })\n .collect();\n\n // Build config\n let config = serde_json::json!({\n \"max_iterations\": thread.config.max_iterations,\n \"max_tool_intent_nudges\": thread.config.max_tool_intent_nudges,\n \"enable_tool_intent_nudge\": thread.config.enable_tool_intent_nudge,\n \"max_consecutive_errors\": thread.config.max_consecutive_errors,\n \"max_tokens_total\": thread.config.max_tokens_total,\n \"max_budget_usd\": thread.config.max_budget_usd,\n \"model_context_limit\": thread.config.model_context_limit,\n \"enable_compaction\": thread.config.enable_compaction,\n \"compaction_threshold\": thread.config.compaction_threshold,\n \"depth\": thread.config.depth,\n \"max_depth\": thread.config.max_depth,\n \"step_count\": thread.step_count,\n });\n\n let values = vec![\n json_to_monty(&serde_json::json!(context)),\n MontyObject::String(thread.goal.clone()),\n json_to_monty(&serde_json::json!([])), // actions loaded dynamically via __get_actions__\n json_to_monty(persisted_state),\n json_to_monty(&config),\n ];\n\n (names, values)\n}\n\n/// JSON shape used to interchange `ActionCall`s with the Python orchestrator.\n///\n/// This is the *single* place that defines the field naming convention used\n/// across the Python boundary. It is intentionally separate from the\n/// canonical `ActionCall` type because:\n///\n/// - `ActionCall` uses Rust-idiomatic field names (`id`, `action_name`,\n/// `parameters`) and is also persisted into Step records and ThreadEvents.\n/// Renaming its serde fields would invalidate every existing row.\n/// - The Python orchestrator uses friendlier names (`call_id`, `name`,\n/// `params`) that read naturally in CodeAct prompts and `default.py`.\n///\n/// Without this type, the round-trip is asymmetric: Rust → Python uses one\n/// shape, Python → Rust used `serde_json::from_value::>`\n/// which silently fails (`.ok()` swallows the error) and produces `None`,\n/// which means assistant messages came back without `action_calls`. The\n/// downstream effect is that every tool result looks orphaned to\n/// `sanitize_tool_messages` and gets rewritten as a user message — losing\n/// the assistant ↔ tool_result linkage the LLM needs to reason about prior\n/// tool calls.\n#[derive(Debug, serde::Serialize, serde::Deserialize)]\nstruct PythonActionCall {\n name: String,\n call_id: String,\n params: serde_json::Value,\n}\n\nimpl From<&ActionCall> for PythonActionCall {\n fn from(c: &ActionCall) -> Self {\n Self {\n name: c.action_name.clone(),\n call_id: c.id.clone(),\n params: c.parameters.clone(),\n }\n }\n}\n\nimpl From for ActionCall {\n fn from(p: PythonActionCall) -> Self {\n Self {\n id: p.call_id,\n action_name: p.name,\n parameters: p.params,\n }\n }\n}\n\n/// Serialize a slice of `ActionCall`s into the Python interchange shape.\n///\n/// On serialization failure (essentially unreachable for `String + String +\n/// Value`, but still possible if the `serde_json::Value` parameters tree\n/// contains a key whose stringification fails), the entry is **dropped**\n/// from the output rather than replaced with `Value::Null`. The previous\n/// `unwrap_or_else(|_| Value::Null)` corrupted the array — Python's\n/// `default.py` accesses `c.get(\"name\")` / `c.get(\"call_id\")` /\n/// `c.get(\"params\")` on each entry, so a `null` would crash with a Python\n/// `AttributeError` and lose the entire LLM step. `filter_map` produces a\n/// shorter array, which Python's tool-result loop handles correctly because\n/// it iterates `range(len(results))` against the shortened call list. The\n/// warn log is preserved so operators have a breadcrumb if it ever fires.\nfn action_calls_to_python_json(calls: &[ActionCall]) -> Vec {\n calls\n .iter()\n .filter_map(|c| match serde_json::to_value(PythonActionCall::from(c)) {\n Ok(value) => Some(value),\n Err(e) => {\n warn!(\n error = %e,\n action_name = %c.action_name,\n \"Failed to serialize ActionCall for Python orchestrator — dropping entry\"\n );\n None\n }\n })\n .collect()\n}\n\n/// Extract the last `n` characters from `s`.\n///\n/// Error tracebacks appear at the end of stdout, after any `print()` output.\n/// Using the head would capture the print statements instead of the error.\nfn tail_chars(s: &str, n: usize) -> String {\n let char_count = s.chars().count();\n if char_count > n {\n s.chars().skip(char_count - n).collect()\n } else {\n s.to_owned()\n }\n}\n\n/// Build a PII-safe summary of an `action_calls` JSON value for log output.\n///\n/// The action_calls payload contains tool parameters, which can carry user\n/// PII (search queries, file names, email content, conversation text).\n/// Dumping the full value into a `warn!` log would leak that PII to log\n/// aggregation systems (Datadog, CloudWatch, Sentry) the moment the parser\n/// fails — and the parser only fails when the Python ↔ Rust shape drifts,\n/// which is exactly when an operator is most likely to be grepping logs.\n///\n/// We emit only the structural information operators actually need to\n/// debug a shape drift: array length and the keys of the first entry. The\n/// keys themselves are not user data — they're field names like\n/// `name`/`call_id`/`params` that are static across all calls.\nfn summarize_action_calls_for_log(value: &serde_json::Value) -> String {\n match value.as_array() {\n Some(arr) if arr.is_empty() => \"empty array\".to_string(),\n Some(arr) => {\n let first_keys = arr\n .first()\n .and_then(|v| v.as_object())\n .map(|obj| {\n let mut keys: Vec<&str> = obj.keys().map(String::as_str).collect();\n keys.sort_unstable();\n keys.join(\",\")\n })\n .unwrap_or_else(|| \"\".to_string());\n format!(\n \"array of {} entries; first entry keys: [{}]\",\n arr.len(),\n first_keys\n )\n }\n None => format!(\"non-array value of type {}\", json_value_type_name(value)),\n }\n}\n\n/// Cheap type-name string for a `serde_json::Value`. Used by\n/// `summarize_action_calls_for_log` to surface the wrong-shape case\n/// (e.g. Python passed a string instead of an array) without leaking the\n/// actual contents.\nfn json_value_type_name(value: &serde_json::Value) -> &'static str {\n match value {\n serde_json::Value::Null => \"null\",\n serde_json::Value::Bool(_) => \"bool\",\n serde_json::Value::Number(_) => \"number\",\n serde_json::Value::String(_) => \"string\",\n serde_json::Value::Array(_) => \"array\",\n serde_json::Value::Object(_) => \"object\",\n }\n}\n\n/// Deserialize an `action_calls` JSON array (in Python interchange shape)\n/// back into canonical `ActionCall`s.\n///\n/// Logs a warning on failure rather than swallowing silently. The whole\n/// commit that introduced this helper exists to undo a `.ok()` swallow that\n/// dropped action_calls without any signal — replacing it with another\n/// `.ok()?` would re-introduce the same trap, just one layer deeper. If the\n/// shape ever drifts again (Python orchestrator field rename, extra\n/// required field, partial migration), the warning is the operator-visible\n/// breadcrumb that explains why subsequent tool results suddenly look\n/// orphaned to `sanitize_tool_messages`.\n///\n/// The warn log emits a structural summary (`summarize_action_calls_for_log`)\n/// instead of the raw value because tool parameters can contain user PII.\nfn python_json_to_action_calls(value: &serde_json::Value) -> Option> {\n match serde_json::from_value::>(value.clone()) {\n Ok(parsed) => Some(parsed.into_iter().map(ActionCall::from).collect()),\n Err(e) => {\n warn!(\n error = %e,\n shape = %summarize_action_calls_for_log(value),\n \"Failed to parse action_calls from Python orchestrator — \\\n assistant message will lose tool_call linkage and downstream \\\n tool results will be rewritten as user messages\"\n );\n None\n }\n }\n}\n\nfn json_to_thread_messages(value: &serde_json::Value) -> Option> {\n let arr = value.as_array()?;\n let mut messages = Vec::with_capacity(arr.len());\n\n for item in arr {\n let role = item.get(\"role\").and_then(|v| v.as_str()).unwrap_or(\"User\");\n let content = item\n .get(\"content\")\n .and_then(|v| v.as_str())\n .unwrap_or_default();\n // Filter out null before calling the parser — `action_calls: null`\n // is Python's legitimate \"this message has no tool calls\" signal (text\n // response), not a parse error. Without this filter, the warn log in\n // python_json_to_action_calls fires on every text-only assistant\n // message with \"invalid type: null, expected a sequence\".\n let action_calls = item\n .get(\"action_calls\")\n .filter(|v| !v.is_null())\n .and_then(python_json_to_action_calls);\n\n let message = match role {\n \"System\" | \"system\" => ThreadMessage::system(content),\n \"Assistant\" | \"assistant\" => {\n if let Some(calls) = action_calls {\n ThreadMessage::assistant_with_actions(Some(content.to_string()), calls)\n } else {\n ThreadMessage::assistant(content)\n }\n }\n \"ActionResult\" | \"action_result\" => ThreadMessage::action_result(\n item.get(\"action_call_id\")\n .and_then(|v| v.as_str())\n .unwrap_or_default(),\n item.get(\"action_name\")\n .and_then(|v| v.as_str())\n .unwrap_or_default(),\n content,\n ),\n _ => ThreadMessage::user(content),\n };\n messages.push(message);\n }\n\n Some(messages)\n}\n\nfn sync_runtime_state(thread: &mut Thread, state: Option<&serde_json::Value>) {\n let Some(state) = state else {\n return;\n };\n if let Some(messages) = state\n .get(\"working_messages\")\n .and_then(json_to_thread_messages)\n {\n thread.internal_messages = messages;\n thread.updated_at = chrono::Utc::now();\n }\n}\n\nfn sync_visible_outcome(thread: &mut Thread, outcome: &ThreadOutcome) {\n if let ThreadOutcome::Completed {\n response: Some(response),\n } = outcome\n {\n let already_present = thread\n .messages\n .last()\n .map(|msg| {\n msg.role == crate::types::message::MessageRole::Assistant\n && msg.content == *response\n })\n .unwrap_or(false);\n if !already_present {\n thread.add_message(ThreadMessage::assistant(response));\n }\n }\n}\n\n/// Parse the orchestrator's return value into a ThreadOutcome.\nfn parse_outcome(result: &serde_json::Value) -> ThreadOutcome {\n let outcome = result\n .get(\"outcome\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"completed\");\n\n match outcome {\n \"completed\" => ThreadOutcome::Completed {\n response: result\n .get(\"response\")\n .and_then(|v| v.as_str())\n .map(String::from),\n },\n \"stopped\" => ThreadOutcome::Stopped,\n \"max_iterations\" => ThreadOutcome::MaxIterations,\n \"failed\" => ThreadOutcome::Failed {\n error: result\n .get(\"error\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"unknown error\")\n .to_string(),\n },\n \"gate_paused\" => {\n let resume_kind_value = result\n .get(\"resume_kind\")\n .cloned()\n .unwrap_or(serde_json::json!({}));\n let resume_kind = serde_json::from_value(resume_kind_value).unwrap_or(\n crate::gate::ResumeKind::Approval {\n allow_always: false,\n },\n );\n ThreadOutcome::GatePaused {\n gate_name: result\n .get(\"gate_name\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"unknown\")\n .to_string(),\n action_name: result\n .get(\"action_name\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"\")\n .to_string(),\n call_id: result\n .get(\"call_id\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"\")\n .to_string(),\n parameters: result\n .get(\"parameters\")\n .cloned()\n .unwrap_or(serde_json::json!({})),\n resume_kind,\n resume_output: result.get(\"resume_output\").cloned(),\n }\n }\n _ => ThreadOutcome::Completed { response: None },\n }\n}\n\nfn extract_string_arg(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n name: &str,\n position: usize,\n) -> Option {\n for (k, v) in kwargs {\n if let MontyObject::String(key) = k\n && key == name\n {\n return Some(monty_to_string(v));\n }\n }\n args.get(position).map(monty_to_string)\n}\n\nfn extract_string_kwarg(kwargs: &[(MontyObject, MontyObject)], name: &str) -> Option {\n for (k, v) in kwargs {\n if let MontyObject::String(key) = k\n && key == name\n {\n return Some(monty_to_string(v));\n }\n }\n None\n}\n\nfn extract_u64_kwarg(kwargs: &[(MontyObject, MontyObject)], name: &str) -> Option {\n for (k, v) in kwargs {\n if let MontyObject::String(key) = k\n && key == name\n && let MontyObject::Int(i) = v\n {\n return Some(*i as u64);\n }\n }\n None\n}\n\n#[cfg(test)]\nmod tests {\n use super::*;\n use crate::types::memory::{DocType, MemoryDoc};\n use crate::types::project::ProjectId;\n\n // ── Python helper unit tests via Monty ──────────────────────\n //\n // Extracts the helper functions from the default orchestrator and\n // evaluates `signals_tool_intent(text)` directly, mirroring the V1\n // Rust unit test suite in src/llm/reasoning.rs.\n\n /// Run a Python expression that returns a bool by prepending the\n /// orchestrator helper definitions and wrapping in `FINAL(expr)`.\n /// Run a Python snippet and drive the Monty VM, returning the FINAL()\n /// value as a `MontyObject`. This is the common core for `eval_python_bool`\n /// and `eval_python_int`.\n fn run_python_final(code: String) -> MontyObject {\n let runner =\n MontyRun::new(code, \"test.py\", vec![]).expect(\"Failed to parse orchestrator helpers\");\n let mut stdout = String::new();\n let tracker = LimitedTracker::new(ResourceLimits::new().max_allocations(500_000));\n\n let mut progress = runner\n .start(vec![], tracker, PrintWriter::Collect(&mut stdout))\n .expect(\"Failed to start orchestrator test\");\n\n loop {\n match progress {\n RunProgress::Complete(obj) => return obj,\n RunProgress::FunctionCall(call) => {\n if call.function_name == \"FINAL\" {\n let val = call.args.first().cloned().unwrap_or(MontyObject::None);\n let _ = call.resume(\n ExtFunctionResult::Return(MontyObject::None),\n PrintWriter::Collect(&mut stdout),\n );\n return val;\n }\n let ext_result = match call.function_name.as_str() {\n \"__regex_match__\" => handle_regex_match(&call.args),\n _ => ExtFunctionResult::Return(MontyObject::None),\n };\n progress = call\n .resume(ext_result, PrintWriter::Collect(&mut stdout))\n .expect(\"resume failed\");\n }\n RunProgress::NameLookup(lookup) => {\n progress = lookup\n .resume(\n NameLookupResult::Undefined,\n PrintWriter::Collect(&mut stdout),\n )\n .expect(\"name lookup resume failed\");\n }\n _ => panic!(\"Unexpected RunProgress variant in test\"),\n }\n }\n }\n\n fn eval_python_bool(expr: &str) -> bool {\n // Extract only the helper functions (everything before run_loop)\n let helpers_end = DEFAULT_ORCHESTRATOR\n .find(\"\\ndef run_loop(\")\n .unwrap_or(DEFAULT_ORCHESTRATOR.len());\n let helpers = &DEFAULT_ORCHESTRATOR[..helpers_end]; // safety: find() returns a char boundary on this ASCII-only constant\n\n let code = format!(\"{helpers}\\nFINAL({expr})\");\n match run_python_final(code) {\n MontyObject::Bool(v) => v,\n other => panic!(\"Expected bool, got: {other:?}\"),\n }\n }\n\n /// Run a Python program (with orchestrator helpers in scope) that ends\n /// with `FINAL(int_expr)` and return the integer value.\n fn eval_python_int(program: &str) -> i64 {\n let helpers_end = DEFAULT_ORCHESTRATOR\n .find(\"\\ndef run_loop(\")\n .unwrap_or(DEFAULT_ORCHESTRATOR.len());\n let helpers = &DEFAULT_ORCHESTRATOR[..helpers_end];\n\n let code = format!(\"{helpers}\\n{program}\");\n match run_python_final(code) {\n MontyObject::Int(v) => v,\n other => panic!(\"Expected int, got: {other:?}\"),\n }\n }\n\n // ── __regex_match__ host function reachability ───────────────\n\n #[test]\n fn regex_match_host_function_is_callable_from_monty() {\n // Regression test for PR #1736 review (serrrfirat, 3059161877):\n // verify that Monty's NameLookup + FunctionCall dispatch actually\n // reaches `handle_regex_match` when default.py calls\n // `__regex_match__(...)`. If Monty ever starts resolving the name\n // before the call, this test will fail with a NameError.\n assert!(eval_python_bool(\n r#\"bool(__regex_match__(\"abc\", \"xxabcxx\"))\"#\n ));\n assert!(!eval_python_bool(\n r#\"bool(__regex_match__(\"zzz\", \"xxabcxx\"))\"#\n ));\n // Invalid pattern should return false silently (the host function\n // swallows the compile error).\n assert!(!eval_python_bool(r#\"bool(__regex_match__(\"[\", \"abc\"))\"#));\n }\n\n // ── True positives (should trigger nudge) ───────────────────\n\n #[test]\n fn signals_tool_intent_true_positives() {\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"Let me search for that file.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'll fetch the data now.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'm going to check the logs.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"Let me add it now.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I will run the tests to verify.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'll look up the documentation.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"Let me read the file contents.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'm going to execute the command.\")\"#\n ));\n }\n\n // ── True negatives: conversational phrases ──────────────────\n\n #[test]\n fn signals_tool_intent_true_negatives_conversational() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me explain how this works.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me know if you need anything.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me think about this.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me summarize the findings.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me clarify what I mean.\")\"#\n ));\n }\n\n // ── Exclusion takes precedence ──────────────────────────────\n\n #[test]\n fn signals_tool_intent_exclusion_takes_precedence() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me explain the approach, then I'll search for the file.\")\"#\n ));\n }\n\n // ── Code blocks are stripped ────────────────────────────────\n\n #[test]\n fn signals_tool_intent_ignores_code_blocks() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Here's the code:\\n\\n```\\nfn main() {\\n println!(\\\"Let me search the database\\\");\\n}\\n```\")\"#\n ));\n }\n\n #[test]\n fn signals_tool_intent_ignores_indented_code() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Here's the code:\\n\\n println!(\\\"I'll fetch the data\\\");\\n\\nThat's it.\")\"#\n ));\n }\n\n // ── Plain informational text ────────────────────────────────\n\n #[test]\n fn signals_tool_intent_ignores_plain_text() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"The task is complete.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Here are the results you asked for.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"I found 3 matching files.\")\"#\n ));\n }\n\n // ── Quoted strings are stripped ─────────────────────────────\n\n #[test]\n fn signals_tool_intent_ignores_quoted_strings() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"The button says \\\"Let me search the database\\\" to the user.\")\"#\n ));\n // But unquoted intent should still trigger\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'll fetch the results for you.\")\"#\n ));\n }\n\n // ── Shadowed prefix (exclusion cancels all) ─────────────────\n\n #[test]\n fn signals_tool_intent_shadowed_prefix() {\n // \"let me think\" is an exclusion → entire text returns false\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Sure, let me think about it. Actually, let me search for the file.\")\"#\n ));\n }\n\n // ── Regression: trace false positive (news content) ─────────\n\n #[test]\n fn signals_tool_intent_no_false_positive_news_content() {\n // \"I can\" + \"call\" in news content triggered false positive in old code\n let news_response = concat!(\n \"The latest headlines suggest this is a fast-moving war.\\n\",\n \"- Reuters: Iran is calling US peace proposals unrealistic.\\n\",\n \"If you want, I can do one of these next:\\n\",\n \"1. give you a 5-bullet update\\n\",\n \"2. focus just on military developments\",\n );\n assert!(!eval_python_bool(&format!(\n \"signals_tool_intent({news_response:?})\"\n )));\n }\n\n #[test]\n fn signals_tool_intent_no_false_positive_past_tense() {\n // \"I fetched\" / \"I already called\" should not trigger\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"I already completed the needed action call by fetching current news feeds.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Current status from the live feeds I fetched:\")\"#\n ));\n }\n\n #[test]\n fn signals_tool_intent_no_false_positive_offer() {\n // \"If you want, I can fetch...\" uses \"I can\" which is not a V1 prefix\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"If you want, I can next fetch a cleaner update.\")\"#\n ));\n }\n\n #[tokio::test]\n async fn load_orchestrator_without_store_returns_default() {\n let (code, version) = load_orchestrator(None, ProjectId::new(), true).await;\n assert_eq!(version, 0);\n assert!(code.contains(\"run_loop\"));\n assert!(code.contains(\"__llm_complete__\"));\n }\n\n #[tokio::test]\n async fn load_orchestrator_with_runtime_version() {\n let project_id = ProjectId::new();\n let mut doc = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"custom_orchestrator_code()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc.metadata = serde_json::json!({\"version\": 1});\n\n let store = Arc::new(crate::tests::InMemoryStore::with_docs(vec![doc]));\n let (code, version) =\n load_orchestrator(Some(&(store as Arc)), project_id, true).await;\n assert_eq!(version, 1);\n assert!(code.contains(\"custom_orchestrator_code\"));\n }\n\n #[tokio::test]\n async fn load_orchestrator_picks_highest_version() {\n let project_id = ProjectId::new();\n let mut doc_v1 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v1_code()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v1.metadata = serde_json::json!({\"version\": 1});\n\n let mut doc_v3 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v3_code()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v3.metadata = serde_json::json!({\"version\": 3});\n\n let mut doc_v2 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v2_code()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v2.metadata = serde_json::json!({\"version\": 2});\n\n let store = Arc::new(crate::tests::InMemoryStore::with_docs(vec![\n doc_v1, doc_v3, doc_v2,\n ]));\n let (code, version) =\n load_orchestrator(Some(&(store as Arc)), project_id, true).await;\n assert_eq!(version, 3);\n assert!(code.contains(\"v3_code\"));\n }\n\n #[tokio::test]\n async fn rollback_after_max_failures() {\n let project_id = ProjectId::new();\n\n // Create v2 orchestrator\n let mut doc_v2 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v2_buggy()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v2.metadata = serde_json::json!({\"version\": 2});\n\n // Create v1 orchestrator (fallback)\n let mut doc_v1 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v1_stable()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v1.metadata = serde_json::json!({\"version\": 1});\n\n // Create failure tracker showing v2 has 3 failures\n let tracker = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n FAILURE_TRACKER_TITLE,\n r#\"{\"version\": 2, \"count\": 3}\"#,\n )\n .with_tags(vec![\"orchestrator_meta\".to_string()]);\n\n let store = Arc::new(crate::tests::InMemoryStore::with_docs(vec![\n doc_v2, doc_v1, tracker,\n ]));\n let (code, version) =\n load_orchestrator(Some(&(store as Arc)), project_id, true).await;\n\n // Should skip v2 (too many failures) and load v1\n assert_eq!(version, 1);\n assert!(code.contains(\"v1_stable\"));\n }\n\n #[tokio::test]\n async fn rollback_to_default_when_all_versions_fail() {\n let project_id = ProjectId::new();\n\n // Single version with 3 failures\n let mut doc_v1 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v1_broken()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v1.metadata = serde_json::json!({\"version\": 1});\n\n let tracker = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n FAILURE_TRACKER_TITLE,\n r#\"{\"version\": 1, \"count\": 5}\"#,\n )\n .with_tags(vec![\"orchestrator_meta\".to_string()]);\n\n let store = Arc::new(crate::tests::InMemoryStore::with_docs(vec![\n doc_v1, tracker,\n ]));\n let (code, version) =\n load_orchestrator(Some(&(store as Arc)), project_id, true).await;\n\n // Should fall back to compiled-in default (v0)\n assert_eq!(version, 0);\n assert!(code.contains(\"run_loop\"));\n }\n\n #[tokio::test]\n async fn record_and_reset_failures() {\n let project_id = ProjectId::new();\n let store: Arc = Arc::new(crate::tests::InMemoryStore::with_docs(vec![]));\n\n // Record 3 failures\n record_orchestrator_failure(&store, project_id, 2).await;\n record_orchestrator_failure(&store, project_id, 2).await;\n record_orchestrator_failure(&store, project_id, 2).await;\n\n let docs = store.list_shared_memory_docs(project_id).await.unwrap();\n let count = load_failure_count(&docs);\n assert_eq!(count, 3);\n\n // Reset\n reset_orchestrator_failures(&store, project_id).await;\n let docs = store.list_shared_memory_docs(project_id).await.unwrap();\n let count = load_failure_count(&docs);\n assert_eq!(count, 0);\n }\n\n #[tokio::test]\n async fn failure_count_resets_on_new_version() {\n let project_id = ProjectId::new();\n let store: Arc = Arc::new(crate::tests::InMemoryStore::with_docs(vec![]));\n\n // Record failures for version 1\n record_orchestrator_failure(&store, project_id, 1).await;\n record_orchestrator_failure(&store, project_id, 1).await;\n\n // Switch to version 2 — count should reset to 1\n record_orchestrator_failure(&store, project_id, 2).await;\n\n let docs = store.list_shared_memory_docs(project_id).await.unwrap();\n let count = load_failure_count(&docs);\n assert_eq!(count, 1);\n }\n\n #[test]\n fn normalize_pause_outcome_transitions_thread_to_waiting() {\n let mut thread = Thread::new(\n \"goal\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n let outcome = ThreadOutcome::GatePaused {\n gate_name: \"approval\".into(),\n action_name: \"shell\".into(),\n call_id: \"call-1\".into(),\n parameters: serde_json::json!({\"cmd\":\"ls\"}),\n resume_kind: crate::gate::ResumeKind::Approval { allow_always: true },\n resume_output: None,\n };\n normalize_pause_outcome(&mut thread, &outcome).unwrap();\n assert_eq!(thread.state, ThreadState::Waiting);\n }\n\n #[test]\n fn parse_outcome_completed() {\n let result = serde_json::json!({\"outcome\": \"completed\", \"response\": \"Hello!\"});\n let outcome = parse_outcome(&result);\n assert!(matches!(outcome, ThreadOutcome::Completed { response: Some(r) } if r == \"Hello!\"));\n }\n\n #[test]\n fn parse_outcome_failed() {\n let result = serde_json::json!({\"outcome\": \"failed\", \"error\": \"boom\"});\n let outcome = parse_outcome(&result);\n assert!(matches!(outcome, ThreadOutcome::Failed { error } if error == \"boom\"));\n }\n\n #[test]\n fn parse_outcome_gate_paused() {\n let result = serde_json::json!({\n \"outcome\": \"gate_paused\",\n \"gate_name\": \"approval\",\n \"action_name\": \"shell\",\n \"call_id\": \"abc\",\n \"parameters\": {\"cmd\": \"rm -rf /\"},\n \"resume_kind\": {\"Approval\": {\"allow_always\": true}}\n });\n let outcome = parse_outcome(&result);\n assert!(\n matches!(outcome, ThreadOutcome::GatePaused { action_name, .. } if action_name == \"shell\")\n );\n }\n\n #[test]\n fn parse_outcome_max_iterations() {\n let result = serde_json::json!({\"outcome\": \"max_iterations\"});\n let outcome = parse_outcome(&result);\n assert!(matches!(outcome, ThreadOutcome::MaxIterations));\n }\n\n #[test]\n fn parse_outcome_stopped() {\n let result = serde_json::json!({\"outcome\": \"stopped\"});\n let outcome = parse_outcome(&result);\n assert!(matches!(outcome, ThreadOutcome::Stopped));\n }\n\n // ── handle_llm_complete model forwarding ────────────────────\n\n /// LLM backend that records the model from each `complete()` call.\n /// Used to verify the orchestrator's __llm_complete__ host fn forwards\n /// `explicit_config[\"model\"]` onto `LlmCallConfig.model`.\n struct ModelCapturingLlm {\n captured: tokio::sync::Mutex>>,\n }\n\n #[async_trait::async_trait]\n impl LlmBackend for ModelCapturingLlm {\n fn model_name(&self) -> &str {\n \"capturing\"\n }\n\n async fn complete(\n &self,\n _messages: &[ThreadMessage],\n _actions: &[crate::types::capability::ActionDef],\n config: &LlmCallConfig,\n ) -> Result {\n self.captured.lock().await.push(config.model.clone());\n Ok(crate::traits::llm::LlmOutput {\n response: crate::types::step::LlmResponse::Text(\"ok\".into()),\n usage: crate::types::step::TokenUsage::default(),\n })\n }\n }\n\n /// No-op effect executor — handle_llm_complete only consults it for\n /// `available_actions(...)`, which we satisfy with an empty list.\n struct NoopEffects;\n\n #[async_trait::async_trait]\n impl EffectExecutor for NoopEffects {\n async fn execute_action(\n &self,\n _: &str,\n _: serde_json::Value,\n _: &crate::types::capability::CapabilityLease,\n _: &ThreadExecutionContext,\n ) -> Result {\n Ok(crate::types::step::ActionResult {\n call_id: String::new(),\n action_name: String::new(),\n output: serde_json::json!({}),\n is_error: false,\n duration: std::time::Duration::from_millis(1),\n })\n }\n\n async fn available_actions(\n &self,\n _: &[crate::types::capability::CapabilityLease],\n ) -> Result, EngineError> {\n Ok(vec![])\n }\n }\n\n #[tokio::test]\n async fn llm_complete_forwards_model_from_explicit_config() {\n let concrete = Arc::new(ModelCapturingLlm {\n captured: tokio::sync::Mutex::new(Vec::new()),\n });\n let llm: Arc = Arc::clone(&concrete) as Arc;\n let effects: Arc = Arc::new(NoopEffects);\n let leases = Arc::new(LeaseManager::new());\n let store: Arc = Arc::new(crate::tests::InMemoryStore::with_docs(vec![]));\n\n let mut thread = Thread::new(\n \"goal\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n // Build the args __llm_complete__ receives from Python:\n // (messages, actions, config). config = {\"model\": \"gpt-4o\"}.\n let mut total_tokens = TokenUsage::default();\n let result = handle_llm_complete(\n &[\n json_to_monty(&serde_json::json!([{\"role\":\"user\",\"content\":\"hi\"}])),\n json_to_monty(&serde_json::json!([])),\n json_to_monty(&serde_json::json!({\"model\": \"gpt-4o\"})),\n ],\n &[],\n &mut thread,\n LlmCompleteDeps {\n llm: &llm,\n effects: &effects,\n leases: &leases,\n store: Some(&store),\n },\n &mut total_tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Return(_)));\n let captured = concrete.captured.lock().await;\n assert_eq!(captured.len(), 1);\n assert_eq!(captured[0].as_deref(), Some(\"gpt-4o\"));\n }\n\n #[tokio::test]\n async fn llm_complete_without_model_passes_none() {\n let concrete = Arc::new(ModelCapturingLlm {\n captured: tokio::sync::Mutex::new(Vec::new()),\n });\n let llm: Arc = Arc::clone(&concrete) as Arc;\n let effects: Arc = Arc::new(NoopEffects);\n let leases = Arc::new(LeaseManager::new());\n let store: Arc = Arc::new(crate::tests::InMemoryStore::with_docs(vec![]));\n\n let mut thread = Thread::new(\n \"goal\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n let mut total_tokens = TokenUsage::default();\n let _ = handle_llm_complete(\n &[\n json_to_monty(&serde_json::json!([{\"role\":\"user\",\"content\":\"hi\"}])),\n json_to_monty(&serde_json::json!([])),\n json_to_monty(&serde_json::json!({\"max_tokens\": 100})),\n ],\n &[],\n &mut thread,\n LlmCompleteDeps {\n llm: &llm,\n effects: &effects,\n leases: &leases,\n store: Some(&store),\n },\n &mut total_tokens,\n )\n .await;\n\n let captured = concrete.captured.lock().await;\n assert_eq!(captured.len(), 1);\n assert_eq!(captured[0], None);\n }\n\n // ── Python ↔ Rust ActionCall round-trip ───────────────────────────────\n //\n // Regression tests for the orphaned-tool-result bug. The Python\n // orchestrator stores `action_calls` on assistant messages using the\n // shape `{name, call_id, params}`, but the canonical Rust `ActionCall`\n // uses `{action_name, id, parameters}`. Without the explicit\n // `PythonActionCall` interchange type, `serde_json::from_value` would\n // silently fail (`.ok()` swallows the error) and the Python-shaped\n // assistant message would be parsed back as a plain assistant message\n // with no tool calls, causing every subsequent ActionResult to be\n // detected as orphaned by `sanitize_tool_messages` in the host crate.\n\n #[test]\n fn python_action_call_round_trips_through_serde() {\n let original = ActionCall {\n id: \"call_abc123\".to_string(),\n action_name: \"google_drive_tool\".to_string(),\n parameters: serde_json::json!({\"query\": \"expenses\"}),\n };\n\n let python_json = serde_json::to_value(PythonActionCall::from(&original))\n .expect(\"PythonActionCall must serialize\");\n // Python-friendly field names — match what default.py reads.\n assert_eq!(python_json[\"name\"], \"google_drive_tool\");\n assert_eq!(python_json[\"call_id\"], \"call_abc123\");\n assert_eq!(\n python_json[\"params\"],\n serde_json::json!({\"query\": \"expenses\"})\n );\n\n let parsed: PythonActionCall =\n serde_json::from_value(python_json).expect(\"must deserialize\");\n let round_tripped: ActionCall = parsed.into();\n assert_eq!(round_tripped.id, original.id);\n assert_eq!(round_tripped.action_name, original.action_name);\n assert_eq!(round_tripped.parameters, original.parameters);\n }\n\n #[test]\n fn action_calls_to_python_json_uses_python_field_names() {\n let calls = vec![\n ActionCall {\n id: \"call_1\".to_string(),\n action_name: \"notion_notion_search\".to_string(),\n parameters: serde_json::json!({\"query\": \"name\"}),\n },\n ActionCall {\n id: \"call_2\".to_string(),\n action_name: \"google_drive_tool\".to_string(),\n parameters: serde_json::json!({\"action\": \"list\"}),\n },\n ];\n let json = action_calls_to_python_json(&calls);\n assert_eq!(json.len(), 2);\n assert_eq!(json[0][\"name\"], \"notion_notion_search\");\n assert_eq!(json[0][\"call_id\"], \"call_1\");\n assert_eq!(json[1][\"name\"], \"google_drive_tool\");\n assert_eq!(json[1][\"call_id\"], \"call_2\");\n }\n\n #[test]\n fn python_json_to_action_calls_parses_python_field_names() {\n // The exact shape default.py produces (and stores on assistant\n // messages via `append_message(..., action_calls=calls)`).\n let python_json = serde_json::json!([\n {\"name\": \"notion_notion_search\", \"call_id\": \"call_xyz\", \"params\": {\"q\": \"foo\"}},\n {\"name\": \"google_drive_tool\", \"call_id\": \"call_abc\", \"params\": {\"action\": \"list\"}},\n ]);\n let parsed = python_json_to_action_calls(&python_json).expect(\"must parse\");\n assert_eq!(parsed.len(), 2);\n assert_eq!(parsed[0].action_name, \"notion_notion_search\");\n assert_eq!(parsed[0].id, \"call_xyz\");\n assert_eq!(parsed[0].parameters, serde_json::json!({\"q\": \"foo\"}));\n assert_eq!(parsed[1].action_name, \"google_drive_tool\");\n assert_eq!(parsed[1].id, \"call_abc\");\n }\n\n #[test]\n fn python_json_to_action_calls_rejects_canonical_field_names() {\n // Sanity check: the parser is strict about Python field names.\n // If `default.py` ever changes the shape, the test must catch it.\n let canonical_json = serde_json::json!([\n {\"action_name\": \"search\", \"id\": \"call_x\", \"parameters\": {}}\n ]);\n // Missing \"name\", \"call_id\", \"params\" → returns None.\n assert!(python_json_to_action_calls(&canonical_json).is_none());\n }\n\n #[test]\n fn summarize_action_calls_for_log_does_not_leak_user_pii() {\n // The whole point of this helper is that the warn log path on a\n // shape-drift failure must NOT dump tool parameters (which can\n // contain user PII like search queries, file names, email content)\n // into log aggregation systems. The summary should expose only\n // structural information: array length and the keys of the first\n // entry. The keys themselves are static (`name`, `call_id`,\n // `params`), not user data.\n let pii_value = serde_json::json!([\n {\n \"name\": \"google_drive_tool\",\n \"call_id\": \"call_xyz\",\n \"params\": {\n \"query\": \"salary spreadsheet for joe\",\n \"secret_token\": \"very-sensitive-token-do-not-log\"\n }\n },\n {\n \"name\": \"gmail\",\n \"call_id\": \"call_abc\",\n \"params\": {\n \"subject\": \"private message about layoffs\"\n }\n }\n ]);\n let summary = summarize_action_calls_for_log(&pii_value);\n\n // Structural info present.\n assert!(summary.contains(\"array of 2 entries\"));\n assert!(summary.contains(\"call_id\"));\n assert!(summary.contains(\"name\"));\n assert!(summary.contains(\"params\"));\n\n // PII fields and their values must NOT appear.\n assert!(\n !summary.contains(\"salary\"),\n \"summary must not leak user PII from params: {summary}\"\n );\n assert!(\n !summary.contains(\"very-sensitive-token\"),\n \"summary must not leak credential-shaped values: {summary}\"\n );\n assert!(\n !summary.contains(\"layoffs\"),\n \"summary must not leak free-text content: {summary}\"\n );\n assert!(\n !summary.contains(\"google_drive_tool\"),\n \"summary must not leak the tool name itself (could expose intent): {summary}\"\n );\n }\n\n #[test]\n fn summarize_action_calls_for_log_handles_edge_cases() {\n assert_eq!(\n summarize_action_calls_for_log(&serde_json::json!([])),\n \"empty array\"\n );\n assert!(\n summarize_action_calls_for_log(&serde_json::json!(\"not an array\")).contains(\"string\")\n );\n assert!(\n summarize_action_calls_for_log(&serde_json::json!({\"foo\": \"bar\"})).contains(\"object\")\n );\n assert!(summarize_action_calls_for_log(&serde_json::json!(null)).contains(\"null\"));\n }\n\n /// Caller-level regression test: feeds `json_to_thread_messages` the\n /// exact JSON shape that `default.py` produces for an assistant message\n /// with tool calls followed by tool results, and asserts that the\n /// resulting `ThreadMessage`s preserve the `action_calls` ↔\n /// `action_call_id` linkage. Without the `PythonActionCall` parser the\n /// assistant message would come back with `action_calls = None` and\n /// every following ActionResult would look orphaned to the bridge.\n #[test]\n fn json_to_thread_messages_preserves_action_calls_from_python_orchestrator() {\n // This is the literal shape `default.py` writes into\n // `state[\"working_messages\"]` after a Tier 0 step:\n //\n // append_message(working_messages, \"Assistant\", \"...\", action_calls=calls)\n // append_message(working_messages, \"ActionResult\", \"...\", action_name=..., action_call_id=...)\n //\n // where `calls` came from the LLM response and has shape\n // `[{\"name\": ..., \"call_id\": ..., \"params\": ...}]`.\n let working_messages = serde_json::json!([\n {\"role\": \"User\", \"content\": \"search in notion for my name\"},\n {\n \"role\": \"Assistant\",\n \"content\": \"\",\n \"action_calls\": [\n {\n \"name\": \"notion_notion_search\",\n \"call_id\": \"call_xyz\",\n \"params\": {\"query\": \"Illia\"}\n }\n ]\n },\n {\n \"role\": \"ActionResult\",\n \"content\": \"found 3 results\",\n \"action_name\": \"notion_notion_search\",\n \"action_call_id\": \"call_xyz\"\n }\n ]);\n\n let messages = json_to_thread_messages(&working_messages).expect(\"must parse\");\n assert_eq!(messages.len(), 3);\n\n // The assistant message MUST have action_calls populated, with\n // matching call_id. If this assertion fails, the bridge layer\n // will treat the following ActionResult as orphaned and rewrite\n // it as a user message — losing the model's ability to reason\n // about prior tool output.\n let assistant = &messages[1];\n assert_eq!(\n assistant.role,\n crate::types::message::MessageRole::Assistant\n );\n let calls = assistant\n .action_calls\n .as_ref()\n .expect(\"assistant message must carry action_calls after round-trip\");\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].id, \"call_xyz\");\n assert_eq!(calls[0].action_name, \"notion_notion_search\");\n assert_eq!(calls[0].parameters, serde_json::json!({\"query\": \"Illia\"}));\n\n // The ActionResult must reference the same call_id so the bridge\n // can pair them.\n let result = &messages[2];\n assert_eq!(\n result.role,\n crate::types::message::MessageRole::ActionResult\n );\n assert_eq!(result.action_call_id.as_deref(), Some(\"call_xyz\"));\n assert_eq!(result.action_name.as_deref(), Some(\"notion_notion_search\"));\n }\n\n /// Regression for the gate-resume / bootstrap path: when a thread\n /// resumes after approval or auth, `build_orchestrator_inputs`\n /// serializes `thread.internal_messages` into the bootstrap context\n /// that Python reads into `working_messages`. If `action_calls` is\n /// serialized with canonical `ActionCall` field names (`action_name`,\n /// `id`, `parameters`) instead of the Python interchange names\n /// (`name`, `call_id`, `params`), the next `__llm_complete__` call\n /// passes them back through `json_to_thread_messages` which fails\n /// with \"missing field `name`\" and orphans every subsequent tool\n /// result.\n ///\n /// This test simulates the full round-trip: build a `ThreadMessage`\n /// with action_calls → serialize through `build_orchestrator_inputs`'s\n /// exact serialization pattern → parse back through\n /// `json_to_thread_messages` → assert the calls survive. If anyone\n /// adds a THIRD serialization path in the future and uses canonical\n /// names, this test documents the pattern they should follow.\n #[test]\n fn bootstrap_context_action_calls_round_trip_through_python_interchange() {\n // Build a thread message the way the engine does: an assistant\n // message with action_calls in canonical ActionCall format (the\n // shape stored in the DB / internal_messages).\n let msg = ThreadMessage::assistant_with_actions(\n Some(\"I'll search for that\".to_string()),\n vec![ActionCall {\n id: \"call_resume_test\".to_string(),\n action_name: \"google_drive_tool\".to_string(),\n parameters: serde_json::json!({\"query\": \"budget\"}),\n }],\n );\n\n // Serialize through the SAME pattern `build_orchestrator_inputs`\n // uses. This is the exact code path that was broken before the\n // fix — it was using `\"action_calls\": m.action_calls` which\n // produced canonical field names.\n let calls_json = msg\n .action_calls\n .as_ref()\n .map(|calls| serde_json::Value::Array(action_calls_to_python_json(calls)));\n let serialized = serde_json::json!([{\n \"role\": \"Assistant\",\n \"content\": msg.content,\n \"action_name\": msg.action_name,\n \"action_call_id\": msg.action_call_id,\n \"action_calls\": calls_json,\n }]);\n\n // Parse back through the same path Python's working_messages\n // takes when it calls __llm_complete__.\n let parsed = json_to_thread_messages(&serialized).expect(\"must parse\");\n assert_eq!(parsed.len(), 1);\n\n let assistant = &parsed[0];\n let calls = assistant.action_calls.as_ref().expect(\n \"bootstrap context action_calls must survive the round-trip. \\\n If this fails, a serialization path is using canonical ActionCall \\\n field names instead of PythonActionCall interchange names.\",\n );\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].id, \"call_resume_test\");\n assert_eq!(calls[0].action_name, \"google_drive_tool\");\n assert_eq!(calls[0].parameters, serde_json::json!({\"query\": \"budget\"}));\n }\n\n /// Negative regression: verify that canonical ActionCall field names\n /// do NOT round-trip. If this test ever PASSES, it means someone\n /// added `#[serde(rename)]` to ActionCall or changed the parser to\n /// accept both formats — which is fine, but the PythonActionCall\n /// interchange type can then be removed. This test documents the\n /// current contract: canonical names are rejected by the parser.\n #[test]\n fn canonical_action_call_field_names_do_not_round_trip() {\n let serialized_with_canonical_names = serde_json::json!([{\n \"role\": \"Assistant\",\n \"content\": \"\",\n \"action_calls\": [{\n \"action_name\": \"search\",\n \"id\": \"call_x\",\n \"parameters\": {}\n }],\n }]);\n let parsed =\n json_to_thread_messages(&serialized_with_canonical_names).expect(\"messages parse\");\n // The assistant message should have NO action_calls because the\n // parser rejects canonical field names.\n assert!(\n parsed[0].action_calls.is_none(),\n \"canonical ActionCall field names must NOT parse as action_calls. \\\n If this assertion fails, the PythonActionCall interchange type \\\n is no longer needed — either remove it or update the contract.\"\n );\n }\n\n /// Regression: `action_calls: null` is Python's legitimate \"this\n /// message has no tool calls\" signal (text-only response). Before the\n /// null filter, `python_json_to_action_calls` would fire a warn log\n /// with \"invalid type: null, expected a sequence\" on every text-only\n /// assistant message — a false alarm that masked real drift issues.\n #[test]\n fn json_to_thread_messages_handles_null_action_calls_gracefully() {\n let messages = serde_json::json!([\n {\n \"role\": \"Assistant\",\n \"content\": \"Here is your answer.\",\n \"action_calls\": null\n }\n ]);\n let parsed = json_to_thread_messages(&messages).expect(\"must parse\");\n assert_eq!(parsed.len(), 1);\n assert_eq!(\n parsed[0].role,\n crate::types::message::MessageRole::Assistant\n );\n assert_eq!(parsed[0].content, \"Here is your answer.\");\n assert!(\n parsed[0].action_calls.is_none(),\n \"null action_calls must produce None, not a parse error\"\n );\n }\n\n /// Verify that messages WITHOUT the action_calls key at all (the most\n /// common case for text responses) also parse correctly — this is the\n /// baseline that the null-filtering regression test extends.\n #[test]\n fn json_to_thread_messages_handles_absent_action_calls() {\n let messages = serde_json::json!([\n {\"role\": \"Assistant\", \"content\": \"Just text, no tools.\"}\n ]);\n let parsed = json_to_thread_messages(&messages).expect(\"must parse\");\n assert_eq!(parsed.len(), 1);\n assert!(parsed[0].action_calls.is_none());\n }\n\n /// Empty action_calls array is valid (LLM decided not to call any\n /// tools this turn but the response still has the array field). Must\n /// produce `Some(vec![])`, not `None`.\n #[test]\n fn json_to_thread_messages_handles_empty_action_calls_array() {\n let messages = serde_json::json!([\n {\n \"role\": \"Assistant\",\n \"content\": \"No tools needed.\",\n \"action_calls\": []\n }\n ]);\n let parsed = json_to_thread_messages(&messages).expect(\"must parse\");\n assert_eq!(parsed.len(), 1);\n let calls = parsed[0]\n .action_calls\n .as_ref()\n .expect(\"empty array should produce Some(vec![])\");\n assert!(calls.is_empty());\n }\n\n // ── Consecutive action error counting (issue #2325) ──────────\n //\n // The run_loop tracks `consecutive_action_errors` for Tier 0 (structured\n // action calls). These tests exercise the counting logic extracted from\n // run_loop into small Python snippets that simulate batch outcomes.\n\n #[test]\n fn action_errors_increment_when_all_actions_fail() {\n // Simulate 3 consecutive batches where all actions fail.\n let count = eval_python_int(\n r#\"\nconsecutive_action_errors = 0\nfor _ in range(3):\n batch_error_count = 2\n batch_success_count = 0\n if batch_success_count > 0:\n consecutive_action_errors = 0\n elif batch_error_count > 0:\n consecutive_action_errors += 1\nFINAL(consecutive_action_errors)\n\"#,\n );\n assert_eq!(count, 3);\n }\n\n #[test]\n fn action_errors_reset_when_any_action_succeeds() {\n // 2 all-fail batches, then 1 batch with a success => resets to 0.\n let count = eval_python_int(\n r#\"\nconsecutive_action_errors = 0\nfor batch in [(0, 2), (0, 1), (1, 1)]:\n batch_success_count = batch[0]\n batch_error_count = batch[1]\n if batch_success_count > 0:\n consecutive_action_errors = 0\n elif batch_error_count > 0:\n consecutive_action_errors += 1\nFINAL(consecutive_action_errors)\n\"#,\n );\n assert_eq!(count, 0);\n }\n\n #[test]\n fn action_errors_partial_success_resets_counter() {\n // A batch with mixed results (some succeed, some fail) should reset.\n let count = eval_python_int(\n r#\"\nconsecutive_action_errors = 5\nbatch_success_count = 1\nbatch_error_count = 3\nif batch_success_count > 0:\n consecutive_action_errors = 0\nelif batch_error_count > 0:\n consecutive_action_errors += 1\nFINAL(consecutive_action_errors)\n\"#,\n );\n assert_eq!(count, 0);\n }\n\n #[test]\n fn action_errors_nudge_injected_at_threshold() {\n // When consecutive_action_errors reaches max_consecutive_errors,\n // a nudge message should be appended. We simulate the branching\n // logic and check whether a nudge would fire.\n // Returns 1 if nudge fires (not failure), 0 otherwise.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = 5\nconsecutive_action_errors = 5\nnudge = False\nfailed = False\nif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nelif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors:\n nudge = True\nif nudge and not failed:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 1, \"nudge should fire at threshold\");\n }\n\n #[test]\n fn action_errors_no_nudge_below_threshold() {\n // Returns 1 if nudge fires, 0 if not.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = 5\nconsecutive_action_errors = 4\nnudge = False\nfailed = False\nif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nelif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors:\n nudge = True\nif nudge:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 0, \"nudge should not fire below threshold\");\n }\n\n #[test]\n fn action_errors_failure_at_threshold_plus_two() {\n // At max_consecutive_errors + 2, the thread should transition to failed.\n // Returns 1 if failed, 0 if not.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = 5\nconsecutive_action_errors = 7\nfailed = False\nif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nif failed:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 1, \"should fail at threshold + 2\");\n }\n\n #[test]\n fn action_errors_nudge_at_threshold_not_failure() {\n // At exactly max_consecutive_errors + 1, we get a nudge but not failure.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = 5\nconsecutive_action_errors = 6\nnudge = False\nfailed = False\nif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nelif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors:\n nudge = True\n# Return 0=nothing, 1=nudge, 2=failed\nif failed:\n FINAL(2)\nelif nudge:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 1, \"should nudge at threshold + 1, not fail\");\n }\n\n #[test]\n fn action_errors_none_limit_skips_check_without_typeerror() {\n // Regression: when max_consecutive_errors is None (meaning \"no limit\"),\n // the arithmetic `max_consecutive_errors + 2` used to crash with\n // TypeError on the first action error. The guard must short-circuit\n // on None and leave both the nudge and failure branches untaken.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = None\nconsecutive_action_errors = 1\nnudge = False\nfailed = False\nif max_consecutive_errors is not None and consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nelif max_consecutive_errors is not None and consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors:\n nudge = True\n# Return 0=nothing, 1=nudge, 2=failed\nif failed:\n FINAL(2)\nelif nudge:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 0, \"None limit should disable the guard entirely\");\n }\n\n #[test]\n fn code_errors_none_limit_skips_failure_check() {\n // Regression: same None-guard for the code-error branch at line 660.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = None\nconsecutive_errors = 99\nfailed = False\nif max_consecutive_errors is not None and consecutive_errors >= max_consecutive_errors:\n failed = True\nif failed:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(\n result, 0,\n \"None limit should not trigger failure regardless of consecutive_errors\"\n );\n }\n\n #[test]\n fn action_error_prefix_added_to_error_output() {\n // Verify that [ACTION FAILED] prefix is prepended to error outputs.\n // Returns 1 if prefix present, 0 if not.\n let result = eval_python_int(\n r#\"\nr = {\"action_name\": \"http\", \"output\": \"connection refused\", \"is_error\": True}\noutput = r.get(\"output\")\noutput_str = str(output) if output is not None else \"[no output]\"\nif r.get(\"is_error\"):\n output_str = \"[ACTION FAILED] \" + output_str\nif output_str.startswith(\"[ACTION FAILED]\"):\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 1, \"error outputs must get [ACTION FAILED] prefix\");\n }\n\n #[test]\n fn action_error_skipped_calls_count_as_errors() {\n // When a call has no result (r is None), it should count as an error.\n let count = eval_python_int(\n r#\"\nbatch_error_count = 0\nbatch_success_count = 0\nr = None\nif r is not None:\n if r.get(\"is_error\"):\n batch_error_count += 1\n else:\n batch_success_count += 1\nelse:\n batch_error_count += 1\nFINAL(batch_error_count)\n\"#,\n );\n assert_eq!(count, 1, \"skipped calls must count as batch errors\");\n }\n\n #[test]\n fn checkpoint_includes_consecutive_action_errors() {\n // Test that handle_save_checkpoint persists consecutive_action_errors\n // in the thread metadata.\n let mut thread = Thread::new(\n \"goal\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n let state = json_to_monty(&serde_json::json!({}));\n let counters = json_to_monty(&serde_json::json!({\n \"nudge_count\": 0,\n \"consecutive_errors\": 1,\n \"consecutive_action_errors\": 4,\n \"compaction_count\": 2,\n }));\n\n handle_save_checkpoint(&[state, counters], &[], &mut thread);\n\n let checkpoint = thread\n .metadata\n .get(\"runtime_checkpoint\")\n .expect(\"checkpoint must exist\");\n assert_eq!(\n checkpoint\n .get(\"consecutive_action_errors\")\n .and_then(|v| v.as_u64()),\n Some(4),\n \"consecutive_action_errors must be persisted in checkpoint\"\n );\n assert_eq!(\n checkpoint\n .get(\"consecutive_errors\")\n .and_then(|v| v.as_u64()),\n Some(1),\n );\n assert_eq!(\n checkpoint.get(\"compaction_count\").and_then(|v| v.as_u64()),\n Some(2),\n );\n }\n\n /// Regression test: every assistant tool_call must have a matching\n /// ActionResult after parsing. If an ActionResult is missing, the LLM\n /// API rejects with \"No tool output found for function call \".\n ///\n /// This was the root cause of the HTTP 400 from the OpenAI Codex\n /// provider: a tool returning null output caused the Python\n /// orchestrator to skip appending the ActionResult.\n #[test]\n fn json_to_thread_messages_every_tool_call_has_action_result() {\n // Simulate working_messages after the Python fix: every call gets\n // an ActionResult, even when the original output was null.\n let messages = serde_json::json!([\n {\"role\": \"System\", \"content\": \"You are a helpful assistant.\"},\n {\"role\": \"User\", \"content\": \"Update all tools.\"},\n {\n \"role\": \"Assistant\",\n \"content\": \"\",\n \"action_calls\": [\n {\"call_id\": \"call_AAA\", \"name\": \"tool_a\", \"params\": {}},\n {\"call_id\": \"call_BBB\", \"name\": \"tool_b\", \"params\": {}},\n {\"call_id\": \"call_CCC\", \"name\": \"tool_c\", \"params\": {}}\n ]\n },\n {\n \"role\": \"ActionResult\",\n \"content\": \"{\\\"ok\\\": true}\",\n \"action_name\": \"tool_a\",\n \"action_call_id\": \"call_AAA\"\n },\n {\n \"role\": \"ActionResult\",\n \"content\": \"[no output]\",\n \"action_name\": \"tool_b\",\n \"action_call_id\": \"call_BBB\"\n },\n {\n \"role\": \"ActionResult\",\n \"content\": \"{\\\"done\\\": true}\",\n \"action_name\": \"tool_c\",\n \"action_call_id\": \"call_CCC\"\n }\n ]);\n\n let parsed = json_to_thread_messages(&messages).expect(\"must parse\");\n assert_eq!(parsed.len(), 6);\n\n // Extract call IDs from the assistant message\n let assistant_calls: std::collections::HashSet = parsed\n .iter()\n .filter_map(|m| m.action_calls.as_ref())\n .flat_map(|calls| calls.iter().map(|c| c.id.clone()))\n .collect();\n\n // Extract call IDs from ActionResult messages\n let result_call_ids: std::collections::HashSet = parsed\n .iter()\n .filter(|m| m.role == crate::types::message::MessageRole::ActionResult)\n .filter_map(|m| m.action_call_id.clone())\n .collect();\n\n // Every tool_call must have a matching ActionResult\n for call_id in &assistant_calls {\n assert!(\n result_call_ids.contains(call_id),\n \"tool_call {call_id} has no matching ActionResult — \\\n this would cause 'No tool output found' from the LLM API\"\n );\n }\n }\n\n // ── CodeExecutionFailed event emission (caller test) ────────\n\n #[tokio::test]\n async fn execute_code_step_emits_code_execution_failed_event() {\n let llm: Arc = Arc::new(ModelCapturingLlm {\n captured: tokio::sync::Mutex::new(Vec::new()),\n });\n let effects: Arc = Arc::new(NoopEffects);\n let leases = Arc::new(LeaseManager::new());\n let policy = Arc::new(PolicyEngine::new());\n\n let mut thread = Thread::new(\n \"test code execution failure instrumentation\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n // Pass intentionally broken Python code (syntax error)\n let args = &[\n json_to_monty(&serde_json::json!(\"def ==\")),\n json_to_monty(&serde_json::json!({})),\n ];\n\n let (tx, _rx) = tokio::sync::broadcast::channel(16);\n let _result = handle_execute_code_step(\n args,\n &[],\n &mut thread,\n &llm,\n &effects,\n &leases,\n &policy,\n Some(&tx),\n )\n .await;\n\n // Verify CodeExecutionFailed event was emitted on thread.events\n let code_failed_events: Vec<_> = thread\n .events\n .iter()\n .filter(|e| matches!(&e.kind, EventKind::CodeExecutionFailed { .. }))\n .collect();\n\n assert_eq!(\n code_failed_events.len(),\n 1,\n \"expected exactly one CodeExecutionFailed event, got {}\",\n code_failed_events.len()\n );\n\n if let EventKind::CodeExecutionFailed {\n category,\n code_hash,\n ..\n } = &code_failed_events[0].kind\n {\n assert_eq!(\n *category,\n crate::types::step::CodeExecutionFailure::SyntaxError\n );\n assert!(code_hash.is_some());\n } else {\n panic!(\"expected CodeExecutionFailed event kind\");\n }\n\n // Also verify ActionFailed was emitted (existing behavior)\n let action_failed = thread\n .events\n .iter()\n .any(|e| matches!(&e.kind, EventKind::ActionFailed { .. }));\n assert!(\n action_failed,\n \"expected ActionFailed event alongside CodeExecutionFailed\"\n );\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/executor/scripting.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:50 GMT" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-xss-protection", + "0" + ], + [ + "server", + "github.com" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "content-length", + "113551" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "x-ratelimit-remaining", + "4970" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "etag", + "\"674fca2750840d6f245ff8676d54df8b83ca74f1\"" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-frame-options", + "deny" + ], + [ + "x-ratelimit-used", + "30" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-github-request-id", + "FB2A:365517:400987:4B674C:69DFAEEE" + ], + [ + "x-ratelimit-limit", + "5000" + ] + ], + "body": "//! Tier 1 executor: embedded Python via Monty.\n//!\n//! Executes LLM-generated Python code using the Monty interpreter. Tool\n//! calls use **async dispatch**: each tool call returns a Monty `ExternalFuture`\n//! via `resume_pending()`, allowing Python code to use `await` and\n//! `asyncio.gather()` for parallel execution. When all tasks are blocked,\n//! Monty yields `ResolveFutures` and we execute pending tools concurrently\n//! via `JoinSet`.\n//!\n//! Follows the RLM (Recursive Language Model) pattern:\n//! - Thread context injected as Python variables (not LLM attention input)\n//! - `llm_query()` / `llm_query_batched()` for recursive subagent spawning\n//! - `FINAL(answer)` / `FINAL_VAR(name)` for explicit termination\n//! - Step 0 orientation preamble for context awareness\n//! - Errors flow back to LLM for self-correction (not step termination)\n//! - Output truncated to configurable limit with variable listing\n//! - `asyncio.gather()` for parallel tool execution (via ResolveFutures)\n\nuse std::collections::HashMap;\nuse std::sync::Arc;\nuse std::time::Duration;\n\nuse monty::{\n ExcType, ExtFunctionResult, LimitedTracker, MontyException, MontyObject, MontyRun,\n NameLookupResult, PrintWriter, ResourceLimits, RunProgress,\n};\nuse tracing::debug;\n\nuse crate::capability::lease::LeaseManager;\nuse crate::capability::policy::{PolicyDecision, PolicyEngine};\nuse crate::traits::effect::{EffectExecutor, ThreadExecutionContext};\nuse crate::traits::llm::{LlmBackend, LlmCallConfig};\nuse crate::types::error::EngineError;\nuse crate::types::event::EventKind;\nuse crate::types::message::{MessageRole, ThreadMessage};\nuse crate::types::step::{ActionResult, CodeExecutionFailure, LlmResponse, TokenUsage};\nuse crate::types::thread::Thread;\nuse ironclaw_common::ValidTimezone;\n\n// ── Configuration ───────────────────────────────────────────\n\n/// Maximum characters of output to include in LLM context between steps.\n/// Matches Prime Intellect's default. Configurable per thread in the future.\nconst OUTPUT_TRUNCATE_LEN: usize = 8_000;\n\n/// Maximum characters for a preview prefix in compact metadata.\nconst OUTPUT_PREVIEW_LEN: usize = 200;\n\n/// Default resource limits for Monty execution.\nfn default_limits() -> ResourceLimits {\n ResourceLimits::new()\n .max_duration(Duration::from_secs(30))\n .max_allocations(1_000_000)\n .max_memory(64 * 1024 * 1024) // 64 MB\n}\n\n// ── Result types ────────────────────────────────────────────\n\n/// Result of executing a code block.\npub struct CodeExecutionResult {\n /// The Python return value, converted to JSON.\n pub return_value: serde_json::Value,\n /// Captured print output.\n pub stdout: String,\n /// All action calls that were made during execution.\n pub action_results: Vec,\n /// Events generated during execution.\n pub events: Vec,\n /// If set, execution was interrupted for approval.\n pub need_approval: Option,\n /// Tokens used by recursive llm_query() calls.\n pub recursive_tokens: TokenUsage,\n /// If set, the code called FINAL() or FINAL_VAR() with this answer.\n pub final_answer: Option,\n /// Classified failure category. `None` when execution succeeded or was\n /// paused by a gate. `Some(category)` when code execution failed —\n /// `failure.is_some()` replaces the former `had_error: bool` field.\n pub failure: Option,\n}\n\n/// Build a compact output summary for inclusion in LLM context between steps.\n///\n/// Truncates to `OUTPUT_TRUNCATE_LEN` (last N chars shown, like fast-rlm).\n/// Includes a list of REPL variable names if available.\npub fn compact_output_metadata(stdout: &str, return_value: &serde_json::Value) -> String {\n let mut parts = Vec::new();\n\n if !stdout.is_empty() {\n let char_count = stdout.chars().count();\n if char_count > OUTPUT_TRUNCATE_LEN {\n let truncated: String = stdout\n .chars()\n .skip(char_count - OUTPUT_TRUNCATE_LEN)\n .collect();\n parts.push(format!(\n \"[TRUNCATED: last {OUTPUT_TRUNCATE_LEN} of {char_count} chars shown]\\n{truncated}\",\n ));\n } else {\n parts.push(format!(\"[FULL OUTPUT: {char_count} chars]\\n{stdout}\"));\n }\n }\n\n if *return_value != serde_json::Value::Null {\n let val_str = serde_json::to_string_pretty(return_value).unwrap_or_default();\n let val_char_count = val_str.chars().count();\n if val_char_count > OUTPUT_PREVIEW_LEN {\n let preview: String = val_str.chars().take(OUTPUT_PREVIEW_LEN).collect();\n parts.push(format!(\n \"Return value ({val_char_count} chars): {preview}...\",\n ));\n } else {\n parts.push(format!(\"Return value: {val_str}\"));\n }\n }\n\n if parts.is_empty() {\n \"[code executed, no output]\".into()\n } else {\n parts.join(\"\\n\")\n }\n}\n\n// ── Step 0 orientation preamble ─────────────────────────────\n\n/// Build the Step 0 orientation preamble that auto-executes before the\n/// first LLM call to give the model structural awareness of the context.\npub fn build_orientation_preamble(thread: &Thread) -> String {\n let msg_count = thread.messages.len();\n let total_chars: usize = thread.messages.iter().map(|m| m.content.len()).sum();\n let user_msgs = thread\n .messages\n .iter()\n .filter(|m| m.role == MessageRole::User)\n .count();\n\n let mut preview = String::new();\n if let Some(last_user) = thread\n .messages\n .iter()\n .rev()\n .find(|m| m.role == MessageRole::User)\n {\n let content_preview: String = last_user.content.chars().take(500).collect();\n let truncated = if last_user.content.chars().count() > 500 {\n \"...\"\n } else {\n \"\"\n };\n preview = format!(\"\\nLast user message preview: {content_preview}{truncated}\");\n }\n\n format!(\n \"[Step 0 — Context Orientation]\\n\\\n Goal: {goal}\\n\\\n Context: {msg_count} messages, {total_chars} total chars, {user_msgs} from user\\n\\\n Step: {step}{preview}\",\n goal = thread.goal,\n step = thread.step_count + 1,\n )\n}\n\n// ── Context injection (RLM 3.4) ────────────────────────────\n\n/// Build Monty input variables from thread state.\n///\n/// `persisted_state` carries variables from previous code steps so the\n/// REPL feels persistent even though each step creates a fresh MontyRun.\nfn build_context_inputs(\n thread: &Thread,\n persisted_state: &serde_json::Value,\n) -> (Vec, Vec) {\n let mut names = Vec::new();\n let mut values = Vec::new();\n\n // `context` — thread messages as a list of dicts\n let messages: Vec = thread\n .messages\n .iter()\n .map(|msg| {\n let mut pairs = vec![\n (\n MontyObject::String(\"role\".into()),\n MontyObject::String(format!(\"{:?}\", msg.role)),\n ),\n (\n MontyObject::String(\"content\".into()),\n MontyObject::String(msg.content.clone()),\n ),\n ];\n if let Some(ref name) = msg.action_name {\n pairs.push((\n MontyObject::String(\"action_name\".into()),\n MontyObject::String(name.clone()),\n ));\n }\n MontyObject::dict(pairs)\n })\n .collect();\n names.push(\"context\".into());\n values.push(MontyObject::List(messages));\n\n // `goal` — the thread's goal string\n names.push(\"goal\".into());\n values.push(MontyObject::String(thread.goal.clone()));\n\n // `step_number` — current step index\n names.push(\"step_number\".into());\n values.push(MontyObject::Int(thread.step_count as i64));\n\n // `state` — persisted variables from previous code steps.\n // This is a dict that accumulates: return values, tool results, etc.\n // The model can read `state[\"results\"]`, `state[\"prev_return\"]`, etc.\n names.push(\"state\".into());\n values.push(json_to_monty(persisted_state));\n\n // `previous_results` — dict of {call_id: result_json} from prior steps\n let result_pairs: Vec<(MontyObject, MontyObject)> = thread\n .messages\n .iter()\n .filter(|m| m.role == MessageRole::ActionResult)\n .filter_map(|m| {\n let call_id = m.action_call_id.as_ref()?;\n Some((\n MontyObject::String(call_id.clone()),\n MontyObject::String(m.content.clone()),\n ))\n })\n .collect();\n names.push(\"previous_results\".into());\n values.push(MontyObject::dict(result_pairs));\n\n // `user_timezone` — validated IANA timezone from the user's channel (e.g. \"America/New_York\")\n let tz = thread\n .metadata\n .get(\"user_timezone\")\n .and_then(|v| v.as_str())\n .and_then(ValidTimezone::parse)\n .map(|vtz| vtz.name().to_string())\n .unwrap_or_else(|| \"UTC\".into());\n names.push(\"user_timezone\".into());\n values.push(MontyObject::String(tz));\n\n (names, values)\n}\n\n// ── Main execution function ─────────────────────────────────\n\n/// Execute a Python code block using Monty.\n///\n/// Handles the full RLM execution pattern: context-as-variables, FINAL()\n/// termination, llm_query() recursive calls, error-to-LLM flow, and\n/// output truncation.\n#[allow(clippy::too_many_arguments)]\npub async fn execute_code(\n code: &str,\n thread: &Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &LeaseManager,\n policy: &PolicyEngine,\n context: &ThreadExecutionContext,\n capability_policies: &[crate::types::capability::PolicyRule],\n persisted_state: &serde_json::Value,\n) -> Result {\n execute_code_with_skills(\n code,\n thread,\n llm,\n effects,\n leases,\n policy,\n context,\n capability_policies,\n persisted_state,\n &[],\n )\n .await\n}\n\n/// Execute a Python code block with optional skill code snippets.\n///\n/// `skill_snippet_names` are registered as additional known functions in the\n/// Monty NameLookup, alongside tool names from capability leases.\n#[allow(clippy::too_many_arguments)]\npub async fn execute_code_with_skills(\n code: &str,\n thread: &Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &LeaseManager,\n policy: &PolicyEngine,\n context: &ThreadExecutionContext,\n capability_policies: &[crate::types::capability::PolicyRule],\n persisted_state: &serde_json::Value,\n skill_snippet_names: &[String],\n) -> Result {\n let mut stdout = String::new();\n let mut action_results = Vec::new();\n let mut events = Vec::new();\n let mut recursive_tokens = TokenUsage::default();\n let mut final_answer: Option = None;\n\n // Build context variables including persisted state from prior steps\n let (input_names, input_values) = build_context_inputs(thread, persisted_state);\n\n // Collect known tool names so NameLookup can return callable stubs.\n // Without this, `mission_list()` in code raises NameError because Monty\n // resolves the name before calling it, and Undefined → NameError.\n let active_leases = leases.active_for_thread(thread.id).await;\n let mut known_actions: std::collections::HashSet = effects\n .available_actions(&active_leases)\n .await\n .unwrap_or_default()\n .into_iter()\n .map(|a| a.name)\n .collect();\n\n // Register skill code snippet function names as additional known actions.\n // These resolve in NameLookup so the LLM can call them as Python functions.\n for name in skill_snippet_names {\n known_actions.insert(name.clone());\n }\n\n // Parse and compile (wrap in catch_unwind — Monty 0.0.x can panic)\n let runner = match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n MontyRun::new(code.to_string(), \"step.py\", input_names)\n })) {\n Ok(Ok(runner)) => runner,\n Ok(Err(e)) => {\n // Parse error flows back to LLM (not a termination)\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"SyntaxError: {e}\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n failure: Some(CodeExecutionFailure::SyntaxError),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"{stdout}\\nVmPanic: Monty VM panicked during code parsing\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n };\n\n // Start execution with resource limits and context inputs\n let tracker = LimitedTracker::new(default_limits());\n\n let run_result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n runner.start(input_values, tracker, PrintWriter::Collect(&mut stdout))\n }));\n\n let mut progress = match run_result {\n Ok(Ok(p)) => p,\n Ok(Err(e)) => {\n // Runtime error flows back to LLM\n let category = classify_runtime_error(&e.to_string());\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"{stdout}\\nError: {e}\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n failure: Some(category),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"{stdout}\\nVmPanic: Monty VM panicked during execution start\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n };\n\n // Pending async tool executions keyed by Monty call_id.\n // When a tool FunctionCall comes in, we spawn a tokio task and store\n // the JoinHandle here. When ResolveFutures yields, we await them.\n let mut pending_futures: HashMap = HashMap::new();\n\n // Drive the execution loop\n let mut call_counter = 0u32;\n loop {\n match progress {\n RunProgress::Complete(obj) => {\n return Ok(CodeExecutionResult {\n return_value: monty_to_json(&obj),\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: None,\n });\n }\n\n RunProgress::FunctionCall(call) => {\n call_counter += 1;\n let str_call_id = format!(\"code_call_{call_counter}\");\n let monty_call_id = call.call_id;\n let action_name = call.function_name.clone();\n let params = monty_args_to_json(&call.args, &call.kwargs);\n\n debug!(action = %action_name, call_id = %str_call_id, monty_id = monty_call_id, \"Monty: function call\");\n\n // Builtins that need synchronous results — resume with value.\n let sync_result = match action_name.as_str() {\n \"FINAL\" => {\n let answer = call.args.first().map(monty_to_string).unwrap_or_default();\n final_answer = Some(answer);\n Some(ExtFunctionResult::Return(MontyObject::None))\n }\n \"FINAL_VAR\" => {\n let var_name = call\n .args\n .first()\n .map(monty_to_string)\n .unwrap_or_else(|| \"result\".into());\n final_answer = Some(format!(\"[FINAL_VAR: {var_name}]\"));\n Some(ExtFunctionResult::Return(MontyObject::None))\n }\n // LLM calls are async — spawn tokio task, resume_pending.\n // This allows asyncio.gather(llm_query(...), tool(...))\n // to run the LLM call and tool call concurrently.\n \"llm_query\" => {\n let args = call.args.clone();\n let kwargs = call.kwargs.clone();\n let llm = llm.clone();\n let handle = tokio::spawn(async move {\n handle_llm_query_standalone(&args, &kwargs, &llm).await\n });\n pending_futures.insert(monty_call_id, PendingFuture::Llm { handle });\n None // handled as async below\n }\n \"llm_query_batched\" => {\n let args = call.args.clone();\n let kwargs = call.kwargs.clone();\n let llm = llm.clone();\n let handle = tokio::spawn(async move {\n handle_llm_query_batched_standalone(&args, &kwargs, &llm).await\n });\n pending_futures.insert(monty_call_id, PendingFuture::Llm { handle });\n None\n }\n // rlm_query stays synchronous — it spawns a child Monty VM\n // which isn't Send, so it can't run in tokio::spawn.\n \"rlm_query\" => Some(\n handle_rlm_query(\n &call.args,\n &call.kwargs,\n thread,\n llm,\n effects,\n leases,\n policy,\n &mut recursive_tokens,\n )\n .await,\n ),\n \"globals\" | \"locals\" => {\n let entries: Vec<(MontyObject, MontyObject)> = known_actions\n .iter()\n .map(|name| {\n (MontyObject::String(name.clone()), MontyObject::Bool(true))\n })\n .collect();\n Some(ExtFunctionResult::Return(MontyObject::Dict(entries.into())))\n }\n _ => None, // tool call — handled async below\n };\n\n if let Some(ext_result) = sync_result {\n // Sync resume for builtins\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume(ext_result, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(classify_runtime_error(&e.to_string())),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during resume\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n continue;\n }\n\n // If an LLM call already inserted a pending future, just\n // resume_pending and continue — no preflight needed.\n if pending_futures.contains_key(&monty_call_id) {\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume_pending(PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(classify_runtime_error(&e.to_string())),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during resume_pending\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n continue;\n }\n\n // ── Async tool dispatch ─────────────────────────────\n // Preflight (lease + policy) is sync. If denied or\n // needs approval, resume with error immediately.\n // If approved, spawn tokio task and resume_pending().\n\n let preflight = preflight_action(\n &action_name,\n ¶ms,\n thread,\n effects,\n leases,\n policy,\n context,\n capability_policies,\n &str_call_id,\n &mut events,\n )\n .await;\n\n match preflight {\n PreflightResult::Approved(lease) => {\n // Spawn async execution\n let effects = effects.clone();\n let name = action_name.clone();\n let params_clone = params.clone();\n let lease_clone = lease.clone();\n let mut ctx = context.clone();\n ctx.current_call_id = Some(str_call_id.clone());\n let ps = crate::types::event::summarize_params(&name, ¶ms);\n\n let handle = tokio::spawn(async move {\n effects\n .execute_action(&name, params_clone, &lease_clone, &ctx)\n .await\n });\n\n pending_futures.insert(\n monty_call_id,\n PendingFuture::Tool {\n handle,\n action_name,\n call_id: str_call_id,\n lease_id: lease.id,\n parameters: params.clone(),\n params_summary: ps,\n },\n );\n\n // Resume with pending future — Python gets ExternalFuture\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume_pending(PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::ToolError),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during resume_pending\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n PreflightResult::Denied(ext_result) => {\n // Resume with error — Python sees an exception\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume(ext_result, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::ToolError),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during resume\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n PreflightResult::GatePaused(outcome) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: Some(outcome),\n recursive_tokens,\n final_answer: None,\n failure: None,\n });\n }\n }\n }\n\n // ── ResolveFutures: parallel execution ────────────────\n // Resolves both tool calls and LLM calls that were deferred\n // via resume_pending(). All pending tokio tasks are awaited\n // and their results fed back to Monty.\n RunProgress::ResolveFutures(resolve) => {\n let pending_ids = resolve.pending_call_ids().to_vec();\n debug!(pending = ?pending_ids, \"Monty: ResolveFutures — resolving {} pending futures\", pending_ids.len());\n\n let mut results: Vec<(u32, ExtFunctionResult)> =\n Vec::with_capacity(pending_ids.len());\n\n for &mid in &pending_ids {\n let ext_result = if let Some(pf) = pending_futures.remove(&mid) {\n match pf {\n PendingFuture::Tool {\n handle,\n action_name,\n call_id,\n lease_id,\n parameters,\n params_summary,\n } => {\n resolve_tool_future(\n handle,\n &action_name,\n &call_id,\n lease_id,\n parameters,\n params_summary,\n leases,\n context,\n &mut action_results,\n &mut events,\n )\n .await\n }\n PendingFuture::Llm { handle } => {\n resolve_llm_future(handle, &mut recursive_tokens).await\n }\n }\n } else {\n debug!(call_id = mid, \"ResolveFutures: unknown pending call_id\");\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"unknown pending call_id {mid}\")),\n ))\n };\n results.push((mid, ext_result));\n }\n\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n resolve.resume(results, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(classify_runtime_error(&e.to_string())),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during ResolveFutures resume\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n\n RunProgress::NameLookup(lookup) => {\n let name = lookup.name.clone();\n\n let result = if known_actions.contains(&name) {\n debug!(name = %name, \"Monty: resolved as tool function\");\n NameLookupResult::Value(MontyObject::Function {\n name: name.clone(),\n docstring: None,\n })\n } else if name == \"globals\" || name == \"locals\" {\n NameLookupResult::Value(MontyObject::Function {\n name: name.clone(),\n docstring: None,\n })\n } else {\n debug!(name = %name, \"Monty: unresolved name\");\n NameLookupResult::Undefined\n };\n\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n lookup.resume(result, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nNameError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::NameLookup),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during name lookup\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n\n RunProgress::OsCall(os_call) => {\n debug!(function = ?os_call.function, \"Monty: OS call denied\");\n let err = ExtFunctionResult::Error(MontyException::new(\n ExcType::OSError,\n Some(\"OS operations are not permitted in CodeAct scripts\".into()),\n ));\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n os_call.resume(err, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nOSError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::OsDenied),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"{stdout}\\nVmPanic: Monty VM panicked during OS call\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n }\n }\n}\n\n// ── Error classification ────────────────────────────────────\n\n/// Classify a runtime error message into a failure category.\n///\n/// Parses the error text from Monty to distinguish between LLM logic bugs\n/// (NameError, TypeError, etc.), resource limit hits, and Monty VM issues.\nfn classify_runtime_error(error_msg: &str) -> CodeExecutionFailure {\n let lower = error_msg.to_ascii_lowercase();\n\n // Most specific checks first to avoid substring false positives.\n if lower.contains(\"timed out\")\n || lower.contains(\"timeout\")\n || lower.contains(\"memory limit\")\n || lower.contains(\"allocation limit\")\n || lower.contains(\"out of fuel\")\n || lower.contains(\"fuel exhausted\")\n || lower.contains(\"resource limit\")\n {\n CodeExecutionFailure::ResourceLimit\n } else if lower.contains(\"os operations are not permitted\") || lower.contains(\"oserror\") {\n CodeExecutionFailure::OsDenied\n } else if lower.contains(\"syntaxerror\") {\n CodeExecutionFailure::SyntaxError\n } else {\n // NameError, TypeError, ValueError, AttributeError, IndexError,\n // KeyError, ModuleNotFoundError, NotImplementedError, etc.\n CodeExecutionFailure::RuntimeError\n }\n}\n\n/// Compute a short hash of Python code for dedup/correlation in events.\n///\n/// Uses FNV-1a (64-bit) which is stable across Rust versions, unlike\n/// `DefaultHasher`. Not cryptographic — collision probability is ~2^-32\n/// at typical usage levels, sufficient for dedup but not for security.\npub fn code_hash(code: &str) -> String {\n const FNV_OFFSET: u64 = 0xcbf29ce484222325;\n const FNV_PRIME: u64 = 0x00000100000001B3;\n let mut hash = FNV_OFFSET;\n for byte in code.as_bytes() {\n hash ^= *byte as u64;\n hash = hash.wrapping_mul(FNV_PRIME);\n }\n format!(\"{hash:016x}\")\n}\n\n// ── Pending future tracking ─────────────────────────────────\n\n/// A deferred computation spawned as a tokio task, pending resolution\n/// via `ResolveFutures`. Can be a tool execution or an LLM call.\nenum PendingFuture {\n /// Tool action execution.\n Tool {\n handle: tokio::task::JoinHandle>,\n action_name: String,\n call_id: String,\n lease_id: crate::types::capability::LeaseId,\n parameters: serde_json::Value,\n params_summary: Option,\n },\n /// LLM call (llm_query / llm_query_batched / rlm_query).\n Llm {\n handle: tokio::task::JoinHandle<(ExtFunctionResult, TokenUsage)>,\n },\n}\n\n/// Result of preflight checks (lease + policy) for a tool call.\nenum PreflightResult {\n /// Tool approved — lease is consumed, ready to execute.\n Approved(crate::types::capability::CapabilityLease),\n /// Tool denied — return this error to Monty.\n Denied(ExtFunctionResult),\n /// Tool is paused by a gate — interrupt the batch.\n GatePaused(crate::runtime::messaging::ThreadOutcome),\n}\n\n/// Run preflight checks for a tool call: find lease, check policy, consume use.\n#[allow(clippy::too_many_arguments)]\nasync fn preflight_action(\n action_name: &str,\n params: &serde_json::Value,\n thread: &Thread,\n effects: &Arc,\n leases: &LeaseManager,\n policy: &PolicyEngine,\n context: &ThreadExecutionContext,\n capability_policies: &[crate::types::capability::PolicyRule],\n call_id: &str,\n events: &mut Vec,\n) -> PreflightResult {\n let lease = match leases.find_lease_for_action(thread.id, action_name).await {\n Some(l) => l,\n None => {\n events.push(EventKind::ActionFailed {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n error: format!(\"no lease for action '{action_name}'\"),\n params_summary: None,\n });\n return PreflightResult::Denied(ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"no lease for action '{action_name}'\")),\n )));\n }\n };\n\n let action_def = effects\n .available_actions(std::slice::from_ref(&lease))\n .await\n .ok()\n .and_then(|actions| actions.into_iter().find(|a| a.name == action_name));\n\n if let Some(ref action_def) = action_def {\n match policy.evaluate(action_def, &lease, capability_policies) {\n PolicyDecision::Deny { reason } => {\n events.push(EventKind::ActionFailed {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n error: reason.clone(),\n params_summary: None,\n });\n return PreflightResult::Denied(ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"denied: {reason}\")),\n )));\n }\n PolicyDecision::RequireApproval { .. } => {\n events.push(EventKind::ApprovalRequested {\n action_name: action_name.into(),\n call_id: call_id.into(),\n parameters: Some(params.clone()),\n description: None,\n allow_always: None,\n gate_name: None,\n params_summary: crate::types::event::summarize_params(action_name, params),\n });\n return PreflightResult::GatePaused(\n crate::runtime::messaging::ThreadOutcome::GatePaused {\n gate_name: \"approval\".into(),\n action_name: action_name.into(),\n call_id: call_id.into(),\n parameters: params.clone(),\n resume_kind: crate::gate::ResumeKind::Approval { allow_always: true },\n resume_output: None,\n },\n );\n }\n PolicyDecision::Allow => {}\n }\n }\n\n if let Err(e) = leases.consume_use(lease.id).await {\n return PreflightResult::Denied(ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"lease exhausted: {e}\")),\n )));\n }\n\n PreflightResult::Approved(lease)\n}\n\n// ── llm_query() — recursive subagent (RLM 3.5) ─────────────\n\n/// Handle `llm_query(prompt, context)` — single recursive sub-call.\nasync fn handle_llm_query(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n llm: &Arc,\n recursive_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n let prompt = extract_string_arg(args, kwargs, \"prompt\", 0);\n let context_arg = extract_string_arg(args, kwargs, \"context\", 1);\n // `model` must be parsed explicitly — `extract_string_arg` coerces via\n // `monty_to_string`, which turns `MontyObject::None` into the literal\n // string \"None\" and stringifies non-string values, both of which would\n // silently route the call to an invalid model ID. Accept only str or None.\n let model_arg = match extract_optional_string_kwarg(args, kwargs, \"model\", 2) {\n Ok(v) => v,\n Err(err) => return err,\n };\n\n let prompt = match prompt {\n Some(p) => p,\n None => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(\"llm_query() requires a 'prompt' argument\".into()),\n ));\n }\n };\n\n let mut messages = Vec::new();\n if let Some(ctx) = context_arg {\n messages.push(ThreadMessage::system(format!(\n \"You are a sub-agent. Answer concisely based on the context.\\n\\n{ctx}\"\n )));\n } else {\n // Some providers (e.g. OpenAI Codex Responses API) require a system\n // message / instructions field. Always include one.\n messages.push(ThreadMessage::system(\n \"You are a helpful sub-agent. Answer concisely.\",\n ));\n }\n messages.push(ThreadMessage::user(prompt));\n\n let config = LlmCallConfig {\n force_text: true,\n model: model_arg,\n ..LlmCallConfig::default()\n };\n\n match llm.complete(&messages, &[], &config).await {\n Ok(output) => {\n recursive_tokens.input_tokens += output.usage.input_tokens;\n recursive_tokens.output_tokens += output.usage.output_tokens;\n let text = match output.response {\n LlmResponse::Text(t) => t,\n LlmResponse::ActionCalls { content, .. } | LlmResponse::Code { content, .. } => {\n content.unwrap_or_default()\n }\n };\n ExtFunctionResult::Return(MontyObject::String(text))\n }\n Err(e) => ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"llm_query failed: {e}\")),\n )),\n }\n}\n\n/// Handle `llm_query_batched(prompts)` — parallel recursive sub-calls.\n///\n/// Takes a list of prompt strings and dispatches them concurrently.\n/// Returns a list of response strings in the same order.\nasync fn handle_llm_query_batched(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n llm: &Arc,\n recursive_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n // Extract prompts list (first arg or kwarg \"prompts\")\n let prompts_obj = args.first().or_else(|| {\n kwargs.iter().find_map(|(k, v)| {\n if let MontyObject::String(key) = k\n && key == \"prompts\"\n {\n return Some(v);\n }\n None\n })\n });\n\n let prompts: Vec = match prompts_obj {\n Some(MontyObject::List(items)) => items.iter().map(monty_to_string).collect(),\n Some(other) => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(format!(\n \"llm_query_batched() expects a list of prompts, got {other:?}\"\n )),\n ));\n }\n None => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(\"llm_query_batched() requires a 'prompts' argument\".into()),\n ));\n }\n };\n\n // Positional/keyword layout (matches the documented signature\n // `llm_query_batched(prompts, context=None, model=None, models=None)`):\n // arg 0 = prompts (already extracted above)\n // arg 1 = context\n // arg 2 = model\n // arg 3 = models\n // All three of context/model/models can also be passed by keyword.\n let context_arg = match extract_optional_string_kwarg(args, kwargs, \"context\", 1) {\n Ok(v) => v,\n Err(err) => return err,\n };\n\n // Optional model overrides:\n // - `model=\"...\"` applies the same model to every prompt\n // - `models=[...]` is a parallel array (must match prompts length); use\n // this to broadcast the same prompt across a council of models by\n // passing `prompts=[same]*N, models=[m1, m2, ...]`. Within `models`,\n // a `None` slot means \"no override for this prompt\" (the caller\n // opted out of routing for that slot); the singular `model=` kwarg\n // does NOT fill those slots, since mixing the two would be surprising.\n // See note in handle_llm_query: `model` must be parsed explicitly so that\n // `model=None` doesn't become the literal string \"None\".\n let single_model = match extract_optional_string_kwarg(args, kwargs, \"model\", 2) {\n Ok(v) => v,\n Err(err) => return err,\n };\n let models_kwarg = kwargs\n .iter()\n .find_map(|(k, v)| match k {\n MontyObject::String(key) if key == \"models\" => Some(v),\n _ => None,\n })\n .or_else(|| args.get(3));\n\n let models_list: Option>> = match models_kwarg {\n None | Some(MontyObject::None) => None,\n Some(MontyObject::List(items)) => {\n let mut out = Vec::with_capacity(items.len());\n for item in items {\n match item {\n MontyObject::String(s) => out.push(Some(s.clone())),\n MontyObject::None => out.push(None),\n other => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(format!(\n \"llm_query_batched(): models list entries must be str or None, got {other:?}\"\n )),\n ));\n }\n }\n }\n Some(out)\n }\n Some(other) => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(format!(\n \"llm_query_batched(): `models` must be a list of str or None, got {other:?}\"\n )),\n ));\n }\n };\n\n if let Some(ref ms) = models_list\n && ms.len() != prompts.len()\n {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::ValueError,\n Some(format!(\n \"llm_query_batched(): models list length ({}) must match prompts length ({})\",\n ms.len(),\n prompts.len()\n )),\n ));\n }\n\n let mut handles = Vec::with_capacity(prompts.len());\n for (i, prompt) in prompts.iter().enumerate() {\n let llm = Arc::clone(llm);\n let ctx = context_arg.clone();\n let prompt = prompt.clone();\n // If `models=` was provided, each slot is authoritative — `None` means\n // \"no override for this prompt\" and is NOT backfilled from `model=`.\n // Otherwise, fall back to the singular `model=` kwarg (or None).\n let model_override = match models_list.as_ref() {\n Some(ms) => ms[i].clone(),\n None => single_model.clone(),\n };\n let config = LlmCallConfig {\n force_text: true,\n model: model_override,\n ..LlmCallConfig::default()\n };\n handles.push(tokio::spawn(async move {\n let mut messages = Vec::new();\n if let Some(ctx) = ctx {\n messages.push(ThreadMessage::system(format!(\n \"You are a sub-agent. Answer concisely.\\n\\n{ctx}\"\n )));\n } else {\n messages.push(ThreadMessage::system(\n \"You are a helpful sub-agent. Answer concisely.\",\n ));\n }\n messages.push(ThreadMessage::user(prompt));\n llm.complete(&messages, &[], &config).await\n }));\n }\n\n // Collect results\n let mut results = Vec::with_capacity(prompts.len());\n let mut total_input = 0u64;\n let mut total_output = 0u64;\n\n for handle in handles {\n match handle.await {\n Ok(Ok(output)) => {\n total_input += output.usage.input_tokens;\n total_output += output.usage.output_tokens;\n let text = match output.response {\n LlmResponse::Text(t) => t,\n LlmResponse::ActionCalls { content, .. }\n | LlmResponse::Code { content, .. } => content.unwrap_or_default(),\n };\n results.push(MontyObject::String(text));\n }\n Ok(Err(e)) => {\n results.push(MontyObject::String(format!(\"Error: {e}\")));\n }\n Err(e) => {\n results.push(MontyObject::String(format!(\"Error: task failed: {e}\")));\n }\n }\n }\n\n recursive_tokens.input_tokens += total_input;\n recursive_tokens.output_tokens += total_output;\n\n ExtFunctionResult::Return(MontyObject::List(results))\n}\n\n// ── rlm_query() — full recursive sub-agent (RLM 3.5) ─────────\n\n/// Handle `rlm_query(prompt)` — spawn a child CodeAct thread with its own\n/// execution loop, tools, and iteration budget.\n///\n/// Unlike `llm_query()` (single-shot LLM call), `rlm_query()` creates a\n/// child thread with full CodeAct capabilities. The child inherits the\n/// parent's remaining budget and tool access.\n#[allow(clippy::too_many_arguments)]\nasync fn handle_rlm_query(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n parent_thread: &Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &LeaseManager,\n policy: &PolicyEngine,\n recursive_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n let prompt = extract_string_arg(args, kwargs, \"prompt\", 0);\n let prompt = match prompt {\n Some(p) => p,\n None => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(\"rlm_query() requires a 'prompt' argument\".into()),\n ));\n }\n };\n\n // Depth check — refuse if at max recursion depth\n let current_depth = parent_thread.config.depth;\n let max_depth = parent_thread.config.max_depth;\n if current_depth >= max_depth {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\n \"rlm_query() depth limit reached: depth {current_depth} >= max {max_depth}\"\n )),\n ));\n }\n\n // Build child thread with inherited budget\n let child_config = crate::types::thread::ThreadConfig {\n max_iterations: parent_thread.config.max_iterations.min(20), // cap child iterations\n enable_tool_intent_nudge: false,\n max_tokens_total: parent_thread\n .config\n .max_tokens_total\n .map(|max| max.saturating_sub(parent_thread.total_tokens_used)),\n max_budget_usd: parent_thread\n .config\n .max_budget_usd\n .map(|max| (max - parent_thread.total_cost_usd).max(0.0)),\n max_duration: parent_thread.config.max_duration,\n depth: current_depth + 1,\n max_depth,\n ..crate::types::thread::ThreadConfig::default()\n };\n\n let mut child_thread = crate::types::thread::Thread::new(\n &prompt,\n crate::types::thread::ThreadType::Research,\n parent_thread.project_id,\n &parent_thread.user_id,\n child_config,\n )\n .with_parent(parent_thread.id);\n\n // Add the prompt as a user message\n child_thread.add_message(ThreadMessage::user(&prompt));\n\n // Create signal channel and child's lease manager\n let (_tx, rx) = crate::runtime::messaging::signal_channel(8);\n let child_leases = Arc::new(LeaseManager::new());\n\n // Grant the child the same leases as the parent (in the child's manager)\n let parent_leases = leases.active_for_thread(parent_thread.id).await;\n let now = chrono::Utc::now();\n for parent_lease in &parent_leases {\n // Convert parent's expires_at to remaining duration\n let remaining_duration = parent_lease\n .expires_at\n .and_then(|exp| (exp - now).to_std().ok())\n .map(|d| chrono::Duration::from_std(d).unwrap_or(chrono::Duration::hours(1)));\n let lease = match child_leases\n .grant(\n child_thread.id,\n &parent_lease.capability_name,\n parent_lease.granted_actions.clone(),\n remaining_duration,\n parent_lease.max_uses,\n )\n .await\n {\n Ok(l) => l,\n Err(e) => {\n debug!(error = %e, \"rlm_query: skipping invalid lease for child thread\");\n continue;\n }\n };\n child_thread.capability_leases.push(lease.id);\n }\n let mut child_policy_engine = PolicyEngine::new();\n // Copy denied effects from parent policy\n for effect in &policy.denied_effects {\n child_policy_engine.deny_effect(*effect);\n }\n let child_policy = Arc::new(child_policy_engine);\n\n let mut child_loop = crate::executor::ExecutionLoop::new(\n child_thread,\n Arc::clone(llm),\n Arc::clone(effects),\n child_leases,\n child_policy,\n rx,\n \"rlm_child\".to_string(),\n );\n\n debug!(\n parent_thread = %parent_thread.id,\n depth = current_depth + 1,\n prompt_len = prompt.len(),\n \"rlm_query: spawning child CodeAct thread\"\n );\n\n // Run the child loop (Box::pin to avoid infinite future size from recursion)\n match Box::pin(child_loop.run()).await {\n Ok(outcome) => {\n // Track child's token usage\n recursive_tokens.input_tokens += child_loop.thread.total_tokens_used;\n recursive_tokens.cost_usd += child_loop.thread.total_cost_usd;\n\n let response = match outcome {\n crate::runtime::messaging::ThreadOutcome::Completed { response } => {\n response.unwrap_or_default()\n }\n crate::runtime::messaging::ThreadOutcome::Failed { error } => {\n format!(\"rlm_query child failed: {error}\")\n }\n crate::runtime::messaging::ThreadOutcome::MaxIterations => {\n \"rlm_query child reached max iterations\".to_string()\n }\n _ => String::new(),\n };\n\n ExtFunctionResult::Return(MontyObject::String(response))\n }\n Err(e) => ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"rlm_query failed: {e}\")),\n )),\n }\n}\n\n// ── Standalone async handlers (for tokio::spawn) ────────────\n\n/// `llm_query()` — standalone version that returns `(ExtFunctionResult, TokenUsage)`.\nasync fn handle_llm_query_standalone(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n llm: &Arc,\n) -> (ExtFunctionResult, TokenUsage) {\n let mut tokens = TokenUsage::default();\n let result = handle_llm_query(args, kwargs, llm, &mut tokens).await;\n (result, tokens)\n}\n\n/// `llm_query_batched()` — standalone version.\nasync fn handle_llm_query_batched_standalone(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n llm: &Arc,\n) -> (ExtFunctionResult, TokenUsage) {\n let mut tokens = TokenUsage::default();\n let result = handle_llm_query_batched(args, kwargs, llm, &mut tokens).await;\n (result, tokens)\n}\n\n// ── Future resolution helpers ───────────────────────────────\n\n/// Resolve a pending tool execution future.\n#[allow(clippy::too_many_arguments)]\nasync fn resolve_tool_future(\n handle: tokio::task::JoinHandle>,\n action_name: &str,\n call_id: &str,\n lease_id: crate::types::capability::LeaseId,\n parameters: serde_json::Value,\n params_summary: Option,\n leases: &LeaseManager,\n context: &ThreadExecutionContext,\n action_results: &mut Vec,\n events: &mut Vec,\n) -> ExtFunctionResult {\n match handle.await {\n Ok(Ok(result)) => {\n // If the effect adapter wrapped a tool error as an Ok(ActionResult)\n // with is_error=true (current convention in\n // `EffectBridgeAdapter::execute_action_internal`), surface it as\n // ActionFailed so traces, observers, and approval flows see the\n // failure correctly. Without this, every wrapped error looked like\n // a successful tool call to downstream consumers.\n if result.is_error {\n let error_msg = result\n .output\n .get(\"error\")\n .and_then(|v| v.as_str())\n .map(String::from)\n .unwrap_or_else(|| result.output.to_string());\n events.push(EventKind::ActionFailed {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n error: error_msg,\n params_summary,\n });\n } else {\n events.push(EventKind::ActionExecuted {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n duration_ms: result.duration.as_millis() as u64,\n params_summary,\n });\n }\n let monty_val = json_to_monty(&result.output);\n action_results.push(result);\n ExtFunctionResult::Return(monty_val)\n }\n Ok(Err(EngineError::GatePaused {\n gate_name,\n action_name,\n call_id,\n resume_kind,\n ..\n })) => {\n let _ = leases.refund_use(lease_id).await;\n events.push(EventKind::ApprovalRequested {\n action_name,\n call_id,\n parameters: Some(parameters),\n description: None,\n allow_always: match *resume_kind {\n crate::gate::ResumeKind::Approval { allow_always } => Some(allow_always),\n _ => None,\n },\n gate_name: Some(gate_name.clone()),\n params_summary,\n });\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"execution paused by gate '{gate_name}'\")),\n ))\n }\n Ok(Err(e)) => {\n events.push(EventKind::ActionFailed {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n error: e.to_string(),\n params_summary,\n });\n action_results.push(ActionResult {\n call_id: call_id.into(),\n action_name: action_name.into(),\n output: serde_json::json!({\"error\": e.to_string()}),\n is_error: true,\n duration: Duration::ZERO,\n });\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(e.to_string()),\n ))\n }\n Err(e) => {\n debug!(\"async tool task panicked: {e}\");\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"tool execution panicked: {e}\")),\n ))\n }\n }\n}\n\n/// Resolve a pending LLM call future, accumulating token usage.\nasync fn resolve_llm_future(\n handle: tokio::task::JoinHandle<(ExtFunctionResult, TokenUsage)>,\n recursive_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n match handle.await {\n Ok((result, tokens)) => {\n recursive_tokens.input_tokens += tokens.input_tokens;\n recursive_tokens.output_tokens += tokens.output_tokens;\n result\n }\n Err(e) => {\n debug!(\"async LLM task panicked: {e}\");\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"LLM call panicked: {e}\")),\n ))\n }\n }\n}\n\n// ── Helpers ─────────────────────────────────────────────────\n\nfn extract_string_arg(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n name: &str,\n position: usize,\n) -> Option {\n for (k, v) in kwargs {\n if let MontyObject::String(key) = k\n && key == name\n {\n return Some(monty_to_string(v));\n }\n }\n args.get(position).map(monty_to_string)\n}\n\n/// Strict optional-string extractor for arguments where silent coercion is\n/// dangerous (e.g. `model=` — passing the wrong type should NOT become an\n/// unintended model ID). Returns:\n/// - `Ok(None)` when the argument is missing or explicitly `None`\n/// - `Ok(Some(s))` when the argument is a string\n/// - `Err(TypeError)` for any other type\nfn extract_optional_string_kwarg(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n name: &str,\n position: usize,\n) -> Result, ExtFunctionResult> {\n let raw = kwargs\n .iter()\n .find_map(|(k, v)| match k {\n MontyObject::String(key) if key == name => Some(v),\n _ => None,\n })\n .or_else(|| args.get(position));\n\n match raw {\n None | Some(MontyObject::None) => Ok(None),\n Some(MontyObject::String(s)) => Ok(Some(s.clone())),\n Some(other) => Err(ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(format!(\"`{name}` must be a string or None, got {other:?}\")),\n ))),\n }\n}\n\npub(crate) fn monty_to_string(obj: &MontyObject) -> String {\n match obj {\n MontyObject::String(s) => s.clone(),\n MontyObject::None => \"None\".into(),\n MontyObject::Bool(b) => b.to_string(),\n MontyObject::Int(i) => i.to_string(),\n MontyObject::Float(f) => f.to_string(),\n other => {\n serde_json::to_string(&monty_to_json(other)).unwrap_or_else(|_| format!(\"{other:?}\"))\n }\n }\n}\n\n// Dispatch logic moved to orchestrator.rs (__execute_action__ handler).\n// GatePaused is handled via EngineError → JSON in orchestrator.rs.\n// ── MontyObject ↔ JSON ──────────────────────────────────────\n\npub(crate) fn monty_to_json(obj: &MontyObject) -> serde_json::Value {\n match obj {\n MontyObject::None => serde_json::Value::Null,\n MontyObject::Bool(b) => serde_json::Value::Bool(*b),\n MontyObject::Int(i) => serde_json::json!(i),\n MontyObject::BigInt(i) => serde_json::Value::String(i.to_string()),\n MontyObject::Float(f) => serde_json::json!(f),\n MontyObject::String(s) => serde_json::Value::String(s.clone()),\n MontyObject::List(items) | MontyObject::Tuple(items) => {\n serde_json::Value::Array(items.iter().map(monty_to_json).collect())\n }\n MontyObject::Dict(pairs) => {\n let map: serde_json::Map = pairs\n .into_iter()\n .map(|(k, v)| {\n let key = match k {\n MontyObject::String(s) => s.clone(),\n other => format!(\"{other:?}\"),\n };\n (key, monty_to_json(v))\n })\n .collect();\n serde_json::Value::Object(map)\n }\n MontyObject::Set(items) | MontyObject::FrozenSet(items) => {\n serde_json::Value::Array(items.iter().map(monty_to_json).collect())\n }\n MontyObject::Bytes(b) => {\n serde_json::Value::String(b.iter().map(|byte| format!(\"{byte:02x}\")).collect())\n }\n other => serde_json::Value::String(format!(\"{other:?}\")),\n }\n}\n\npub(crate) fn json_to_monty(val: &serde_json::Value) -> MontyObject {\n match val {\n serde_json::Value::Null => MontyObject::None,\n serde_json::Value::Bool(b) => MontyObject::Bool(*b),\n serde_json::Value::Number(n) => {\n if let Some(i) = n.as_i64() {\n MontyObject::Int(i)\n } else if let Some(f) = n.as_f64() {\n MontyObject::Float(f)\n } else {\n MontyObject::String(n.to_string())\n }\n }\n serde_json::Value::String(s) => MontyObject::String(s.clone()),\n serde_json::Value::Array(arr) => MontyObject::List(arr.iter().map(json_to_monty).collect()),\n serde_json::Value::Object(map) => MontyObject::dict(\n map.iter()\n .map(|(k, v)| (MontyObject::String(k.clone()), json_to_monty(v)))\n .collect::>(),\n ),\n }\n}\n\nfn monty_args_to_json(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n) -> serde_json::Value {\n let mut map = serde_json::Map::new();\n if !args.is_empty() {\n map.insert(\n \"_args\".into(),\n serde_json::Value::Array(args.iter().map(monty_to_json).collect()),\n );\n }\n for (k, v) in kwargs {\n let key = match k {\n MontyObject::String(s) => s.clone(),\n other => format!(\"{other:?}\"),\n };\n map.insert(key, monty_to_json(v));\n }\n serde_json::Value::Object(map)\n}\n\n#[cfg(test)]\nmod tests {\n use super::*;\n use crate::capability::lease::LeaseManager;\n use crate::capability::policy::PolicyEngine;\n use crate::traits::effect::ThreadExecutionContext;\n use crate::types::capability::{ActionDef, CapabilityLease, EffectType, GrantedActions};\n use crate::types::project::ProjectId;\n use crate::types::step::{ActionResult, StepId};\n use crate::types::thread::{Thread, ThreadConfig, ThreadType};\n use std::sync::Mutex;\n\n /// Truncate a string to at most `max_bytes`, snapping to a UTF-8 char\n /// boundary so assertion messages never panic on multibyte output.\n fn truncate_for_assert(s: &str, max_bytes: usize) -> &str {\n if s.len() <= max_bytes {\n return s;\n }\n let mut end = max_bytes;\n while end > 0 && !s.is_char_boundary(end) {\n end -= 1;\n }\n &s[..end] // safety: end is walked down to a valid char boundary above\n }\n\n struct MockEffects {\n results: Mutex>>,\n actions: Vec,\n }\n\n impl MockEffects {\n fn new(actions: Vec, results: Vec>) -> Self {\n Self {\n results: Mutex::new(results),\n actions,\n }\n }\n }\n\n #[async_trait::async_trait]\n impl EffectExecutor for MockEffects {\n async fn execute_action(\n &self,\n name: &str,\n _params: serde_json::Value,\n _lease: &CapabilityLease,\n _ctx: &ThreadExecutionContext,\n ) -> Result {\n let mut results = self.results.lock().unwrap();\n if results.is_empty() {\n Ok(ActionResult {\n call_id: String::new(),\n action_name: name.into(),\n output: serde_json::json!({\"result\": \"ok\"}),\n is_error: false,\n duration: Duration::from_millis(1),\n })\n } else {\n results.remove(0)\n }\n }\n\n async fn available_actions(\n &self,\n _leases: &[CapabilityLease],\n ) -> Result, EngineError> {\n Ok(self.actions.clone())\n }\n }\n\n fn test_action(name: &str) -> ActionDef {\n ActionDef {\n name: name.into(),\n description: \"Test tool\".into(),\n parameters_schema: serde_json::json!({\"type\": \"object\"}),\n effects: vec![EffectType::ReadLocal],\n requires_approval: false,\n }\n }\n\n fn make_test_thread() -> Thread {\n Thread::new(\n \"test goal\",\n ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n ThreadConfig::default(),\n )\n }\n\n fn make_exec_context(thread: &Thread) -> ThreadExecutionContext {\n ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: \"test\".into(),\n step_id: StepId::new(),\n current_call_id: None,\n source_channel: None,\n user_timezone: None,\n }\n }\n\n /// Stub LLM that always returns text \"stub\". Only used so execute_code\n /// doesn't need a real LLM — our tests exercise tool dispatch, not LLM calls.\n struct StubLlm;\n\n #[async_trait::async_trait]\n impl crate::traits::llm::LlmBackend for StubLlm {\n fn model_name(&self) -> &str {\n \"stub\"\n }\n\n async fn complete(\n &self,\n _messages: &[crate::types::message::ThreadMessage],\n _actions: &[ActionDef],\n _config: &crate::traits::llm::LlmCallConfig,\n ) -> Result {\n Ok(crate::traits::llm::LlmOutput {\n response: crate::types::step::LlmResponse::Text(\"stub\".into()),\n usage: crate::types::step::TokenUsage::default(),\n })\n }\n }\n\n async fn run_code(\n code: &str,\n effects: Arc,\n thread: &Thread,\n ) -> Result {\n let leases = LeaseManager::new();\n let policy = PolicyEngine::new();\n let ctx = make_exec_context(thread);\n\n // Grant a wildcard lease\n leases\n .grant(thread.id, \"tools\", GrantedActions::All, None, None)\n .await\n .unwrap();\n\n execute_code(\n code,\n thread,\n &(Arc::new(StubLlm) as Arc),\n &effects,\n &leases,\n &policy,\n &ctx,\n &[],\n &serde_json::json!({}),\n )\n .await\n }\n\n // ── Single await tool call ──────────────────────────────\n\n #[tokio::test]\n async fn single_await_tool_call() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"echo\")],\n vec![Ok(ActionResult {\n call_id: String::new(),\n action_name: \"echo\".into(),\n output: serde_json::json!(\"hello world\"),\n is_error: false,\n duration: Duration::from_millis(1),\n })],\n ));\n\n let code = r#\"\nresult = await echo(message=\"hello\")\nFINAL(str(result))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(\n result.final_answer.is_some(),\n \"should have final answer, stdout: {}\",\n result.stdout\n );\n assert!(\n result.failure.is_none(),\n \"should not error, stdout: {}\",\n result.stdout\n );\n assert_eq!(result.action_results.len(), 1);\n }\n\n // ── asyncio.gather parallel execution ───────────────────\n\n #[tokio::test]\n async fn asyncio_gather_two_tools() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"tool_a\"), test_action(\"tool_b\")],\n vec![\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"tool_a\".into(),\n output: serde_json::json!(10),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"tool_b\".into(),\n output: serde_json::json!(32),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n ],\n ));\n\n let code = r#\"\nimport asyncio\na, b = await asyncio.gather(tool_a(), tool_b())\nFINAL(str(a + b))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(\n result.final_answer.is_some(),\n \"should have final answer, stdout: {}\",\n result.stdout\n );\n assert_eq!(\n result.final_answer.as_deref(),\n Some(\"42\"),\n \"10 + 32 = 42, got: {:?}, stdout: {}\",\n result.final_answer,\n result.stdout\n );\n assert_eq!(result.action_results.len(), 2);\n assert!(result.failure.is_none());\n }\n\n // ── asyncio.gather three tools ──────────────────────────\n\n #[tokio::test]\n async fn asyncio_gather_three_tools() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![\n test_action(\"web_search\"),\n test_action(\"http\"),\n test_action(\"memory_search\"),\n ],\n vec![\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"web_search\".into(),\n output: serde_json::json!(\"search results\"),\n is_error: false,\n duration: Duration::from_millis(50),\n }),\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"http\".into(),\n output: serde_json::json!(\"page content\"),\n is_error: false,\n duration: Duration::from_millis(100),\n }),\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"memory_search\".into(),\n output: serde_json::json!(\"memories\"),\n is_error: false,\n duration: Duration::from_millis(25),\n }),\n ],\n ));\n\n let code = r#\"\nimport asyncio\ns, h, m = await asyncio.gather(\n web_search(query=\"test\"),\n http(url=\"https://example.com\"),\n memory_search(query=\"prior\"),\n)\nFINAL(str(s) + \"|\" + str(h) + \"|\" + str(m))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.action_results.len(), 3);\n let answer = result.final_answer.unwrap();\n assert!(answer.contains(\"search results\"), \"got: {answer}\");\n assert!(answer.contains(\"page content\"), \"got: {answer}\");\n assert!(answer.contains(\"memories\"), \"got: {answer}\");\n }\n\n // ── Data-dependent chain (sequential await) ─────────────\n\n #[tokio::test]\n async fn sequential_dependent_calls() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"step1\"), test_action(\"step2\")],\n vec![\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"step1\".into(),\n output: serde_json::json!(\"intermediate\"),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"step2\".into(),\n output: serde_json::json!(\"final\"),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n ],\n ));\n\n let code = r#\"\na = await step1()\nb = await step2(input=a)\nFINAL(str(b))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.action_results.len(), 2);\n assert_eq!(result.final_answer.as_deref(), Some(\"final\"));\n }\n\n // ── Error in one gathered tool ──────────────────────────\n\n #[tokio::test]\n async fn gather_with_error_propagates() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"good\"), test_action(\"bad\")],\n vec![\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"good\".into(),\n output: serde_json::json!(\"ok\"),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n Err(EngineError::Effect {\n reason: \"tool exploded\".into(),\n }),\n ],\n ));\n\n let code = r#\"\nimport asyncio\na, b = await asyncio.gather(good(), bad())\nFINAL(\"should not reach\")\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n // Error in gather propagates as exception — code should error\n assert!(\n result.failure.is_some(),\n \"should have error, stdout: {}\",\n result.stdout\n );\n assert!(\n result.final_answer.is_none()\n || result.final_answer.as_deref() != Some(\"should not reach\")\n );\n }\n\n // ── Tool with no lease (denied in preflight) ────────────\n\n #[tokio::test]\n async fn denied_tool_raises_exception() {\n let thread = make_test_thread();\n // No actions registered — tool has no lease\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n\n let code = r#\"\ntry:\n result = await unknown_tool()\n FINAL(\"should not reach\")\nexcept:\n FINAL(\"caught error\")\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n // Tool not found raises NameError before we even get to dispatch\n assert!(result.final_answer.is_some(), \"stdout: {}\", result.stdout);\n }\n\n // ── FINAL works without await ───────────────────────────\n\n #[tokio::test]\n async fn final_is_sync() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n\n let code = r#\"\nFINAL(\"hello from sync\")\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert_eq!(result.final_answer.as_deref(), Some(\"hello from sync\"));\n assert!(result.failure.is_none());\n }\n\n // ── globals() still works ───────────────────────────────\n\n #[tokio::test]\n async fn globals_returns_known_tools() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"web_search\"), test_action(\"http\")],\n vec![],\n ));\n\n let code = r#\"\ng = globals()\nhas_search = \"web_search\" in g\nhas_http = \"http\" in g\nFINAL(str(has_search) + \"|\" + str(has_http))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.final_answer.as_deref(), Some(\"True|True\"));\n }\n\n // ── Empty gather ────────────────────────────────────────\n\n #[tokio::test]\n async fn empty_gather() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n\n let code = r#\"\nimport asyncio\nresults = await asyncio.gather()\nFINAL(str(len(results)))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.final_answer.as_deref(), Some(\"0\"));\n }\n\n // ── Single-item gather ──────────────────────────────────\n\n #[tokio::test]\n async fn single_item_gather() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"echo\")],\n vec![Ok(ActionResult {\n call_id: String::new(),\n action_name: \"echo\".into(),\n output: serde_json::json!(\"gathered\"),\n is_error: false,\n duration: Duration::from_millis(1),\n })],\n ));\n\n let code = r#\"\nimport asyncio\nresults = await asyncio.gather(echo())\nFINAL(str(results[0]))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.final_answer.as_deref(), Some(\"gathered\"));\n assert_eq!(result.action_results.len(), 1);\n }\n\n // ── Sandbox security negative tests ────────────────────────\n\n /// OS-level operations must be denied or restricted by the Monty VM.\n #[tokio::test]\n async fn sandbox_denies_os_operations() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n // Try to import os and call os.system — should fail\n let code = r#\"\ntry:\n import os\n os.system(\"echo pwned\")\n FINAL(\"ESCAPED: os.system ran\")\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"os.system should be blocked, got: {answer}\",\n );\n }\n\n /// Resource limits must be enforced — infinite loops should be terminated.\n #[tokio::test]\n async fn sandbox_enforces_resource_limits() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n // Infinite allocation loop — should hit allocation or memory limit\n let code = r#\"\ndata = []\nwhile True:\n data.append(\"x\" * 10000)\n\"#;\n let result = run_code(code, effects, &thread).await;\n // Either returns an error or the stdout contains an error message —\n // the key assertion is that it DOES NOT run forever.\n if let Ok(r) = result {\n assert!(\n r.failure.is_some() || r.stdout.contains(\"Error\") || r.stdout.contains(\"limit\"),\n \"resource limit should terminate infinite loop, got stdout: {}\",\n truncate_for_assert(&r.stdout, 500),\n );\n }\n // Err(_) is also acceptable — means the VM was killed by resource limits\n }\n\n /// Python `import` of system modules must be restricted.\n #[tokio::test]\n async fn sandbox_restricts_imports() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n // Try to import subprocess — should fail\n let code = r#\"\ntry:\n import subprocess\n result = subprocess.run([\"echo\", \"escaped\"], capture_output=True, text=True)\n FINAL(\"ESCAPED: \" + result.stdout)\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"subprocess import should be blocked, got: {answer}\",\n );\n }\n\n /// File system access via open() must be blocked.\n #[tokio::test]\n async fn sandbox_denies_file_access() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n let code = r#\"\ntry:\n f = open(\"/etc/passwd\", \"r\")\n content = f.read()\n f.close()\n FINAL(\"ESCAPED: \" + content[:50])\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"open() should be blocked, got: {answer}\",\n );\n }\n\n /// Network access via socket must be blocked.\n #[tokio::test]\n async fn sandbox_denies_socket_access() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n let code = r#\"\ntry:\n import socket\n s = socket.socket(socket.AF_INET, socket.SOCK_STREAM)\n s.connect((\"127.0.0.1\", 80))\n FINAL(\"ESCAPED: connected\")\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"socket access should be blocked, got: {answer}\",\n );\n }\n\n /// Calls to tools not covered by the lease must be denied.\n #[tokio::test]\n async fn sandbox_unlicensed_tool_denied() {\n let effects: Arc =\n Arc::new(MockEffects::new(vec![test_action(\"allowed_tool\")], vec![]));\n let thread = make_test_thread();\n let leases = LeaseManager::new();\n let policy = PolicyEngine::new();\n let ctx = make_exec_context(&thread);\n\n // Grant a restricted lease — only \"allowed_tool\" is permitted.\n leases\n .grant(\n thread.id,\n \"tools\",\n GrantedActions::Specific(vec![\"allowed_tool\".into()]),\n None,\n None,\n )\n .await\n .unwrap();\n\n let code = r#\"\ntry:\n result = await secret_admin_tool(data=\"pwn\")\n FINAL(\"ESCAPED: \" + str(result))\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = execute_code(\n code,\n &thread,\n &(Arc::new(StubLlm) as Arc),\n &effects,\n &leases,\n &policy,\n &ctx,\n &[],\n &serde_json::json!({}),\n )\n .await\n .unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"unlicensed tool should be denied by preflight, got: {answer}\",\n );\n }\n\n /// CPU-bound infinite loops must be terminated by allocation/duration limits.\n #[tokio::test]\n async fn sandbox_enforces_cpu_limits() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n // Tight CPU-bound loop (no allocations to trip allocation limit)\n let code = r#\"\nx = 0\nwhile True:\n x += 1\n\"#;\n let result = run_code(code, effects, &thread).await;\n // Must terminate — either via error or resource limit\n if let Ok(r) = result {\n assert!(\n r.failure.is_some() || r.stdout.contains(\"Error\") || r.stdout.contains(\"limit\"),\n \"cpu-bound loop should be terminated, stdout: {}\",\n truncate_for_assert(&r.stdout, 500),\n );\n }\n // Err(_) is also acceptable — means the VM was killed by resource limits\n }\n\n /// FINAL() must capture the answer from the code.\n #[tokio::test]\n async fn sandbox_final_captures_answer() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n let code = r#\"\nx = 2 + 3\nFINAL(str(x))\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n assert_eq!(\n result.final_answer.as_deref(),\n Some(\"5\"),\n \"FINAL should capture the computed answer\"\n );\n }\n\n /// Syntax errors flow back as errors, not panics.\n #[tokio::test]\n async fn sandbox_handles_syntax_error() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n let code = \"def broken(\\nFINAL('nope')\";\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_some(), \"syntax error should set failure\");\n assert!(\n result.stdout.contains(\"SyntaxError\") || result.stdout.contains(\"Error\"),\n \"should contain SyntaxError, got: {}\",\n result.stdout,\n );\n }\n\n // ── llm_query model parameter plumbing ─────────────────────\n\n /// LLM backend that records every call's model + prompt for assertions.\n struct CapturingLlm {\n calls: tokio::sync::Mutex, String)>>,\n }\n\n impl CapturingLlm {\n fn new() -> Self {\n Self {\n calls: tokio::sync::Mutex::new(Vec::new()),\n }\n }\n }\n\n #[async_trait::async_trait]\n impl crate::traits::llm::LlmBackend for CapturingLlm {\n fn model_name(&self) -> &str {\n \"capturing\"\n }\n\n async fn complete(\n &self,\n messages: &[crate::types::message::ThreadMessage],\n _actions: &[ActionDef],\n config: &crate::traits::llm::LlmCallConfig,\n ) -> Result {\n let user_prompt = messages\n .iter()\n .rev()\n .find(|m| matches!(m.role, crate::types::message::MessageRole::User))\n .map(|m| m.content.clone())\n .unwrap_or_default();\n self.calls\n .lock()\n .await\n .push((config.model.clone(), user_prompt.clone()));\n Ok(crate::traits::llm::LlmOutput {\n response: crate::types::step::LlmResponse::Text(format!(\n \"ack:{}:{user_prompt}\",\n config.model.as_deref().unwrap_or(\"default\")\n )),\n usage: crate::types::step::TokenUsage::default(),\n })\n }\n }\n\n #[tokio::test]\n async fn llm_query_forwards_model_kwarg() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query(\n &[],\n &[\n (\n MontyObject::String(\"prompt\".into()),\n MontyObject::String(\"what is 2+2?\".into()),\n ),\n (\n MontyObject::String(\"model\".into()),\n MontyObject::String(\"gpt-4o\".into()),\n ),\n ],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n match result {\n ExtFunctionResult::Return(MontyObject::String(s)) => {\n assert!(s.contains(\"gpt-4o\"), \"got: {s}\");\n }\n other => panic!(\"expected string return, got {other:?}\"),\n }\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].0.as_deref(), Some(\"gpt-4o\"));\n assert_eq!(calls[0].1, \"what is 2+2?\");\n }\n\n #[tokio::test]\n async fn llm_query_without_model_passes_none() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let _ = handle_llm_query(\n &[MontyObject::String(\"hello\".into())],\n &[],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].0, None);\n }\n\n #[tokio::test]\n async fn llm_query_batched_broadcasts_with_models_list() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"Q\".into()),\n MontyObject::String(\"Q\".into()),\n MontyObject::String(\"Q\".into()),\n ]);\n let models = MontyObject::List(vec![\n MontyObject::String(\"gpt-4o\".into()),\n MontyObject::String(\"claude-sonnet-4-20250514\".into()),\n MontyObject::String(\"llama-3.1-70b-instruct\".into()),\n ]);\n let result = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"models\".into()), models)],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n match result {\n ExtFunctionResult::Return(MontyObject::List(items)) => {\n assert_eq!(items.len(), 3);\n }\n other => panic!(\"expected list return, got {other:?}\"),\n }\n\n let mut calls = llm.calls.lock().await;\n calls.sort_by(|a, b| a.0.cmp(&b.0));\n assert_eq!(calls.len(), 3);\n assert_eq!(calls[0].0.as_deref(), Some(\"claude-sonnet-4-20250514\"));\n assert_eq!(calls[1].0.as_deref(), Some(\"gpt-4o\"));\n assert_eq!(calls[2].0.as_deref(), Some(\"llama-3.1-70b-instruct\"));\n }\n\n #[tokio::test]\n async fn llm_query_batched_single_model_applies_to_all() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n let _ = handle_llm_query_batched(\n &[prompts],\n &[(\n MontyObject::String(\"model\".into()),\n MontyObject::String(\"gpt-4o\".into()),\n )],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 2);\n assert!(calls.iter().all(|(m, _)| m.as_deref() == Some(\"gpt-4o\")));\n }\n\n #[tokio::test]\n async fn llm_query_model_none_kwarg_is_no_override_not_literal_none_string() {\n // Regression: `extract_string_arg` would have coerced\n // MontyObject::None to the literal string \"None\", silently routing\n // every model=None call to an invalid model ID. Must stay None.\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let _ = handle_llm_query(\n &[],\n &[\n (\n MontyObject::String(\"prompt\".into()),\n MontyObject::String(\"hi\".into()),\n ),\n (MontyObject::String(\"model\".into()), MontyObject::None),\n ],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].0, None);\n }\n\n #[tokio::test]\n async fn llm_query_rejects_non_string_model_kwarg() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query(\n &[],\n &[\n (\n MontyObject::String(\"prompt\".into()),\n MontyObject::String(\"hi\".into()),\n ),\n (MontyObject::String(\"model\".into()), MontyObject::Int(42)),\n ],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Error(_)));\n assert!(llm.calls.lock().await.is_empty());\n }\n\n #[tokio::test]\n async fn llm_query_batched_single_model_none_kwarg_is_no_override() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n let _ = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"model\".into()), MontyObject::None)],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 2);\n assert!(calls.iter().all(|(m, _)| m.is_none()));\n }\n\n #[tokio::test]\n async fn llm_query_batched_honors_positional_context_and_model() {\n // Regression: `context`, `model`, and `models` used to be kwarg-only.\n // A positional call matching the documented signature\n // `llm_query_batched(prompts, context=None, model=None, models=None)`\n // silently dropped the model, violating the preamble.\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query_batched(\n &[\n MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]),\n MontyObject::String(\"shared context\".into()), // position 1: context\n MontyObject::String(\"gpt-4o\".into()), // position 2: model\n ],\n &[],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n match result {\n ExtFunctionResult::Return(MontyObject::List(items)) => assert_eq!(items.len(), 2),\n other => panic!(\"expected list return, got {other:?}\"),\n }\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 2);\n assert!(calls.iter().all(|(m, _)| m.as_deref() == Some(\"gpt-4o\")));\n }\n\n #[tokio::test]\n async fn llm_query_batched_honors_positional_models_list() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query_batched(\n &[\n MontyObject::List(vec![\n MontyObject::String(\"q\".into()),\n MontyObject::String(\"q\".into()),\n ]),\n MontyObject::None, // position 1: context = None\n MontyObject::None, // position 2: model = None\n MontyObject::List(vec![\n // position 3: models\n MontyObject::String(\"gpt-4o\".into()),\n MontyObject::String(\"claude-sonnet-4-6\".into()),\n ]),\n ],\n &[],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Return(_)));\n let mut calls = llm.calls.lock().await;\n calls.sort_by(|a, b| a.0.cmp(&b.0));\n assert_eq!(calls.len(), 2);\n assert_eq!(calls[0].0.as_deref(), Some(\"claude-sonnet-4-6\"));\n assert_eq!(calls[1].0.as_deref(), Some(\"gpt-4o\"));\n }\n\n #[tokio::test]\n async fn llm_query_batched_positional_none_for_models_is_no_override() {\n // `llm_query_batched(prompts, None, None, None)` should run with no\n // model overrides, not error on the positional None at slot 3.\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query_batched(\n &[\n MontyObject::List(vec![MontyObject::String(\"a\".into())]),\n MontyObject::None,\n MontyObject::None,\n MontyObject::None,\n ],\n &[],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Return(_)));\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].0, None);\n }\n\n #[tokio::test]\n async fn llm_query_batched_rejects_non_string_single_model() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![MontyObject::String(\"a\".into())]);\n let result = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"model\".into()), MontyObject::Int(7))],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Error(_)));\n assert!(llm.calls.lock().await.is_empty());\n }\n\n #[tokio::test]\n async fn llm_query_batched_rejects_non_string_models_entries() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n // Integers in the models list should fail loudly, not be coerced to \"1\"/\"2\".\n let models = MontyObject::List(vec![MontyObject::Int(1), MontyObject::Int(2)]);\n let result = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"models\".into()), models)],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Error(_)));\n assert!(llm.calls.lock().await.is_empty());\n }\n\n #[tokio::test]\n async fn llm_query_batched_none_in_models_list_does_not_backfill_from_model_kwarg() {\n // Regression: when `models=[None, \"gpt-4o\"]` and `model=\"claude-...\"`\n // are both passed, the None slot must NOT be backfilled by the\n // singular `model=` kwarg. Each slot is authoritative.\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n let models = MontyObject::List(vec![\n MontyObject::None,\n MontyObject::String(\"gpt-4o\".into()),\n ]);\n let _ = handle_llm_query_batched(\n &[prompts],\n &[\n (MontyObject::String(\"models\".into()), models),\n (\n MontyObject::String(\"model\".into()),\n MontyObject::String(\"claude-sonnet-4-20250514\".into()),\n ),\n ],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 2);\n // Slot 0 was None — must remain None, not become \"claude-sonnet-4-20250514\".\n let slot_a = calls.iter().find(|(_, p)| p == \"a\").expect(\"call for a\");\n let slot_b = calls.iter().find(|(_, p)| p == \"b\").expect(\"call for b\");\n assert_eq!(slot_a.0, None);\n assert_eq!(slot_b.0.as_deref(), Some(\"gpt-4o\"));\n }\n\n #[tokio::test]\n async fn llm_query_batched_models_length_mismatch_errors() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n let models = MontyObject::List(vec![MontyObject::String(\"only-one\".into())]);\n let result = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"models\".into()), models)],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Error(_)));\n assert!(llm.calls.lock().await.is_empty());\n }\n\n // ── Error classification tests ──────────────────────────────\n\n #[test]\n fn classify_syntax_error() {\n let cat = classify_runtime_error(\"SyntaxError: unexpected token\");\n assert_eq!(cat, CodeExecutionFailure::SyntaxError);\n }\n\n #[test]\n fn classify_timeout() {\n let cat = classify_runtime_error(\"execution timed out after 30s\");\n assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\n }\n\n #[test]\n fn classify_memory_limit() {\n let cat = classify_runtime_error(\"memory limit exceeded\");\n assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\n }\n\n #[test]\n fn classify_fuel_exhaustion() {\n let cat = classify_runtime_error(\"fuel exhausted during execution\");\n assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\n }\n\n #[test]\n fn classify_os_denied() {\n let cat = classify_runtime_error(\"OS operations are not permitted in CodeAct scripts\");\n assert_eq!(cat, CodeExecutionFailure::OsDenied);\n }\n\n #[test]\n fn classify_name_error_as_runtime() {\n // NameError from Monty (not NameLookup) is classified as RuntimeError\n let cat = classify_runtime_error(\"NameError: name 'foo' is not defined\");\n assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n }\n\n #[test]\n fn classify_type_error_as_runtime() {\n let cat = classify_runtime_error(\"TypeError: unsupported operand\");\n assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n }\n\n #[test]\n fn classify_module_not_found_as_runtime() {\n let cat = classify_runtime_error(\"ModuleNotFoundError: No module named 'csv'\");\n assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n }\n\n #[test]\n fn classify_syntax_word_is_not_syntaxerror() {\n // \"syntax\" alone should not trigger SyntaxError — only \"syntaxerror\" should.\n let cat = classify_runtime_error(\"unexpected syntax in expression\");\n assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n }\n\n #[test]\n fn vm_panic_variant_serializes_as_snake_case() {\n // VmPanic is set directly by catch_unwind paths, not by classify_runtime_error.\n // Verify it serializes consistently with Display (both snake_case).\n let failure = CodeExecutionFailure::VmPanic;\n assert_eq!(failure.to_string(), \"vm_panic\");\n let json = serde_json::to_value(&failure).unwrap();\n assert_eq!(json, serde_json::json!(\"vm_panic\"));\n }\n\n #[test]\n fn code_hash_deterministic() {\n let h1 = code_hash(\"print('hello')\");\n let h2 = code_hash(\"print('hello')\");\n assert_eq!(h1, h2);\n }\n\n #[test]\n fn code_hash_differs_for_different_code() {\n let h1 = code_hash(\"print('hello')\");\n let h2 = code_hash(\"print('world')\");\n assert_ne!(h1, h2);\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/executor/trace.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "x-ratelimit-used", + "31" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "etag", + "\"563d5e8dd6835053c514ff45fcebbf5e94bf283f\"" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-frame-options", + "deny" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "x-xss-protection", + "0" + ], + [ + "server", + "github.com" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-ratelimit-remaining", + "4969" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "x-github-request-id", + "FB5B:91279:41CD5A:4D2BE4:69DFAEEF" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:51 GMT" + ], + [ + "content-length", + "26791" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ] + ], + "body": "//! Execution trace analysis.\n//!\n//! Builds an in-memory `ExecutionTrace` from a completed `Thread` and runs a\n//! retrospective analyzer that flags common failure patterns. Used by the\n//! self-improvement mission and surfaced in debug logs.\n//!\n//! **There is no separate engine trace file.** Live trace recording for the\n//! whole system is handled by `RecordingLlm` in the host crate\n//! (`src/llm/recording.rs`), gated by `IRONCLAW_RECORD_TRACE`. Because the\n//! engine's `LlmBackend` is wired to the same provider chain, engine LLM\n//! interactions are captured by that single recorder — no engine-side env var\n//! and no second JSON file.\n\nuse chrono::Utc;\nuse serde::Serialize;\nuse tracing::debug;\n\nuse crate::types::event::ThreadEvent;\nuse crate::types::thread::{Thread, ThreadId, ThreadState};\n\n/// A complete execution trace for a single thread.\n#[derive(Debug, Serialize)]\npub struct ExecutionTrace {\n pub thread_id: ThreadId,\n pub goal: String,\n pub final_state: ThreadState,\n pub step_count: usize,\n pub total_tokens: u64,\n pub messages: Vec,\n pub events: Vec,\n pub issues: Vec,\n pub timestamp: chrono::DateTime,\n}\n\n/// A single doc record, for the trace.\n#[derive(Debug, Serialize)]\npub struct DocRecord {\n pub doc_type: String,\n pub title: String,\n pub content: String,\n}\n\n/// A message in the trace with role labeling.\n#[derive(Debug, Serialize)]\npub struct MessageRecord {\n pub role: String,\n pub content_length: usize,\n pub content_preview: String,\n pub full_content: String,\n pub action_name: Option,\n pub action_call_id: Option,\n}\n\n/// An issue detected by the retrospective analyzer.\n#[derive(Debug, Serialize)]\npub struct TraceIssue {\n pub severity: IssueSeverity,\n pub category: String,\n pub description: String,\n pub step: Option,\n}\n\n#[derive(Debug, PartialEq, Serialize)]\npub enum IssueSeverity {\n Error,\n Warning,\n Info,\n}\n\n/// Build a trace from a completed thread.\npub fn build_trace(thread: &Thread) -> ExecutionTrace {\n let messages: Vec = thread\n .messages\n .iter()\n .map(|m| {\n let preview: String = m.content.chars().take(300).collect();\n MessageRecord {\n role: format!(\"{:?}\", m.role),\n content_length: m.content.chars().count(),\n content_preview: if m.content.chars().count() > 300 {\n format!(\"{preview}...\")\n } else {\n preview\n },\n full_content: m.content.clone(),\n action_name: m.action_name.clone(),\n action_call_id: m.action_call_id.clone(),\n }\n })\n .collect();\n\n let issues = analyze_trace(thread);\n\n ExecutionTrace {\n thread_id: thread.id,\n goal: thread.goal.clone(),\n final_state: thread.state,\n step_count: thread.step_count,\n total_tokens: thread.total_tokens_used,\n messages,\n events: thread.events.clone(),\n issues,\n timestamp: Utc::now(),\n }\n}\n\n/// Print a summary of the trace to the log.\npub fn log_trace_summary(trace: &ExecutionTrace) {\n debug!(\n thread_id = %trace.thread_id,\n goal = %trace.goal,\n state = ?trace.final_state,\n steps = trace.step_count,\n tokens = trace.total_tokens,\n messages = trace.messages.len(),\n events = trace.events.len(),\n issues = trace.issues.len(),\n \"=== Engine V2 Trace Summary ===\"\n );\n\n for issue in &trace.issues {\n match issue.severity {\n IssueSeverity::Error => debug!(\n category = %issue.category,\n step = ?issue.step,\n \"ISSUE: {}\",\n issue.description\n ),\n IssueSeverity::Warning => debug!(\n category = %issue.category,\n step = ?issue.step,\n \"WARNING: {}\",\n issue.description\n ),\n IssueSeverity::Info => debug!(\n category = %issue.category,\n step = ?issue.step,\n \"NOTE: {}\",\n issue.description\n ),\n }\n }\n}\n\n// ── Retrospective analysis ──────────────────────────────────\n\n/// Analyze a completed thread for common issues.\nfn analyze_trace(thread: &Thread) -> Vec {\n let mut issues = Vec::new();\n\n // 1. Check if the thread failed\n if thread.state == ThreadState::Failed {\n issues.push(TraceIssue {\n severity: IssueSeverity::Error,\n category: \"thread_failure\".into(),\n description: \"Thread ended in Failed state\".into(),\n step: None,\n });\n }\n\n // 2. Check for empty response (no FINAL, no useful output)\n let has_assistant_response = thread\n .messages\n .iter()\n .any(|m| m.role == crate::types::message::MessageRole::Assistant && !m.content.is_empty());\n if !has_assistant_response {\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"no_response\".into(),\n description: \"No assistant message in thread — model may not have generated output\"\n .into(),\n step: None,\n });\n }\n\n // 3. Check for tool errors\n let tool_errors: Vec<&ThreadEvent> = thread\n .events\n .iter()\n .filter(|e| matches!(e.kind, crate::types::event::EventKind::ActionFailed { .. }))\n .collect();\n if !tool_errors.is_empty() {\n for event in &tool_errors {\n if let crate::types::event::EventKind::ActionFailed {\n action_name, error, ..\n } = &event.kind\n {\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"tool_error\".into(),\n description: format!(\"Tool '{action_name}' failed: {error}\"),\n step: None,\n });\n }\n }\n }\n\n // 4. Check for code execution errors via structured CodeExecutionFailed events.\n // These carry a classified failure category that tells us exactly what kind\n // of error occurred (syntax, runtime, name lookup, VM panic, resource limit,\n // tool error, OS denied, gate pause).\n let code_failures: Vec<&ThreadEvent> = thread\n .events\n .iter()\n .filter(|e| {\n matches!(\n e.kind,\n crate::types::event::EventKind::CodeExecutionFailed { .. }\n )\n })\n .collect();\n for event in &code_failures {\n if let crate::types::event::EventKind::CodeExecutionFailed {\n category, error, ..\n } = &event.kind\n {\n let preview: String = error.chars().take(200).collect();\n let severity = match category {\n crate::types::step::CodeExecutionFailure::VmPanic => IssueSeverity::Error,\n crate::types::step::CodeExecutionFailure::ResourceLimit => IssueSeverity::Error,\n _ => IssueSeverity::Warning,\n };\n issues.push(TraceIssue {\n severity,\n category: format!(\"code_{category}\"),\n description: format!(\"Code execution failed ({category}): {preview}\"),\n step: None,\n });\n }\n }\n\n // Fallback: also check message-level patterns for backward compatibility\n // with threads that ran before the CodeExecutionFailed instrumentation\n // was added (PR #2483). Note: threads from mixed eras (some steps\n // instrumented, some not) will only report structured events when any\n // exist, silently skipping message-level errors from uninstrumented steps.\n if code_failures.is_empty() {\n let error_patterns = [\n \"NameError\",\n \"SyntaxError\",\n \"TypeError\",\n \"NotImplementedError\",\n \"ValueError\",\n \"AttributeError\",\n \"IndexError\",\n \"KeyError\",\n \"ModuleNotFoundError\",\n \"RuntimeError\",\n ];\n for (i, msg) in thread.messages.iter().enumerate() {\n let is_code_output = msg.role == crate::types::message::MessageRole::User\n && (msg.content.starts_with(\"[stdout]\")\n || msg.content.starts_with(\"[stderr]\")\n || msg.content.starts_with(\"[code \")\n || msg.content.starts_with(\"Traceback\"));\n if is_code_output && error_patterns.iter().any(|p| msg.content.contains(p)) {\n let preview: String = msg.content.chars().take(200).collect();\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"code_error\".into(),\n description: format!(\"Code execution error in message {i}: {preview}\"),\n step: None,\n });\n }\n }\n }\n\n // 5. Check for empty call_id on ActionResult messages (causes LLM API rejection).\n for (i, msg) in thread.messages.iter().enumerate() {\n if msg.role == crate::types::message::MessageRole::ActionResult {\n let call_id_empty = msg.action_call_id.as_ref().is_none_or(|id| id.is_empty());\n if call_id_empty {\n let name = msg.action_name.as_deref().unwrap_or(\"unknown\");\n issues.push(TraceIssue {\n severity: IssueSeverity::Error,\n category: \"empty_call_id\".into(),\n description: format!(\n \"ActionResult message {i} (tool '{name}') has empty call_id — will cause LLM API rejection\"\n ),\n step: None,\n });\n }\n }\n }\n\n // 6. Check for model ignoring tool results (hallucination risk).\n // In Tier 0 (structured), results appear as ActionResult messages.\n // In Tier 1 (CodeAct), results appear as User messages with \"[tool result]\" prefixes.\n let has_tool_results = thread\n .messages\n .iter()\n .any(|m| m.role == crate::types::message::MessageRole::ActionResult);\n let has_tool_output_in_context = thread.messages.iter().any(|m| {\n m.role == crate::types::message::MessageRole::User\n && (m.content.contains(\" result]\") || m.content.contains(\" error]\"))\n });\n if has_tool_results && !has_tool_output_in_context {\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"missing_tool_output\".into(),\n description:\n \"Tool results exist but no tool output in messages — model may not see tool results\"\n .into(),\n step: None,\n });\n }\n\n // 7. Check for excessive iterations\n if thread.step_count > 10 {\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"excessive_steps\".into(),\n description: format!(\n \"Thread took {} steps — may be stuck in a loop\",\n thread.step_count\n ),\n step: None,\n });\n }\n\n // 8. Check for text response without FINAL (model answered from memory)\n let text_without_code = thread.events.iter().all(|e| {\n !matches!(\n e.kind,\n crate::types::event::EventKind::ActionExecuted { .. }\n )\n });\n if text_without_code && thread.step_count == 1 && has_assistant_response {\n issues.push(TraceIssue {\n severity: IssueSeverity::Info,\n category: \"no_tools_used\".into(),\n description: \"Model answered in one step without using any tools — may be answering from training data\".into(),\n step: Some(1),\n });\n }\n\n // 9. Check for LLM not producing code blocks\n let code_steps = thread\n .events\n .iter()\n .filter(|e| matches!(e.kind, crate::types::event::EventKind::StepStarted { .. }))\n .count();\n let text_responses_without_code = thread\n .messages\n .iter()\n .filter(|m| {\n m.role == crate::types::message::MessageRole::Assistant\n && !m.content.contains(\"```\")\n && !m.content.contains(\"FINAL(\")\n })\n .count();\n if text_responses_without_code > 0 && code_steps > 0 {\n issues.push(TraceIssue {\n severity: IssueSeverity::Info,\n category: \"mixed_mode\".into(),\n description: format!(\n \"{text_responses_without_code} text response(s) without code blocks — model may not be following CodeAct prompt\"\n ),\n step: None,\n });\n }\n\n // 10. Extract failure reason from StateChanged → Failed events\n for event in &thread.events {\n if let crate::types::event::EventKind::StateChanged {\n to: ThreadState::Failed,\n reason: Some(reason),\n ..\n } = &event.kind\n {\n if reason.contains(\"LLM\") || reason.contains(\"Provider\") {\n issues.push(TraceIssue {\n severity: IssueSeverity::Error,\n category: \"llm_error\".into(),\n description: format!(\"LLM provider error: {}\", truncate(reason, 300)),\n step: None,\n });\n } else if reason.contains(\"orchestrator\") {\n issues.push(TraceIssue {\n severity: IssueSeverity::Error,\n category: \"orchestrator_error\".into(),\n description: format!(\"Orchestrator error: {}\", truncate(reason, 300)),\n step: None,\n });\n }\n }\n }\n\n issues\n}\n\nfn truncate(s: &str, max_chars: usize) -> String {\n let chars: String = s.chars().take(max_chars).collect();\n if s.chars().count() > max_chars {\n format!(\"{chars}...\")\n } else {\n chars\n }\n}\n\n#[cfg(test)]\nmod tests {\n use super::*;\n use crate::types::event::EventKind;\n use crate::types::message::ThreadMessage;\n use crate::types::project::ProjectId;\n use crate::types::step::StepId;\n use crate::types::thread::{ThreadConfig, ThreadType};\n\n fn make_thread() -> Thread {\n Thread::new(\n \"test goal\",\n ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n ThreadConfig::default(),\n )\n }\n\n // ── empty_call_id detection (OpenAI / Codex rejection) ───\n\n /// OpenAI and Codex reject ActionResult messages with empty call_id.\n /// The trace analyzer must flag these as errors.\n #[test]\n fn detects_empty_call_id_on_action_result() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"calling tool\"));\n // Simulate the bug: empty call_id\n thread.add_message(ThreadMessage::action_result(\"\", \"web_search\", \"result\"));\n\n let issues = analyze_trace(&thread);\n let empty_id_issues: Vec<_> = issues\n .iter()\n .filter(|i| i.category == \"empty_call_id\")\n .collect();\n\n assert_eq!(empty_id_issues.len(), 1);\n assert_eq!(empty_id_issues[0].severity, IssueSeverity::Error);\n assert!(empty_id_issues[0].description.contains(\"web_search\"));\n }\n\n /// ActionResult with None call_id should also be flagged.\n #[test]\n fn detects_none_call_id_on_action_result() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"calling tool\"));\n // Manually construct a message with None call_id\n thread.add_message(ThreadMessage {\n role: crate::types::message::MessageRole::ActionResult,\n content: \"result\".into(),\n provenance: crate::types::provenance::Provenance::ToolOutput {\n action_name: \"shell\".into(),\n },\n action_call_id: None,\n action_name: Some(\"shell\".into()),\n action_calls: None,\n timestamp: chrono::Utc::now(),\n });\n\n let issues = analyze_trace(&thread);\n assert!(issues.iter().any(|i| i.category == \"empty_call_id\"));\n }\n\n /// No false positive: valid call_id should not be flagged.\n #[test]\n fn no_false_positive_for_valid_call_id() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"calling tool\"));\n thread.add_message(ThreadMessage::action_result(\n \"call_abc123\",\n \"web_search\",\n \"result\",\n ));\n\n let issues = analyze_trace(&thread);\n assert!(\n !issues.iter().any(|i| i.category == \"empty_call_id\"),\n \"valid call_id should not be flagged\"\n );\n }\n\n // ── tool_error detection ─────────────────────────────────\n\n /// ActionFailed events should produce tool_error warnings.\n #[test]\n fn detects_tool_failures_in_events() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"ok\"));\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::ActionFailed {\n step_id: StepId::new(),\n action_name: \"web_search\".into(),\n call_id: \"call_123\".into(),\n error: \"No lease for action 'web_search'\".into(),\n params_summary: None,\n },\n ));\n\n let issues = analyze_trace(&thread);\n let tool_errors: Vec<_> = issues\n .iter()\n .filter(|i| i.category == \"tool_error\")\n .collect();\n assert_eq!(tool_errors.len(), 1);\n assert!(tool_errors[0].description.contains(\"web_search\"));\n }\n\n // ── thread_failure detection ─────────────────────────────\n\n #[test]\n fn detects_failed_thread_state() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"trying\"));\n thread.state = ThreadState::Failed;\n\n let issues = analyze_trace(&thread);\n assert!(issues.iter().any(|i| i.category == \"thread_failure\"));\n }\n\n // ── LLM error detection from StateChanged events ─────────\n\n /// Reproduces the exact pattern from the trace: OpenAI rejects empty call_id.\n #[test]\n fn detects_llm_error_from_state_changed() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"ok\"));\n thread.state = ThreadState::Failed;\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::StateChanged {\n from: ThreadState::Running,\n to: ThreadState::Failed,\n reason: Some(\n \"LLM error: Provider openai_codex request failed: HTTP 400 Bad Request: \\\n Invalid 'input[5].call_id': empty string\"\n .into(),\n ),\n },\n ));\n\n let issues = analyze_trace(&thread);\n assert!(\n issues.iter().any(|i| i.category == \"llm_error\"),\n \"should detect LLM provider error in StateChanged reason\"\n );\n }\n\n // ── Multiple empty call_ids ──────────────────────────────\n\n /// Anthropic sends consecutive tool results merged into one User message.\n /// If multiple ActionResults have empty call_ids, each must be flagged.\n #[test]\n fn flags_each_empty_call_id_separately() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"parallel calls\"));\n thread.add_message(ThreadMessage::action_result(\"\", \"tool_a\", \"result_a\"));\n thread.add_message(ThreadMessage::action_result(\"\", \"tool_b\", \"result_b\"));\n thread.add_message(ThreadMessage::action_result(\n \"call_ok\", \"tool_c\", \"result_c\",\n ));\n\n let issues = analyze_trace(&thread);\n let empty_issues: Vec<_> = issues\n .iter()\n .filter(|i| i.category == \"empty_call_id\")\n .collect();\n assert_eq!(\n empty_issues.len(),\n 2,\n \"should flag exactly the 2 empty call_ids\"\n );\n }\n\n // ── CodeExecutionFailed event detection ────────────────────\n\n #[test]\n fn detects_code_execution_failure_from_event() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"```repl\\nimport csv\\n```\"));\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::CodeExecutionFailed {\n step_id: StepId::new(),\n category: crate::types::step::CodeExecutionFailure::RuntimeError,\n error: \"ModuleNotFoundError: No module named 'csv'\".into(),\n code_hash: Some(\"abc123\".into()),\n duration_ms: 42,\n },\n ));\n\n let issues = analyze_trace(&thread);\n let code_issues: Vec<_> = issues\n .iter()\n .filter(|i| i.category.starts_with(\"code_\"))\n .collect();\n assert_eq!(code_issues.len(), 1);\n assert_eq!(code_issues[0].category, \"code_runtime_error\");\n assert_eq!(code_issues[0].severity, IssueSeverity::Warning);\n assert!(code_issues[0].description.contains(\"ModuleNotFoundError\"));\n }\n\n #[test]\n fn vm_panic_is_error_severity() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"code\"));\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::CodeExecutionFailed {\n step_id: StepId::new(),\n category: crate::types::step::CodeExecutionFailure::VmPanic,\n error: \"Monty panicked: unreachable\".into(),\n code_hash: None,\n duration_ms: 0,\n },\n ));\n\n let issues = analyze_trace(&thread);\n let panic_issues: Vec<_> = issues\n .iter()\n .filter(|i| i.category == \"code_vm_panic\")\n .collect();\n assert_eq!(panic_issues.len(), 1);\n assert_eq!(panic_issues[0].severity, IssueSeverity::Error);\n }\n\n #[test]\n fn fallback_message_detection_when_no_events() {\n // Threads from before instrumentation should still be detected\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"code\"));\n thread.add_message(ThreadMessage::user(\n \"[stdout]\\nNameError: name 'foo' is not defined\",\n ));\n\n let issues = analyze_trace(&thread);\n assert!(\n issues.iter().any(|i| i.category == \"code_error\"),\n \"should detect code error from message when no CodeExecutionFailed events exist\"\n );\n }\n\n #[test]\n fn trace_serializes_approval_request_payload() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"installing notion\"));\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::ApprovalRequested {\n action_name: \"tool_install\".into(),\n call_id: \"call_install_1\".into(),\n parameters: Some(serde_json::json!({\"name\": \"notion\", \"kind\": \"mcp_server\"})),\n description: Some(\"Install an extension\".into()),\n allow_always: Some(true),\n gate_name: Some(\"approval\".into()),\n params_summary: Some(\"notion\".into()),\n },\n ));\n\n let trace = build_trace(&thread);\n // `Thread::add_message` records a `MessageAdded` event for each\n // message, so the `ApprovalRequested` event is no longer at index 0\n // — it's mixed in with the message events. Find it by kind.\n let approval = trace\n .events\n .iter()\n .find(|e| matches!(&e.kind, EventKind::ApprovalRequested { .. }))\n .expect(\"trace should contain an ApprovalRequested event\");\n match &approval.kind {\n EventKind::ApprovalRequested {\n action_name,\n call_id,\n parameters,\n description,\n allow_always,\n gate_name,\n params_summary,\n } => {\n assert_eq!(action_name, \"tool_install\");\n assert_eq!(call_id, \"call_install_1\");\n assert_eq!(\n parameters.as_ref().and_then(|p| p.get(\"name\")),\n Some(&serde_json::json!(\"notion\"))\n );\n assert_eq!(description.as_deref(), Some(\"Install an extension\"));\n assert_eq!(*allow_always, Some(true));\n assert_eq!(gate_name.as_deref(), Some(\"approval\"));\n assert_eq!(params_summary.as_deref(), Some(\"notion\"));\n }\n other => panic!(\"unexpected event kind: {other:?}\"),\n }\n\n let json = serde_json::to_string(&trace).expect(\"trace serializes\");\n assert!(json.contains(\"\\\"ApprovalRequested\\\"\"));\n assert!(json.contains(\"\\\"action_name\\\":\\\"tool_install\\\"\"));\n assert!(json.contains(\"\\\"call_id\\\":\\\"call_install_1\\\"\"));\n // Parameter map key order isn't stable across serde_json versions; check\n // both required keys are present rather than the exact serialized form.\n assert!(json.contains(\"\\\"name\\\":\\\"notion\\\"\"));\n assert!(json.contains(\"\\\"kind\\\":\\\"mcp_server\\\"\"));\n assert!(json.contains(\"\\\"description\\\":\\\"Install an extension\\\"\"));\n assert!(json.contains(\"\\\"allow_always\\\":true\"));\n assert!(json.contains(\"\\\"gate_name\\\":\\\"approval\\\"\"));\n assert!(json.contains(\"\\\"params_summary\\\":\\\"notion\\\"\"));\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/lib.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "x-ratelimit-remaining", + "4968" + ], + [ + "x-github-request-id", + "FB7C:16ECB3:441CDE:4F7C44:69DFAEEF" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "server", + "github.com" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-frame-options", + "deny" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "content-length", + "19797" + ], + [ + "x-xss-protection", + "0" + ], + [ + "x-ratelimit-used", + "32" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:51 GMT" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "etag", + "\"9ecaa6b575a369535e657b5f922187d3ce5149fa\"" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ] + ], + "body": "//! IronClaw Engine — unified thread-capability-CodeAct execution model.\n//!\n//! This crate provides the core execution engine for IronClaw, unifying\n//! ~10 separate abstractions (Session, Job, Routine, Channel, Tool, Skill,\n//! Hook, Observer, Extension, LoopDelegate) around 5 primitives:\n//!\n//! - **Thread** — unit of work (replaces Session + Job + Routine + Sub-agent)\n//! - **Step** — unit of execution (replaces agentic loop iteration + tool calls)\n//! - **Capability** — unit of effect (replaces Tool + Skill + Hook + Extension)\n//! - **MemoryDoc** — unit of durable knowledge (replaces workspace memory blobs)\n//! - **Project** — unit of context (replaces flat workspace namespace)\n//!\n//! The engine defines traits for external dependencies ([`LlmBackend`],\n//! [`Store`], [`EffectExecutor`]) that the host crate implements via bridge\n//! adapters over existing infrastructure.\n\n// Security: `__regex_match__` (in `executor/orchestrator.rs`) accepts\n// arbitrary patterns from the Python orchestrator and runs them on\n// user-supplied text. The default `regex` crate is linear-time. The\n// `fancy-regex` crate supports backreferences and is NOT linear-time, which\n// would turn that handler into a ReDoS vector. Cargo.toml depends on\n// `regex = \"1\"` with default features only — do NOT add `fancy-regex` to\n// this crate's dependency tree without first redesigning `__regex_match__`\n// to enforce a wall-clock matching budget.\n\npub mod capability;\npub mod executor;\npub mod gate;\npub mod memory;\npub mod reliability;\npub mod runtime;\npub mod traits;\npub mod types;\n\n// ── Re-exports: types ───────────────────────────────────────\n\npub use types::capability::{\n ActionDef, Capability, CapabilityLease, EffectType, GrantedActions, LeaseId, PolicyCondition,\n PolicyEffect, PolicyRule,\n};\npub use types::error::{CapabilityError, EngineError, StepError, ThreadError};\npub use types::event::{EventId, EventKind, ThreadEvent};\npub use types::memory::{DocId, DocType, MemoryDoc};\npub use types::message::{MessageRole, ThreadMessage};\npub use types::mission::{Mission, MissionCadence, MissionId, MissionStatus, ValidTimezone};\npub use types::project::{Project, ProjectId};\npub use types::provenance::Provenance;\npub use types::step::{\n ActionCall, ActionResult, CodeExecutionFailure, ExecutionTier, LlmResponse, Step, StepId,\n StepStatus, TokenUsage,\n};\npub use types::thread::{\n ActiveSkillProvenance, Thread, ThreadConfig, ThreadId, ThreadState, ThreadType,\n};\n\n// ── Re-exports: traits ──────────────────────────────────────\n\npub use traits::effect::{EffectExecutor, ThreadExecutionContext};\npub use traits::llm::{LlmBackend, LlmCallConfig, LlmOutput};\npub use traits::store::Store;\npub use traits::workspace::WorkspaceReader;\n\n// ── Re-exports: capability ────────────────────────────────────\n\npub use capability::lease::LeaseManager;\npub use capability::planner::{CapabilityGrantPlan, LeasePlanner};\npub use capability::policy::{PolicyDecision, PolicyEngine};\npub use capability::registry::CapabilityRegistry;\n\n// ── Re-exports: gate ─────────────────────────────────────────\n\npub use gate::lease::LeaseGate;\npub use gate::pipeline::GatePipeline;\npub use gate::tool_tier::{ToolTier, classify_tool_tier};\npub use gate::{\n ExecutionGate, ExecutionMode, GateContext, GateDecision, GateResolution, ResumeKind,\n};\n\n// ── Re-exports: runtime ───────────────────────────────────────\n\npub use executor::prompt::PlatformInfo;\npub use runtime::conversation::ConversationManager;\npub use runtime::manager::ThreadManager;\npub use runtime::messaging::ThreadOutcome;\npub use runtime::mission::{\n BudgetGate, FireRateLimit, MissionManager, MissionNotification, MissionUpdate,\n};\npub use runtime::tree::ThreadTree;\n\npub use types::conversation::{\n ConversationEntry, ConversationId, ConversationSurface, EntrySender,\n};\n\n// ── Re-exports: executor ──────────────────────────────────────\n\npub use executor::ExecutionLoop;\n\n// ── Re-exports: memory ────────────────────────────────────────\n\npub use memory::MemoryStore;\npub use memory::RetrievalEngine;\n\n// ── Re-exports: reliability ──────────────────────────────────\n\npub use reliability::ReliabilityTracker;\n\n// ── Test utilities ──────────────────────────────────────────\n\n#[cfg(test)]\npub(crate) mod tests {\n use tokio::sync::RwLock;\n\n use crate::traits::store::Store;\n use crate::types::capability::{CapabilityLease, LeaseId};\n use crate::types::conversation::{ConversationId, ConversationSurface};\n use crate::types::error::EngineError;\n use crate::types::event::ThreadEvent;\n use crate::types::memory::{DocId, MemoryDoc};\n use crate::types::mission::{Mission, MissionId, MissionStatus};\n use crate::types::project::{Project, ProjectId};\n use crate::types::step::Step;\n use crate::types::thread::{Thread, ThreadId, ThreadState};\n\n /// Shared in-memory Store implementation for tests.\n ///\n /// Stores all entity types with proper CRUD semantics and filtering by\n /// project_id / user_id. Use this instead of defining per-module mocks.\n pub struct InMemoryStore {\n threads: RwLock>,\n steps: RwLock>,\n events: RwLock>,\n projects: RwLock>,\n conversations: RwLock>,\n docs: RwLock>,\n leases: RwLock>,\n missions: RwLock>,\n }\n\n impl InMemoryStore {\n pub fn new() -> Self {\n Self {\n threads: RwLock::new(Vec::new()),\n steps: RwLock::new(Vec::new()),\n events: RwLock::new(Vec::new()),\n projects: RwLock::new(Vec::new()),\n conversations: RwLock::new(Vec::new()),\n docs: RwLock::new(Vec::new()),\n leases: RwLock::new(Vec::new()),\n missions: RwLock::new(Vec::new()),\n }\n }\n\n pub fn with_docs(docs: Vec) -> Self {\n Self {\n docs: RwLock::new(docs),\n ..Self::new()\n }\n }\n }\n\n #[async_trait::async_trait]\n impl Store for InMemoryStore {\n async fn save_thread(&self, thread: &Thread) -> Result<(), EngineError> {\n let mut threads = self.threads.write().await;\n threads.retain(|t| t.id != thread.id);\n threads.push(thread.clone());\n Ok(())\n }\n async fn load_thread(&self, id: ThreadId) -> Result, EngineError> {\n Ok(self\n .threads\n .read()\n .await\n .iter()\n .find(|t| t.id == id)\n .cloned())\n }\n async fn list_threads(\n &self,\n project_id: ProjectId,\n user_id: &str,\n ) -> Result, EngineError> {\n Ok(self\n .threads\n .read()\n .await\n .iter()\n .filter(|t| t.project_id == project_id && t.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn update_thread_state(\n &self,\n id: ThreadId,\n state: ThreadState,\n ) -> Result<(), EngineError> {\n let mut threads = self.threads.write().await;\n if let Some(t) = threads.iter_mut().find(|t| t.id == id) {\n t.state = state;\n }\n Ok(())\n }\n async fn save_step(&self, step: &Step) -> Result<(), EngineError> {\n let mut steps = self.steps.write().await;\n steps.retain(|s| s.id != step.id);\n steps.push(step.clone());\n Ok(())\n }\n async fn load_steps(&self, thread_id: ThreadId) -> Result, EngineError> {\n Ok(self\n .steps\n .read()\n .await\n .iter()\n .filter(|s| s.thread_id == thread_id)\n .cloned()\n .collect())\n }\n async fn append_events(&self, events: &[ThreadEvent]) -> Result<(), EngineError> {\n self.events.write().await.extend(events.iter().cloned());\n Ok(())\n }\n async fn load_events(&self, thread_id: ThreadId) -> Result, EngineError> {\n Ok(self\n .events\n .read()\n .await\n .iter()\n .filter(|e| e.thread_id == thread_id)\n .cloned()\n .collect())\n }\n async fn save_project(&self, project: &Project) -> Result<(), EngineError> {\n let mut projects = self.projects.write().await;\n projects.retain(|p| p.id != project.id);\n projects.push(project.clone());\n Ok(())\n }\n async fn load_project(&self, id: ProjectId) -> Result, EngineError> {\n Ok(self\n .projects\n .read()\n .await\n .iter()\n .find(|p| p.id == id)\n .cloned())\n }\n async fn list_projects(&self, user_id: &str) -> Result, EngineError> {\n Ok(self\n .projects\n .read()\n .await\n .iter()\n .filter(|p| p.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn list_all_projects(&self) -> Result, EngineError> {\n Ok(self.projects.read().await.iter().cloned().collect())\n }\n async fn save_conversation(\n &self,\n conversation: &ConversationSurface,\n ) -> Result<(), EngineError> {\n let mut conversations = self.conversations.write().await;\n conversations.retain(|c| c.id != conversation.id);\n conversations.push(conversation.clone());\n Ok(())\n }\n async fn load_conversation(\n &self,\n id: ConversationId,\n ) -> Result, EngineError> {\n Ok(self\n .conversations\n .read()\n .await\n .iter()\n .find(|c| c.id == id)\n .cloned())\n }\n async fn list_conversations(\n &self,\n user_id: &str,\n ) -> Result, EngineError> {\n Ok(self\n .conversations\n .read()\n .await\n .iter()\n .filter(|c| c.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn save_memory_doc(&self, doc: &MemoryDoc) -> Result<(), EngineError> {\n let mut docs = self.docs.write().await;\n docs.retain(|d| d.id != doc.id);\n docs.push(doc.clone());\n Ok(())\n }\n async fn load_memory_doc(&self, id: DocId) -> Result, EngineError> {\n Ok(self.docs.read().await.iter().find(|d| d.id == id).cloned())\n }\n async fn list_memory_docs(\n &self,\n project_id: ProjectId,\n user_id: &str,\n ) -> Result, EngineError> {\n Ok(self\n .docs\n .read()\n .await\n .iter()\n .filter(|d| d.project_id == project_id && d.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn save_lease(&self, lease: &CapabilityLease) -> Result<(), EngineError> {\n let mut leases = self.leases.write().await;\n leases.retain(|l| l.id != lease.id);\n leases.push(lease.clone());\n Ok(())\n }\n async fn load_active_leases(\n &self,\n thread_id: ThreadId,\n ) -> Result, EngineError> {\n Ok(self\n .leases\n .read()\n .await\n .iter()\n .filter(|l| l.thread_id == thread_id && !l.revoked)\n .cloned()\n .collect())\n }\n async fn revoke_lease(&self, lease_id: LeaseId, _reason: &str) -> Result<(), EngineError> {\n let mut leases = self.leases.write().await;\n if let Some(l) = leases.iter_mut().find(|l| l.id == lease_id) {\n l.revoked = true;\n }\n Ok(())\n }\n async fn save_mission(&self, mission: &Mission) -> Result<(), EngineError> {\n let mut missions = self.missions.write().await;\n missions.retain(|m| m.id != mission.id);\n missions.push(mission.clone());\n Ok(())\n }\n async fn load_mission(&self, id: MissionId) -> Result, EngineError> {\n Ok(self\n .missions\n .read()\n .await\n .iter()\n .find(|m| m.id == id)\n .cloned())\n }\n async fn list_missions(\n &self,\n project_id: ProjectId,\n user_id: &str,\n ) -> Result, EngineError> {\n Ok(self\n .missions\n .read()\n .await\n .iter()\n .filter(|m| m.project_id == project_id && m.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn update_mission_status(\n &self,\n id: MissionId,\n status: MissionStatus,\n ) -> Result<(), EngineError> {\n let mut missions = self.missions.write().await;\n if let Some(m) = missions.iter_mut().find(|m| m.id == id) {\n m.status = status;\n }\n Ok(())\n }\n async fn list_all_threads(\n &self,\n project_id: ProjectId,\n ) -> Result, EngineError> {\n Ok(self\n .threads\n .read()\n .await\n .iter()\n .filter(|t| t.project_id == project_id)\n .cloned()\n .collect())\n }\n async fn list_all_missions(\n &self,\n project_id: ProjectId,\n ) -> Result, EngineError> {\n Ok(self\n .missions\n .read()\n .await\n .iter()\n .filter(|m| m.project_id == project_id)\n .cloned()\n .collect())\n }\n }\n\n struct MinimalStore;\n\n #[async_trait::async_trait]\n impl Store for MinimalStore {\n async fn save_thread(&self, _thread: &Thread) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_thread(&self, _id: ThreadId) -> Result, EngineError> {\n Ok(None)\n }\n async fn list_threads(\n &self,\n _project_id: ProjectId,\n _user_id: &str,\n ) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn update_thread_state(\n &self,\n _id: ThreadId,\n _state: ThreadState,\n ) -> Result<(), EngineError> {\n Ok(())\n }\n async fn save_step(&self, _step: &Step) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_steps(&self, _thread_id: ThreadId) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn append_events(&self, _events: &[ThreadEvent]) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_events(&self, _thread_id: ThreadId) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn save_project(&self, _project: &Project) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_project(&self, _id: ProjectId) -> Result, EngineError> {\n Ok(None)\n }\n async fn save_memory_doc(&self, _doc: &MemoryDoc) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_memory_doc(&self, _id: DocId) -> Result, EngineError> {\n Ok(None)\n }\n async fn list_memory_docs(\n &self,\n _project_id: ProjectId,\n _user_id: &str,\n ) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn save_lease(&self, _lease: &CapabilityLease) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_active_leases(\n &self,\n _thread_id: ThreadId,\n ) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn revoke_lease(&self, _lease_id: LeaseId, _reason: &str) -> Result<(), EngineError> {\n Ok(())\n }\n async fn save_mission(&self, _mission: &Mission) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_mission(&self, _id: MissionId) -> Result, EngineError> {\n Ok(None)\n }\n async fn list_missions(\n &self,\n _project_id: ProjectId,\n _user_id: &str,\n ) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn update_mission_status(\n &self,\n _id: MissionId,\n _status: MissionStatus,\n ) -> Result<(), EngineError> {\n Ok(())\n }\n }\n\n #[tokio::test]\n async fn store_defaults_fail_closed() {\n let store = MinimalStore;\n assert!(matches!(\n store.list_projects(\"alice\").await,\n Err(EngineError::Store { .. })\n ));\n assert!(matches!(\n store.list_all_projects().await,\n Err(EngineError::Store { .. })\n ));\n assert!(matches!(\n store.load_conversation(ConversationId::new()).await,\n Err(EngineError::Store { .. })\n ));\n assert!(matches!(\n store.list_all_threads(ProjectId::new()).await,\n Err(EngineError::Store { .. })\n ));\n }\n\n #[tokio::test]\n async fn shared_queries_include_legacy_and_current_shared_owner() {\n use crate::types::memory::DocType;\n use crate::types::{LEGACY_SHARED_OWNER_ID, shared_owner_id};\n\n let project_id = ProjectId::new();\n let mut legacy = MemoryDoc::new(\n project_id,\n LEGACY_SHARED_OWNER_ID,\n DocType::Note,\n \"legacy\",\n \"a\",\n );\n let current = MemoryDoc::new(project_id, shared_owner_id(), DocType::Note, \"current\", \"b\");\n legacy.id = DocId::new();\n let store = InMemoryStore::with_docs(vec![legacy, current]);\n\n let docs = store\n .list_memory_docs_with_shared(project_id, \"alice\")\n .await\n .unwrap();\n assert_eq!(docs.len(), 2);\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/types/event.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "x-ratelimit-remaining", + "4967" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:52 GMT" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "server", + "github.com" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-ratelimit-used", + "33" + ], + [ + "x-github-request-id", + "FB91:17B00F:4740F3:52A42C:69DFAEF0" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-xss-protection", + "0" + ], + [ + "etag", + "\"b61185f07131d737272f1068c899f7f6b5821960\"" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "content-length", + "8364" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "x-frame-options", + "deny" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-oauth-scopes", + "repo" + ] + ], + "body": "//! Event sourcing types.\n//!\n//! Every significant action within a thread is recorded as an event.\n//! This enables replay, debugging, reflection, and trace-based testing.\n\nuse chrono::{DateTime, Utc};\nuse serde::{Deserialize, Serialize};\nuse uuid::Uuid;\n\nuse crate::types::capability::LeaseId;\n\n/// Generate a short human-readable summary of tool parameters for display.\n///\n/// For `http`: shows the URL. For `web_search`: shows the query.\n/// For other tools: shows the first string argument, truncated.\n/// Returns `None` for empty or unrecognizable params.\npub fn summarize_params(action_name: &str, params: &serde_json::Value) -> Option {\n let summary = match action_name {\n \"http\" | \"web_fetch\" => params\n .get(\"url\")\n .and_then(|v| v.as_str())\n .map(|u| truncate(u, 80)),\n \"web_search\" | \"llm_context\" => params\n .get(\"query\")\n .and_then(|v| v.as_str())\n .map(|q| truncate(q, 60)),\n \"memory_search\" => params\n .get(\"query\")\n .and_then(|v| v.as_str())\n .map(|q| truncate(q, 60)),\n \"memory_write\" => params\n .get(\"target\")\n .and_then(|v| v.as_str())\n .map(|t| t.to_string()),\n \"memory_read\" => params\n .get(\"path\")\n .and_then(|v| v.as_str())\n .map(|p| p.to_string()),\n \"shell\" => params\n .get(\"command\")\n .and_then(|v| v.as_str())\n .map(|c| truncate(c, 60)),\n \"message\" => params\n .get(\"content\")\n .and_then(|v| v.as_str())\n .map(|c| truncate(c, 40)),\n _ => {\n // Generic: show first string value\n if let Some(obj) = params.as_object() {\n obj.values()\n .find_map(|v| v.as_str())\n .map(|s| truncate(s, 50))\n } else {\n None\n }\n }\n };\n summary.filter(|s| !s.is_empty())\n}\n\nfn truncate(s: &str, max: usize) -> String {\n if s.len() <= max {\n s.to_string()\n } else {\n // Find a safe UTF-8 boundary\n let mut end = max.min(s.len());\n while end > 0 && !s.is_char_boundary(end) {\n end -= 1;\n }\n format!(\"{}...\", &s[..end]) // safety: end is validated by is_char_boundary loop above\n }\n}\nuse crate::types::step::{StepId, TokenUsage};\nuse crate::types::thread::{ThreadId, ThreadState};\n\n/// Strongly-typed event identifier.\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]\npub struct EventId(pub Uuid);\n\nimpl EventId {\n pub fn new() -> Self {\n Self(Uuid::new_v4())\n }\n}\n\nimpl Default for EventId {\n fn default() -> Self {\n Self::new()\n }\n}\n\n/// A recorded event in a thread's execution history.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct ThreadEvent {\n pub id: EventId,\n pub thread_id: ThreadId,\n pub timestamp: DateTime,\n pub kind: EventKind,\n}\n\nimpl ThreadEvent {\n pub fn new(thread_id: ThreadId, kind: EventKind) -> Self {\n Self {\n id: EventId::new(),\n thread_id,\n timestamp: Utc::now(),\n kind,\n }\n }\n}\n\n/// The specific kind of event that occurred.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub enum EventKind {\n // ── Thread lifecycle ────────────────────────────────────\n StateChanged {\n from: ThreadState,\n to: ThreadState,\n reason: Option,\n },\n\n // ── Step lifecycle ──────────────────────────────────────\n StepStarted {\n step_id: StepId,\n },\n StepCompleted {\n step_id: StepId,\n tokens: TokenUsage,\n },\n StepFailed {\n step_id: StepId,\n error: String,\n },\n\n // ── Action execution ────────────────────────────────────\n ActionExecuted {\n step_id: StepId,\n action_name: String,\n call_id: String,\n duration_ms: u64,\n /// Short human-readable summary of parameters (e.g., URL for http tool).\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n params_summary: Option,\n },\n ActionFailed {\n step_id: StepId,\n action_name: String,\n call_id: String,\n error: String,\n /// Short human-readable summary of parameters.\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n params_summary: Option,\n },\n\n // ── Capability leases ───────────────────────────────────\n LeaseGranted {\n lease_id: LeaseId,\n capability_name: String,\n },\n LeaseRevoked {\n lease_id: LeaseId,\n reason: String,\n },\n LeaseExpired {\n lease_id: LeaseId,\n },\n\n // ── Messages ────────────────────────────────────────────\n MessageAdded {\n role: String,\n content_preview: String,\n },\n\n // ── Thread tree ─────────────────────────────────────────\n ChildSpawned {\n child_id: ThreadId,\n goal: String,\n },\n ChildCompleted {\n child_id: ThreadId,\n },\n\n // ── Approval flow ───────────────────────────────────────\n ApprovalRequested {\n action_name: String,\n call_id: String,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n parameters: Option,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n description: Option,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n allow_always: Option,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n gate_name: Option,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n params_summary: Option,\n },\n ApprovalReceived {\n call_id: String,\n approved: bool,\n },\n\n // ── Self-improvement ──────────────────────────────────────\n SelfImprovementStarted,\n SelfImprovementComplete {\n prompt_updated: bool,\n patterns_added: usize,\n },\n SelfImprovementFailed {\n error: String,\n },\n\n // ── Skill activation ───────────────────────────────────────\n SkillActivated {\n skill_names: Vec,\n },\n\n // ── Code execution instrumentation ────────────────────────\n /// Emitted when a code (REPL) execution attempt fails. Enables aggregate\n /// analysis of code execution failure modes to determine whether the\n /// runtime (Monty), the LLM, or tool dispatch is the primary source of\n /// failures.\n CodeExecutionFailed {\n step_id: StepId,\n /// Classified failure category.\n category: crate::types::step::CodeExecutionFailure,\n /// The error message text (truncated to 500 chars).\n error: String,\n /// Hash of the Python code that was executed, for dedup/correlation.\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n code_hash: Option,\n /// Duration of the code execution attempt in milliseconds.\n #[serde(default)]\n duration_ms: u64,\n },\n\n // ── Orchestrator versioning ───────────────────────────────\n OrchestratorRollback {\n from_version: u64,\n to_version: u64,\n reason: String,\n },\n\n /// Unknown event kind — catch-all for forward compatibility during\n /// rolling deploys. Older binaries deserializing events written by\n /// newer binaries will produce this variant instead of failing.\n #[serde(other)]\n Unknown,\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/types/step.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-frame-options", + "deny" + ], + [ + "x-github-request-id", + "FBA9:16ECB3:441D6B:4F7CF8:69DFAEF0" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "etag", + "\"161096e9b02238d78d86f611befa72d32fbe0a76\"" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "x-ratelimit-remaining", + "4966" + ], + [ + "server", + "github.com" + ], + [ + "date", + "Wed, 15 Apr 2026 15:29:52 GMT" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-xss-protection", + "0" + ], + [ + "content-length", + "6490" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "x-ratelimit-used", + "34" + ] + ], + "body": "//! Step — the unit of execution within a thread.\n//!\n//! Each step corresponds to one LLM call plus its subsequent action\n//! executions. This replaces the implicit \"iteration\" counter in the\n//! existing `run_agentic_loop`.\n\nuse std::time::Duration;\n\nuse chrono::{DateTime, Utc};\nuse serde::{Deserialize, Serialize};\nuse uuid::Uuid;\n\nuse crate::types::thread::ThreadId;\n\n/// Strongly-typed step identifier.\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]\npub struct StepId(pub Uuid);\n\nimpl StepId {\n pub fn new() -> Self {\n Self(Uuid::new_v4())\n }\n}\n\nimpl Default for StepId {\n fn default() -> Self {\n Self::new()\n }\n}\n\n/// Status of a step within its lifecycle.\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]\npub enum StepStatus {\n Pending,\n LlmCalling,\n Executing,\n Completed,\n Failed,\n}\n\n/// Which execution tier handles the step's code/actions.\n///\n/// Monty is the sole CodeAct/RLM executor. WASM and Docker are used for\n/// third-party tool isolation and thread sandboxing (Phase 8), not for\n/// running LLM-generated Python.\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]\npub enum ExecutionTier {\n /// Structured tool calls (JSON action calls from LLM).\n Structured,\n /// Embedded Python via Monty (CodeAct/RLM pattern).\n Scripting,\n}\n\n/// A single execution step within a thread.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct Step {\n pub id: StepId,\n pub thread_id: ThreadId,\n /// 1-indexed sequence within the thread.\n pub sequence: usize,\n pub status: StepStatus,\n pub tier: ExecutionTier,\n pub llm_response: Option,\n pub action_results: Vec,\n pub tokens_used: TokenUsage,\n pub started_at: DateTime,\n pub completed_at: Option>,\n}\n\nimpl Step {\n pub fn new(thread_id: ThreadId, sequence: usize) -> Self {\n Self {\n id: StepId::new(),\n thread_id,\n sequence,\n status: StepStatus::Pending,\n tier: ExecutionTier::Structured,\n llm_response: None,\n action_results: Vec::new(),\n tokens_used: TokenUsage::default(),\n started_at: Utc::now(),\n completed_at: None,\n }\n }\n}\n\n// ── LLM response types ─────────────────────────────────────\n\n/// Response from the LLM: text, action calls, or executable code.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub enum LlmResponse {\n /// Final text response.\n Text(String),\n /// One or more action calls (with optional reasoning text).\n ActionCalls {\n calls: Vec,\n content: Option,\n },\n /// Executable Python code (CodeAct). Tool calls happen as function\n /// calls within the code; the runtime suspends at each one and\n /// delegates to the EffectExecutor.\n Code {\n code: String,\n content: Option,\n },\n}\n\n/// A request from the LLM to execute a capability action.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct ActionCall {\n /// Unique call identifier (echoed in the result).\n pub id: String,\n /// Action name (e.g. \"web_fetch\", \"create_issue\").\n pub action_name: String,\n /// Action parameters as JSON.\n pub parameters: serde_json::Value,\n}\n\n/// Result of executing a capability action.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct ActionResult {\n /// The call ID this result corresponds to.\n pub call_id: String,\n /// The action that was executed.\n pub action_name: String,\n /// Output value.\n pub output: serde_json::Value,\n /// Whether this result represents an error.\n pub is_error: bool,\n /// How long the action took.\n #[serde(with = \"duration_millis\")]\n pub duration: Duration,\n}\n\n/// Classification of code execution failures.\n///\n/// Used by the instrumentation layer to distinguish Monty VM limitations\n/// from LLM logic errors, tool dispatch failures, and resource exhaustion.\n/// This data enables informed decisions about runtime alternatives.\n#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]\n#[serde(rename_all = \"snake_case\")]\npub enum CodeExecutionFailure {\n /// Python parse error — LLM generated invalid syntax.\n SyntaxError,\n /// Python runtime error (NameError, TypeError, ValueError, etc.) —\n /// LLM logic bug or use of unsupported feature.\n RuntimeError,\n /// Name lookup failed — function/variable not in scope and not a known tool.\n NameLookup,\n /// Monty VM panicked (catch_unwind caught it). Indicates a Monty bug,\n /// not a user code issue.\n VmPanic,\n /// Resource limit hit (timeout, memory, or allocation cap).\n ResourceLimit,\n /// A tool call inside code returned an error.\n ToolError,\n /// OS operation attempted (blocked by sandbox).\n OsDenied,\n}\n\nimpl std::fmt::Display for CodeExecutionFailure {\n fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {\n match self {\n Self::SyntaxError => write!(f, \"syntax_error\"),\n Self::RuntimeError => write!(f, \"runtime_error\"),\n Self::NameLookup => write!(f, \"name_lookup\"),\n Self::VmPanic => write!(f, \"vm_panic\"),\n Self::ResourceLimit => write!(f, \"resource_limit\"),\n Self::ToolError => write!(f, \"tool_error\"),\n Self::OsDenied => write!(f, \"os_denied\"),\n }\n }\n}\n\n/// Token usage for a single LLM call.\n#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize)]\npub struct TokenUsage {\n pub input_tokens: u64,\n pub output_tokens: u64,\n pub cache_read_tokens: u64,\n pub cache_write_tokens: u64,\n /// USD cost for this call (populated by LlmBackend if cost data is available).\n pub cost_usd: f64,\n}\n\nimpl TokenUsage {\n pub fn total(&self) -> u64 {\n self.input_tokens + self.output_tokens\n }\n}\n\n/// Serde helper for Duration as milliseconds.\nmod duration_millis {\n use std::time::Duration;\n\n use serde::{Deserialize, Deserializer, Serializer};\n\n pub fn serialize(d: &Duration, s: S) -> Result {\n s.serialize_u64(d.as_millis() as u64)\n }\n\n pub fn deserialize<'de, D: Deserializer<'de>>(d: D) -> Result {\n let millis = u64::deserialize(d)?;\n Ok(Duration::from_millis(millis))\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/pulls/2483" + }, + "response": { + "status": 200, + "headers": [ + [ + "last-modified", + "Wed, 15 Apr 2026 14:23:08 GMT" + ], + [ + "x-ratelimit-remaining", + "4965" + ], + [ + "date", + "Wed, 15 Apr 2026 15:30:27 GMT" + ], + [ + "etag", + "W/\"9e16e6116f0660a9e6887ab1801d96ff4f14bb9a0f40491f309b0afb108e8b95\"" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-ratelimit-used", + "35" + ], + [ + "x-github-request-id", + "13C2:8A635:4223F1:4D86B1:69DFAF12" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "x-frame-options", + "deny" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "content-type", + "application/json; charset=utf-8" + ], + [ + "x-xss-protection", + "0" + ], + [ + "server", + "github.com" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-github-media-type", + "github.v3; format=json" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP,Accept-Encoding, Accept, X-Requested-With" + ] + ], + "body": "{\"url\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483\",\"id\":3532069864,\"node_id\":\"PR_kwDORHZ7Z87Shxvo\",\"html_url\":\"https://github.com/nearai/ironclaw/pull/2483\",\"diff_url\":\"https://github.com/nearai/ironclaw/pull/2483.diff\",\"patch_url\":\"https://github.com/nearai/ironclaw/pull/2483.patch\",\"issue_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/2483\",\"number\":2483,\"state\":\"open\",\"locked\":false,\"title\":\"feat(engine): add code execution failure categorization instrumentation\",\"user\":{\"login\":\"serrrfirat\",\"id\":5748809,\"node_id\":\"MDQ6VXNlcjU3NDg4MDk=\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/5748809?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/serrrfirat\",\"html_url\":\"https://github.com/serrrfirat\",\"followers_url\":\"https://api.github.com/users/serrrfirat/followers\",\"following_url\":\"https://api.github.com/users/serrrfirat/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/serrrfirat/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/serrrfirat/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/serrrfirat/subscriptions\",\"organizations_url\":\"https://api.github.com/users/serrrfirat/orgs\",\"repos_url\":\"https://api.github.com/users/serrrfirat/repos\",\"events_url\":\"https://api.github.com/users/serrrfirat/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/serrrfirat/received_events\",\"type\":\"User\",\"user_view_type\":\"public\",\"site_admin\":false},\"body\":\"## Summary\\n- Adds `CodeExecutionFailure` enum (7 variants: SyntaxError, RuntimeError, NameLookup, VmPanic, ResourceLimit, ToolError, OsDenied) to classify why code execution fails\\n- Threads `failure` through `CodeExecutionResult` (replacing `had_error: bool`) and emits structured `CodeExecutionFailed` events for aggregate analysis of failure modes (Monty limitation vs LLM logic error vs tool dispatch failure)\\n- Upgrades trace analysis to use structured events with severity escalation (VmPanic/ResourceLimit → Error), with backward-compatible fallback for pre-instrumentation threads\\n\\n### Caller audit (Err → Ok(failure=VmPanic) shift)\\n`execute_code` / `execute_code_with_skills` previously returned `Err(EngineError::Effect)` on VM panic; now returns `Ok(CodeExecutionResult { failure: Some(VmPanic) })`. The only production call site is `orchestrator.rs:handle_execute_code_step` which correctly dispatches on `result.failure`. Test-only callers in `scripting.rs` were also updated. No other callers exist.\\n\\n### Known limitation: mixed-era fallback\\nThe trace analyzer's backward-compatible fallback (message-scraping for pre-instrumentation threads) is all-or-nothing: if a thread has *any* `CodeExecutionFailed` event, message-scraping is skipped entirely. Errors from pre-instrumentation steps in a mixed-era thread will go unreported. This is acceptable during the transition period since all new threads will have full instrumentation.\\n\\n## Test plan\\n- [x] 11 new unit tests for error classifier, code hash, and trace detection\\n- [x] `cargo test -p ironclaw_engine --lib` — 395 pass\\n- [x] `cargo clippy --all --all-features` — zero engine warnings\\n\\n🤖 Generated with [Claude Code](https://claude.com/claude-code)\",\"created_at\":\"2026-04-15T05:50:13Z\",\"updated_at\":\"2026-04-15T14:23:08Z\",\"closed_at\":null,\"merged_at\":null,\"merge_commit_sha\":\"35b0564ebf9bd9aef17e24c821ea98158844030d\",\"assignees\":[],\"requested_reviewers\":[{\"login\":\"ilblackdragon\",\"id\":175486,\"node_id\":\"MDQ6VXNlcjE3NTQ4Ng==\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/175486?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/ilblackdragon\",\"html_url\":\"https://github.com/ilblackdragon\",\"followers_url\":\"https://api.github.com/users/ilblackdragon/followers\",\"following_url\":\"https://api.github.com/users/ilblackdragon/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/ilblackdragon/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/ilblackdragon/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/ilblackdragon/subscriptions\",\"organizations_url\":\"https://api.github.com/users/ilblackdragon/orgs\",\"repos_url\":\"https://api.github.com/users/ilblackdragon/repos\",\"events_url\":\"https://api.github.com/users/ilblackdragon/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/ilblackdragon/received_events\",\"type\":\"User\",\"user_view_type\":\"public\",\"site_admin\":false}],\"requested_teams\":[],\"labels\":[{\"id\":10248775942,\"node_id\":\"LA_kwDORHZ7Z88AAAACYt_pBg\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/size:%20XL\",\"name\":\"size: XL\",\"color\":\"B71C1C\",\"default\":false,\"description\":\"500+ changed lines\"},{\"id\":10248776001,\"node_id\":\"LA_kwDORHZ7Z88AAAACYt_pQQ\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/risk:%20low\",\"name\":\"risk: low\",\"color\":\"4CAF50\",\"default\":false,\"description\":\"Changes to docs, tests, or low-risk modules\"},{\"id\":10248776633,\"node_id\":\"LA_kwDORHZ7Z88AAAACYt_ruQ\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/scope:%20db/postgres\",\"name\":\"scope: db/postgres\",\"color\":\"6A1B9A\",\"default\":false,\"description\":\"PostgreSQL backend\"},{\"id\":10248777536,\"node_id\":\"LA_kwDORHZ7Z88AAAACYt_vQA\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/contributor:%20core\",\"name\":\"contributor: core\",\"color\":\"FF8A65\",\"default\":false,\"description\":\"20+ merged PRs\"},{\"id\":10665130192,\"node_id\":\"LA_kwDORHZ7Z88AAAACe7D40A\",\"url\":\"https://api.github.com/repos/nearai/ironclaw/labels/DB%20MIGRATION\",\"name\":\"DB MIGRATION\",\"color\":\"C62828\",\"default\":false,\"description\":\"PR adds or modifies PostgreSQL or libSQL migration definitions\"}],\"milestone\":null,\"draft\":false,\"commits_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483/commits\",\"review_comments_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483/comments\",\"review_comment_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls/comments{/number}\",\"comments_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/2483/comments\",\"statuses_url\":\"https://api.github.com/repos/nearai/ironclaw/statuses/5496341c49b2abaa9842e2f939ea349c75fd7340\",\"head\":{\"label\":\"nearai:claude/audit-v2-engine-usage-I7dIz\",\"ref\":\"claude/audit-v2-engine-usage-I7dIz\",\"sha\":\"5496341c49b2abaa9842e2f939ea349c75fd7340\",\"user\":{\"login\":\"nearai\",\"id\":29134221,\"node_id\":\"MDEyOk9yZ2FuaXphdGlvbjI5MTM0MjIx\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/29134221?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/nearai\",\"html_url\":\"https://github.com/nearai\",\"followers_url\":\"https://api.github.com/users/nearai/followers\",\"following_url\":\"https://api.github.com/users/nearai/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/nearai/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/nearai/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/nearai/subscriptions\",\"organizations_url\":\"https://api.github.com/users/nearai/orgs\",\"repos_url\":\"https://api.github.com/users/nearai/repos\",\"events_url\":\"https://api.github.com/users/nearai/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/nearai/received_events\",\"type\":\"Organization\",\"user_view_type\":\"public\",\"site_admin\":false},\"repo\":{\"id\":1148615527,\"node_id\":\"R_kgDORHZ7Zw\",\"name\":\"ironclaw\",\"full_name\":\"nearai/ironclaw\",\"private\":false,\"owner\":{\"login\":\"nearai\",\"id\":29134221,\"node_id\":\"MDEyOk9yZ2FuaXphdGlvbjI5MTM0MjIx\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/29134221?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/nearai\",\"html_url\":\"https://github.com/nearai\",\"followers_url\":\"https://api.github.com/users/nearai/followers\",\"following_url\":\"https://api.github.com/users/nearai/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/nearai/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/nearai/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/nearai/subscriptions\",\"organizations_url\":\"https://api.github.com/users/nearai/orgs\",\"repos_url\":\"https://api.github.com/users/nearai/repos\",\"events_url\":\"https://api.github.com/users/nearai/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/nearai/received_events\",\"type\":\"Organization\",\"user_view_type\":\"public\",\"site_admin\":false},\"html_url\":\"https://github.com/nearai/ironclaw\",\"description\":\"IronClaw is OpenClaw inspired implementation in Rust focused on privacy and security\",\"fork\":false,\"url\":\"https://api.github.com/repos/nearai/ironclaw\",\"forks_url\":\"https://api.github.com/repos/nearai/ironclaw/forks\",\"keys_url\":\"https://api.github.com/repos/nearai/ironclaw/keys{/key_id}\",\"collaborators_url\":\"https://api.github.com/repos/nearai/ironclaw/collaborators{/collaborator}\",\"teams_url\":\"https://api.github.com/repos/nearai/ironclaw/teams\",\"hooks_url\":\"https://api.github.com/repos/nearai/ironclaw/hooks\",\"issue_events_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/events{/number}\",\"events_url\":\"https://api.github.com/repos/nearai/ironclaw/events\",\"assignees_url\":\"https://api.github.com/repos/nearai/ironclaw/assignees{/user}\",\"branches_url\":\"https://api.github.com/repos/nearai/ironclaw/branches{/branch}\",\"tags_url\":\"https://api.github.com/repos/nearai/ironclaw/tags\",\"blobs_url\":\"https://api.github.com/repos/nearai/ironclaw/git/blobs{/sha}\",\"git_tags_url\":\"https://api.github.com/repos/nearai/ironclaw/git/tags{/sha}\",\"git_refs_url\":\"https://api.github.com/repos/nearai/ironclaw/git/refs{/sha}\",\"trees_url\":\"https://api.github.com/repos/nearai/ironclaw/git/trees{/sha}\",\"statuses_url\":\"https://api.github.com/repos/nearai/ironclaw/statuses/{sha}\",\"languages_url\":\"https://api.github.com/repos/nearai/ironclaw/languages\",\"stargazers_url\":\"https://api.github.com/repos/nearai/ironclaw/stargazers\",\"contributors_url\":\"https://api.github.com/repos/nearai/ironclaw/contributors\",\"subscribers_url\":\"https://api.github.com/repos/nearai/ironclaw/subscribers\",\"subscription_url\":\"https://api.github.com/repos/nearai/ironclaw/subscription\",\"commits_url\":\"https://api.github.com/repos/nearai/ironclaw/commits{/sha}\",\"git_commits_url\":\"https://api.github.com/repos/nearai/ironclaw/git/commits{/sha}\",\"comments_url\":\"https://api.github.com/repos/nearai/ironclaw/comments{/number}\",\"issue_comment_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/comments{/number}\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/{+path}\",\"compare_url\":\"https://api.github.com/repos/nearai/ironclaw/compare/{base}...{head}\",\"merges_url\":\"https://api.github.com/repos/nearai/ironclaw/merges\",\"archive_url\":\"https://api.github.com/repos/nearai/ironclaw/{archive_format}{/ref}\",\"downloads_url\":\"https://api.github.com/repos/nearai/ironclaw/downloads\",\"issues_url\":\"https://api.github.com/repos/nearai/ironclaw/issues{/number}\",\"pulls_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls{/number}\",\"milestones_url\":\"https://api.github.com/repos/nearai/ironclaw/milestones{/number}\",\"notifications_url\":\"https://api.github.com/repos/nearai/ironclaw/notifications{?since,all,participating}\",\"labels_url\":\"https://api.github.com/repos/nearai/ironclaw/labels{/name}\",\"releases_url\":\"https://api.github.com/repos/nearai/ironclaw/releases{/id}\",\"deployments_url\":\"https://api.github.com/repos/nearai/ironclaw/deployments\",\"created_at\":\"2026-02-03T06:57:10Z\",\"updated_at\":\"2026-04-15T15:21:34Z\",\"pushed_at\":\"2026-04-15T15:27:33Z\",\"git_url\":\"git://github.com/nearai/ironclaw.git\",\"ssh_url\":\"git@github.com:nearai/ironclaw.git\",\"clone_url\":\"https://github.com/nearai/ironclaw.git\",\"svn_url\":\"https://github.com/nearai/ironclaw\",\"homepage\":\"https://www.ironclaw.com\",\"size\":30129,\"stargazers_count\":11789,\"watchers_count\":11789,\"language\":\"Rust\",\"has_issues\":true,\"has_projects\":false,\"has_downloads\":true,\"has_wiki\":false,\"has_pages\":false,\"has_discussions\":false,\"forks_count\":1349,\"mirror_url\":null,\"archived\":false,\"disabled\":false,\"open_issues_count\":640,\"license\":{\"key\":\"apache-2.0\",\"name\":\"Apache License 2.0\",\"spdx_id\":\"Apache-2.0\",\"url\":\"https://api.github.com/licenses/apache-2.0\",\"node_id\":\"MDc6TGljZW5zZTI=\"},\"allow_forking\":true,\"is_template\":false,\"web_commit_signoff_required\":false,\"has_pull_requests\":true,\"pull_request_creation_policy\":\"all\",\"topics\":[],\"visibility\":\"public\",\"forks\":1349,\"open_issues\":640,\"watchers\":11789,\"default_branch\":\"staging\"}},\"base\":{\"label\":\"nearai:staging\",\"ref\":\"staging\",\"sha\":\"16a07316d03430066f420e0d19e98a7506fec032\",\"user\":{\"login\":\"nearai\",\"id\":29134221,\"node_id\":\"MDEyOk9yZ2FuaXphdGlvbjI5MTM0MjIx\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/29134221?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/nearai\",\"html_url\":\"https://github.com/nearai\",\"followers_url\":\"https://api.github.com/users/nearai/followers\",\"following_url\":\"https://api.github.com/users/nearai/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/nearai/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/nearai/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/nearai/subscriptions\",\"organizations_url\":\"https://api.github.com/users/nearai/orgs\",\"repos_url\":\"https://api.github.com/users/nearai/repos\",\"events_url\":\"https://api.github.com/users/nearai/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/nearai/received_events\",\"type\":\"Organization\",\"user_view_type\":\"public\",\"site_admin\":false},\"repo\":{\"id\":1148615527,\"node_id\":\"R_kgDORHZ7Zw\",\"name\":\"ironclaw\",\"full_name\":\"nearai/ironclaw\",\"private\":false,\"owner\":{\"login\":\"nearai\",\"id\":29134221,\"node_id\":\"MDEyOk9yZ2FuaXphdGlvbjI5MTM0MjIx\",\"avatar_url\":\"https://avatars.githubusercontent.com/u/29134221?v=4\",\"gravatar_id\":\"\",\"url\":\"https://api.github.com/users/nearai\",\"html_url\":\"https://github.com/nearai\",\"followers_url\":\"https://api.github.com/users/nearai/followers\",\"following_url\":\"https://api.github.com/users/nearai/following{/other_user}\",\"gists_url\":\"https://api.github.com/users/nearai/gists{/gist_id}\",\"starred_url\":\"https://api.github.com/users/nearai/starred{/owner}{/repo}\",\"subscriptions_url\":\"https://api.github.com/users/nearai/subscriptions\",\"organizations_url\":\"https://api.github.com/users/nearai/orgs\",\"repos_url\":\"https://api.github.com/users/nearai/repos\",\"events_url\":\"https://api.github.com/users/nearai/events{/privacy}\",\"received_events_url\":\"https://api.github.com/users/nearai/received_events\",\"type\":\"Organization\",\"user_view_type\":\"public\",\"site_admin\":false},\"html_url\":\"https://github.com/nearai/ironclaw\",\"description\":\"IronClaw is OpenClaw inspired implementation in Rust focused on privacy and security\",\"fork\":false,\"url\":\"https://api.github.com/repos/nearai/ironclaw\",\"forks_url\":\"https://api.github.com/repos/nearai/ironclaw/forks\",\"keys_url\":\"https://api.github.com/repos/nearai/ironclaw/keys{/key_id}\",\"collaborators_url\":\"https://api.github.com/repos/nearai/ironclaw/collaborators{/collaborator}\",\"teams_url\":\"https://api.github.com/repos/nearai/ironclaw/teams\",\"hooks_url\":\"https://api.github.com/repos/nearai/ironclaw/hooks\",\"issue_events_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/events{/number}\",\"events_url\":\"https://api.github.com/repos/nearai/ironclaw/events\",\"assignees_url\":\"https://api.github.com/repos/nearai/ironclaw/assignees{/user}\",\"branches_url\":\"https://api.github.com/repos/nearai/ironclaw/branches{/branch}\",\"tags_url\":\"https://api.github.com/repos/nearai/ironclaw/tags\",\"blobs_url\":\"https://api.github.com/repos/nearai/ironclaw/git/blobs{/sha}\",\"git_tags_url\":\"https://api.github.com/repos/nearai/ironclaw/git/tags{/sha}\",\"git_refs_url\":\"https://api.github.com/repos/nearai/ironclaw/git/refs{/sha}\",\"trees_url\":\"https://api.github.com/repos/nearai/ironclaw/git/trees{/sha}\",\"statuses_url\":\"https://api.github.com/repos/nearai/ironclaw/statuses/{sha}\",\"languages_url\":\"https://api.github.com/repos/nearai/ironclaw/languages\",\"stargazers_url\":\"https://api.github.com/repos/nearai/ironclaw/stargazers\",\"contributors_url\":\"https://api.github.com/repos/nearai/ironclaw/contributors\",\"subscribers_url\":\"https://api.github.com/repos/nearai/ironclaw/subscribers\",\"subscription_url\":\"https://api.github.com/repos/nearai/ironclaw/subscription\",\"commits_url\":\"https://api.github.com/repos/nearai/ironclaw/commits{/sha}\",\"git_commits_url\":\"https://api.github.com/repos/nearai/ironclaw/git/commits{/sha}\",\"comments_url\":\"https://api.github.com/repos/nearai/ironclaw/comments{/number}\",\"issue_comment_url\":\"https://api.github.com/repos/nearai/ironclaw/issues/comments{/number}\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/{+path}\",\"compare_url\":\"https://api.github.com/repos/nearai/ironclaw/compare/{base}...{head}\",\"merges_url\":\"https://api.github.com/repos/nearai/ironclaw/merges\",\"archive_url\":\"https://api.github.com/repos/nearai/ironclaw/{archive_format}{/ref}\",\"downloads_url\":\"https://api.github.com/repos/nearai/ironclaw/downloads\",\"issues_url\":\"https://api.github.com/repos/nearai/ironclaw/issues{/number}\",\"pulls_url\":\"https://api.github.com/repos/nearai/ironclaw/pulls{/number}\",\"milestones_url\":\"https://api.github.com/repos/nearai/ironclaw/milestones{/number}\",\"notifications_url\":\"https://api.github.com/repos/nearai/ironclaw/notifications{?since,all,participating}\",\"labels_url\":\"https://api.github.com/repos/nearai/ironclaw/labels{/name}\",\"releases_url\":\"https://api.github.com/repos/nearai/ironclaw/releases{/id}\",\"deployments_url\":\"https://api.github.com/repos/nearai/ironclaw/deployments\",\"created_at\":\"2026-02-03T06:57:10Z\",\"updated_at\":\"2026-04-15T15:21:34Z\",\"pushed_at\":\"2026-04-15T15:27:33Z\",\"git_url\":\"git://github.com/nearai/ironclaw.git\",\"ssh_url\":\"git@github.com:nearai/ironclaw.git\",\"clone_url\":\"https://github.com/nearai/ironclaw.git\",\"svn_url\":\"https://github.com/nearai/ironclaw\",\"homepage\":\"https://www.ironclaw.com\",\"size\":30129,\"stargazers_count\":11789,\"watchers_count\":11789,\"language\":\"Rust\",\"has_issues\":true,\"has_projects\":false,\"has_downloads\":true,\"has_wiki\":false,\"has_pages\":false,\"has_discussions\":false,\"forks_count\":1349,\"mirror_url\":null,\"archived\":false,\"disabled\":false,\"open_issues_count\":640,\"license\":{\"key\":\"apache-2.0\",\"name\":\"Apache License 2.0\",\"spdx_id\":\"Apache-2.0\",\"url\":\"https://api.github.com/licenses/apache-2.0\",\"node_id\":\"MDc6TGljZW5zZTI=\"},\"allow_forking\":true,\"is_template\":false,\"web_commit_signoff_required\":false,\"has_pull_requests\":true,\"pull_request_creation_policy\":\"all\",\"topics\":[],\"visibility\":\"public\",\"forks\":1349,\"open_issues\":640,\"watchers\":11789,\"default_branch\":\"staging\"}},\"_links\":{\"self\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483\"},\"html\":{\"href\":\"https://github.com/nearai/ironclaw/pull/2483\"},\"issue\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/issues/2483\"},\"comments\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/issues/2483/comments\"},\"review_comments\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483/comments\"},\"review_comment\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/pulls/comments{/number}\"},\"commits\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/pulls/2483/commits\"},\"statuses\":{\"href\":\"https://api.github.com/repos/nearai/ironclaw/statuses/5496341c49b2abaa9842e2f939ea349c75fd7340\"}},\"author_association\":\"MEMBER\",\"auto_merge\":null,\"assignee\":null,\"active_lock_reason\":null,\"merged\":false,\"mergeable\":true,\"rebaseable\":false,\"mergeable_state\":\"blocked\",\"merged_by\":null,\"comments\":5,\"review_comments\":4,\"maintainer_can_modify\":false,\"commits\":5,\"additions\":570,\"deletions\":78,\"changed_files\":6}" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/pulls/2483/files?per_page=100" + }, + "response": { + "status": 200, + "headers": [ + [ + "x-frame-options", + "deny" + ], + [ + "content-type", + "application/json; charset=utf-8" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-ratelimit-used", + "36" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP,Accept-Encoding, Accept, X-Requested-With" + ], + [ + "etag", + "\"684a195e98a258be2a092400e150c420a630966f3f2d516f4f9ff7a9ce4d8fa9\"" + ], + [ + "date", + "Wed, 15 Apr 2026 15:30:28 GMT" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "x-xss-protection", + "0" + ], + [ + "x-github-media-type", + "github.v3; format=json" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "strict-transport-security", + "max-age=31536000; includeSubdomains; preload" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-ratelimit-remaining", + "4964" + ], + [ + "x-github-request-id", + "13F3:DBAD2:40CE9F:4C32B5:69DFAF13" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:23:08 GMT" + ], + [ + "x-accepted-oauth-scopes", + "" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "server", + "github.com" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "content-length", + "49621" + ] + ], + "body": "[{\"sha\":\"76bf321deece47e13ba3044a33ee11da44d6130c\",\"filename\":\"crates/ironclaw_engine/src/executor/orchestrator.rs\",\"status\":\"modified\",\"additions\":118,\"deletions\":4,\"changes\":122,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Forchestrator.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Forchestrator.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Forchestrator.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -720,6 +720,7 @@ async fn handle_execute_code_step(\\n };\\n \\n // Run user code in a nested Monty VM (same pattern as rlm_query)\\n+ let code_start = std::time::Instant::now();\\n match Box::pin(execute_code(\\n &code,\\n thread,\\n@@ -748,10 +749,12 @@ async fn handle_execute_code_step(\\n // error, etc.), surface it as an ActionFailed event so traces and\\n // observers see the failure. Without this, parse errors silently\\n // fall back to the LLM via the result dict and never warn callers.\\n- if result.had_error {\\n+ if let Some(ref category) = result.failure {\\n let error_msg = if !result.stdout.is_empty() {\\n- let snippet: String = result.stdout.chars().take(500).collect();\\n- format!(\\\"CodeAct execution failed: {snippet}\\\")\\n+ format!(\\n+ \\\"CodeAct execution failed: {}\\\",\\n+ tail_chars(&result.stdout, 500)\\n+ )\\n } else {\\n \\\"CodeAct execution failed (no stdout)\\\".to_string()\\n };\\n@@ -774,6 +777,25 @@ async fn handle_execute_code_step(\\n let _ = tx.send(failed_event.clone());\\n }\\n thread.events.push(failed_event);\\n+\\n+ // Emit structured CodeExecutionFailed event for instrumentation.\\n+ // This enables aggregate analysis of WHY code execution fails\\n+ // (Monty limitation vs LLM logic error vs tool dispatch failure).\\n+ let error_text = tail_chars(&result.stdout, 500);\\n+ let instrumentation_event = ThreadEvent::new(\\n+ thread.id,\\n+ EventKind::CodeExecutionFailed {\\n+ step_id: exec_ctx.step_id,\\n+ category: category.clone(),\\n+ error: error_text,\\n+ code_hash: Some(crate::executor::scripting::code_hash(&code)),\\n+ duration_ms: code_start.elapsed().as_millis() as u64,\\n+ },\\n+ );\\n+ if let Some(tx) = event_tx {\\n+ let _ = tx.send(instrumentation_event.clone());\\n+ }\\n+ thread.events.push(instrumentation_event);\\n }\\n thread.updated_at = chrono::Utc::now();\\n \\n@@ -795,7 +817,7 @@ async fn handle_execute_code_step(\\n \\\"stdout\\\": result.stdout,\\n \\\"action_results\\\": action_results,\\n \\\"final_answer\\\": result.final_answer,\\n- \\\"had_error\\\": result.had_error,\\n+ \\\"had_error\\\": result.failure.is_some(),\\n \\\"pending_gate\\\": result.need_approval.as_ref().map(|na| {\\n match na {\\n ThreadOutcome::GatePaused { gate_name, action_name, call_id, parameters, resume_kind, resume_output } => serde_json::json!({\\n@@ -2196,6 +2218,19 @@ fn action_calls_to_python_json(calls: &[ActionCall]) -> Vec {\\n .collect()\\n }\\n \\n+/// Extract the last `n` characters from `s`.\\n+///\\n+/// Error tracebacks appear at the end of stdout, after any `print()` output.\\n+/// Using the head would capture the print statements instead of the error.\\n+fn tail_chars(s: &str, n: usize) -> String {\\n+ let char_count = s.chars().count();\\n+ if char_count > n {\\n+ s.chars().skip(char_count - n).collect()\\n+ } else {\\n+ s.to_owned()\\n+ }\\n+}\\n+\\n /// Build a PII-safe summary of an `action_calls` JSON value for log output.\\n ///\\n /// The action_calls payload contains tool parameters, which can carry user\\n@@ -3860,4 +3895,83 @@ FINAL(batch_error_count)\\n );\\n }\\n }\\n+\\n+ // ── CodeExecutionFailed event emission (caller test) ────────\\n+\\n+ #[tokio::test]\\n+ async fn execute_code_step_emits_code_execution_failed_event() {\\n+ let llm: Arc = Arc::new(ModelCapturingLlm {\\n+ captured: tokio::sync::Mutex::new(Vec::new()),\\n+ });\\n+ let effects: Arc = Arc::new(NoopEffects);\\n+ let leases = Arc::new(LeaseManager::new());\\n+ let policy = Arc::new(PolicyEngine::new());\\n+\\n+ let mut thread = Thread::new(\\n+ \\\"test code execution failure instrumentation\\\",\\n+ crate::types::thread::ThreadType::Foreground,\\n+ ProjectId::new(),\\n+ \\\"test-user\\\",\\n+ crate::types::thread::ThreadConfig::default(),\\n+ );\\n+ thread.transition_to(ThreadState::Running, None).unwrap();\\n+\\n+ // Pass intentionally broken Python code (syntax error)\\n+ let args = &[\\n+ json_to_monty(&serde_json::json!(\\\"def ==\\\")),\\n+ json_to_monty(&serde_json::json!({})),\\n+ ];\\n+\\n+ let (tx, _rx) = tokio::sync::broadcast::channel(16);\\n+ let _result = handle_execute_code_step(\\n+ args,\\n+ &[],\\n+ &mut thread,\\n+ &llm,\\n+ &effects,\\n+ &leases,\\n+ &policy,\\n+ Some(&tx),\\n+ )\\n+ .await;\\n+\\n+ // Verify CodeExecutionFailed event was emitted on thread.events\\n+ let code_failed_events: Vec<_> = thread\\n+ .events\\n+ .iter()\\n+ .filter(|e| matches!(&e.kind, EventKind::CodeExecutionFailed { .. }))\\n+ .collect();\\n+\\n+ assert_eq!(\\n+ code_failed_events.len(),\\n+ 1,\\n+ \\\"expected exactly one CodeExecutionFailed event, got {}\\\",\\n+ code_failed_events.len()\\n+ );\\n+\\n+ if let EventKind::CodeExecutionFailed {\\n+ category,\\n+ code_hash,\\n+ ..\\n+ } = &code_failed_events[0].kind\\n+ {\\n+ assert_eq!(\\n+ *category,\\n+ crate::types::step::CodeExecutionFailure::SyntaxError\\n+ );\\n+ assert!(code_hash.is_some());\\n+ } else {\\n+ panic!(\\\"expected CodeExecutionFailed event kind\\\");\\n+ }\\n+\\n+ // Also verify ActionFailed was emitted (existing behavior)\\n+ let action_failed = thread\\n+ .events\\n+ .iter()\\n+ .any(|e| matches!(&e.kind, EventKind::ActionFailed { .. }));\\n+ assert!(\\n+ action_failed,\\n+ \\\"expected ActionFailed event alongside CodeExecutionFailed\\\"\\n+ );\\n+ }\\n }\"},{\"sha\":\"674fca2750840d6f245ff8676d54df8b83ca74f1\",\"filename\":\"crates/ironclaw_engine/src/executor/scripting.rs\",\"status\":\"modified\",\"additions\":250,\"deletions\":52,\"changes\":302,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Fscripting.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Fscripting.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Fscripting.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -33,7 +33,7 @@ use crate::traits::llm::{LlmBackend, LlmCallConfig};\\n use crate::types::error::EngineError;\\n use crate::types::event::EventKind;\\n use crate::types::message::{MessageRole, ThreadMessage};\\n-use crate::types::step::{ActionResult, LlmResponse, TokenUsage};\\n+use crate::types::step::{ActionResult, CodeExecutionFailure, LlmResponse, TokenUsage};\\n use crate::types::thread::Thread;\\n use ironclaw_common::ValidTimezone;\\n \\n@@ -72,8 +72,10 @@ pub struct CodeExecutionResult {\\n pub recursive_tokens: TokenUsage,\\n /// If set, the code called FINAL() or FINAL_VAR() with this answer.\\n pub final_answer: Option,\\n- /// Whether the code execution hit an error (traceback included in stdout).\\n- pub had_error: bool,\\n+ /// Classified failure category. `None` when execution succeeded or was\\n+ /// paused by a gate. `Some(category)` when code execution failed —\\n+ /// `failure.is_some()` replaces the former `had_error: bool` field.\\n+ pub failure: Option,\\n }\\n \\n /// Build a compact output summary for inclusion in LLM context between steps.\\n@@ -297,7 +299,6 @@ pub async fn execute_code_with_skills(\\n let mut events = Vec::new();\\n let mut recursive_tokens = TokenUsage::default();\\n let mut final_answer: Option = None;\\n- let mut had_error = false;\\n \\n // Build context variables including persisted state from prior steps\\n let (input_names, input_values) = build_context_inputs(thread, persisted_state);\\n@@ -335,12 +336,19 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer: None,\\n- had_error: true,\\n+ failure: Some(CodeExecutionFailure::SyntaxError),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during code parsing\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\\"{stdout}\\\\nVmPanic: Monty VM panicked during code parsing\\\"),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer: None,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n };\\n@@ -356,6 +364,7 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => p,\\n Ok(Err(e)) => {\\n // Runtime error flows back to LLM\\n+ let category = classify_runtime_error(&e.to_string());\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout: format!(\\\"{stdout}\\\\nError: {e}\\\"),\\n@@ -364,12 +373,19 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer: None,\\n- had_error: true,\\n+ failure: Some(category),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during execution start\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\\"{stdout}\\\\nVmPanic: Monty VM panicked during execution start\\\"),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer: None,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n };\\n@@ -392,7 +408,7 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: None,\\n });\\n }\\n \\n@@ -479,7 +495,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -488,12 +503,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(classify_runtime_error(&e.to_string())),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during resume\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during resume\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -509,7 +533,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -518,12 +541,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(classify_runtime_error(&e.to_string())),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during resume_pending\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during resume_pending\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -585,7 +617,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -594,12 +625,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(CodeExecutionFailure::ToolError),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during resume_pending\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during resume_pending\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -612,7 +652,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -621,12 +660,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(CodeExecutionFailure::ToolError),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during resume\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during resume\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -640,7 +688,7 @@ pub async fn execute_code_with_skills(\\n need_approval: Some(outcome),\\n recursive_tokens,\\n final_answer: None,\\n- had_error,\\n+ failure: None,\\n });\\n }\\n }\\n@@ -702,7 +750,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -711,12 +758,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(classify_runtime_error(&e.to_string())),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during ResolveFutures resume\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during ResolveFutures resume\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -747,7 +803,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nNameError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -756,12 +811,21 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(CodeExecutionFailure::NameLookup),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during name lookup\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\n+ \\\"{stdout}\\\\nVmPanic: Monty VM panicked during name lookup\\\"\\n+ ),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -779,7 +843,6 @@ pub async fn execute_code_with_skills(\\n Ok(Ok(p)) => progress = p,\\n Ok(Err(e)) => {\\n stdout.push_str(&format!(\\\"\\\\nOSError: {e}\\\"));\\n- had_error = true;\\n return Ok(CodeExecutionResult {\\n return_value: serde_json::Value::Null,\\n stdout,\\n@@ -788,12 +851,19 @@ pub async fn execute_code_with_skills(\\n need_approval: None,\\n recursive_tokens,\\n final_answer,\\n- had_error,\\n+ failure: Some(CodeExecutionFailure::OsDenied),\\n });\\n }\\n Err(_) => {\\n- return Err(EngineError::Effect {\\n- reason: \\\"Monty VM panicked during OS call\\\".into(),\\n+ return Ok(CodeExecutionResult {\\n+ return_value: serde_json::Value::Null,\\n+ stdout: format!(\\\"{stdout}\\\\nVmPanic: Monty VM panicked during OS call\\\"),\\n+ action_results,\\n+ events,\\n+ need_approval: None,\\n+ recursive_tokens,\\n+ final_answer,\\n+ failure: Some(CodeExecutionFailure::VmPanic),\\n });\\n }\\n }\\n@@ -802,6 +872,52 @@ pub async fn execute_code_with_skills(\\n }\\n }\\n \\n+// ── Error classification ────────────────────────────────────\\n+\\n+/// Classify a runtime error message into a failure category.\\n+///\\n+/// Parses the error text from Monty to distinguish between LLM logic bugs\\n+/// (NameError, TypeError, etc.), resource limit hits, and Monty VM issues.\\n+fn classify_runtime_error(error_msg: &str) -> CodeExecutionFailure {\\n+ let lower = error_msg.to_ascii_lowercase();\\n+\\n+ // Most specific checks first to avoid substring false positives.\\n+ if lower.contains(\\\"timed out\\\")\\n+ || lower.contains(\\\"timeout\\\")\\n+ || lower.contains(\\\"memory limit\\\")\\n+ || lower.contains(\\\"allocation limit\\\")\\n+ || lower.contains(\\\"out of fuel\\\")\\n+ || lower.contains(\\\"fuel exhausted\\\")\\n+ || lower.contains(\\\"resource limit\\\")\\n+ {\\n+ CodeExecutionFailure::ResourceLimit\\n+ } else if lower.contains(\\\"os operations are not permitted\\\") || lower.contains(\\\"oserror\\\") {\\n+ CodeExecutionFailure::OsDenied\\n+ } else if lower.contains(\\\"syntaxerror\\\") {\\n+ CodeExecutionFailure::SyntaxError\\n+ } else {\\n+ // NameError, TypeError, ValueError, AttributeError, IndexError,\\n+ // KeyError, ModuleNotFoundError, NotImplementedError, etc.\\n+ CodeExecutionFailure::RuntimeError\\n+ }\\n+}\\n+\\n+/// Compute a short hash of Python code for dedup/correlation in events.\\n+///\\n+/// Uses FNV-1a (64-bit) which is stable across Rust versions, unlike\\n+/// `DefaultHasher`. Not cryptographic — collision probability is ~2^-32\\n+/// at typical usage levels, sufficient for dedup but not for security.\\n+pub fn code_hash(code: &str) -> String {\\n+ const FNV_OFFSET: u64 = 0xcbf29ce484222325;\\n+ const FNV_PRIME: u64 = 0x00000100000001B3;\\n+ let mut hash = FNV_OFFSET;\\n+ for byte in code.as_bytes() {\\n+ hash ^= *byte as u64;\\n+ hash = hash.wrapping_mul(FNV_PRIME);\\n+ }\\n+ format!(\\\"{hash:016x}\\\")\\n+}\\n+\\n // ── Pending future tracking ─────────────────────────────────\\n \\n /// A deferred computation spawned as a tokio task, pending resolution\\n@@ -1803,7 +1919,7 @@ FINAL(str(result))\\n result.stdout\\n );\\n assert!(\\n- !result.had_error,\\n+ result.failure.is_none(),\\n \\\"should not error, stdout: {}\\\",\\n result.stdout\\n );\\n@@ -1855,7 +1971,7 @@ FINAL(str(a + b))\\n result.stdout\\n );\\n assert_eq!(result.action_results.len(), 2);\\n- assert!(!result.had_error);\\n+ assert!(result.failure.is_none());\\n }\\n \\n // ── asyncio.gather three tools ──────────────────────────\\n@@ -1905,7 +2021,7 @@ FINAL(str(s) + \\\"|\\\" + str(h) + \\\"|\\\" + str(m))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.action_results.len(), 3);\\n let answer = result.final_answer.unwrap();\\n assert!(answer.contains(\\\"search results\\\"), \\\"got: {answer}\\\");\\n@@ -1945,7 +2061,7 @@ FINAL(str(b))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.action_results.len(), 2);\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"final\\\"));\\n }\\n@@ -1980,7 +2096,7 @@ FINAL(\\\"should not reach\\\")\\n let result = run_code(code, effects, &thread).await.unwrap();\\n // Error in gather propagates as exception — code should error\\n assert!(\\n- result.had_error,\\n+ result.failure.is_some(),\\n \\\"should have error, stdout: {}\\\",\\n result.stdout\\n );\\n@@ -2024,7 +2140,7 @@ FINAL(\\\"hello from sync\\\")\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"hello from sync\\\"));\\n- assert!(!result.had_error);\\n+ assert!(result.failure.is_none());\\n }\\n \\n // ── globals() still works ───────────────────────────────\\n@@ -2045,7 +2161,7 @@ FINAL(str(has_search) + \\\"|\\\" + str(has_http))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"True|True\\\"));\\n }\\n \\n@@ -2063,7 +2179,7 @@ FINAL(str(len(results)))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"0\\\"));\\n }\\n \\n@@ -2090,7 +2206,7 @@ FINAL(str(results[0]))\\n \\\"#;\\n \\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(!result.had_error, \\\"stdout: {}\\\", result.stdout);\\n+ assert!(result.failure.is_none(), \\\"stdout: {}\\\", result.stdout);\\n assert_eq!(result.final_answer.as_deref(), Some(\\\"gathered\\\"));\\n assert_eq!(result.action_results.len(), 1);\\n }\\n@@ -2137,7 +2253,7 @@ while True:\\n // the key assertion is that it DOES NOT run forever.\\n if let Ok(r) = result {\\n assert!(\\n- r.had_error || r.stdout.contains(\\\"Error\\\") || r.stdout.contains(\\\"limit\\\"),\\n+ r.failure.is_some() || r.stdout.contains(\\\"Error\\\") || r.stdout.contains(\\\"limit\\\"),\\n \\\"resource limit should terminate infinite loop, got stdout: {}\\\",\\n truncate_for_assert(&r.stdout, 500),\\n );\\n@@ -2279,7 +2395,7 @@ while True:\\n // Must terminate — either via error or resource limit\\n if let Ok(r) = result {\\n assert!(\\n- r.had_error || r.stdout.contains(\\\"Error\\\") || r.stdout.contains(\\\"limit\\\"),\\n+ r.failure.is_some() || r.stdout.contains(\\\"Error\\\") || r.stdout.contains(\\\"limit\\\"),\\n \\\"cpu-bound loop should be terminated, stdout: {}\\\",\\n truncate_for_assert(&r.stdout, 500),\\n );\\n@@ -2313,7 +2429,7 @@ FINAL(str(x))\\n \\n let code = \\\"def broken(\\\\nFINAL('nope')\\\";\\n let result = run_code(code, effects, &thread).await.unwrap();\\n- assert!(result.had_error, \\\"syntax error should set had_error\\\");\\n+ assert!(result.failure.is_some(), \\\"syntax error should set failure\\\");\\n assert!(\\n result.stdout.contains(\\\"SyntaxError\\\") || result.stdout.contains(\\\"Error\\\"),\\n \\\"should contain SyntaxError, got: {}\\\",\\n@@ -2736,4 +2852,86 @@ FINAL(str(x))\\n assert!(matches!(result, ExtFunctionResult::Error(_)));\\n assert!(llm.calls.lock().await.is_empty());\\n }\\n+\\n+ // ── Error classification tests ──────────────────────────────\\n+\\n+ #[test]\\n+ fn classify_syntax_error() {\\n+ let cat = classify_runtime_error(\\\"SyntaxError: unexpected token\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::SyntaxError);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_timeout() {\\n+ let cat = classify_runtime_error(\\\"execution timed out after 30s\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_memory_limit() {\\n+ let cat = classify_runtime_error(\\\"memory limit exceeded\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_fuel_exhaustion() {\\n+ let cat = classify_runtime_error(\\\"fuel exhausted during execution\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_os_denied() {\\n+ let cat = classify_runtime_error(\\\"OS operations are not permitted in CodeAct scripts\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::OsDenied);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_name_error_as_runtime() {\\n+ // NameError from Monty (not NameLookup) is classified as RuntimeError\\n+ let cat = classify_runtime_error(\\\"NameError: name 'foo' is not defined\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_type_error_as_runtime() {\\n+ let cat = classify_runtime_error(\\\"TypeError: unsupported operand\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_module_not_found_as_runtime() {\\n+ let cat = classify_runtime_error(\\\"ModuleNotFoundError: No module named 'csv'\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\\n+ }\\n+\\n+ #[test]\\n+ fn classify_syntax_word_is_not_syntaxerror() {\\n+ // \\\"syntax\\\" alone should not trigger SyntaxError — only \\\"syntaxerror\\\" should.\\n+ let cat = classify_runtime_error(\\\"unexpected syntax in expression\\\");\\n+ assert_eq!(cat, CodeExecutionFailure::RuntimeError);\\n+ }\\n+\\n+ #[test]\\n+ fn vm_panic_variant_serializes_as_snake_case() {\\n+ // VmPanic is set directly by catch_unwind paths, not by classify_runtime_error.\\n+ // Verify it serializes consistently with Display (both snake_case).\\n+ let failure = CodeExecutionFailure::VmPanic;\\n+ assert_eq!(failure.to_string(), \\\"vm_panic\\\");\\n+ let json = serde_json::to_value(&failure).unwrap();\\n+ assert_eq!(json, serde_json::json!(\\\"vm_panic\\\"));\\n+ }\\n+\\n+ #[test]\\n+ fn code_hash_deterministic() {\\n+ let h1 = code_hash(\\\"print('hello')\\\");\\n+ let h2 = code_hash(\\\"print('hello')\\\");\\n+ assert_eq!(h1, h2);\\n+ }\\n+\\n+ #[test]\\n+ fn code_hash_differs_for_different_code() {\\n+ let h1 = code_hash(\\\"print('hello')\\\");\\n+ let h2 = code_hash(\\\"print('world')\\\");\\n+ assert_ne!(h1, h2);\\n+ }\\n }\"},{\"sha\":\"563d5e8dd6835053c514ff45fcebbf5e94bf283f\",\"filename\":\"crates/ironclaw_engine/src/executor/trace.rs\",\"status\":\"modified\",\"additions\":135,\"deletions\":21,\"changes\":156,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Ftrace.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Ftrace.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Ftrace.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -195,33 +195,76 @@ fn analyze_trace(thread: &Thread) -> Vec {\\n }\\n }\\n \\n- // 4. Check for code execution errors in output messages.\\n- // Code output appears as User-role messages (Monty stdout/stderr) with\\n- // prefixes like \\\"[stdout]\\\" or \\\"[stderr]\\\". Skip the System prompt (index 0)\\n- // and Assistant messages to avoid false positives from example text.\\n- let error_patterns = [\\n- \\\"NameError\\\",\\n- \\\"SyntaxError\\\",\\n- \\\"TypeError\\\",\\n- \\\"NotImplementedError\\\",\\n- ];\\n- for (i, msg) in thread.messages.iter().enumerate() {\\n- let is_code_output = msg.role == crate::types::message::MessageRole::User\\n- && (msg.content.starts_with(\\\"[stdout]\\\")\\n- || msg.content.starts_with(\\\"[stderr]\\\")\\n- || msg.content.starts_with(\\\"[code \\\")\\n- || msg.content.starts_with(\\\"Traceback\\\"));\\n- if is_code_output && error_patterns.iter().any(|p| msg.content.contains(p)) {\\n- let preview: String = msg.content.chars().take(200).collect();\\n+ // 4. Check for code execution errors via structured CodeExecutionFailed events.\\n+ // These carry a classified failure category that tells us exactly what kind\\n+ // of error occurred (syntax, runtime, name lookup, VM panic, resource limit,\\n+ // tool error, OS denied, gate pause).\\n+ let code_failures: Vec<&ThreadEvent> = thread\\n+ .events\\n+ .iter()\\n+ .filter(|e| {\\n+ matches!(\\n+ e.kind,\\n+ crate::types::event::EventKind::CodeExecutionFailed { .. }\\n+ )\\n+ })\\n+ .collect();\\n+ for event in &code_failures {\\n+ if let crate::types::event::EventKind::CodeExecutionFailed {\\n+ category, error, ..\\n+ } = &event.kind\\n+ {\\n+ let preview: String = error.chars().take(200).collect();\\n+ let severity = match category {\\n+ crate::types::step::CodeExecutionFailure::VmPanic => IssueSeverity::Error,\\n+ crate::types::step::CodeExecutionFailure::ResourceLimit => IssueSeverity::Error,\\n+ _ => IssueSeverity::Warning,\\n+ };\\n issues.push(TraceIssue {\\n- severity: IssueSeverity::Warning,\\n- category: \\\"code_error\\\".into(),\\n- description: format!(\\\"Code execution error in message {i}: {preview}\\\"),\\n+ severity,\\n+ category: format!(\\\"code_{category}\\\"),\\n+ description: format!(\\\"Code execution failed ({category}): {preview}\\\"),\\n step: None,\\n });\\n }\\n }\\n \\n+ // Fallback: also check message-level patterns for backward compatibility\\n+ // with threads that ran before the CodeExecutionFailed instrumentation\\n+ // was added (PR #2483). Note: threads from mixed eras (some steps\\n+ // instrumented, some not) will only report structured events when any\\n+ // exist, silently skipping message-level errors from uninstrumented steps.\\n+ if code_failures.is_empty() {\\n+ let error_patterns = [\\n+ \\\"NameError\\\",\\n+ \\\"SyntaxError\\\",\\n+ \\\"TypeError\\\",\\n+ \\\"NotImplementedError\\\",\\n+ \\\"ValueError\\\",\\n+ \\\"AttributeError\\\",\\n+ \\\"IndexError\\\",\\n+ \\\"KeyError\\\",\\n+ \\\"ModuleNotFoundError\\\",\\n+ \\\"RuntimeError\\\",\\n+ ];\\n+ for (i, msg) in thread.messages.iter().enumerate() {\\n+ let is_code_output = msg.role == crate::types::message::MessageRole::User\\n+ && (msg.content.starts_with(\\\"[stdout]\\\")\\n+ || msg.content.starts_with(\\\"[stderr]\\\")\\n+ || msg.content.starts_with(\\\"[code \\\")\\n+ || msg.content.starts_with(\\\"Traceback\\\"));\\n+ if is_code_output && error_patterns.iter().any(|p| msg.content.contains(p)) {\\n+ let preview: String = msg.content.chars().take(200).collect();\\n+ issues.push(TraceIssue {\\n+ severity: IssueSeverity::Warning,\\n+ category: \\\"code_error\\\".into(),\\n+ description: format!(\\\"Code execution error in message {i}: {preview}\\\"),\\n+ step: None,\\n+ });\\n+ }\\n+ }\\n+ }\\n+\\n // 5. Check for empty call_id on ActionResult messages (causes LLM API rejection).\\n for (i, msg) in thread.messages.iter().enumerate() {\\n if msg.role == crate::types::message::MessageRole::ActionResult {\\n@@ -536,6 +579,77 @@ mod tests {\\n );\\n }\\n \\n+ // ── CodeExecutionFailed event detection ────────────────────\\n+\\n+ #[test]\\n+ fn detects_code_execution_failure_from_event() {\\n+ let mut thread = make_thread();\\n+ thread.add_message(ThreadMessage::system(\\\"sys\\\"));\\n+ thread.add_message(ThreadMessage::assistant(\\\"```repl\\\\nimport csv\\\\n```\\\"));\\n+ thread.events.push(ThreadEvent::new(\\n+ thread.id,\\n+ EventKind::CodeExecutionFailed {\\n+ step_id: StepId::new(),\\n+ category: crate::types::step::CodeExecutionFailure::RuntimeError,\\n+ error: \\\"ModuleNotFoundError: No module named 'csv'\\\".into(),\\n+ code_hash: Some(\\\"abc123\\\".into()),\\n+ duration_ms: 42,\\n+ },\\n+ ));\\n+\\n+ let issues = analyze_trace(&thread);\\n+ let code_issues: Vec<_> = issues\\n+ .iter()\\n+ .filter(|i| i.category.starts_with(\\\"code_\\\"))\\n+ .collect();\\n+ assert_eq!(code_issues.len(), 1);\\n+ assert_eq!(code_issues[0].category, \\\"code_runtime_error\\\");\\n+ assert_eq!(code_issues[0].severity, IssueSeverity::Warning);\\n+ assert!(code_issues[0].description.contains(\\\"ModuleNotFoundError\\\"));\\n+ }\\n+\\n+ #[test]\\n+ fn vm_panic_is_error_severity() {\\n+ let mut thread = make_thread();\\n+ thread.add_message(ThreadMessage::system(\\\"sys\\\"));\\n+ thread.add_message(ThreadMessage::assistant(\\\"code\\\"));\\n+ thread.events.push(ThreadEvent::new(\\n+ thread.id,\\n+ EventKind::CodeExecutionFailed {\\n+ step_id: StepId::new(),\\n+ category: crate::types::step::CodeExecutionFailure::VmPanic,\\n+ error: \\\"Monty panicked: unreachable\\\".into(),\\n+ code_hash: None,\\n+ duration_ms: 0,\\n+ },\\n+ ));\\n+\\n+ let issues = analyze_trace(&thread);\\n+ let panic_issues: Vec<_> = issues\\n+ .iter()\\n+ .filter(|i| i.category == \\\"code_vm_panic\\\")\\n+ .collect();\\n+ assert_eq!(panic_issues.len(), 1);\\n+ assert_eq!(panic_issues[0].severity, IssueSeverity::Error);\\n+ }\\n+\\n+ #[test]\\n+ fn fallback_message_detection_when_no_events() {\\n+ // Threads from before instrumentation should still be detected\\n+ let mut thread = make_thread();\\n+ thread.add_message(ThreadMessage::system(\\\"sys\\\"));\\n+ thread.add_message(ThreadMessage::assistant(\\\"code\\\"));\\n+ thread.add_message(ThreadMessage::user(\\n+ \\\"[stdout]\\\\nNameError: name 'foo' is not defined\\\",\\n+ ));\\n+\\n+ let issues = analyze_trace(&thread);\\n+ assert!(\\n+ issues.iter().any(|i| i.category == \\\"code_error\\\"),\\n+ \\\"should detect code error from message when no CodeExecutionFailed events exist\\\"\\n+ );\\n+ }\\n+\\n #[test]\\n fn trace_serializes_approval_request_payload() {\\n let mut thread = make_thread();\"},{\"sha\":\"9ecaa6b575a369535e657b5f922187d3ce5149fa\",\"filename\":\"crates/ironclaw_engine/src/lib.rs\",\"status\":\"modified\",\"additions\":2,\"deletions\":1,\"changes\":3,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Flib.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Flib.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Flib.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -46,7 +46,8 @@ pub use types::mission::{Mission, MissionCadence, MissionId, MissionStatus, Vali\\n pub use types::project::{Project, ProjectId};\\n pub use types::provenance::Provenance;\\n pub use types::step::{\\n- ActionCall, ActionResult, ExecutionTier, LlmResponse, Step, StepId, StepStatus, TokenUsage,\\n+ ActionCall, ActionResult, CodeExecutionFailure, ExecutionTier, LlmResponse, Step, StepId,\\n+ StepStatus, TokenUsage,\\n };\\n pub use types::thread::{\\n ActiveSkillProvenance, Thread, ThreadConfig, ThreadId, ThreadState, ThreadType,\"},{\"sha\":\"b61185f07131d737272f1068c899f7f6b5821960\",\"filename\":\"crates/ironclaw_engine/src/types/event.rs\",\"status\":\"modified\",\"additions\":25,\"deletions\":0,\"changes\":25,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fevent.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fevent.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fevent.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -215,10 +215,35 @@ pub enum EventKind {\\n skill_names: Vec,\\n },\\n \\n+ // ── Code execution instrumentation ────────────────────────\\n+ /// Emitted when a code (REPL) execution attempt fails. Enables aggregate\\n+ /// analysis of code execution failure modes to determine whether the\\n+ /// runtime (Monty), the LLM, or tool dispatch is the primary source of\\n+ /// failures.\\n+ CodeExecutionFailed {\\n+ step_id: StepId,\\n+ /// Classified failure category.\\n+ category: crate::types::step::CodeExecutionFailure,\\n+ /// The error message text (truncated to 500 chars).\\n+ error: String,\\n+ /// Hash of the Python code that was executed, for dedup/correlation.\\n+ #[serde(default, skip_serializing_if = \\\"Option::is_none\\\")]\\n+ code_hash: Option,\\n+ /// Duration of the code execution attempt in milliseconds.\\n+ #[serde(default)]\\n+ duration_ms: u64,\\n+ },\\n+\\n // ── Orchestrator versioning ───────────────────────────────\\n OrchestratorRollback {\\n from_version: u64,\\n to_version: u64,\\n reason: String,\\n },\\n+\\n+ /// Unknown event kind — catch-all for forward compatibility during\\n+ /// rolling deploys. Older binaries deserializing events written by\\n+ /// newer binaries will produce this variant instead of failing.\\n+ #[serde(other)]\\n+ Unknown,\\n }\"},{\"sha\":\"161096e9b02238d78d86f611befa72d32fbe0a76\",\"filename\":\"crates/ironclaw_engine/src/types/step.rs\",\"status\":\"modified\",\"additions\":40,\"deletions\":0,\"changes\":40,\"blob_url\":\"https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fstep.rs\",\"raw_url\":\"https://github.com/nearai/ironclaw/raw/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fstep.rs\",\"contents_url\":\"https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Ftypes%2Fstep.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340\",\"patch\":\"@@ -132,6 +132,46 @@ pub struct ActionResult {\\n pub duration: Duration,\\n }\\n \\n+/// Classification of code execution failures.\\n+///\\n+/// Used by the instrumentation layer to distinguish Monty VM limitations\\n+/// from LLM logic errors, tool dispatch failures, and resource exhaustion.\\n+/// This data enables informed decisions about runtime alternatives.\\n+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]\\n+#[serde(rename_all = \\\"snake_case\\\")]\\n+pub enum CodeExecutionFailure {\\n+ /// Python parse error — LLM generated invalid syntax.\\n+ SyntaxError,\\n+ /// Python runtime error (NameError, TypeError, ValueError, etc.) —\\n+ /// LLM logic bug or use of unsupported feature.\\n+ RuntimeError,\\n+ /// Name lookup failed — function/variable not in scope and not a known tool.\\n+ NameLookup,\\n+ /// Monty VM panicked (catch_unwind caught it). Indicates a Monty bug,\\n+ /// not a user code issue.\\n+ VmPanic,\\n+ /// Resource limit hit (timeout, memory, or allocation cap).\\n+ ResourceLimit,\\n+ /// A tool call inside code returned an error.\\n+ ToolError,\\n+ /// OS operation attempted (blocked by sandbox).\\n+ OsDenied,\\n+}\\n+\\n+impl std::fmt::Display for CodeExecutionFailure {\\n+ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {\\n+ match self {\\n+ Self::SyntaxError => write!(f, \\\"syntax_error\\\"),\\n+ Self::RuntimeError => write!(f, \\\"runtime_error\\\"),\\n+ Self::NameLookup => write!(f, \\\"name_lookup\\\"),\\n+ Self::VmPanic => write!(f, \\\"vm_panic\\\"),\\n+ Self::ResourceLimit => write!(f, \\\"resource_limit\\\"),\\n+ Self::ToolError => write!(f, \\\"tool_error\\\"),\\n+ Self::OsDenied => write!(f, \\\"os_denied\\\"),\\n+ }\\n+ }\\n+}\\n+\\n /// Token usage for a single LLM call.\\n #[derive(Debug, Clone, Copy, Default, Serialize, Deserialize)]\\n pub struct TokenUsage {\"}]" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/executor/orchestrator.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-xss-protection", + "0" + ], + [ + "content-length", + "152011" + ], + [ + "etag", + "\"76bf321deece47e13ba3044a33ee11da44d6130c\"" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-ratelimit-used", + "37" + ], + [ + "server", + "github.com" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "date", + "Wed, 15 Apr 2026 15:30:28 GMT" + ], + [ + "x-github-request-id", + "140E:13AC25:3FC4E7:4B26A5:69DFAF14" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-frame-options", + "deny" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-ratelimit-remaining", + "4963" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ] + ], + "body": "//! Python orchestrator — the self-modifiable execution loop.\n//!\n//! Replaces the Rust `ExecutionLoop::run()` with versioned Python code\n//! executed via Monty. The orchestrator is the \"glue layer\" between the\n//! LLM and tools — tool dispatch, output formatting, state management,\n//! truncation — all in Python, patchable by the self-improvement Mission.\n//!\n//! Host functions exposed to the orchestrator Python:\n//! - `__llm_complete__` — make an LLM call\n//! - `__execute_code_step__` — run user CodeAct code in a nested Monty VM\n//! - `__execute_action__` — execute a single tool action\n//! - `__execute_actions_parallel__` — execute multiple tool actions concurrently\n//! - `__check_signals__` — poll for stop/inject signals\n//! - `__emit_event__` — broadcast a ThreadEvent\n//! - `__save_checkpoint__` — persist thread state\n//! - `__transition_to__` — change thread state (validated)\n//! - `__retrieve_docs__` — query memory docs\n//! - `__check_budget__` — remaining tokens/time/USD\n//! - `__get_actions__` — available tool definitions\n\nuse std::sync::Arc;\nuse std::sync::atomic::{AtomicU64, Ordering};\n\nuse std::collections::HashMap;\n\nuse monty::{\n ExtFunctionResult, LimitedTracker, MontyObject, MontyRun, NameLookupResult, PrintWriter,\n ResourceLimits, RunProgress,\n};\nuse tracing::{debug, warn};\n\nuse crate::capability::lease::LeaseManager;\nuse crate::capability::policy::PolicyEngine;\nuse crate::memory::RetrievalEngine;\nuse crate::runtime::lease_refresh::reconcile_dynamic_tool_lease;\nuse crate::runtime::messaging::{SignalReceiver, ThreadOutcome, ThreadSignal};\nuse crate::traits::effect::{EffectExecutor, ThreadExecutionContext};\nuse crate::traits::llm::{LlmBackend, LlmCallConfig};\nuse crate::traits::store::Store;\nuse crate::types::error::EngineError;\nuse crate::types::event::{EventKind, ThreadEvent, summarize_params};\nuse crate::types::message::ThreadMessage;\nuse crate::types::project::ProjectId;\nuse crate::types::shared_owner_id;\nuse crate::types::step::{ActionCall, StepId, TokenUsage};\nuse crate::types::thread::{ActiveSkillProvenance, Thread, ThreadState};\nuse ironclaw_common::ValidTimezone;\n\nuse super::scripting::{execute_code, json_to_monty, monty_to_json, monty_to_string};\n\n/// The compiled-in default orchestrator (v0).\npub(crate) const DEFAULT_ORCHESTRATOR: &str = include_str!(\"../../orchestrator/default.py\");\n\n/// Well-known title for orchestrator code in the Store.\npub const ORCHESTRATOR_TITLE: &str = \"orchestrator:main\";\n\n/// Well-known tag for orchestrator code docs.\npub const ORCHESTRATOR_TAG: &str = \"orchestrator_code\";\n\n/// Result of running the orchestrator.\npub struct OrchestratorResult {\n /// The thread outcome parsed from the orchestrator's return value.\n pub outcome: ThreadOutcome,\n /// Total tokens used by LLM calls within the orchestrator.\n pub tokens_used: TokenUsage,\n}\n\n/// Extract source_channel from thread metadata (set by ConversationManager).\nfn thread_source_channel(thread: &Thread) -> Option {\n thread\n .metadata\n .get(\"source_channel\")\n .and_then(|v| v.as_str())\n .map(String::from)\n}\n\n/// Extract and validate user_timezone from thread metadata (set by bridge router).\nfn thread_user_timezone(thread: &Thread) -> Option {\n thread\n .metadata\n .get(\"user_timezone\")\n .and_then(|v| v.as_str())\n .and_then(ValidTimezone::parse)\n}\n\nfn normalize_pause_outcome(\n thread: &mut Thread,\n outcome: &ThreadOutcome,\n) -> Result<(), EngineError> {\n if matches!(outcome, ThreadOutcome::GatePaused { .. }) && thread.state != ThreadState::Waiting {\n thread.transition_to(\n ThreadState::Waiting,\n Some(\"waiting on external gate resolution\".into()),\n )?;\n }\n Ok(())\n}\n\n/// Resource limits for the orchestrator VM.\nfn orchestrator_limits() -> ResourceLimits {\n ResourceLimits::new()\n .max_duration(std::time::Duration::from_secs(300)) // 5 min (longer than user code)\n .max_allocations(5_000_000)\n .max_memory(128 * 1024 * 1024) // 128 MB\n}\n\n/// Maximum consecutive failures before auto-rollback.\nconst MAX_FAILURES_BEFORE_ROLLBACK: u64 = 3;\n\n/// Well-known title for orchestrator failure tracking.\nconst FAILURE_TRACKER_TITLE: &str = \"orchestrator:failures\";\nconst LEASE_REFRESH_WARN_INTERVAL_SECS: u64 = 60;\n\nfn warn_on_lease_refresh_failure(context: &'static str, error: &crate::types::error::EngineError) {\n static LAST_WARN_TS: AtomicU64 = AtomicU64::new(0);\n\n let now = chrono::Utc::now().timestamp().max(0) as u64;\n let last = LAST_WARN_TS.load(Ordering::Relaxed);\n if now.saturating_sub(last) >= LEASE_REFRESH_WARN_INTERVAL_SECS\n && LAST_WARN_TS\n .compare_exchange(last, now, Ordering::Relaxed, Ordering::Relaxed)\n .is_ok()\n {\n warn!(context, error = %error, \"dynamic lease refresh failed\");\n } else {\n debug!(context, error = %error, \"dynamic lease refresh failed\");\n }\n}\n\n/// Load orchestrator code: runtime version from Store, or compiled-in default.\n///\n/// When `allow_self_modify` is false, always uses the compiled-in default\n/// regardless of any runtime versions in the Store. This is the safe default\n/// for production — runtime orchestrator patching is opt-in.\n///\n/// Checks the failure tracker — if the latest version has >= 3 consecutive\n/// failures, falls back to the previous version (or compiled-in default).\npub async fn load_orchestrator(\n store: Option<&Arc>,\n project_id: ProjectId,\n allow_self_modify: bool,\n) -> (String, u64) {\n if !allow_self_modify {\n debug!(\"orchestrator self-modification disabled, using compiled-in default (v0)\");\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n }\n\n let Some(store) = store else {\n debug!(\"using compiled-in default orchestrator (v0, no store)\");\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n };\n\n let docs = match store.list_shared_memory_docs(project_id).await {\n Ok(d) => d,\n Err(_) => {\n debug!(\"using compiled-in default orchestrator (v0, store error)\");\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n }\n };\n\n load_orchestrator_from_docs(&docs, allow_self_modify)\n}\n\n/// Load orchestrator from pre-fetched system memory docs.\n///\n/// When the caller already has the `list_memory_docs` result, use this to\n/// avoid a duplicate Store query. Returns `(code, version)`.\n///\n/// Respects `allow_self_modify` — when false, always returns the compiled-in\n/// default. The caller in `loop_engine.rs` passes this from engine config.\npub fn load_orchestrator_from_docs(\n docs: &[crate::types::memory::MemoryDoc],\n allow_self_modify: bool,\n) -> (String, u64) {\n if !allow_self_modify {\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n }\n\n // Find all orchestrator versions, sorted by version number descending\n let mut versions: Vec<_> = docs\n .iter()\n .filter(|d| d.title == ORCHESTRATOR_TITLE && d.tags.contains(&ORCHESTRATOR_TAG.to_string()))\n .collect();\n versions.sort_by(|a, b| {\n let va = a\n .metadata\n .get(\"version\")\n .and_then(|v| v.as_u64())\n .unwrap_or(0);\n let vb = b\n .metadata\n .get(\"version\")\n .and_then(|v| v.as_u64())\n .unwrap_or(0);\n vb.cmp(&va) // descending\n });\n\n if versions.is_empty() {\n debug!(\"using compiled-in default orchestrator (v0)\");\n return (DEFAULT_ORCHESTRATOR.to_string(), 0);\n }\n\n // Check failure count for the latest version\n let failures = load_failure_count(docs);\n\n for doc in &versions {\n let version = doc\n .metadata\n .get(\"version\")\n .and_then(|v| v.as_u64())\n .unwrap_or(1);\n\n // Skip versions with too many failures (only check the latest)\n if version\n == versions[0]\n .metadata\n .get(\"version\")\n .and_then(|v| v.as_u64())\n .unwrap_or(1)\n && failures >= MAX_FAILURES_BEFORE_ROLLBACK\n {\n debug!(\n version,\n failures, \"orchestrator version has too many failures, skipping\"\n );\n continue;\n }\n\n debug!(version, \"loaded runtime orchestrator\");\n return (doc.content.clone(), version);\n }\n\n // All versions failed — fall back to compiled-in default\n debug!(\"all orchestrator versions failed, using compiled-in default (v0)\");\n (DEFAULT_ORCHESTRATOR.to_string(), 0)\n}\n\n/// Record a failure for the current orchestrator version.\npub async fn record_orchestrator_failure(\n store: &Arc,\n project_id: ProjectId,\n version: u64,\n) {\n use crate::types::memory::{DocType, MemoryDoc};\n\n let docs = match store.list_shared_memory_docs(project_id).await {\n Ok(docs) => docs,\n Err(e) => {\n debug!(\"failed to list memory docs for failure tracker: {e}\");\n return;\n }\n };\n let existing = docs.iter().find(|d| d.title == FAILURE_TRACKER_TITLE);\n\n let mut tracker = if let Some(doc) = existing {\n doc.clone()\n } else {\n MemoryDoc::new(\n project_id,\n shared_owner_id(),\n DocType::Note,\n FAILURE_TRACKER_TITLE,\n \"\",\n )\n .with_tags(vec![\"orchestrator_meta\".to_string()])\n };\n\n // Store failure count as JSON in content: {\"version\": N, \"count\": M}\n let current: serde_json::Value =\n serde_json::from_str(&tracker.content).unwrap_or(serde_json::json!({}));\n let current_version = current.get(\"version\").and_then(|v| v.as_u64()).unwrap_or(0);\n let current_count = current.get(\"count\").and_then(|v| v.as_u64()).unwrap_or(0);\n\n let new_count = if current_version == version {\n current_count + 1\n } else {\n 1 // new version, reset count\n };\n\n tracker.content = serde_json::json!({\n \"version\": version,\n \"count\": new_count,\n })\n .to_string();\n tracker.updated_at = chrono::Utc::now();\n\n if let Err(e) = store.save_memory_doc(&tracker).await {\n debug!(\"failed to save orchestrator failure tracker: {e}\");\n }\n\n debug!(version, count = new_count, \"recorded orchestrator failure\");\n}\n\n/// Reset the failure counter (called after successful execution).\npub async fn reset_orchestrator_failures(store: &Arc, project_id: ProjectId) {\n let docs = store\n .list_shared_memory_docs(project_id)\n .await\n .unwrap_or_default();\n let existing = docs.iter().find(|d| d.title == FAILURE_TRACKER_TITLE);\n\n if let Some(doc) = existing {\n let mut tracker = doc.clone();\n tracker.content = serde_json::json!({\"version\": 0, \"count\": 0}).to_string();\n tracker.updated_at = chrono::Utc::now();\n let _ = store.save_memory_doc(&tracker).await;\n }\n}\n\n/// Load failure count for the latest orchestrator version.\nfn load_failure_count(docs: &[crate::types::memory::MemoryDoc]) -> u64 {\n docs.iter()\n .find(|d| d.title == FAILURE_TRACKER_TITLE)\n .and_then(|d| serde_json::from_str::(&d.content).ok())\n .and_then(|v| v.get(\"count\").and_then(|c| c.as_u64()))\n .unwrap_or(0)\n}\n\n/// Execute the orchestrator Python code with host function dispatch.\n///\n/// This is the core function that replaces `ExecutionLoop::run()`'s inner loop.\n/// The orchestrator Python calls host functions via Monty's suspension mechanism,\n/// and this function handles each suspension by delegating to the appropriate\n/// Rust implementation.\n#[allow(clippy::too_many_arguments)]\npub async fn execute_orchestrator(\n code: &str,\n thread: &mut Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &Arc,\n policy: &Arc,\n signal_rx: &mut SignalReceiver,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n retrieval: Option<&RetrievalEngine>,\n store: Option<&Arc>,\n persisted_state: &serde_json::Value,\n) -> Result {\n let mut total_tokens = TokenUsage::default();\n\n // Build context variables for the orchestrator\n let (input_names, input_values) = build_orchestrator_inputs(thread, persisted_state);\n\n // Parse and compile\n let runner = match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n MontyRun::new(code.to_string(), \"orchestrator.py\", input_names)\n })) {\n Ok(Ok(runner)) => runner,\n Ok(Err(e)) => {\n return Err(EngineError::Effect {\n reason: format!(\"Orchestrator parse error: {e}\"),\n });\n }\n Err(_) => {\n return Err(EngineError::Effect {\n reason: \"Monty VM panicked during orchestrator parsing\".into(),\n });\n }\n };\n\n // Start execution\n let mut stdout = String::new();\n let tracker = LimitedTracker::new(orchestrator_limits());\n\n let run_result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n runner.start(input_values, tracker, PrintWriter::Collect(&mut stdout))\n }));\n\n let mut progress = match run_result {\n Ok(Ok(p)) => p,\n Ok(Err(e)) => {\n return Err(EngineError::Effect {\n reason: format!(\"Orchestrator runtime error: {e}\"),\n });\n }\n Err(_) => {\n return Err(EngineError::Effect {\n reason: \"Monty VM panicked during orchestrator start\".into(),\n });\n }\n };\n\n // Drive the orchestrator dispatch loop\n let mut final_result: Option = None;\n\n loop {\n match progress {\n RunProgress::Complete(obj) => {\n // Use FINAL result if set, otherwise fall back to VM return value\n let result = if let Some(ref fr) = final_result {\n fr.clone()\n } else {\n monty_to_json(&obj)\n };\n sync_runtime_state(thread, result.get(\"state\"));\n let outcome = parse_outcome(&result);\n sync_visible_outcome(thread, &outcome);\n normalize_pause_outcome(thread, &outcome)?;\n return Ok(OrchestratorResult {\n outcome,\n tokens_used: total_tokens,\n });\n }\n\n RunProgress::FunctionCall(call) => {\n let action_name = call.function_name.clone();\n let args = &call.args;\n let kwargs = &call.kwargs;\n\n debug!(action = %action_name, \"orchestrator: host function call\");\n\n let ext_result = match action_name.as_str() {\n // FINAL(result) — orchestrator returns its outcome\n \"FINAL\" => {\n let val = args.first().map(monty_to_json).unwrap_or_default();\n final_result = Some(val);\n ExtFunctionResult::Return(MontyObject::None)\n }\n\n // __llm_complete__(messages, actions, config)\n \"__llm_complete__\" => {\n handle_llm_complete(\n args,\n kwargs,\n thread,\n LlmCompleteDeps {\n llm,\n effects,\n leases,\n store,\n },\n &mut total_tokens,\n )\n .await\n }\n\n // __execute_code_step__(code, state)\n \"__execute_code_step__\" => {\n handle_execute_code_step(\n args, kwargs, thread, llm, effects, leases, policy, event_tx,\n )\n .await\n }\n\n // __execute_action__(name, params, call_id=...)\n \"__execute_action__\" => {\n handle_execute_action(\n args, kwargs, thread, effects, leases, policy, event_tx,\n )\n .await\n }\n\n // __execute_actions_parallel__(calls)\n \"__execute_actions_parallel__\" => {\n handle_execute_actions_parallel(\n args, thread, effects, leases, policy, event_tx,\n )\n .await\n }\n\n // __check_signals__()\n \"__check_signals__\" => handle_check_signals(signal_rx, thread),\n\n // __emit_event__(kind, **data)\n \"__emit_event__\" => handle_emit_event(args, kwargs, thread, event_tx),\n\n // __save_checkpoint__(state, counters)\n \"__save_checkpoint__\" => handle_save_checkpoint(args, kwargs, thread),\n\n // __transition_to__(state, reason)\n \"__transition_to__\" => handle_transition_to(args, kwargs, thread),\n\n // __retrieve_docs__(goal, max_docs)\n \"__retrieve_docs__\" => {\n handle_retrieve_docs(args, kwargs, thread, retrieval).await\n }\n\n // __check_budget__()\"\n \"__check_budget__\" => handle_check_budget(thread),\n\n // __get_actions__()\n \"__get_actions__\" => handle_get_actions(thread, effects, leases, store).await,\n\n // __list_skills__(max_candidates, max_tokens)\n \"__list_skills__\" => handle_list_skills(args, thread, store).await,\n\n // __record_skill_usage__(doc_id, success)\n \"__record_skill_usage__\" => handle_record_skill_usage(args, store).await,\n\n // __regex_match__(pattern, text) -> bool\n // Evaluates a regex against text using Rust's regex crate.\n // Invalid patterns return False silently. Monty has no `re`\n // module, so this host function bridges the gap for the\n // skill selector's pattern-based scoring.\n \"__regex_match__\" => handle_regex_match(args),\n\n // __set_active_skills__(skills)\n \"__set_active_skills__\" => handle_set_active_skills(args, thread),\n\n // Unknown — let Monty resolve it (user-defined functions, builtins)\n other => ExtFunctionResult::NotFound(other.to_string()),\n };\n\n // Resume the orchestrator VM\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume(ext_result, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n return Err(EngineError::Effect {\n reason: format!(\"Orchestrator error after resume: {e}\"),\n });\n }\n Err(_) => {\n return Err(EngineError::Effect {\n reason: \"Monty VM panicked during orchestrator resume\".into(),\n });\n }\n }\n\n // If FINAL was called, the VM should complete on next iteration\n if final_result.is_some() {\n continue;\n }\n }\n\n RunProgress::NameLookup(lookup) => {\n // Undefined variable — resume with NameError\n let name = lookup.name.clone();\n debug!(name = %name, \"orchestrator: unresolved name\");\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n lookup.resume(\n NameLookupResult::Undefined,\n PrintWriter::Collect(&mut stdout),\n )\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n return Err(EngineError::Effect {\n reason: format!(\"Orchestrator NameError '{name}': {e}\"),\n });\n }\n Err(_) => {\n return Err(EngineError::Effect {\n reason: format!(\"Monty panic on NameLookup '{name}'\"),\n });\n }\n }\n }\n\n RunProgress::OsCall(_) => {\n return Err(EngineError::Effect {\n reason: \"Orchestrator attempted OS call (blocked)\".into(),\n });\n }\n\n RunProgress::ResolveFutures(_) => {\n return Err(EngineError::Effect {\n reason: \"Orchestrator attempted async (not supported)\".into(),\n });\n }\n }\n }\n}\n\n// ── Host function handlers ──────────────────────────────────\n\nstruct LlmCompleteDeps<'a> {\n llm: &'a Arc,\n effects: &'a Arc,\n leases: &'a Arc,\n store: Option<&'a Arc>,\n}\n\n/// Handle `__llm_complete__(messages, actions, config)`.\n///\n/// Calls the LLM and returns the response as a dict:\n/// `{type: \"text\"|\"code\"|\"actions\", content/code/calls: ..., usage: {...}}`\n///\nasync fn handle_llm_complete(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n deps: LlmCompleteDeps<'_>,\n total_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n use crate::types::step::LlmResponse;\n\n let explicit_messages = args.first().map(monty_to_json).filter(|v| !v.is_null());\n let explicit_config = args.get(2).map(monty_to_json).filter(|v| !v.is_null());\n let messages = explicit_messages\n .as_ref()\n .and_then(json_to_thread_messages)\n .unwrap_or_else(|| thread.messages.clone());\n\n if let Err(e) = reconcile_dynamic_tool_lease(\n thread,\n deps.effects,\n deps.leases,\n deps.store,\n &crate::LeasePlanner::new(),\n )\n .await\n {\n warn_on_lease_refresh_failure(\"llm_complete\", &e);\n }\n\n let active_leases = deps.leases.active_for_thread(thread.id).await;\n let actions = deps\n .effects\n .available_actions(&active_leases)\n .await\n .unwrap_or_default();\n\n let config = LlmCallConfig {\n max_tokens: explicit_config\n .as_ref()\n .and_then(|cfg| cfg.get(\"max_tokens\"))\n .and_then(|v| v.as_u64())\n .and_then(|v| u32::try_from(v).ok()),\n temperature: explicit_config\n .as_ref()\n .and_then(|cfg| cfg.get(\"temperature\"))\n .and_then(|v| v.as_f64())\n .map(|v| v as f32),\n force_text: explicit_config\n .as_ref()\n .and_then(|cfg| cfg.get(\"force_text\"))\n .and_then(|v| v.as_bool())\n .unwrap_or(false),\n depth: thread.config.depth,\n model: explicit_config\n .as_ref()\n .and_then(|cfg| cfg.get(\"model\"))\n .and_then(|v| v.as_str())\n .map(String::from),\n metadata: HashMap::new(),\n };\n\n match deps.llm.complete(&messages, &actions, &config).await {\n Ok(output) => {\n total_tokens.input_tokens += output.usage.input_tokens;\n total_tokens.output_tokens += output.usage.output_tokens;\n total_tokens.cost_usd += output.usage.cost_usd;\n\n let usage = serde_json::json!({\n \"input_tokens\": output.usage.input_tokens,\n \"output_tokens\": output.usage.output_tokens,\n \"cost_usd\": output.usage.cost_usd,\n });\n\n let result = match output.response {\n LlmResponse::Text(text) => {\n serde_json::json!({\"type\": \"text\", \"content\": text, \"usage\": usage})\n }\n LlmResponse::Code { code, .. } => {\n serde_json::json!({\"type\": \"code\", \"code\": code, \"usage\": usage})\n }\n LlmResponse::ActionCalls { calls, content } => {\n // Single source of truth for the Python interchange\n // shape — must round-trip via `python_json_to_action_calls`.\n let calls_json = action_calls_to_python_json(&calls);\n serde_json::json!({\n \"type\": \"actions\",\n \"content\": content,\n \"calls\": calls_json,\n \"usage\": usage\n })\n }\n };\n\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n Err(e) => ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::RuntimeError,\n Some(format!(\"LLM call failed: {e}\")),\n )),\n }\n}\n\n/// Handle `__execute_code_step__(code, state)`.\n///\n/// Runs user CodeAct code in a nested Monty VM with full tool dispatch.\n/// Returns a dict with stdout, return_value, action_results, etc.\n#[allow(clippy::too_many_arguments)]\nasync fn handle_execute_code_step(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &Arc,\n policy: &Arc,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n) -> ExtFunctionResult {\n let code = match args.first() {\n Some(obj) => monty_to_string(obj),\n None => {\n return ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::TypeError,\n Some(\"__execute_code_step__ requires a code string\".into()),\n ));\n }\n };\n\n let state = args\n .get(1)\n .map(monty_to_json)\n .unwrap_or(serde_json::json!({}));\n\n let exec_ctx = ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: thread.user_id.clone(),\n step_id: StepId::new(),\n current_call_id: None,\n source_channel: thread_source_channel(thread),\n user_timezone: thread_user_timezone(thread),\n };\n\n // Run user code in a nested Monty VM (same pattern as rlm_query)\n let code_start = std::time::Instant::now();\n match Box::pin(execute_code(\n &code,\n thread,\n llm,\n effects,\n leases,\n policy,\n &exec_ctx,\n &[],\n &state,\n ))\n .await\n {\n Ok(result) => {\n // Broadcast events from code execution to the thread and event channel.\n // Without this, ActionExecuted events from CodeAct tool calls are lost\n // and never appear in traces.\n for event_kind in &result.events {\n let event = ThreadEvent::new(thread.id, event_kind.clone());\n if let Some(tx) = event_tx {\n let _ = tx.send(event.clone());\n }\n thread.events.push(event);\n }\n // If the CodeAct snippet itself failed (Python SyntaxError, runtime\n // error, etc.), surface it as an ActionFailed event so traces and\n // observers see the failure. Without this, parse errors silently\n // fall back to the LLM via the result dict and never warn callers.\n if let Some(ref category) = result.failure {\n let error_msg = if !result.stdout.is_empty() {\n format!(\n \"CodeAct execution failed: {}\",\n tail_chars(&result.stdout, 500)\n )\n } else {\n \"CodeAct execution failed (no stdout)\".to_string()\n };\n let failed_event = ThreadEvent::new(\n thread.id,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: \"__codeact__\".to_string(),\n // Synthetic call_id derived from the step id —\n // CodeAct snippet failures don't have an LLM-provided\n // call_id, but `loop_engine.rs:1277` asserts that\n // ActionFailed events carry a non-empty call_id for\n // trace correlation.\n call_id: format!(\"codeact-step-{}\", exec_ctx.step_id.0),\n error: error_msg,\n params_summary: None,\n },\n );\n if let Some(tx) = event_tx {\n let _ = tx.send(failed_event.clone());\n }\n thread.events.push(failed_event);\n\n // Emit structured CodeExecutionFailed event for instrumentation.\n // This enables aggregate analysis of WHY code execution fails\n // (Monty limitation vs LLM logic error vs tool dispatch failure).\n let error_text = tail_chars(&result.stdout, 500);\n let instrumentation_event = ThreadEvent::new(\n thread.id,\n EventKind::CodeExecutionFailed {\n step_id: exec_ctx.step_id,\n category: category.clone(),\n error: error_text,\n code_hash: Some(crate::executor::scripting::code_hash(&code)),\n duration_ms: code_start.elapsed().as_millis() as u64,\n },\n );\n if let Some(tx) = event_tx {\n let _ = tx.send(instrumentation_event.clone());\n }\n thread.events.push(instrumentation_event);\n }\n thread.updated_at = chrono::Utc::now();\n\n let action_results: Vec = result\n .action_results\n .iter()\n .map(|r| {\n serde_json::json!({\n \"action_name\": r.action_name,\n \"output\": r.output,\n \"is_error\": r.is_error,\n \"duration_ms\": r.duration.as_millis(),\n })\n })\n .collect();\n\n let result_json = serde_json::json!({\n \"return_value\": result.return_value,\n \"stdout\": result.stdout,\n \"action_results\": action_results,\n \"final_answer\": result.final_answer,\n \"had_error\": result.failure.is_some(),\n \"pending_gate\": result.need_approval.as_ref().map(|na| {\n match na {\n ThreadOutcome::GatePaused { gate_name, action_name, call_id, parameters, resume_kind, resume_output } => serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": gate_name,\n \"action_name\": action_name,\n \"call_id\": call_id,\n \"parameters\": parameters,\n \"resume_kind\": serde_json::to_value(resume_kind).unwrap_or_default(),\n \"resume_output\": resume_output,\n }),\n _ => serde_json::Value::Null,\n }\n }),\n });\n\n ExtFunctionResult::Return(json_to_monty(&result_json))\n }\n Err(e) => ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::RuntimeError,\n Some(format!(\"Code execution failed: {e}\")),\n )),\n }\n}\n\n/// Handle `__execute_action__(name, params, call_id=...)`.\n///\n/// Single source of truth for action execution. Performs:\n/// 1. Lease lookup\n/// 2. Policy check\n/// 3. Lease consumption\n/// 4. Action execution via EffectExecutor\n/// 5. Event emission (ActionExecuted/ActionFailed)\n///\n/// Python owns the working transcript and decides how tool outputs are\n/// represented in internal message history.\nasync fn handle_execute_action(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n effects: &Arc,\n leases: &Arc,\n policy: &Arc,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n) -> ExtFunctionResult {\n let name = match extract_string_arg(args, kwargs, \"name\", 0) {\n Some(n) => n,\n None => {\n return ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::TypeError,\n Some(\"__execute_action__ requires a name argument\".into()),\n ));\n }\n };\n\n let params = args\n .get(1)\n .map(monty_to_json)\n .unwrap_or(serde_json::json!({}));\n\n let call_id = extract_string_kwarg(kwargs, \"call_id\").unwrap_or_default();\n\n let exec_ctx = ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: thread.user_id.clone(),\n step_id: StepId::new(),\n current_call_id: Some(call_id.clone()),\n source_channel: thread_source_channel(thread),\n user_timezone: thread_user_timezone(thread),\n };\n\n // Helper: emit event only. The orchestrator owns transcript recording.\n let emit_and_record = |thread: &mut Thread,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n event_kind: EventKind,\n _call_id: &str,\n _action_name: &str,\n _output: &serde_json::Value| {\n let event = ThreadEvent::new(thread.id, event_kind);\n if let Some(tx) = event_tx {\n let _ = tx.send(event.clone());\n }\n thread.events.push(event);\n thread.updated_at = chrono::Utc::now();\n };\n\n // 1. Find lease for this action\n let lease = match leases.find_lease_for_action(thread.id, &name).await {\n Some(l) => l,\n None => {\n let error = format!(\"No lease for action '{name}'\");\n let output = serde_json::json!({\"error\": &error});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error,\n params_summary: None,\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"output\": output,\n \"is_error\": true,\n });\n return ExtFunctionResult::Return(json_to_monty(&result));\n }\n };\n\n // 2. Check policy\n let action_def = effects\n .available_actions(std::slice::from_ref(&lease))\n .await\n .ok()\n .and_then(|actions| actions.into_iter().find(|a| a.name == name));\n\n if let Some(ref ad) = action_def {\n match policy.evaluate(ad, &lease, &[]) {\n crate::capability::policy::PolicyDecision::Deny { reason } => {\n let output = serde_json::json!({\"error\": format!(\"Denied: {reason}\")});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error: reason,\n params_summary: None,\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"output\": output,\n \"is_error\": true,\n });\n return ExtFunctionResult::Return(json_to_monty(&result));\n }\n crate::capability::policy::PolicyDecision::RequireApproval { .. } => {\n let output = serde_json::json!({\"status\": \"gate_paused\", \"gate_name\": \"approval\"});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ApprovalRequested {\n action_name: name.clone(),\n call_id: call_id.clone(),\n parameters: Some(params.clone()),\n description: None,\n allow_always: None,\n gate_name: None,\n params_summary: summarize_params(&name, ¶ms),\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": \"approval\",\n \"action_name\": name,\n \"call_id\": call_id,\n \"parameters\": params,\n \"resume_kind\": serde_json::to_value(crate::gate::ResumeKind::Approval {\n allow_always: true,\n })\n .unwrap_or_default(),\n });\n return ExtFunctionResult::Return(json_to_monty(&result));\n }\n crate::capability::policy::PolicyDecision::Allow => {}\n }\n }\n\n // 3. Atomically re-find + consume a lease use under a single write\n // lock. This closes the TOCTOU window between the read-only\n // `find_lease_for_action` (used above for the policy check) and the\n // consume — without it, two concurrent calls could both observe a\n // lease with one remaining use and both proceed to execute. Mirrors\n // `structured.rs::execute_action_batch_with_results`.\n let lease = match leases.find_and_consume(thread.id, &name).await {\n Ok(l) => l,\n Err(e) => {\n debug!(error = %e, \"atomic lease find_and_consume failed\");\n let error = format!(\"lease consumption failed for action '{name}': {e}\");\n let output = serde_json::json!({\"error\": &error});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error,\n params_summary: None,\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"output\": output,\n \"is_error\": true,\n });\n return ExtFunctionResult::Return(json_to_monty(&result));\n }\n };\n\n // 4. Execute\n let ps = summarize_params(&name, ¶ms);\n match effects\n .execute_action(&name, params, &lease, &exec_ctx)\n .await\n {\n Ok(r) => {\n // Effect adapters wrap tool errors as `Ok(ActionResult { is_error: true })`\n // — surface them as `ActionFailed` so traces and observers see the\n // failure. See `resolve_tool_future` in `scripting.rs` for the same\n // pattern on the structured-tool path.\n if r.is_error {\n let error_msg = r\n .output\n .get(\"error\")\n .and_then(|v| v.as_str())\n .map(String::from)\n .unwrap_or_else(|| r.output.to_string());\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error: error_msg,\n params_summary: ps.clone(),\n },\n &call_id,\n &name,\n &r.output,\n );\n } else {\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionExecuted {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n duration_ms: r.duration.as_millis() as u64,\n params_summary: ps.clone(),\n },\n &call_id,\n &name,\n &r.output,\n );\n }\n let result = serde_json::json!({\n \"action_name\": r.action_name,\n \"output\": r.output,\n \"is_error\": r.is_error,\n \"duration_ms\": r.duration.as_millis(),\n });\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n Err(EngineError::GatePaused {\n gate_name,\n action_name: _,\n call_id: _,\n parameters,\n resume_kind,\n resume_output,\n }) => {\n let _ = leases.refund_use(lease.id).await;\n let output = serde_json::json!({\"status\": \"gate_paused\", \"gate_name\": gate_name});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ApprovalRequested {\n action_name: name.clone(),\n call_id: call_id.clone(),\n parameters: Some((*parameters).clone()),\n description: None,\n allow_always: match resume_kind.as_ref() {\n crate::gate::ResumeKind::Approval { allow_always } => Some(*allow_always),\n _ => None,\n },\n gate_name: Some(gate_name.clone()),\n params_summary: summarize_params(&name, ¶meters),\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": gate_name,\n \"action_name\": name,\n \"call_id\": call_id,\n \"parameters\": parameters,\n \"resume_kind\": serde_json::to_value(&*resume_kind).unwrap_or_default(),\n \"resume_output\": resume_output,\n });\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n Err(e) => {\n let output = serde_json::json!({\"error\": e.to_string()});\n emit_and_record(\n thread,\n event_tx,\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.clone(),\n call_id: call_id.clone(),\n error: e.to_string(),\n params_summary: ps,\n },\n &call_id,\n &name,\n &output,\n );\n let result = serde_json::json!({\n \"output\": output,\n \"is_error\": true,\n });\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n }\n}\n\n/// Handle `__execute_actions_parallel__(calls)`.\n///\n/// Batch host function that receives a list of action calls and executes them\n/// concurrently. Each call is a dict with `name`, `params`, and optionally `call_id`.\n///\n/// Returns a list of result dicts (one per call, in order). Each result has the\n/// same shape as `__execute_action__` output, plus an optional gate pause payload.\n///\n/// Events are emitted in original call order after all parallel executions complete.\nasync fn handle_execute_actions_parallel(\n args: &[MontyObject],\n thread: &mut Thread,\n effects: &Arc,\n leases: &Arc,\n policy: &Arc,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n) -> ExtFunctionResult {\n // Parse the calls list from the first argument (list of dicts)\n let calls_json = args\n .first()\n .map(monty_to_json)\n .unwrap_or(serde_json::json!([]));\n let calls_array = match calls_json.as_array() {\n Some(arr) => arr.clone(),\n None => {\n return ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::TypeError,\n Some(\"__execute_actions_parallel__ requires a list of call dicts\".into()),\n ));\n }\n };\n\n if calls_array.is_empty() {\n return ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])));\n }\n\n // Parse each call dict into (name, params, call_id)\n struct ParsedCall {\n name: String,\n params: serde_json::Value,\n call_id: String,\n }\n\n let mut parsed: Vec = Vec::with_capacity(calls_array.len());\n for c in &calls_array {\n let name = c\n .get(\"name\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"\")\n .to_string();\n let params = c.get(\"params\").cloned().unwrap_or(serde_json::json!({}));\n let call_id = c\n .get(\"call_id\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"\")\n .to_string();\n parsed.push(ParsedCall {\n name,\n params,\n call_id,\n });\n }\n\n let step_id = StepId::new();\n\n // ── Phase 1: Preflight (sequential) ─────────────────────────\n // Check leases and policies. Denied → error result. Approval → interrupt.\n\n enum PfOutcome {\n Runnable {\n lease: crate::types::capability::CapabilityLease,\n },\n Error {\n result_json: serde_json::Value,\n event: EventKind,\n output: serde_json::Value,\n },\n }\n\n let mut preflight: Vec> = Vec::with_capacity(parsed.len());\n\n for pc in &parsed {\n // Find lease\n let lease = match leases.find_lease_for_action(thread.id, &pc.name).await {\n Some(l) => l,\n None => {\n let error = format!(\"No lease for action '{}'\", pc.name);\n let output = serde_json::json!({\"error\": &error});\n let result_json = serde_json::json!({\n \"output\": &output,\n \"is_error\": true,\n });\n let event = EventKind::ActionFailed {\n step_id,\n action_name: pc.name.clone(),\n call_id: pc.call_id.clone(),\n error,\n params_summary: None,\n };\n preflight.push(Some(PfOutcome::Error {\n result_json,\n event,\n output,\n }));\n continue;\n }\n };\n\n // Check policy\n let action_def = effects\n .available_actions(std::slice::from_ref(&lease))\n .await\n .ok()\n .and_then(|actions| actions.into_iter().find(|a| a.name == pc.name));\n\n if let Some(ref ad) = action_def {\n match policy.evaluate(ad, &lease, &[]) {\n crate::capability::policy::PolicyDecision::Deny { reason } => {\n let output = serde_json::json!({\"error\": format!(\"Denied: {reason}\")});\n let result_json = serde_json::json!({\n \"output\": &output,\n \"is_error\": true,\n });\n let event = EventKind::ActionFailed {\n step_id,\n action_name: pc.name.clone(),\n call_id: pc.call_id.clone(),\n error: reason,\n params_summary: None,\n };\n preflight.push(Some(PfOutcome::Error {\n result_json,\n event,\n output,\n }));\n continue;\n }\n crate::capability::policy::PolicyDecision::RequireApproval { .. } => {\n // Emit events for earlier errors, then interrupt\n let mut results_json = Vec::with_capacity(preflight.len() + 1);\n for pf in preflight {\n match pf {\n Some(PfOutcome::Error {\n result_json,\n event,\n output: _,\n }) => {\n let ev = ThreadEvent::new(thread.id, event);\n if let Some(tx) = event_tx {\n let _ = tx.send(ev.clone());\n }\n thread.events.push(ev);\n results_json.push(result_json);\n }\n Some(PfOutcome::Runnable { .. }) | None => {\n results_json.push(serde_json::json!(null));\n }\n }\n }\n // Add the approval entry\n let ev = ThreadEvent::new(\n thread.id,\n EventKind::ApprovalRequested {\n action_name: pc.name.clone(),\n call_id: pc.call_id.clone(),\n parameters: Some(pc.params.clone()),\n description: None,\n allow_always: None,\n gate_name: None,\n params_summary: summarize_params(&pc.name, &pc.params),\n },\n );\n if let Some(tx) = event_tx {\n let _ = tx.send(ev.clone());\n }\n thread.events.push(ev);\n thread.updated_at = chrono::Utc::now();\n\n results_json.push(serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": \"approval\",\n \"action_name\": &pc.name,\n \"call_id\": &pc.call_id,\n \"parameters\": &pc.params,\n \"resume_kind\": serde_json::to_value(crate::gate::ResumeKind::Approval {\n allow_always: true,\n })\n .unwrap_or_default(),\n }));\n // Pad with nulls for calls that weren't reached so the\n // Python-side loop can emit ActionResult placeholders for\n // every tool call in the assistant message.\n while results_json.len() < parsed.len() {\n results_json.push(serde_json::json!(null));\n }\n return ExtFunctionResult::Return(json_to_monty(&serde_json::json!(\n results_json\n )));\n }\n crate::capability::policy::PolicyDecision::Allow => {}\n }\n }\n\n // Atomically re-find + consume a lease use under a single write\n // lock, closing the TOCTOU window between the read-only\n // `find_lease_for_action` above and the consume. Mirrors\n // `structured.rs::execute_action_batch_with_results`.\n let lease = match leases.find_and_consume(thread.id, &pc.name).await {\n Ok(l) => l,\n Err(e) => {\n debug!(error = %e, \"atomic lease find_and_consume failed\");\n let error = format!(\"lease consumption failed for action '{}': {e}\", pc.name);\n let output = serde_json::json!({\"error\": &error});\n let result_json = serde_json::json!({\n \"output\": &output,\n \"is_error\": true,\n });\n let event = EventKind::ActionFailed {\n step_id,\n action_name: pc.name.clone(),\n call_id: pc.call_id.clone(),\n error,\n params_summary: None,\n };\n preflight.push(Some(PfOutcome::Error {\n result_json,\n event,\n output,\n }));\n continue;\n }\n };\n\n preflight.push(Some(PfOutcome::Runnable { lease }));\n }\n\n // ── Phase 2: Execute in parallel ────────────────────────────\n\n // Slot array: index → execution result\n let mut slot_results: Vec> = vec![None; parsed.len()];\n let mut slot_events: Vec> = vec![None; parsed.len()];\n let mut slot_outputs: Vec> = vec![None; parsed.len()];\n\n // Separate runnable from errors\n let mut runnable: Vec<(usize, crate::types::capability::CapabilityLease)> = Vec::new();\n for (idx, pf) in preflight.into_iter().enumerate() {\n match pf {\n Some(PfOutcome::Error {\n result_json,\n event,\n output,\n }) => {\n slot_results[idx] = Some(result_json);\n slot_events[idx] = Some(event);\n slot_outputs[idx] = Some(output);\n }\n Some(PfOutcome::Runnable { lease }) => {\n runnable.push((idx, lease));\n }\n None => {}\n }\n }\n\n if runnable.len() == 1 {\n // Single call: execute directly\n let (idx, lease) = runnable.into_iter().next().unwrap(); // safety: len()==1 checked above\n let pc = &parsed[idx];\n let exec_ctx = ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: thread.user_id.clone(),\n step_id,\n current_call_id: Some(pc.call_id.clone()),\n // Read source_channel from thread metadata so downstream tools\n // (e.g. mission_create) can default notify_channels to the\n // originating channel. Hardcoding `None` here was a bug — it\n // silently dropped the gateway routing for any tool dispatched\n // through the parallel batch path.\n source_channel: thread_source_channel(thread),\n user_timezone: thread_user_timezone(thread),\n };\n let ps = summarize_params(&pc.name, &pc.params);\n let (result_json, event, output) = execute_single_action(\n effects,\n &pc.name,\n pc.params.clone(),\n &pc.call_id,\n &lease,\n &exec_ctx,\n ps,\n )\n .await;\n if interrupted_result_needs_refund(&result_json) {\n let _ = leases.refund_use(lease.id).await;\n }\n slot_results[idx] = Some(result_json);\n slot_events[idx] = Some(event);\n slot_outputs[idx] = Some(output);\n } else if runnable.len() > 1 {\n // Multiple calls: execute in parallel via JoinSet\n let mut join_set = tokio::task::JoinSet::new();\n let effects = effects.clone();\n // Capture once outside the loop — the thread's metadata is stable\n // for the duration of the parallel batch.\n let parallel_source_channel = thread_source_channel(thread);\n let parallel_user_timezone = thread_user_timezone(thread);\n\n for (idx, lease) in runnable {\n let pc_name = parsed[idx].name.clone();\n let pc_params = parsed[idx].params.clone();\n let pc_call_id = parsed[idx].call_id.clone();\n let effects = effects.clone();\n let lease = lease.clone();\n let exec_ctx = ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: thread.user_id.clone(),\n step_id,\n current_call_id: Some(pc_call_id.clone()),\n // See comment above — read from thread metadata, not None.\n source_channel: parallel_source_channel.clone(),\n user_timezone: parallel_user_timezone,\n };\n let ps = summarize_params(&pc_name, &pc_params);\n\n join_set.spawn(async move {\n let (result_json, event, output) = execute_single_action(\n &effects,\n &pc_name,\n pc_params,\n &pc_call_id,\n &lease,\n &exec_ctx,\n ps,\n )\n .await;\n (idx, lease.id, result_json, event, output)\n });\n }\n\n while let Some(join_result) = join_set.join_next().await {\n match join_result {\n Ok((idx, lease_id, result_json, event, output)) => {\n if interrupted_result_needs_refund(&result_json) {\n let _ = leases.refund_use(lease_id).await;\n }\n slot_results[idx] = Some(result_json);\n slot_events[idx] = Some(event);\n slot_outputs[idx] = Some(output);\n }\n Err(e) => {\n debug!(\"parallel action execution task panicked: {e}\");\n }\n }\n }\n }\n\n // ── Phase 3: Emit events in order ───────────────────────────\n\n let mut results_json = Vec::with_capacity(parsed.len());\n for idx in 0..parsed.len() {\n let result_json = slot_results[idx].take().unwrap_or(\n serde_json::json!({\"is_error\": true, \"output\": {\"error\": \"execution slot empty\"}}),\n );\n let _output = slot_outputs[idx]\n .take()\n .unwrap_or(serde_json::json!({\"error\": \"no output\"}));\n\n if let Some(event) = slot_events[idx].take() {\n let ev = ThreadEvent::new(thread.id, event);\n if let Some(tx) = event_tx {\n let _ = tx.send(ev.clone());\n }\n thread.events.push(ev);\n }\n\n results_json.push(result_json.clone());\n }\n\n thread.updated_at = chrono::Utc::now();\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!(results_json)))\n}\n\n/// Execute a single action and return (result_json, event, output) for the\n/// batch handler to record. Shared by both single-call and parallel paths.\nasync fn execute_single_action(\n effects: &Arc,\n name: &str,\n params: serde_json::Value,\n call_id: &str,\n lease: &crate::types::capability::CapabilityLease,\n exec_ctx: &ThreadExecutionContext,\n params_summary: Option,\n) -> (serde_json::Value, EventKind, serde_json::Value) {\n match effects.execute_action(name, params, lease, exec_ctx).await {\n Ok(r) => {\n // Surface wrapped errors as ActionFailed (see resolve_tool_future\n // and the parallel execute path for the same pattern).\n let event = if r.is_error {\n let error_msg = r\n .output\n .get(\"error\")\n .and_then(|v| v.as_str())\n .map(String::from)\n .unwrap_or_else(|| r.output.to_string());\n EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.to_string(),\n call_id: call_id.to_string(),\n error: error_msg,\n params_summary: params_summary.clone(),\n }\n } else {\n EventKind::ActionExecuted {\n step_id: exec_ctx.step_id,\n action_name: name.to_string(),\n call_id: call_id.to_string(),\n duration_ms: r.duration.as_millis() as u64,\n params_summary: params_summary.clone(),\n }\n };\n let result_json = serde_json::json!({\n \"action_name\": r.action_name,\n \"output\": r.output,\n \"is_error\": r.is_error,\n \"duration_ms\": r.duration.as_millis(),\n });\n (result_json, event, r.output)\n }\n Err(EngineError::GatePaused {\n gate_name,\n action_name: _,\n call_id: _,\n parameters,\n resume_kind,\n resume_output,\n }) => {\n let output = serde_json::json!({\"status\": \"gate_paused\", \"gate_name\": &gate_name});\n let event = EventKind::ApprovalRequested {\n action_name: name.to_string(),\n call_id: call_id.to_string(),\n parameters: Some((*parameters).clone()),\n description: None,\n allow_always: match resume_kind.as_ref() {\n crate::gate::ResumeKind::Approval { allow_always } => Some(*allow_always),\n _ => None,\n },\n gate_name: Some(gate_name.clone()),\n params_summary: summarize_params(name, ¶meters),\n };\n let result_json = serde_json::json!({\n \"gate_paused\": true,\n \"gate_name\": gate_name,\n \"action_name\": name,\n \"call_id\": call_id,\n \"parameters\": parameters,\n \"resume_kind\": serde_json::to_value(&*resume_kind).unwrap_or_default(),\n \"resume_output\": resume_output,\n });\n (result_json, event, output)\n }\n Err(e) => {\n let output = serde_json::json!({\"error\": e.to_string()});\n let event = EventKind::ActionFailed {\n step_id: exec_ctx.step_id,\n action_name: name.to_string(),\n call_id: call_id.to_string(),\n error: e.to_string(),\n params_summary,\n };\n let result_json = serde_json::json!({\n \"output\": &output,\n \"is_error\": true,\n });\n (result_json, event, output)\n }\n }\n}\n\nfn interrupted_result_needs_refund(result: &serde_json::Value) -> bool {\n result.get(\"gate_paused\").and_then(|v| v.as_bool()) == Some(true)\n}\n\n/// Handle `__check_signals__()`.\nfn handle_check_signals(signal_rx: &mut SignalReceiver, thread: &mut Thread) -> ExtFunctionResult {\n match signal_rx.try_recv() {\n Ok(ThreadSignal::Stop) | Ok(ThreadSignal::Suspend) => {\n ExtFunctionResult::Return(MontyObject::String(\"stop\".into()))\n }\n Ok(ThreadSignal::InjectMessage(msg)) => {\n thread.add_message(msg.clone());\n let result = serde_json::json!({\"inject\": msg.content});\n ExtFunctionResult::Return(json_to_monty(&result))\n }\n Ok(ThreadSignal::Resume) | Ok(ThreadSignal::ChildCompleted { .. }) => {\n ExtFunctionResult::Return(MontyObject::None)\n }\n Err(_) => ExtFunctionResult::Return(MontyObject::None),\n }\n}\n\n/// Handle `__emit_event__(kind, **data)`.\nfn handle_emit_event(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n event_tx: Option<&tokio::sync::broadcast::Sender>,\n) -> ExtFunctionResult {\n let kind_str = args.first().map(monty_to_string).unwrap_or_default();\n\n let kind = match kind_str.as_str() {\n \"step_started\" => {\n let _step = extract_u64_kwarg(kwargs, \"step\").unwrap_or(0);\n EventKind::StepStarted {\n step_id: StepId::new(),\n }\n }\n \"step_completed\" => {\n let input = extract_u64_kwarg(kwargs, \"input_tokens\").unwrap_or(0);\n let output = extract_u64_kwarg(kwargs, \"output_tokens\").unwrap_or(0);\n // Increment step count (mirrors the old Rust loop's step_count += 1)\n thread.step_count += 1;\n // Track token usage\n thread.total_tokens_used += input + output;\n EventKind::StepCompleted {\n step_id: StepId::new(),\n tokens: TokenUsage {\n input_tokens: input,\n output_tokens: output,\n ..Default::default()\n },\n }\n }\n \"action_executed\" => {\n let action_name = extract_string_kwarg(kwargs, \"action_name\").unwrap_or_default();\n let call_id = extract_string_kwarg(kwargs, \"call_id\").unwrap_or_default();\n EventKind::ActionExecuted {\n step_id: StepId::new(),\n action_name,\n call_id,\n duration_ms: 0,\n params_summary: None,\n }\n }\n \"action_failed\" => {\n let action_name = extract_string_kwarg(kwargs, \"action_name\").unwrap_or_default();\n let call_id = extract_string_kwarg(kwargs, \"call_id\").unwrap_or_default();\n let error = extract_string_kwarg(kwargs, \"error\").unwrap_or_default();\n EventKind::ActionFailed {\n step_id: StepId::new(),\n action_name,\n call_id,\n error,\n params_summary: None,\n }\n }\n \"skill_activated\" => {\n let names_str = extract_string_kwarg(kwargs, \"skill_names\").unwrap_or_default();\n let skill_names: Vec = names_str\n .split(',')\n .map(|s| s.trim().to_string())\n .filter(|s| !s.is_empty())\n .collect();\n EventKind::SkillActivated { skill_names }\n }\n _ => {\n debug!(kind = %kind_str, \"orchestrator: unknown event kind, skipping\");\n return ExtFunctionResult::Return(MontyObject::None);\n }\n };\n\n let event = ThreadEvent::new(thread.id, kind);\n if let Some(tx) = event_tx {\n let _ = tx.send(event.clone());\n }\n thread.events.push(event);\n thread.updated_at = chrono::Utc::now();\n\n ExtFunctionResult::Return(MontyObject::None)\n}\n\n/// Handle `__save_checkpoint__(state, counters)`.\nfn handle_save_checkpoint(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n) -> ExtFunctionResult {\n let state = args\n .first()\n .map(monty_to_json)\n .unwrap_or(serde_json::json!({}));\n let counters = args\n .get(1)\n .map(monty_to_json)\n .unwrap_or(serde_json::json!({}));\n\n sync_runtime_state(thread, Some(&state));\n\n if let Some(metadata) = thread.metadata.as_object_mut() {\n metadata.insert(\n \"runtime_checkpoint\".into(),\n serde_json::json!({\n \"persisted_state\": state,\n \"nudge_count\": counters.get(\"nudge_count\").and_then(|v| v.as_u64()).unwrap_or(0),\n \"consecutive_errors\": counters.get(\"consecutive_errors\").and_then(|v| v.as_u64()).unwrap_or(0),\n \"consecutive_action_errors\": counters.get(\"consecutive_action_errors\").and_then(|v| v.as_u64()).unwrap_or(0),\n \"compaction_count\": counters.get(\"compaction_count\").and_then(|v| v.as_u64()).unwrap_or(0),\n }),\n );\n }\n thread.updated_at = chrono::Utc::now();\n\n ExtFunctionResult::Return(MontyObject::None)\n}\n\n/// Handle `__transition_to__(state, reason)`.\nfn handle_transition_to(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &mut Thread,\n) -> ExtFunctionResult {\n let state_str = args.first().map(monty_to_string).unwrap_or_default();\n let reason = args.get(1).map(monty_to_string);\n\n let target = match state_str.as_str() {\n \"running\" => crate::types::thread::ThreadState::Running,\n \"completed\" => crate::types::thread::ThreadState::Completed,\n \"failed\" => crate::types::thread::ThreadState::Failed,\n \"waiting\" => crate::types::thread::ThreadState::Waiting,\n \"suspended\" => crate::types::thread::ThreadState::Suspended,\n other => {\n return ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::ValueError,\n Some(format!(\"Unknown thread state: {other}\")),\n ));\n }\n };\n\n match thread.transition_to(target, reason) {\n Ok(()) => ExtFunctionResult::Return(MontyObject::None),\n Err(e) => ExtFunctionResult::Error(monty::MontyException::new(\n monty::ExcType::RuntimeError,\n Some(format!(\"State transition failed: {e}\")),\n )),\n }\n}\n\n/// Handle `__retrieve_docs__(goal, max_docs)`.\nasync fn handle_retrieve_docs(\n args: &[MontyObject],\n _kwargs: &[(MontyObject, MontyObject)],\n thread: &Thread,\n retrieval: Option<&RetrievalEngine>,\n) -> ExtFunctionResult {\n let retrieval = match retrieval {\n Some(r) => r,\n None => return ExtFunctionResult::Return(json_to_monty(&serde_json::json!([]))),\n };\n\n let goal = args.first().map(monty_to_string).unwrap_or_default();\n let max_docs = args\n .get(1)\n .and_then(|v| match v {\n MontyObject::Int(i) => Some(*i as usize),\n _ => None,\n })\n .unwrap_or(5);\n\n match retrieval\n .retrieve_context(thread.project_id, &thread.user_id, &goal, max_docs)\n .await\n {\n Ok(docs) => {\n let docs_json: Vec = docs\n .iter()\n .map(|d| {\n serde_json::json!({\n \"type\": format!(\"{:?}\", d.doc_type),\n \"title\": d.title,\n \"content\": d.content,\n })\n })\n .collect();\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!(docs_json)))\n }\n Err(e) => {\n debug!(\"retrieve_docs failed: {e}\");\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])))\n }\n }\n}\n\n/// Handle `__check_budget__()`.\nfn handle_check_budget(thread: &Thread) -> ExtFunctionResult {\n let tokens_remaining = thread\n .config\n .max_tokens_total\n .map(|max| max.saturating_sub(thread.total_tokens_used))\n .unwrap_or(u64::MAX);\n\n let time_remaining_ms = thread\n .config\n .max_duration\n .map(|dur| {\n let elapsed = chrono::Utc::now()\n .signed_duration_since(thread.created_at)\n .num_milliseconds()\n .max(0) as u64;\n dur.as_millis() as u64 - elapsed.min(dur.as_millis() as u64)\n })\n .unwrap_or(u64::MAX);\n\n let usd_remaining = thread\n .config\n .max_budget_usd\n .map(|max| (max - thread.total_cost_usd).max(0.0));\n\n let result = serde_json::json!({\n \"tokens_remaining\": tokens_remaining,\n \"time_remaining_ms\": time_remaining_ms,\n \"usd_remaining\": usd_remaining,\n });\n\n ExtFunctionResult::Return(json_to_monty(&result))\n}\n\n/// Handle `__get_actions__()`.\nasync fn handle_get_actions(\n thread: &mut Thread,\n effects: &Arc,\n leases: &Arc,\n store: Option<&Arc>,\n) -> ExtFunctionResult {\n if let Err(e) =\n reconcile_dynamic_tool_lease(thread, effects, leases, store, &crate::LeasePlanner::new())\n .await\n {\n warn_on_lease_refresh_failure(\"get_actions\", &e);\n }\n\n let active_leases = leases.active_for_thread(thread.id).await;\n match effects.available_actions(&active_leases).await {\n Ok(actions) => {\n let actions_json: Vec = actions\n .iter()\n .map(|a| {\n serde_json::json!({\n \"name\": a.name,\n \"description\": a.description,\n \"params\": a.parameters_schema,\n })\n })\n .collect();\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!(actions_json)))\n }\n Err(e) => {\n debug!(\"get_actions failed: {e}\");\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])))\n }\n }\n}\n\n/// Handle `__list_skills__()`.\n///\n/// Loads all `DocType::Skill` MemoryDocs from the project and returns them\n/// as a list of Python dicts. The Python orchestrator handles scoring,\n/// selection, and injection — Rust just provides data access.\nasync fn handle_list_skills(\n _args: &[MontyObject],\n thread: &Thread,\n store: Option<&Arc>,\n) -> ExtFunctionResult {\n let Some(store) = store else {\n return ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])));\n };\n\n // Use shared listing: user's own skills + system/admin-installed skills.\n let docs = match store\n .list_memory_docs_with_shared(thread.project_id, &thread.user_id)\n .await\n {\n Ok(docs) => docs,\n Err(e) => {\n debug!(\"__list_skills__: failed to load docs: {e}\");\n return ExtFunctionResult::Return(json_to_monty(&serde_json::json!([])));\n }\n };\n\n let skills: Vec = docs\n .into_iter()\n .filter(|d| d.doc_type == crate::types::memory::DocType::Skill)\n .map(|d| {\n serde_json::json!({\n \"doc_id\": d.id.0.to_string(),\n \"title\": d.title,\n \"content\": d.content,\n \"metadata\": d.metadata,\n })\n })\n .collect();\n\n ExtFunctionResult::Return(json_to_monty(&serde_json::json!(skills)))\n}\n\n/// Handle `__record_skill_usage__(doc_id, success)`.\n///\n/// Records that a skill was used in this thread. Called by the Python\n/// orchestrator after skill-assisted execution completes.\nasync fn handle_record_skill_usage(\n args: &[MontyObject],\n store: Option<&Arc>,\n) -> ExtFunctionResult {\n let Some(store) = store else {\n return ExtFunctionResult::Return(MontyObject::None);\n };\n\n let doc_id_str = args.first().map(monty_to_string).unwrap_or_default();\n let success = args\n .get(1)\n .map(|o| matches!(o, MontyObject::Bool(true)))\n .unwrap_or(false);\n\n let Ok(uuid) = uuid::Uuid::parse_str(&doc_id_str) else {\n debug!(\"__record_skill_usage__: invalid doc_id: {doc_id_str}\");\n return ExtFunctionResult::Return(MontyObject::None);\n };\n\n let tracker = crate::memory::SkillTracker::new(Arc::clone(store));\n if let Err(e) = tracker\n .record_usage(crate::types::memory::DocId(uuid), success)\n .await\n {\n debug!(\"__record_skill_usage__: failed: {e}\");\n }\n\n ExtFunctionResult::Return(MontyObject::None)\n}\n\n/// Handle `__regex_match__(pattern, text) -> bool`.\n///\n/// Compiles `pattern` with a bounded size limit and returns whether it\n/// matches anywhere in `text`. Invalid regex or a size-limit violation\n/// returns `False` silently. Used by the Python skill selector for regex\n/// pattern scoring (Monty has no `re` module).\n///\n/// **Security: ReDoS safety.** This handler accepts arbitrary patterns from\n/// the Python orchestrator (which itself receives them from skill manifests)\n/// and runs them on user-supplied text. Safety relies on the `regex` crate's\n/// linear-time matching guarantee (no backreferences, no lookaround) plus the\n/// 64 KiB compiled-size cap and DFA-size cap below. If the `regex` crate is\n/// ever swapped for `fancy-regex` (which supports backreferences and is NOT\n/// linear-time), this becomes a real ReDoS vector. This is enforced by\n/// convention and documentation only — see the top-of-crate comment in\n/// `crates/ironclaw_engine/src/lib.rs`. (A `#[cfg(feature = \"fancy-regex\")]\n/// compile_error!` tripwire was evaluated but conflicts with\n/// `cargo clippy --all-features` which is the standard CI command.)\nfn handle_regex_match(args: &[MontyObject]) -> ExtFunctionResult {\n let pattern = args.first().map(monty_to_string).unwrap_or_default();\n let text = args.get(1).map(monty_to_string).unwrap_or_default();\n if pattern.is_empty() {\n return ExtFunctionResult::Return(MontyObject::Bool(false));\n }\n // Cap compiled regex size to prevent ReDoS (matches the 64 KiB limit used\n // by `LoadedSkill::compile_patterns` in `ironclaw_skills`). Also cap the\n // lazy-DFA cache: the `regex` crate's DFA can grow beyond `size_limit`\n // during matching, so `dfa_size_limit` is a separate defensive cap on\n // memory allocation from a crafted pattern over untrusted skill manifests.\n const MAX_REGEX_SIZE: usize = 1 << 16;\n let matched = match regex::RegexBuilder::new(&pattern)\n .size_limit(MAX_REGEX_SIZE)\n .dfa_size_limit(MAX_REGEX_SIZE)\n .build()\n {\n Ok(re) => re.is_match(&text),\n Err(e) => {\n debug!(\"__regex_match__: invalid pattern '{pattern}': {e}\");\n false\n }\n };\n ExtFunctionResult::Return(MontyObject::Bool(matched))\n}\n\n/// Handle `__set_active_skills__(skills)`.\n///\n/// Persists the selected skill provenance onto the thread so post-run learning\n/// flows can reason about the exact skill versions and snippets that were active.\nfn handle_set_active_skills(args: &[MontyObject], thread: &mut Thread) -> ExtFunctionResult {\n let skills_json = args\n .first()\n .map(monty_to_json)\n .unwrap_or_else(|| serde_json::json!([]));\n\n let skills = match serde_json::from_value::>(skills_json) {\n Ok(skills) => skills,\n Err(e) => {\n debug!(\"__set_active_skills__: invalid payload: {e}\");\n return ExtFunctionResult::Return(MontyObject::None);\n }\n };\n\n if let Err(e) = thread.set_active_skills(&skills) {\n debug!(\"__set_active_skills__: failed to persist active skills: {e}\");\n }\n\n ExtFunctionResult::Return(MontyObject::None)\n}\n\n// ── Helpers ─────────────────────────────────────────────────\n\n/// Build the context variables injected into the orchestrator Python.\nfn build_orchestrator_inputs(\n thread: &Thread,\n persisted_state: &serde_json::Value,\n) -> (Vec, Vec) {\n let names = vec![\n \"context\".into(),\n \"goal\".into(),\n \"actions\".into(),\n \"state\".into(),\n \"config\".into(),\n ];\n\n // Build orchestrator bootstrap context. Prefer the internal execution\n // transcript when present, otherwise fall back to the user-visible transcript.\n let bootstrap_messages = if thread.internal_messages.is_empty() {\n &thread.messages\n } else {\n &thread.internal_messages\n };\n let context: Vec = bootstrap_messages\n .iter()\n .map(|m| {\n // Serialize action_calls through the Python interchange shape\n // (`{name, call_id, params}`) so the bootstrap context is\n // round-trip compatible with `python_json_to_action_calls`.\n // Using bare `m.action_calls` here produces the canonical Rust\n // serde format (`{action_name, id, parameters}`), which the\n // Python orchestrator passes back verbatim on the next\n // `__llm_complete__` call — and `python_json_to_action_calls`\n // then fails with \"missing field `name`\", orphaning every\n // subsequent tool result. This is the SECOND code path (after\n // `handle_llm_complete`) that feeds action_calls into the\n // Python working transcript; both must use the same shape.\n let calls_json = m\n .action_calls\n .as_ref()\n .map(|calls| serde_json::Value::Array(action_calls_to_python_json(calls)));\n serde_json::json!({\n \"role\": format!(\"{:?}\", m.role),\n \"content\": m.content,\n \"action_name\": m.action_name,\n \"action_call_id\": m.action_call_id,\n \"action_calls\": calls_json,\n })\n })\n .collect();\n\n // Build config\n let config = serde_json::json!({\n \"max_iterations\": thread.config.max_iterations,\n \"max_tool_intent_nudges\": thread.config.max_tool_intent_nudges,\n \"enable_tool_intent_nudge\": thread.config.enable_tool_intent_nudge,\n \"max_consecutive_errors\": thread.config.max_consecutive_errors,\n \"max_tokens_total\": thread.config.max_tokens_total,\n \"max_budget_usd\": thread.config.max_budget_usd,\n \"model_context_limit\": thread.config.model_context_limit,\n \"enable_compaction\": thread.config.enable_compaction,\n \"compaction_threshold\": thread.config.compaction_threshold,\n \"depth\": thread.config.depth,\n \"max_depth\": thread.config.max_depth,\n \"step_count\": thread.step_count,\n });\n\n let values = vec![\n json_to_monty(&serde_json::json!(context)),\n MontyObject::String(thread.goal.clone()),\n json_to_monty(&serde_json::json!([])), // actions loaded dynamically via __get_actions__\n json_to_monty(persisted_state),\n json_to_monty(&config),\n ];\n\n (names, values)\n}\n\n/// JSON shape used to interchange `ActionCall`s with the Python orchestrator.\n///\n/// This is the *single* place that defines the field naming convention used\n/// across the Python boundary. It is intentionally separate from the\n/// canonical `ActionCall` type because:\n///\n/// - `ActionCall` uses Rust-idiomatic field names (`id`, `action_name`,\n/// `parameters`) and is also persisted into Step records and ThreadEvents.\n/// Renaming its serde fields would invalidate every existing row.\n/// - The Python orchestrator uses friendlier names (`call_id`, `name`,\n/// `params`) that read naturally in CodeAct prompts and `default.py`.\n///\n/// Without this type, the round-trip is asymmetric: Rust → Python uses one\n/// shape, Python → Rust used `serde_json::from_value::>`\n/// which silently fails (`.ok()` swallows the error) and produces `None`,\n/// which means assistant messages came back without `action_calls`. The\n/// downstream effect is that every tool result looks orphaned to\n/// `sanitize_tool_messages` and gets rewritten as a user message — losing\n/// the assistant ↔ tool_result linkage the LLM needs to reason about prior\n/// tool calls.\n#[derive(Debug, serde::Serialize, serde::Deserialize)]\nstruct PythonActionCall {\n name: String,\n call_id: String,\n params: serde_json::Value,\n}\n\nimpl From<&ActionCall> for PythonActionCall {\n fn from(c: &ActionCall) -> Self {\n Self {\n name: c.action_name.clone(),\n call_id: c.id.clone(),\n params: c.parameters.clone(),\n }\n }\n}\n\nimpl From for ActionCall {\n fn from(p: PythonActionCall) -> Self {\n Self {\n id: p.call_id,\n action_name: p.name,\n parameters: p.params,\n }\n }\n}\n\n/// Serialize a slice of `ActionCall`s into the Python interchange shape.\n///\n/// On serialization failure (essentially unreachable for `String + String +\n/// Value`, but still possible if the `serde_json::Value` parameters tree\n/// contains a key whose stringification fails), the entry is **dropped**\n/// from the output rather than replaced with `Value::Null`. The previous\n/// `unwrap_or_else(|_| Value::Null)` corrupted the array — Python's\n/// `default.py` accesses `c.get(\"name\")` / `c.get(\"call_id\")` /\n/// `c.get(\"params\")` on each entry, so a `null` would crash with a Python\n/// `AttributeError` and lose the entire LLM step. `filter_map` produces a\n/// shorter array, which Python's tool-result loop handles correctly because\n/// it iterates `range(len(results))` against the shortened call list. The\n/// warn log is preserved so operators have a breadcrumb if it ever fires.\nfn action_calls_to_python_json(calls: &[ActionCall]) -> Vec {\n calls\n .iter()\n .filter_map(|c| match serde_json::to_value(PythonActionCall::from(c)) {\n Ok(value) => Some(value),\n Err(e) => {\n warn!(\n error = %e,\n action_name = %c.action_name,\n \"Failed to serialize ActionCall for Python orchestrator — dropping entry\"\n );\n None\n }\n })\n .collect()\n}\n\n/// Extract the last `n` characters from `s`.\n///\n/// Error tracebacks appear at the end of stdout, after any `print()` output.\n/// Using the head would capture the print statements instead of the error.\nfn tail_chars(s: &str, n: usize) -> String {\n let char_count = s.chars().count();\n if char_count > n {\n s.chars().skip(char_count - n).collect()\n } else {\n s.to_owned()\n }\n}\n\n/// Build a PII-safe summary of an `action_calls` JSON value for log output.\n///\n/// The action_calls payload contains tool parameters, which can carry user\n/// PII (search queries, file names, email content, conversation text).\n/// Dumping the full value into a `warn!` log would leak that PII to log\n/// aggregation systems (Datadog, CloudWatch, Sentry) the moment the parser\n/// fails — and the parser only fails when the Python ↔ Rust shape drifts,\n/// which is exactly when an operator is most likely to be grepping logs.\n///\n/// We emit only the structural information operators actually need to\n/// debug a shape drift: array length and the keys of the first entry. The\n/// keys themselves are not user data — they're field names like\n/// `name`/`call_id`/`params` that are static across all calls.\nfn summarize_action_calls_for_log(value: &serde_json::Value) -> String {\n match value.as_array() {\n Some(arr) if arr.is_empty() => \"empty array\".to_string(),\n Some(arr) => {\n let first_keys = arr\n .first()\n .and_then(|v| v.as_object())\n .map(|obj| {\n let mut keys: Vec<&str> = obj.keys().map(String::as_str).collect();\n keys.sort_unstable();\n keys.join(\",\")\n })\n .unwrap_or_else(|| \"\".to_string());\n format!(\n \"array of {} entries; first entry keys: [{}]\",\n arr.len(),\n first_keys\n )\n }\n None => format!(\"non-array value of type {}\", json_value_type_name(value)),\n }\n}\n\n/// Cheap type-name string for a `serde_json::Value`. Used by\n/// `summarize_action_calls_for_log` to surface the wrong-shape case\n/// (e.g. Python passed a string instead of an array) without leaking the\n/// actual contents.\nfn json_value_type_name(value: &serde_json::Value) -> &'static str {\n match value {\n serde_json::Value::Null => \"null\",\n serde_json::Value::Bool(_) => \"bool\",\n serde_json::Value::Number(_) => \"number\",\n serde_json::Value::String(_) => \"string\",\n serde_json::Value::Array(_) => \"array\",\n serde_json::Value::Object(_) => \"object\",\n }\n}\n\n/// Deserialize an `action_calls` JSON array (in Python interchange shape)\n/// back into canonical `ActionCall`s.\n///\n/// Logs a warning on failure rather than swallowing silently. The whole\n/// commit that introduced this helper exists to undo a `.ok()` swallow that\n/// dropped action_calls without any signal — replacing it with another\n/// `.ok()?` would re-introduce the same trap, just one layer deeper. If the\n/// shape ever drifts again (Python orchestrator field rename, extra\n/// required field, partial migration), the warning is the operator-visible\n/// breadcrumb that explains why subsequent tool results suddenly look\n/// orphaned to `sanitize_tool_messages`.\n///\n/// The warn log emits a structural summary (`summarize_action_calls_for_log`)\n/// instead of the raw value because tool parameters can contain user PII.\nfn python_json_to_action_calls(value: &serde_json::Value) -> Option> {\n match serde_json::from_value::>(value.clone()) {\n Ok(parsed) => Some(parsed.into_iter().map(ActionCall::from).collect()),\n Err(e) => {\n warn!(\n error = %e,\n shape = %summarize_action_calls_for_log(value),\n \"Failed to parse action_calls from Python orchestrator — \\\n assistant message will lose tool_call linkage and downstream \\\n tool results will be rewritten as user messages\"\n );\n None\n }\n }\n}\n\nfn json_to_thread_messages(value: &serde_json::Value) -> Option> {\n let arr = value.as_array()?;\n let mut messages = Vec::with_capacity(arr.len());\n\n for item in arr {\n let role = item.get(\"role\").and_then(|v| v.as_str()).unwrap_or(\"User\");\n let content = item\n .get(\"content\")\n .and_then(|v| v.as_str())\n .unwrap_or_default();\n // Filter out null before calling the parser — `action_calls: null`\n // is Python's legitimate \"this message has no tool calls\" signal (text\n // response), not a parse error. Without this filter, the warn log in\n // python_json_to_action_calls fires on every text-only assistant\n // message with \"invalid type: null, expected a sequence\".\n let action_calls = item\n .get(\"action_calls\")\n .filter(|v| !v.is_null())\n .and_then(python_json_to_action_calls);\n\n let message = match role {\n \"System\" | \"system\" => ThreadMessage::system(content),\n \"Assistant\" | \"assistant\" => {\n if let Some(calls) = action_calls {\n ThreadMessage::assistant_with_actions(Some(content.to_string()), calls)\n } else {\n ThreadMessage::assistant(content)\n }\n }\n \"ActionResult\" | \"action_result\" => ThreadMessage::action_result(\n item.get(\"action_call_id\")\n .and_then(|v| v.as_str())\n .unwrap_or_default(),\n item.get(\"action_name\")\n .and_then(|v| v.as_str())\n .unwrap_or_default(),\n content,\n ),\n _ => ThreadMessage::user(content),\n };\n messages.push(message);\n }\n\n Some(messages)\n}\n\nfn sync_runtime_state(thread: &mut Thread, state: Option<&serde_json::Value>) {\n let Some(state) = state else {\n return;\n };\n if let Some(messages) = state\n .get(\"working_messages\")\n .and_then(json_to_thread_messages)\n {\n thread.internal_messages = messages;\n thread.updated_at = chrono::Utc::now();\n }\n}\n\nfn sync_visible_outcome(thread: &mut Thread, outcome: &ThreadOutcome) {\n if let ThreadOutcome::Completed {\n response: Some(response),\n } = outcome\n {\n let already_present = thread\n .messages\n .last()\n .map(|msg| {\n msg.role == crate::types::message::MessageRole::Assistant\n && msg.content == *response\n })\n .unwrap_or(false);\n if !already_present {\n thread.add_message(ThreadMessage::assistant(response));\n }\n }\n}\n\n/// Parse the orchestrator's return value into a ThreadOutcome.\nfn parse_outcome(result: &serde_json::Value) -> ThreadOutcome {\n let outcome = result\n .get(\"outcome\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"completed\");\n\n match outcome {\n \"completed\" => ThreadOutcome::Completed {\n response: result\n .get(\"response\")\n .and_then(|v| v.as_str())\n .map(String::from),\n },\n \"stopped\" => ThreadOutcome::Stopped,\n \"max_iterations\" => ThreadOutcome::MaxIterations,\n \"failed\" => ThreadOutcome::Failed {\n error: result\n .get(\"error\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"unknown error\")\n .to_string(),\n },\n \"gate_paused\" => {\n let resume_kind_value = result\n .get(\"resume_kind\")\n .cloned()\n .unwrap_or(serde_json::json!({}));\n let resume_kind = serde_json::from_value(resume_kind_value).unwrap_or(\n crate::gate::ResumeKind::Approval {\n allow_always: false,\n },\n );\n ThreadOutcome::GatePaused {\n gate_name: result\n .get(\"gate_name\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"unknown\")\n .to_string(),\n action_name: result\n .get(\"action_name\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"\")\n .to_string(),\n call_id: result\n .get(\"call_id\")\n .and_then(|v| v.as_str())\n .unwrap_or(\"\")\n .to_string(),\n parameters: result\n .get(\"parameters\")\n .cloned()\n .unwrap_or(serde_json::json!({})),\n resume_kind,\n resume_output: result.get(\"resume_output\").cloned(),\n }\n }\n _ => ThreadOutcome::Completed { response: None },\n }\n}\n\nfn extract_string_arg(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n name: &str,\n position: usize,\n) -> Option {\n for (k, v) in kwargs {\n if let MontyObject::String(key) = k\n && key == name\n {\n return Some(monty_to_string(v));\n }\n }\n args.get(position).map(monty_to_string)\n}\n\nfn extract_string_kwarg(kwargs: &[(MontyObject, MontyObject)], name: &str) -> Option {\n for (k, v) in kwargs {\n if let MontyObject::String(key) = k\n && key == name\n {\n return Some(monty_to_string(v));\n }\n }\n None\n}\n\nfn extract_u64_kwarg(kwargs: &[(MontyObject, MontyObject)], name: &str) -> Option {\n for (k, v) in kwargs {\n if let MontyObject::String(key) = k\n && key == name\n && let MontyObject::Int(i) = v\n {\n return Some(*i as u64);\n }\n }\n None\n}\n\n#[cfg(test)]\nmod tests {\n use super::*;\n use crate::types::memory::{DocType, MemoryDoc};\n use crate::types::project::ProjectId;\n\n // ── Python helper unit tests via Monty ──────────────────────\n //\n // Extracts the helper functions from the default orchestrator and\n // evaluates `signals_tool_intent(text)` directly, mirroring the V1\n // Rust unit test suite in src/llm/reasoning.rs.\n\n /// Run a Python expression that returns a bool by prepending the\n /// orchestrator helper definitions and wrapping in `FINAL(expr)`.\n /// Run a Python snippet and drive the Monty VM, returning the FINAL()\n /// value as a `MontyObject`. This is the common core for `eval_python_bool`\n /// and `eval_python_int`.\n fn run_python_final(code: String) -> MontyObject {\n let runner =\n MontyRun::new(code, \"test.py\", vec![]).expect(\"Failed to parse orchestrator helpers\");\n let mut stdout = String::new();\n let tracker = LimitedTracker::new(ResourceLimits::new().max_allocations(500_000));\n\n let mut progress = runner\n .start(vec![], tracker, PrintWriter::Collect(&mut stdout))\n .expect(\"Failed to start orchestrator test\");\n\n loop {\n match progress {\n RunProgress::Complete(obj) => return obj,\n RunProgress::FunctionCall(call) => {\n if call.function_name == \"FINAL\" {\n let val = call.args.first().cloned().unwrap_or(MontyObject::None);\n let _ = call.resume(\n ExtFunctionResult::Return(MontyObject::None),\n PrintWriter::Collect(&mut stdout),\n );\n return val;\n }\n let ext_result = match call.function_name.as_str() {\n \"__regex_match__\" => handle_regex_match(&call.args),\n _ => ExtFunctionResult::Return(MontyObject::None),\n };\n progress = call\n .resume(ext_result, PrintWriter::Collect(&mut stdout))\n .expect(\"resume failed\");\n }\n RunProgress::NameLookup(lookup) => {\n progress = lookup\n .resume(\n NameLookupResult::Undefined,\n PrintWriter::Collect(&mut stdout),\n )\n .expect(\"name lookup resume failed\");\n }\n _ => panic!(\"Unexpected RunProgress variant in test\"),\n }\n }\n }\n\n fn eval_python_bool(expr: &str) -> bool {\n // Extract only the helper functions (everything before run_loop)\n let helpers_end = DEFAULT_ORCHESTRATOR\n .find(\"\\ndef run_loop(\")\n .unwrap_or(DEFAULT_ORCHESTRATOR.len());\n let helpers = &DEFAULT_ORCHESTRATOR[..helpers_end]; // safety: find() returns a char boundary on this ASCII-only constant\n\n let code = format!(\"{helpers}\\nFINAL({expr})\");\n match run_python_final(code) {\n MontyObject::Bool(v) => v,\n other => panic!(\"Expected bool, got: {other:?}\"),\n }\n }\n\n /// Run a Python program (with orchestrator helpers in scope) that ends\n /// with `FINAL(int_expr)` and return the integer value.\n fn eval_python_int(program: &str) -> i64 {\n let helpers_end = DEFAULT_ORCHESTRATOR\n .find(\"\\ndef run_loop(\")\n .unwrap_or(DEFAULT_ORCHESTRATOR.len());\n let helpers = &DEFAULT_ORCHESTRATOR[..helpers_end];\n\n let code = format!(\"{helpers}\\n{program}\");\n match run_python_final(code) {\n MontyObject::Int(v) => v,\n other => panic!(\"Expected int, got: {other:?}\"),\n }\n }\n\n // ── __regex_match__ host function reachability ───────────────\n\n #[test]\n fn regex_match_host_function_is_callable_from_monty() {\n // Regression test for PR #1736 review (serrrfirat, 3059161877):\n // verify that Monty's NameLookup + FunctionCall dispatch actually\n // reaches `handle_regex_match` when default.py calls\n // `__regex_match__(...)`. If Monty ever starts resolving the name\n // before the call, this test will fail with a NameError.\n assert!(eval_python_bool(\n r#\"bool(__regex_match__(\"abc\", \"xxabcxx\"))\"#\n ));\n assert!(!eval_python_bool(\n r#\"bool(__regex_match__(\"zzz\", \"xxabcxx\"))\"#\n ));\n // Invalid pattern should return false silently (the host function\n // swallows the compile error).\n assert!(!eval_python_bool(r#\"bool(__regex_match__(\"[\", \"abc\"))\"#));\n }\n\n // ── True positives (should trigger nudge) ───────────────────\n\n #[test]\n fn signals_tool_intent_true_positives() {\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"Let me search for that file.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'll fetch the data now.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'm going to check the logs.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"Let me add it now.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I will run the tests to verify.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'll look up the documentation.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"Let me read the file contents.\")\"#\n ));\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'm going to execute the command.\")\"#\n ));\n }\n\n // ── True negatives: conversational phrases ──────────────────\n\n #[test]\n fn signals_tool_intent_true_negatives_conversational() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me explain how this works.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me know if you need anything.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me think about this.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me summarize the findings.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me clarify what I mean.\")\"#\n ));\n }\n\n // ── Exclusion takes precedence ──────────────────────────────\n\n #[test]\n fn signals_tool_intent_exclusion_takes_precedence() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Let me explain the approach, then I'll search for the file.\")\"#\n ));\n }\n\n // ── Code blocks are stripped ────────────────────────────────\n\n #[test]\n fn signals_tool_intent_ignores_code_blocks() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Here's the code:\\n\\n```\\nfn main() {\\n println!(\\\"Let me search the database\\\");\\n}\\n```\")\"#\n ));\n }\n\n #[test]\n fn signals_tool_intent_ignores_indented_code() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Here's the code:\\n\\n println!(\\\"I'll fetch the data\\\");\\n\\nThat's it.\")\"#\n ));\n }\n\n // ── Plain informational text ────────────────────────────────\n\n #[test]\n fn signals_tool_intent_ignores_plain_text() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"The task is complete.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Here are the results you asked for.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"I found 3 matching files.\")\"#\n ));\n }\n\n // ── Quoted strings are stripped ─────────────────────────────\n\n #[test]\n fn signals_tool_intent_ignores_quoted_strings() {\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"The button says \\\"Let me search the database\\\" to the user.\")\"#\n ));\n // But unquoted intent should still trigger\n assert!(eval_python_bool(\n r#\"signals_tool_intent(\"I'll fetch the results for you.\")\"#\n ));\n }\n\n // ── Shadowed prefix (exclusion cancels all) ─────────────────\n\n #[test]\n fn signals_tool_intent_shadowed_prefix() {\n // \"let me think\" is an exclusion → entire text returns false\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Sure, let me think about it. Actually, let me search for the file.\")\"#\n ));\n }\n\n // ── Regression: trace false positive (news content) ─────────\n\n #[test]\n fn signals_tool_intent_no_false_positive_news_content() {\n // \"I can\" + \"call\" in news content triggered false positive in old code\n let news_response = concat!(\n \"The latest headlines suggest this is a fast-moving war.\\n\",\n \"- Reuters: Iran is calling US peace proposals unrealistic.\\n\",\n \"If you want, I can do one of these next:\\n\",\n \"1. give you a 5-bullet update\\n\",\n \"2. focus just on military developments\",\n );\n assert!(!eval_python_bool(&format!(\n \"signals_tool_intent({news_response:?})\"\n )));\n }\n\n #[test]\n fn signals_tool_intent_no_false_positive_past_tense() {\n // \"I fetched\" / \"I already called\" should not trigger\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"I already completed the needed action call by fetching current news feeds.\")\"#\n ));\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"Current status from the live feeds I fetched:\")\"#\n ));\n }\n\n #[test]\n fn signals_tool_intent_no_false_positive_offer() {\n // \"If you want, I can fetch...\" uses \"I can\" which is not a V1 prefix\n assert!(!eval_python_bool(\n r#\"signals_tool_intent(\"If you want, I can next fetch a cleaner update.\")\"#\n ));\n }\n\n #[tokio::test]\n async fn load_orchestrator_without_store_returns_default() {\n let (code, version) = load_orchestrator(None, ProjectId::new(), true).await;\n assert_eq!(version, 0);\n assert!(code.contains(\"run_loop\"));\n assert!(code.contains(\"__llm_complete__\"));\n }\n\n #[tokio::test]\n async fn load_orchestrator_with_runtime_version() {\n let project_id = ProjectId::new();\n let mut doc = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"custom_orchestrator_code()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc.metadata = serde_json::json!({\"version\": 1});\n\n let store = Arc::new(crate::tests::InMemoryStore::with_docs(vec![doc]));\n let (code, version) =\n load_orchestrator(Some(&(store as Arc)), project_id, true).await;\n assert_eq!(version, 1);\n assert!(code.contains(\"custom_orchestrator_code\"));\n }\n\n #[tokio::test]\n async fn load_orchestrator_picks_highest_version() {\n let project_id = ProjectId::new();\n let mut doc_v1 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v1_code()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v1.metadata = serde_json::json!({\"version\": 1});\n\n let mut doc_v3 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v3_code()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v3.metadata = serde_json::json!({\"version\": 3});\n\n let mut doc_v2 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v2_code()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v2.metadata = serde_json::json!({\"version\": 2});\n\n let store = Arc::new(crate::tests::InMemoryStore::with_docs(vec![\n doc_v1, doc_v3, doc_v2,\n ]));\n let (code, version) =\n load_orchestrator(Some(&(store as Arc)), project_id, true).await;\n assert_eq!(version, 3);\n assert!(code.contains(\"v3_code\"));\n }\n\n #[tokio::test]\n async fn rollback_after_max_failures() {\n let project_id = ProjectId::new();\n\n // Create v2 orchestrator\n let mut doc_v2 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v2_buggy()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v2.metadata = serde_json::json!({\"version\": 2});\n\n // Create v1 orchestrator (fallback)\n let mut doc_v1 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v1_stable()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v1.metadata = serde_json::json!({\"version\": 1});\n\n // Create failure tracker showing v2 has 3 failures\n let tracker = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n FAILURE_TRACKER_TITLE,\n r#\"{\"version\": 2, \"count\": 3}\"#,\n )\n .with_tags(vec![\"orchestrator_meta\".to_string()]);\n\n let store = Arc::new(crate::tests::InMemoryStore::with_docs(vec![\n doc_v2, doc_v1, tracker,\n ]));\n let (code, version) =\n load_orchestrator(Some(&(store as Arc)), project_id, true).await;\n\n // Should skip v2 (too many failures) and load v1\n assert_eq!(version, 1);\n assert!(code.contains(\"v1_stable\"));\n }\n\n #[tokio::test]\n async fn rollback_to_default_when_all_versions_fail() {\n let project_id = ProjectId::new();\n\n // Single version with 3 failures\n let mut doc_v1 = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n ORCHESTRATOR_TITLE,\n \"v1_broken()\",\n )\n .with_tags(vec![ORCHESTRATOR_TAG.to_string()]);\n doc_v1.metadata = serde_json::json!({\"version\": 1});\n\n let tracker = MemoryDoc::new(\n project_id,\n \"system\",\n DocType::Note,\n FAILURE_TRACKER_TITLE,\n r#\"{\"version\": 1, \"count\": 5}\"#,\n )\n .with_tags(vec![\"orchestrator_meta\".to_string()]);\n\n let store = Arc::new(crate::tests::InMemoryStore::with_docs(vec![\n doc_v1, tracker,\n ]));\n let (code, version) =\n load_orchestrator(Some(&(store as Arc)), project_id, true).await;\n\n // Should fall back to compiled-in default (v0)\n assert_eq!(version, 0);\n assert!(code.contains(\"run_loop\"));\n }\n\n #[tokio::test]\n async fn record_and_reset_failures() {\n let project_id = ProjectId::new();\n let store: Arc = Arc::new(crate::tests::InMemoryStore::with_docs(vec![]));\n\n // Record 3 failures\n record_orchestrator_failure(&store, project_id, 2).await;\n record_orchestrator_failure(&store, project_id, 2).await;\n record_orchestrator_failure(&store, project_id, 2).await;\n\n let docs = store.list_shared_memory_docs(project_id).await.unwrap();\n let count = load_failure_count(&docs);\n assert_eq!(count, 3);\n\n // Reset\n reset_orchestrator_failures(&store, project_id).await;\n let docs = store.list_shared_memory_docs(project_id).await.unwrap();\n let count = load_failure_count(&docs);\n assert_eq!(count, 0);\n }\n\n #[tokio::test]\n async fn failure_count_resets_on_new_version() {\n let project_id = ProjectId::new();\n let store: Arc = Arc::new(crate::tests::InMemoryStore::with_docs(vec![]));\n\n // Record failures for version 1\n record_orchestrator_failure(&store, project_id, 1).await;\n record_orchestrator_failure(&store, project_id, 1).await;\n\n // Switch to version 2 — count should reset to 1\n record_orchestrator_failure(&store, project_id, 2).await;\n\n let docs = store.list_shared_memory_docs(project_id).await.unwrap();\n let count = load_failure_count(&docs);\n assert_eq!(count, 1);\n }\n\n #[test]\n fn normalize_pause_outcome_transitions_thread_to_waiting() {\n let mut thread = Thread::new(\n \"goal\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n let outcome = ThreadOutcome::GatePaused {\n gate_name: \"approval\".into(),\n action_name: \"shell\".into(),\n call_id: \"call-1\".into(),\n parameters: serde_json::json!({\"cmd\":\"ls\"}),\n resume_kind: crate::gate::ResumeKind::Approval { allow_always: true },\n resume_output: None,\n };\n normalize_pause_outcome(&mut thread, &outcome).unwrap();\n assert_eq!(thread.state, ThreadState::Waiting);\n }\n\n #[test]\n fn parse_outcome_completed() {\n let result = serde_json::json!({\"outcome\": \"completed\", \"response\": \"Hello!\"});\n let outcome = parse_outcome(&result);\n assert!(matches!(outcome, ThreadOutcome::Completed { response: Some(r) } if r == \"Hello!\"));\n }\n\n #[test]\n fn parse_outcome_failed() {\n let result = serde_json::json!({\"outcome\": \"failed\", \"error\": \"boom\"});\n let outcome = parse_outcome(&result);\n assert!(matches!(outcome, ThreadOutcome::Failed { error } if error == \"boom\"));\n }\n\n #[test]\n fn parse_outcome_gate_paused() {\n let result = serde_json::json!({\n \"outcome\": \"gate_paused\",\n \"gate_name\": \"approval\",\n \"action_name\": \"shell\",\n \"call_id\": \"abc\",\n \"parameters\": {\"cmd\": \"rm -rf /\"},\n \"resume_kind\": {\"Approval\": {\"allow_always\": true}}\n });\n let outcome = parse_outcome(&result);\n assert!(\n matches!(outcome, ThreadOutcome::GatePaused { action_name, .. } if action_name == \"shell\")\n );\n }\n\n #[test]\n fn parse_outcome_max_iterations() {\n let result = serde_json::json!({\"outcome\": \"max_iterations\"});\n let outcome = parse_outcome(&result);\n assert!(matches!(outcome, ThreadOutcome::MaxIterations));\n }\n\n #[test]\n fn parse_outcome_stopped() {\n let result = serde_json::json!({\"outcome\": \"stopped\"});\n let outcome = parse_outcome(&result);\n assert!(matches!(outcome, ThreadOutcome::Stopped));\n }\n\n // ── handle_llm_complete model forwarding ────────────────────\n\n /// LLM backend that records the model from each `complete()` call.\n /// Used to verify the orchestrator's __llm_complete__ host fn forwards\n /// `explicit_config[\"model\"]` onto `LlmCallConfig.model`.\n struct ModelCapturingLlm {\n captured: tokio::sync::Mutex>>,\n }\n\n #[async_trait::async_trait]\n impl LlmBackend for ModelCapturingLlm {\n fn model_name(&self) -> &str {\n \"capturing\"\n }\n\n async fn complete(\n &self,\n _messages: &[ThreadMessage],\n _actions: &[crate::types::capability::ActionDef],\n config: &LlmCallConfig,\n ) -> Result {\n self.captured.lock().await.push(config.model.clone());\n Ok(crate::traits::llm::LlmOutput {\n response: crate::types::step::LlmResponse::Text(\"ok\".into()),\n usage: crate::types::step::TokenUsage::default(),\n })\n }\n }\n\n /// No-op effect executor — handle_llm_complete only consults it for\n /// `available_actions(...)`, which we satisfy with an empty list.\n struct NoopEffects;\n\n #[async_trait::async_trait]\n impl EffectExecutor for NoopEffects {\n async fn execute_action(\n &self,\n _: &str,\n _: serde_json::Value,\n _: &crate::types::capability::CapabilityLease,\n _: &ThreadExecutionContext,\n ) -> Result {\n Ok(crate::types::step::ActionResult {\n call_id: String::new(),\n action_name: String::new(),\n output: serde_json::json!({}),\n is_error: false,\n duration: std::time::Duration::from_millis(1),\n })\n }\n\n async fn available_actions(\n &self,\n _: &[crate::types::capability::CapabilityLease],\n ) -> Result, EngineError> {\n Ok(vec![])\n }\n }\n\n #[tokio::test]\n async fn llm_complete_forwards_model_from_explicit_config() {\n let concrete = Arc::new(ModelCapturingLlm {\n captured: tokio::sync::Mutex::new(Vec::new()),\n });\n let llm: Arc = Arc::clone(&concrete) as Arc;\n let effects: Arc = Arc::new(NoopEffects);\n let leases = Arc::new(LeaseManager::new());\n let store: Arc = Arc::new(crate::tests::InMemoryStore::with_docs(vec![]));\n\n let mut thread = Thread::new(\n \"goal\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n // Build the args __llm_complete__ receives from Python:\n // (messages, actions, config). config = {\"model\": \"gpt-4o\"}.\n let mut total_tokens = TokenUsage::default();\n let result = handle_llm_complete(\n &[\n json_to_monty(&serde_json::json!([{\"role\":\"user\",\"content\":\"hi\"}])),\n json_to_monty(&serde_json::json!([])),\n json_to_monty(&serde_json::json!({\"model\": \"gpt-4o\"})),\n ],\n &[],\n &mut thread,\n LlmCompleteDeps {\n llm: &llm,\n effects: &effects,\n leases: &leases,\n store: Some(&store),\n },\n &mut total_tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Return(_)));\n let captured = concrete.captured.lock().await;\n assert_eq!(captured.len(), 1);\n assert_eq!(captured[0].as_deref(), Some(\"gpt-4o\"));\n }\n\n #[tokio::test]\n async fn llm_complete_without_model_passes_none() {\n let concrete = Arc::new(ModelCapturingLlm {\n captured: tokio::sync::Mutex::new(Vec::new()),\n });\n let llm: Arc = Arc::clone(&concrete) as Arc;\n let effects: Arc = Arc::new(NoopEffects);\n let leases = Arc::new(LeaseManager::new());\n let store: Arc = Arc::new(crate::tests::InMemoryStore::with_docs(vec![]));\n\n let mut thread = Thread::new(\n \"goal\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n let mut total_tokens = TokenUsage::default();\n let _ = handle_llm_complete(\n &[\n json_to_monty(&serde_json::json!([{\"role\":\"user\",\"content\":\"hi\"}])),\n json_to_monty(&serde_json::json!([])),\n json_to_monty(&serde_json::json!({\"max_tokens\": 100})),\n ],\n &[],\n &mut thread,\n LlmCompleteDeps {\n llm: &llm,\n effects: &effects,\n leases: &leases,\n store: Some(&store),\n },\n &mut total_tokens,\n )\n .await;\n\n let captured = concrete.captured.lock().await;\n assert_eq!(captured.len(), 1);\n assert_eq!(captured[0], None);\n }\n\n // ── Python ↔ Rust ActionCall round-trip ───────────────────────────────\n //\n // Regression tests for the orphaned-tool-result bug. The Python\n // orchestrator stores `action_calls` on assistant messages using the\n // shape `{name, call_id, params}`, but the canonical Rust `ActionCall`\n // uses `{action_name, id, parameters}`. Without the explicit\n // `PythonActionCall` interchange type, `serde_json::from_value` would\n // silently fail (`.ok()` swallows the error) and the Python-shaped\n // assistant message would be parsed back as a plain assistant message\n // with no tool calls, causing every subsequent ActionResult to be\n // detected as orphaned by `sanitize_tool_messages` in the host crate.\n\n #[test]\n fn python_action_call_round_trips_through_serde() {\n let original = ActionCall {\n id: \"call_abc123\".to_string(),\n action_name: \"google_drive_tool\".to_string(),\n parameters: serde_json::json!({\"query\": \"expenses\"}),\n };\n\n let python_json = serde_json::to_value(PythonActionCall::from(&original))\n .expect(\"PythonActionCall must serialize\");\n // Python-friendly field names — match what default.py reads.\n assert_eq!(python_json[\"name\"], \"google_drive_tool\");\n assert_eq!(python_json[\"call_id\"], \"call_abc123\");\n assert_eq!(\n python_json[\"params\"],\n serde_json::json!({\"query\": \"expenses\"})\n );\n\n let parsed: PythonActionCall =\n serde_json::from_value(python_json).expect(\"must deserialize\");\n let round_tripped: ActionCall = parsed.into();\n assert_eq!(round_tripped.id, original.id);\n assert_eq!(round_tripped.action_name, original.action_name);\n assert_eq!(round_tripped.parameters, original.parameters);\n }\n\n #[test]\n fn action_calls_to_python_json_uses_python_field_names() {\n let calls = vec![\n ActionCall {\n id: \"call_1\".to_string(),\n action_name: \"notion_notion_search\".to_string(),\n parameters: serde_json::json!({\"query\": \"name\"}),\n },\n ActionCall {\n id: \"call_2\".to_string(),\n action_name: \"google_drive_tool\".to_string(),\n parameters: serde_json::json!({\"action\": \"list\"}),\n },\n ];\n let json = action_calls_to_python_json(&calls);\n assert_eq!(json.len(), 2);\n assert_eq!(json[0][\"name\"], \"notion_notion_search\");\n assert_eq!(json[0][\"call_id\"], \"call_1\");\n assert_eq!(json[1][\"name\"], \"google_drive_tool\");\n assert_eq!(json[1][\"call_id\"], \"call_2\");\n }\n\n #[test]\n fn python_json_to_action_calls_parses_python_field_names() {\n // The exact shape default.py produces (and stores on assistant\n // messages via `append_message(..., action_calls=calls)`).\n let python_json = serde_json::json!([\n {\"name\": \"notion_notion_search\", \"call_id\": \"call_xyz\", \"params\": {\"q\": \"foo\"}},\n {\"name\": \"google_drive_tool\", \"call_id\": \"call_abc\", \"params\": {\"action\": \"list\"}},\n ]);\n let parsed = python_json_to_action_calls(&python_json).expect(\"must parse\");\n assert_eq!(parsed.len(), 2);\n assert_eq!(parsed[0].action_name, \"notion_notion_search\");\n assert_eq!(parsed[0].id, \"call_xyz\");\n assert_eq!(parsed[0].parameters, serde_json::json!({\"q\": \"foo\"}));\n assert_eq!(parsed[1].action_name, \"google_drive_tool\");\n assert_eq!(parsed[1].id, \"call_abc\");\n }\n\n #[test]\n fn python_json_to_action_calls_rejects_canonical_field_names() {\n // Sanity check: the parser is strict about Python field names.\n // If `default.py` ever changes the shape, the test must catch it.\n let canonical_json = serde_json::json!([\n {\"action_name\": \"search\", \"id\": \"call_x\", \"parameters\": {}}\n ]);\n // Missing \"name\", \"call_id\", \"params\" → returns None.\n assert!(python_json_to_action_calls(&canonical_json).is_none());\n }\n\n #[test]\n fn summarize_action_calls_for_log_does_not_leak_user_pii() {\n // The whole point of this helper is that the warn log path on a\n // shape-drift failure must NOT dump tool parameters (which can\n // contain user PII like search queries, file names, email content)\n // into log aggregation systems. The summary should expose only\n // structural information: array length and the keys of the first\n // entry. The keys themselves are static (`name`, `call_id`,\n // `params`), not user data.\n let pii_value = serde_json::json!([\n {\n \"name\": \"google_drive_tool\",\n \"call_id\": \"call_xyz\",\n \"params\": {\n \"query\": \"salary spreadsheet for joe\",\n \"secret_token\": \"very-sensitive-token-do-not-log\"\n }\n },\n {\n \"name\": \"gmail\",\n \"call_id\": \"call_abc\",\n \"params\": {\n \"subject\": \"private message about layoffs\"\n }\n }\n ]);\n let summary = summarize_action_calls_for_log(&pii_value);\n\n // Structural info present.\n assert!(summary.contains(\"array of 2 entries\"));\n assert!(summary.contains(\"call_id\"));\n assert!(summary.contains(\"name\"));\n assert!(summary.contains(\"params\"));\n\n // PII fields and their values must NOT appear.\n assert!(\n !summary.contains(\"salary\"),\n \"summary must not leak user PII from params: {summary}\"\n );\n assert!(\n !summary.contains(\"very-sensitive-token\"),\n \"summary must not leak credential-shaped values: {summary}\"\n );\n assert!(\n !summary.contains(\"layoffs\"),\n \"summary must not leak free-text content: {summary}\"\n );\n assert!(\n !summary.contains(\"google_drive_tool\"),\n \"summary must not leak the tool name itself (could expose intent): {summary}\"\n );\n }\n\n #[test]\n fn summarize_action_calls_for_log_handles_edge_cases() {\n assert_eq!(\n summarize_action_calls_for_log(&serde_json::json!([])),\n \"empty array\"\n );\n assert!(\n summarize_action_calls_for_log(&serde_json::json!(\"not an array\")).contains(\"string\")\n );\n assert!(\n summarize_action_calls_for_log(&serde_json::json!({\"foo\": \"bar\"})).contains(\"object\")\n );\n assert!(summarize_action_calls_for_log(&serde_json::json!(null)).contains(\"null\"));\n }\n\n /// Caller-level regression test: feeds `json_to_thread_messages` the\n /// exact JSON shape that `default.py` produces for an assistant message\n /// with tool calls followed by tool results, and asserts that the\n /// resulting `ThreadMessage`s preserve the `action_calls` ↔\n /// `action_call_id` linkage. Without the `PythonActionCall` parser the\n /// assistant message would come back with `action_calls = None` and\n /// every following ActionResult would look orphaned to the bridge.\n #[test]\n fn json_to_thread_messages_preserves_action_calls_from_python_orchestrator() {\n // This is the literal shape `default.py` writes into\n // `state[\"working_messages\"]` after a Tier 0 step:\n //\n // append_message(working_messages, \"Assistant\", \"...\", action_calls=calls)\n // append_message(working_messages, \"ActionResult\", \"...\", action_name=..., action_call_id=...)\n //\n // where `calls` came from the LLM response and has shape\n // `[{\"name\": ..., \"call_id\": ..., \"params\": ...}]`.\n let working_messages = serde_json::json!([\n {\"role\": \"User\", \"content\": \"search in notion for my name\"},\n {\n \"role\": \"Assistant\",\n \"content\": \"\",\n \"action_calls\": [\n {\n \"name\": \"notion_notion_search\",\n \"call_id\": \"call_xyz\",\n \"params\": {\"query\": \"Illia\"}\n }\n ]\n },\n {\n \"role\": \"ActionResult\",\n \"content\": \"found 3 results\",\n \"action_name\": \"notion_notion_search\",\n \"action_call_id\": \"call_xyz\"\n }\n ]);\n\n let messages = json_to_thread_messages(&working_messages).expect(\"must parse\");\n assert_eq!(messages.len(), 3);\n\n // The assistant message MUST have action_calls populated, with\n // matching call_id. If this assertion fails, the bridge layer\n // will treat the following ActionResult as orphaned and rewrite\n // it as a user message — losing the model's ability to reason\n // about prior tool output.\n let assistant = &messages[1];\n assert_eq!(\n assistant.role,\n crate::types::message::MessageRole::Assistant\n );\n let calls = assistant\n .action_calls\n .as_ref()\n .expect(\"assistant message must carry action_calls after round-trip\");\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].id, \"call_xyz\");\n assert_eq!(calls[0].action_name, \"notion_notion_search\");\n assert_eq!(calls[0].parameters, serde_json::json!({\"query\": \"Illia\"}));\n\n // The ActionResult must reference the same call_id so the bridge\n // can pair them.\n let result = &messages[2];\n assert_eq!(\n result.role,\n crate::types::message::MessageRole::ActionResult\n );\n assert_eq!(result.action_call_id.as_deref(), Some(\"call_xyz\"));\n assert_eq!(result.action_name.as_deref(), Some(\"notion_notion_search\"));\n }\n\n /// Regression for the gate-resume / bootstrap path: when a thread\n /// resumes after approval or auth, `build_orchestrator_inputs`\n /// serializes `thread.internal_messages` into the bootstrap context\n /// that Python reads into `working_messages`. If `action_calls` is\n /// serialized with canonical `ActionCall` field names (`action_name`,\n /// `id`, `parameters`) instead of the Python interchange names\n /// (`name`, `call_id`, `params`), the next `__llm_complete__` call\n /// passes them back through `json_to_thread_messages` which fails\n /// with \"missing field `name`\" and orphans every subsequent tool\n /// result.\n ///\n /// This test simulates the full round-trip: build a `ThreadMessage`\n /// with action_calls → serialize through `build_orchestrator_inputs`'s\n /// exact serialization pattern → parse back through\n /// `json_to_thread_messages` → assert the calls survive. If anyone\n /// adds a THIRD serialization path in the future and uses canonical\n /// names, this test documents the pattern they should follow.\n #[test]\n fn bootstrap_context_action_calls_round_trip_through_python_interchange() {\n // Build a thread message the way the engine does: an assistant\n // message with action_calls in canonical ActionCall format (the\n // shape stored in the DB / internal_messages).\n let msg = ThreadMessage::assistant_with_actions(\n Some(\"I'll search for that\".to_string()),\n vec![ActionCall {\n id: \"call_resume_test\".to_string(),\n action_name: \"google_drive_tool\".to_string(),\n parameters: serde_json::json!({\"query\": \"budget\"}),\n }],\n );\n\n // Serialize through the SAME pattern `build_orchestrator_inputs`\n // uses. This is the exact code path that was broken before the\n // fix — it was using `\"action_calls\": m.action_calls` which\n // produced canonical field names.\n let calls_json = msg\n .action_calls\n .as_ref()\n .map(|calls| serde_json::Value::Array(action_calls_to_python_json(calls)));\n let serialized = serde_json::json!([{\n \"role\": \"Assistant\",\n \"content\": msg.content,\n \"action_name\": msg.action_name,\n \"action_call_id\": msg.action_call_id,\n \"action_calls\": calls_json,\n }]);\n\n // Parse back through the same path Python's working_messages\n // takes when it calls __llm_complete__.\n let parsed = json_to_thread_messages(&serialized).expect(\"must parse\");\n assert_eq!(parsed.len(), 1);\n\n let assistant = &parsed[0];\n let calls = assistant.action_calls.as_ref().expect(\n \"bootstrap context action_calls must survive the round-trip. \\\n If this fails, a serialization path is using canonical ActionCall \\\n field names instead of PythonActionCall interchange names.\",\n );\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].id, \"call_resume_test\");\n assert_eq!(calls[0].action_name, \"google_drive_tool\");\n assert_eq!(calls[0].parameters, serde_json::json!({\"query\": \"budget\"}));\n }\n\n /// Negative regression: verify that canonical ActionCall field names\n /// do NOT round-trip. If this test ever PASSES, it means someone\n /// added `#[serde(rename)]` to ActionCall or changed the parser to\n /// accept both formats — which is fine, but the PythonActionCall\n /// interchange type can then be removed. This test documents the\n /// current contract: canonical names are rejected by the parser.\n #[test]\n fn canonical_action_call_field_names_do_not_round_trip() {\n let serialized_with_canonical_names = serde_json::json!([{\n \"role\": \"Assistant\",\n \"content\": \"\",\n \"action_calls\": [{\n \"action_name\": \"search\",\n \"id\": \"call_x\",\n \"parameters\": {}\n }],\n }]);\n let parsed =\n json_to_thread_messages(&serialized_with_canonical_names).expect(\"messages parse\");\n // The assistant message should have NO action_calls because the\n // parser rejects canonical field names.\n assert!(\n parsed[0].action_calls.is_none(),\n \"canonical ActionCall field names must NOT parse as action_calls. \\\n If this assertion fails, the PythonActionCall interchange type \\\n is no longer needed — either remove it or update the contract.\"\n );\n }\n\n /// Regression: `action_calls: null` is Python's legitimate \"this\n /// message has no tool calls\" signal (text-only response). Before the\n /// null filter, `python_json_to_action_calls` would fire a warn log\n /// with \"invalid type: null, expected a sequence\" on every text-only\n /// assistant message — a false alarm that masked real drift issues.\n #[test]\n fn json_to_thread_messages_handles_null_action_calls_gracefully() {\n let messages = serde_json::json!([\n {\n \"role\": \"Assistant\",\n \"content\": \"Here is your answer.\",\n \"action_calls\": null\n }\n ]);\n let parsed = json_to_thread_messages(&messages).expect(\"must parse\");\n assert_eq!(parsed.len(), 1);\n assert_eq!(\n parsed[0].role,\n crate::types::message::MessageRole::Assistant\n );\n assert_eq!(parsed[0].content, \"Here is your answer.\");\n assert!(\n parsed[0].action_calls.is_none(),\n \"null action_calls must produce None, not a parse error\"\n );\n }\n\n /// Verify that messages WITHOUT the action_calls key at all (the most\n /// common case for text responses) also parse correctly — this is the\n /// baseline that the null-filtering regression test extends.\n #[test]\n fn json_to_thread_messages_handles_absent_action_calls() {\n let messages = serde_json::json!([\n {\"role\": \"Assistant\", \"content\": \"Just text, no tools.\"}\n ]);\n let parsed = json_to_thread_messages(&messages).expect(\"must parse\");\n assert_eq!(parsed.len(), 1);\n assert!(parsed[0].action_calls.is_none());\n }\n\n /// Empty action_calls array is valid (LLM decided not to call any\n /// tools this turn but the response still has the array field). Must\n /// produce `Some(vec![])`, not `None`.\n #[test]\n fn json_to_thread_messages_handles_empty_action_calls_array() {\n let messages = serde_json::json!([\n {\n \"role\": \"Assistant\",\n \"content\": \"No tools needed.\",\n \"action_calls\": []\n }\n ]);\n let parsed = json_to_thread_messages(&messages).expect(\"must parse\");\n assert_eq!(parsed.len(), 1);\n let calls = parsed[0]\n .action_calls\n .as_ref()\n .expect(\"empty array should produce Some(vec![])\");\n assert!(calls.is_empty());\n }\n\n // ── Consecutive action error counting (issue #2325) ──────────\n //\n // The run_loop tracks `consecutive_action_errors` for Tier 0 (structured\n // action calls). These tests exercise the counting logic extracted from\n // run_loop into small Python snippets that simulate batch outcomes.\n\n #[test]\n fn action_errors_increment_when_all_actions_fail() {\n // Simulate 3 consecutive batches where all actions fail.\n let count = eval_python_int(\n r#\"\nconsecutive_action_errors = 0\nfor _ in range(3):\n batch_error_count = 2\n batch_success_count = 0\n if batch_success_count > 0:\n consecutive_action_errors = 0\n elif batch_error_count > 0:\n consecutive_action_errors += 1\nFINAL(consecutive_action_errors)\n\"#,\n );\n assert_eq!(count, 3);\n }\n\n #[test]\n fn action_errors_reset_when_any_action_succeeds() {\n // 2 all-fail batches, then 1 batch with a success => resets to 0.\n let count = eval_python_int(\n r#\"\nconsecutive_action_errors = 0\nfor batch in [(0, 2), (0, 1), (1, 1)]:\n batch_success_count = batch[0]\n batch_error_count = batch[1]\n if batch_success_count > 0:\n consecutive_action_errors = 0\n elif batch_error_count > 0:\n consecutive_action_errors += 1\nFINAL(consecutive_action_errors)\n\"#,\n );\n assert_eq!(count, 0);\n }\n\n #[test]\n fn action_errors_partial_success_resets_counter() {\n // A batch with mixed results (some succeed, some fail) should reset.\n let count = eval_python_int(\n r#\"\nconsecutive_action_errors = 5\nbatch_success_count = 1\nbatch_error_count = 3\nif batch_success_count > 0:\n consecutive_action_errors = 0\nelif batch_error_count > 0:\n consecutive_action_errors += 1\nFINAL(consecutive_action_errors)\n\"#,\n );\n assert_eq!(count, 0);\n }\n\n #[test]\n fn action_errors_nudge_injected_at_threshold() {\n // When consecutive_action_errors reaches max_consecutive_errors,\n // a nudge message should be appended. We simulate the branching\n // logic and check whether a nudge would fire.\n // Returns 1 if nudge fires (not failure), 0 otherwise.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = 5\nconsecutive_action_errors = 5\nnudge = False\nfailed = False\nif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nelif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors:\n nudge = True\nif nudge and not failed:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 1, \"nudge should fire at threshold\");\n }\n\n #[test]\n fn action_errors_no_nudge_below_threshold() {\n // Returns 1 if nudge fires, 0 if not.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = 5\nconsecutive_action_errors = 4\nnudge = False\nfailed = False\nif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nelif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors:\n nudge = True\nif nudge:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 0, \"nudge should not fire below threshold\");\n }\n\n #[test]\n fn action_errors_failure_at_threshold_plus_two() {\n // At max_consecutive_errors + 2, the thread should transition to failed.\n // Returns 1 if failed, 0 if not.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = 5\nconsecutive_action_errors = 7\nfailed = False\nif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nif failed:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 1, \"should fail at threshold + 2\");\n }\n\n #[test]\n fn action_errors_nudge_at_threshold_not_failure() {\n // At exactly max_consecutive_errors + 1, we get a nudge but not failure.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = 5\nconsecutive_action_errors = 6\nnudge = False\nfailed = False\nif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nelif consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors:\n nudge = True\n# Return 0=nothing, 1=nudge, 2=failed\nif failed:\n FINAL(2)\nelif nudge:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 1, \"should nudge at threshold + 1, not fail\");\n }\n\n #[test]\n fn action_errors_none_limit_skips_check_without_typeerror() {\n // Regression: when max_consecutive_errors is None (meaning \"no limit\"),\n // the arithmetic `max_consecutive_errors + 2` used to crash with\n // TypeError on the first action error. The guard must short-circuit\n // on None and leave both the nudge and failure branches untaken.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = None\nconsecutive_action_errors = 1\nnudge = False\nfailed = False\nif max_consecutive_errors is not None and consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors + 2:\n failed = True\nelif max_consecutive_errors is not None and consecutive_action_errors > 0 and consecutive_action_errors >= max_consecutive_errors:\n nudge = True\n# Return 0=nothing, 1=nudge, 2=failed\nif failed:\n FINAL(2)\nelif nudge:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 0, \"None limit should disable the guard entirely\");\n }\n\n #[test]\n fn code_errors_none_limit_skips_failure_check() {\n // Regression: same None-guard for the code-error branch at line 660.\n let result = eval_python_int(\n r#\"\nmax_consecutive_errors = None\nconsecutive_errors = 99\nfailed = False\nif max_consecutive_errors is not None and consecutive_errors >= max_consecutive_errors:\n failed = True\nif failed:\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(\n result, 0,\n \"None limit should not trigger failure regardless of consecutive_errors\"\n );\n }\n\n #[test]\n fn action_error_prefix_added_to_error_output() {\n // Verify that [ACTION FAILED] prefix is prepended to error outputs.\n // Returns 1 if prefix present, 0 if not.\n let result = eval_python_int(\n r#\"\nr = {\"action_name\": \"http\", \"output\": \"connection refused\", \"is_error\": True}\noutput = r.get(\"output\")\noutput_str = str(output) if output is not None else \"[no output]\"\nif r.get(\"is_error\"):\n output_str = \"[ACTION FAILED] \" + output_str\nif output_str.startswith(\"[ACTION FAILED]\"):\n FINAL(1)\nelse:\n FINAL(0)\n\"#,\n );\n assert_eq!(result, 1, \"error outputs must get [ACTION FAILED] prefix\");\n }\n\n #[test]\n fn action_error_skipped_calls_count_as_errors() {\n // When a call has no result (r is None), it should count as an error.\n let count = eval_python_int(\n r#\"\nbatch_error_count = 0\nbatch_success_count = 0\nr = None\nif r is not None:\n if r.get(\"is_error\"):\n batch_error_count += 1\n else:\n batch_success_count += 1\nelse:\n batch_error_count += 1\nFINAL(batch_error_count)\n\"#,\n );\n assert_eq!(count, 1, \"skipped calls must count as batch errors\");\n }\n\n #[test]\n fn checkpoint_includes_consecutive_action_errors() {\n // Test that handle_save_checkpoint persists consecutive_action_errors\n // in the thread metadata.\n let mut thread = Thread::new(\n \"goal\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n let state = json_to_monty(&serde_json::json!({}));\n let counters = json_to_monty(&serde_json::json!({\n \"nudge_count\": 0,\n \"consecutive_errors\": 1,\n \"consecutive_action_errors\": 4,\n \"compaction_count\": 2,\n }));\n\n handle_save_checkpoint(&[state, counters], &[], &mut thread);\n\n let checkpoint = thread\n .metadata\n .get(\"runtime_checkpoint\")\n .expect(\"checkpoint must exist\");\n assert_eq!(\n checkpoint\n .get(\"consecutive_action_errors\")\n .and_then(|v| v.as_u64()),\n Some(4),\n \"consecutive_action_errors must be persisted in checkpoint\"\n );\n assert_eq!(\n checkpoint\n .get(\"consecutive_errors\")\n .and_then(|v| v.as_u64()),\n Some(1),\n );\n assert_eq!(\n checkpoint.get(\"compaction_count\").and_then(|v| v.as_u64()),\n Some(2),\n );\n }\n\n /// Regression test: every assistant tool_call must have a matching\n /// ActionResult after parsing. If an ActionResult is missing, the LLM\n /// API rejects with \"No tool output found for function call \".\n ///\n /// This was the root cause of the HTTP 400 from the OpenAI Codex\n /// provider: a tool returning null output caused the Python\n /// orchestrator to skip appending the ActionResult.\n #[test]\n fn json_to_thread_messages_every_tool_call_has_action_result() {\n // Simulate working_messages after the Python fix: every call gets\n // an ActionResult, even when the original output was null.\n let messages = serde_json::json!([\n {\"role\": \"System\", \"content\": \"You are a helpful assistant.\"},\n {\"role\": \"User\", \"content\": \"Update all tools.\"},\n {\n \"role\": \"Assistant\",\n \"content\": \"\",\n \"action_calls\": [\n {\"call_id\": \"call_AAA\", \"name\": \"tool_a\", \"params\": {}},\n {\"call_id\": \"call_BBB\", \"name\": \"tool_b\", \"params\": {}},\n {\"call_id\": \"call_CCC\", \"name\": \"tool_c\", \"params\": {}}\n ]\n },\n {\n \"role\": \"ActionResult\",\n \"content\": \"{\\\"ok\\\": true}\",\n \"action_name\": \"tool_a\",\n \"action_call_id\": \"call_AAA\"\n },\n {\n \"role\": \"ActionResult\",\n \"content\": \"[no output]\",\n \"action_name\": \"tool_b\",\n \"action_call_id\": \"call_BBB\"\n },\n {\n \"role\": \"ActionResult\",\n \"content\": \"{\\\"done\\\": true}\",\n \"action_name\": \"tool_c\",\n \"action_call_id\": \"call_CCC\"\n }\n ]);\n\n let parsed = json_to_thread_messages(&messages).expect(\"must parse\");\n assert_eq!(parsed.len(), 6);\n\n // Extract call IDs from the assistant message\n let assistant_calls: std::collections::HashSet = parsed\n .iter()\n .filter_map(|m| m.action_calls.as_ref())\n .flat_map(|calls| calls.iter().map(|c| c.id.clone()))\n .collect();\n\n // Extract call IDs from ActionResult messages\n let result_call_ids: std::collections::HashSet = parsed\n .iter()\n .filter(|m| m.role == crate::types::message::MessageRole::ActionResult)\n .filter_map(|m| m.action_call_id.clone())\n .collect();\n\n // Every tool_call must have a matching ActionResult\n for call_id in &assistant_calls {\n assert!(\n result_call_ids.contains(call_id),\n \"tool_call {call_id} has no matching ActionResult — \\\n this would cause 'No tool output found' from the LLM API\"\n );\n }\n }\n\n // ── CodeExecutionFailed event emission (caller test) ────────\n\n #[tokio::test]\n async fn execute_code_step_emits_code_execution_failed_event() {\n let llm: Arc = Arc::new(ModelCapturingLlm {\n captured: tokio::sync::Mutex::new(Vec::new()),\n });\n let effects: Arc = Arc::new(NoopEffects);\n let leases = Arc::new(LeaseManager::new());\n let policy = Arc::new(PolicyEngine::new());\n\n let mut thread = Thread::new(\n \"test code execution failure instrumentation\",\n crate::types::thread::ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n crate::types::thread::ThreadConfig::default(),\n );\n thread.transition_to(ThreadState::Running, None).unwrap();\n\n // Pass intentionally broken Python code (syntax error)\n let args = &[\n json_to_monty(&serde_json::json!(\"def ==\")),\n json_to_monty(&serde_json::json!({})),\n ];\n\n let (tx, _rx) = tokio::sync::broadcast::channel(16);\n let _result = handle_execute_code_step(\n args,\n &[],\n &mut thread,\n &llm,\n &effects,\n &leases,\n &policy,\n Some(&tx),\n )\n .await;\n\n // Verify CodeExecutionFailed event was emitted on thread.events\n let code_failed_events: Vec<_> = thread\n .events\n .iter()\n .filter(|e| matches!(&e.kind, EventKind::CodeExecutionFailed { .. }))\n .collect();\n\n assert_eq!(\n code_failed_events.len(),\n 1,\n \"expected exactly one CodeExecutionFailed event, got {}\",\n code_failed_events.len()\n );\n\n if let EventKind::CodeExecutionFailed {\n category,\n code_hash,\n ..\n } = &code_failed_events[0].kind\n {\n assert_eq!(\n *category,\n crate::types::step::CodeExecutionFailure::SyntaxError\n );\n assert!(code_hash.is_some());\n } else {\n panic!(\"expected CodeExecutionFailed event kind\");\n }\n\n // Also verify ActionFailed was emitted (existing behavior)\n let action_failed = thread\n .events\n .iter()\n .any(|e| matches!(&e.kind, EventKind::ActionFailed { .. }));\n assert!(\n action_failed,\n \"expected ActionFailed event alongside CodeExecutionFailed\"\n );\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/executor/scripting.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "x-content-type-options", + "nosniff" + ], + [ + "content-length", + "113551" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "server", + "github.com" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-github-request-id", + "1449:17B00F:4754C5:52BC6C:69DFAF15" + ], + [ + "date", + "Wed, 15 Apr 2026 15:30:29 GMT" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "etag", + "\"674fca2750840d6f245ff8676d54df8b83ca74f1\"" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "x-ratelimit-used", + "38" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "x-xss-protection", + "0" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-frame-options", + "deny" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "x-ratelimit-remaining", + "4962" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ] + ], + "body": "//! Tier 1 executor: embedded Python via Monty.\n//!\n//! Executes LLM-generated Python code using the Monty interpreter. Tool\n//! calls use **async dispatch**: each tool call returns a Monty `ExternalFuture`\n//! via `resume_pending()`, allowing Python code to use `await` and\n//! `asyncio.gather()` for parallel execution. When all tasks are blocked,\n//! Monty yields `ResolveFutures` and we execute pending tools concurrently\n//! via `JoinSet`.\n//!\n//! Follows the RLM (Recursive Language Model) pattern:\n//! - Thread context injected as Python variables (not LLM attention input)\n//! - `llm_query()` / `llm_query_batched()` for recursive subagent spawning\n//! - `FINAL(answer)` / `FINAL_VAR(name)` for explicit termination\n//! - Step 0 orientation preamble for context awareness\n//! - Errors flow back to LLM for self-correction (not step termination)\n//! - Output truncated to configurable limit with variable listing\n//! - `asyncio.gather()` for parallel tool execution (via ResolveFutures)\n\nuse std::collections::HashMap;\nuse std::sync::Arc;\nuse std::time::Duration;\n\nuse monty::{\n ExcType, ExtFunctionResult, LimitedTracker, MontyException, MontyObject, MontyRun,\n NameLookupResult, PrintWriter, ResourceLimits, RunProgress,\n};\nuse tracing::debug;\n\nuse crate::capability::lease::LeaseManager;\nuse crate::capability::policy::{PolicyDecision, PolicyEngine};\nuse crate::traits::effect::{EffectExecutor, ThreadExecutionContext};\nuse crate::traits::llm::{LlmBackend, LlmCallConfig};\nuse crate::types::error::EngineError;\nuse crate::types::event::EventKind;\nuse crate::types::message::{MessageRole, ThreadMessage};\nuse crate::types::step::{ActionResult, CodeExecutionFailure, LlmResponse, TokenUsage};\nuse crate::types::thread::Thread;\nuse ironclaw_common::ValidTimezone;\n\n// ── Configuration ───────────────────────────────────────────\n\n/// Maximum characters of output to include in LLM context between steps.\n/// Matches Prime Intellect's default. Configurable per thread in the future.\nconst OUTPUT_TRUNCATE_LEN: usize = 8_000;\n\n/// Maximum characters for a preview prefix in compact metadata.\nconst OUTPUT_PREVIEW_LEN: usize = 200;\n\n/// Default resource limits for Monty execution.\nfn default_limits() -> ResourceLimits {\n ResourceLimits::new()\n .max_duration(Duration::from_secs(30))\n .max_allocations(1_000_000)\n .max_memory(64 * 1024 * 1024) // 64 MB\n}\n\n// ── Result types ────────────────────────────────────────────\n\n/// Result of executing a code block.\npub struct CodeExecutionResult {\n /// The Python return value, converted to JSON.\n pub return_value: serde_json::Value,\n /// Captured print output.\n pub stdout: String,\n /// All action calls that were made during execution.\n pub action_results: Vec,\n /// Events generated during execution.\n pub events: Vec,\n /// If set, execution was interrupted for approval.\n pub need_approval: Option,\n /// Tokens used by recursive llm_query() calls.\n pub recursive_tokens: TokenUsage,\n /// If set, the code called FINAL() or FINAL_VAR() with this answer.\n pub final_answer: Option,\n /// Classified failure category. `None` when execution succeeded or was\n /// paused by a gate. `Some(category)` when code execution failed —\n /// `failure.is_some()` replaces the former `had_error: bool` field.\n pub failure: Option,\n}\n\n/// Build a compact output summary for inclusion in LLM context between steps.\n///\n/// Truncates to `OUTPUT_TRUNCATE_LEN` (last N chars shown, like fast-rlm).\n/// Includes a list of REPL variable names if available.\npub fn compact_output_metadata(stdout: &str, return_value: &serde_json::Value) -> String {\n let mut parts = Vec::new();\n\n if !stdout.is_empty() {\n let char_count = stdout.chars().count();\n if char_count > OUTPUT_TRUNCATE_LEN {\n let truncated: String = stdout\n .chars()\n .skip(char_count - OUTPUT_TRUNCATE_LEN)\n .collect();\n parts.push(format!(\n \"[TRUNCATED: last {OUTPUT_TRUNCATE_LEN} of {char_count} chars shown]\\n{truncated}\",\n ));\n } else {\n parts.push(format!(\"[FULL OUTPUT: {char_count} chars]\\n{stdout}\"));\n }\n }\n\n if *return_value != serde_json::Value::Null {\n let val_str = serde_json::to_string_pretty(return_value).unwrap_or_default();\n let val_char_count = val_str.chars().count();\n if val_char_count > OUTPUT_PREVIEW_LEN {\n let preview: String = val_str.chars().take(OUTPUT_PREVIEW_LEN).collect();\n parts.push(format!(\n \"Return value ({val_char_count} chars): {preview}...\",\n ));\n } else {\n parts.push(format!(\"Return value: {val_str}\"));\n }\n }\n\n if parts.is_empty() {\n \"[code executed, no output]\".into()\n } else {\n parts.join(\"\\n\")\n }\n}\n\n// ── Step 0 orientation preamble ─────────────────────────────\n\n/// Build the Step 0 orientation preamble that auto-executes before the\n/// first LLM call to give the model structural awareness of the context.\npub fn build_orientation_preamble(thread: &Thread) -> String {\n let msg_count = thread.messages.len();\n let total_chars: usize = thread.messages.iter().map(|m| m.content.len()).sum();\n let user_msgs = thread\n .messages\n .iter()\n .filter(|m| m.role == MessageRole::User)\n .count();\n\n let mut preview = String::new();\n if let Some(last_user) = thread\n .messages\n .iter()\n .rev()\n .find(|m| m.role == MessageRole::User)\n {\n let content_preview: String = last_user.content.chars().take(500).collect();\n let truncated = if last_user.content.chars().count() > 500 {\n \"...\"\n } else {\n \"\"\n };\n preview = format!(\"\\nLast user message preview: {content_preview}{truncated}\");\n }\n\n format!(\n \"[Step 0 — Context Orientation]\\n\\\n Goal: {goal}\\n\\\n Context: {msg_count} messages, {total_chars} total chars, {user_msgs} from user\\n\\\n Step: {step}{preview}\",\n goal = thread.goal,\n step = thread.step_count + 1,\n )\n}\n\n// ── Context injection (RLM 3.4) ────────────────────────────\n\n/// Build Monty input variables from thread state.\n///\n/// `persisted_state` carries variables from previous code steps so the\n/// REPL feels persistent even though each step creates a fresh MontyRun.\nfn build_context_inputs(\n thread: &Thread,\n persisted_state: &serde_json::Value,\n) -> (Vec, Vec) {\n let mut names = Vec::new();\n let mut values = Vec::new();\n\n // `context` — thread messages as a list of dicts\n let messages: Vec = thread\n .messages\n .iter()\n .map(|msg| {\n let mut pairs = vec![\n (\n MontyObject::String(\"role\".into()),\n MontyObject::String(format!(\"{:?}\", msg.role)),\n ),\n (\n MontyObject::String(\"content\".into()),\n MontyObject::String(msg.content.clone()),\n ),\n ];\n if let Some(ref name) = msg.action_name {\n pairs.push((\n MontyObject::String(\"action_name\".into()),\n MontyObject::String(name.clone()),\n ));\n }\n MontyObject::dict(pairs)\n })\n .collect();\n names.push(\"context\".into());\n values.push(MontyObject::List(messages));\n\n // `goal` — the thread's goal string\n names.push(\"goal\".into());\n values.push(MontyObject::String(thread.goal.clone()));\n\n // `step_number` — current step index\n names.push(\"step_number\".into());\n values.push(MontyObject::Int(thread.step_count as i64));\n\n // `state` — persisted variables from previous code steps.\n // This is a dict that accumulates: return values, tool results, etc.\n // The model can read `state[\"results\"]`, `state[\"prev_return\"]`, etc.\n names.push(\"state\".into());\n values.push(json_to_monty(persisted_state));\n\n // `previous_results` — dict of {call_id: result_json} from prior steps\n let result_pairs: Vec<(MontyObject, MontyObject)> = thread\n .messages\n .iter()\n .filter(|m| m.role == MessageRole::ActionResult)\n .filter_map(|m| {\n let call_id = m.action_call_id.as_ref()?;\n Some((\n MontyObject::String(call_id.clone()),\n MontyObject::String(m.content.clone()),\n ))\n })\n .collect();\n names.push(\"previous_results\".into());\n values.push(MontyObject::dict(result_pairs));\n\n // `user_timezone` — validated IANA timezone from the user's channel (e.g. \"America/New_York\")\n let tz = thread\n .metadata\n .get(\"user_timezone\")\n .and_then(|v| v.as_str())\n .and_then(ValidTimezone::parse)\n .map(|vtz| vtz.name().to_string())\n .unwrap_or_else(|| \"UTC\".into());\n names.push(\"user_timezone\".into());\n values.push(MontyObject::String(tz));\n\n (names, values)\n}\n\n// ── Main execution function ─────────────────────────────────\n\n/// Execute a Python code block using Monty.\n///\n/// Handles the full RLM execution pattern: context-as-variables, FINAL()\n/// termination, llm_query() recursive calls, error-to-LLM flow, and\n/// output truncation.\n#[allow(clippy::too_many_arguments)]\npub async fn execute_code(\n code: &str,\n thread: &Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &LeaseManager,\n policy: &PolicyEngine,\n context: &ThreadExecutionContext,\n capability_policies: &[crate::types::capability::PolicyRule],\n persisted_state: &serde_json::Value,\n) -> Result {\n execute_code_with_skills(\n code,\n thread,\n llm,\n effects,\n leases,\n policy,\n context,\n capability_policies,\n persisted_state,\n &[],\n )\n .await\n}\n\n/// Execute a Python code block with optional skill code snippets.\n///\n/// `skill_snippet_names` are registered as additional known functions in the\n/// Monty NameLookup, alongside tool names from capability leases.\n#[allow(clippy::too_many_arguments)]\npub async fn execute_code_with_skills(\n code: &str,\n thread: &Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &LeaseManager,\n policy: &PolicyEngine,\n context: &ThreadExecutionContext,\n capability_policies: &[crate::types::capability::PolicyRule],\n persisted_state: &serde_json::Value,\n skill_snippet_names: &[String],\n) -> Result {\n let mut stdout = String::new();\n let mut action_results = Vec::new();\n let mut events = Vec::new();\n let mut recursive_tokens = TokenUsage::default();\n let mut final_answer: Option = None;\n\n // Build context variables including persisted state from prior steps\n let (input_names, input_values) = build_context_inputs(thread, persisted_state);\n\n // Collect known tool names so NameLookup can return callable stubs.\n // Without this, `mission_list()` in code raises NameError because Monty\n // resolves the name before calling it, and Undefined → NameError.\n let active_leases = leases.active_for_thread(thread.id).await;\n let mut known_actions: std::collections::HashSet = effects\n .available_actions(&active_leases)\n .await\n .unwrap_or_default()\n .into_iter()\n .map(|a| a.name)\n .collect();\n\n // Register skill code snippet function names as additional known actions.\n // These resolve in NameLookup so the LLM can call them as Python functions.\n for name in skill_snippet_names {\n known_actions.insert(name.clone());\n }\n\n // Parse and compile (wrap in catch_unwind — Monty 0.0.x can panic)\n let runner = match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n MontyRun::new(code.to_string(), \"step.py\", input_names)\n })) {\n Ok(Ok(runner)) => runner,\n Ok(Err(e)) => {\n // Parse error flows back to LLM (not a termination)\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"SyntaxError: {e}\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n failure: Some(CodeExecutionFailure::SyntaxError),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"{stdout}\\nVmPanic: Monty VM panicked during code parsing\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n };\n\n // Start execution with resource limits and context inputs\n let tracker = LimitedTracker::new(default_limits());\n\n let run_result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n runner.start(input_values, tracker, PrintWriter::Collect(&mut stdout))\n }));\n\n let mut progress = match run_result {\n Ok(Ok(p)) => p,\n Ok(Err(e)) => {\n // Runtime error flows back to LLM\n let category = classify_runtime_error(&e.to_string());\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"{stdout}\\nError: {e}\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n failure: Some(category),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"{stdout}\\nVmPanic: Monty VM panicked during execution start\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer: None,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n };\n\n // Pending async tool executions keyed by Monty call_id.\n // When a tool FunctionCall comes in, we spawn a tokio task and store\n // the JoinHandle here. When ResolveFutures yields, we await them.\n let mut pending_futures: HashMap = HashMap::new();\n\n // Drive the execution loop\n let mut call_counter = 0u32;\n loop {\n match progress {\n RunProgress::Complete(obj) => {\n return Ok(CodeExecutionResult {\n return_value: monty_to_json(&obj),\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: None,\n });\n }\n\n RunProgress::FunctionCall(call) => {\n call_counter += 1;\n let str_call_id = format!(\"code_call_{call_counter}\");\n let monty_call_id = call.call_id;\n let action_name = call.function_name.clone();\n let params = monty_args_to_json(&call.args, &call.kwargs);\n\n debug!(action = %action_name, call_id = %str_call_id, monty_id = monty_call_id, \"Monty: function call\");\n\n // Builtins that need synchronous results — resume with value.\n let sync_result = match action_name.as_str() {\n \"FINAL\" => {\n let answer = call.args.first().map(monty_to_string).unwrap_or_default();\n final_answer = Some(answer);\n Some(ExtFunctionResult::Return(MontyObject::None))\n }\n \"FINAL_VAR\" => {\n let var_name = call\n .args\n .first()\n .map(monty_to_string)\n .unwrap_or_else(|| \"result\".into());\n final_answer = Some(format!(\"[FINAL_VAR: {var_name}]\"));\n Some(ExtFunctionResult::Return(MontyObject::None))\n }\n // LLM calls are async — spawn tokio task, resume_pending.\n // This allows asyncio.gather(llm_query(...), tool(...))\n // to run the LLM call and tool call concurrently.\n \"llm_query\" => {\n let args = call.args.clone();\n let kwargs = call.kwargs.clone();\n let llm = llm.clone();\n let handle = tokio::spawn(async move {\n handle_llm_query_standalone(&args, &kwargs, &llm).await\n });\n pending_futures.insert(monty_call_id, PendingFuture::Llm { handle });\n None // handled as async below\n }\n \"llm_query_batched\" => {\n let args = call.args.clone();\n let kwargs = call.kwargs.clone();\n let llm = llm.clone();\n let handle = tokio::spawn(async move {\n handle_llm_query_batched_standalone(&args, &kwargs, &llm).await\n });\n pending_futures.insert(monty_call_id, PendingFuture::Llm { handle });\n None\n }\n // rlm_query stays synchronous — it spawns a child Monty VM\n // which isn't Send, so it can't run in tokio::spawn.\n \"rlm_query\" => Some(\n handle_rlm_query(\n &call.args,\n &call.kwargs,\n thread,\n llm,\n effects,\n leases,\n policy,\n &mut recursive_tokens,\n )\n .await,\n ),\n \"globals\" | \"locals\" => {\n let entries: Vec<(MontyObject, MontyObject)> = known_actions\n .iter()\n .map(|name| {\n (MontyObject::String(name.clone()), MontyObject::Bool(true))\n })\n .collect();\n Some(ExtFunctionResult::Return(MontyObject::Dict(entries.into())))\n }\n _ => None, // tool call — handled async below\n };\n\n if let Some(ext_result) = sync_result {\n // Sync resume for builtins\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume(ext_result, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(classify_runtime_error(&e.to_string())),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during resume\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n continue;\n }\n\n // If an LLM call already inserted a pending future, just\n // resume_pending and continue — no preflight needed.\n if pending_futures.contains_key(&monty_call_id) {\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume_pending(PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(classify_runtime_error(&e.to_string())),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during resume_pending\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n continue;\n }\n\n // ── Async tool dispatch ─────────────────────────────\n // Preflight (lease + policy) is sync. If denied or\n // needs approval, resume with error immediately.\n // If approved, spawn tokio task and resume_pending().\n\n let preflight = preflight_action(\n &action_name,\n ¶ms,\n thread,\n effects,\n leases,\n policy,\n context,\n capability_policies,\n &str_call_id,\n &mut events,\n )\n .await;\n\n match preflight {\n PreflightResult::Approved(lease) => {\n // Spawn async execution\n let effects = effects.clone();\n let name = action_name.clone();\n let params_clone = params.clone();\n let lease_clone = lease.clone();\n let mut ctx = context.clone();\n ctx.current_call_id = Some(str_call_id.clone());\n let ps = crate::types::event::summarize_params(&name, ¶ms);\n\n let handle = tokio::spawn(async move {\n effects\n .execute_action(&name, params_clone, &lease_clone, &ctx)\n .await\n });\n\n pending_futures.insert(\n monty_call_id,\n PendingFuture::Tool {\n handle,\n action_name,\n call_id: str_call_id,\n lease_id: lease.id,\n parameters: params.clone(),\n params_summary: ps,\n },\n );\n\n // Resume with pending future — Python gets ExternalFuture\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume_pending(PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::ToolError),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during resume_pending\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n PreflightResult::Denied(ext_result) => {\n // Resume with error — Python sees an exception\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n call.resume(ext_result, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::ToolError),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during resume\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n PreflightResult::GatePaused(outcome) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: Some(outcome),\n recursive_tokens,\n final_answer: None,\n failure: None,\n });\n }\n }\n }\n\n // ── ResolveFutures: parallel execution ────────────────\n // Resolves both tool calls and LLM calls that were deferred\n // via resume_pending(). All pending tokio tasks are awaited\n // and their results fed back to Monty.\n RunProgress::ResolveFutures(resolve) => {\n let pending_ids = resolve.pending_call_ids().to_vec();\n debug!(pending = ?pending_ids, \"Monty: ResolveFutures — resolving {} pending futures\", pending_ids.len());\n\n let mut results: Vec<(u32, ExtFunctionResult)> =\n Vec::with_capacity(pending_ids.len());\n\n for &mid in &pending_ids {\n let ext_result = if let Some(pf) = pending_futures.remove(&mid) {\n match pf {\n PendingFuture::Tool {\n handle,\n action_name,\n call_id,\n lease_id,\n parameters,\n params_summary,\n } => {\n resolve_tool_future(\n handle,\n &action_name,\n &call_id,\n lease_id,\n parameters,\n params_summary,\n leases,\n context,\n &mut action_results,\n &mut events,\n )\n .await\n }\n PendingFuture::Llm { handle } => {\n resolve_llm_future(handle, &mut recursive_tokens).await\n }\n }\n } else {\n debug!(call_id = mid, \"ResolveFutures: unknown pending call_id\");\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"unknown pending call_id {mid}\")),\n ))\n };\n results.push((mid, ext_result));\n }\n\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n resolve.resume(results, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(classify_runtime_error(&e.to_string())),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during ResolveFutures resume\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n\n RunProgress::NameLookup(lookup) => {\n let name = lookup.name.clone();\n\n let result = if known_actions.contains(&name) {\n debug!(name = %name, \"Monty: resolved as tool function\");\n NameLookupResult::Value(MontyObject::Function {\n name: name.clone(),\n docstring: None,\n })\n } else if name == \"globals\" || name == \"locals\" {\n NameLookupResult::Value(MontyObject::Function {\n name: name.clone(),\n docstring: None,\n })\n } else {\n debug!(name = %name, \"Monty: unresolved name\");\n NameLookupResult::Undefined\n };\n\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n lookup.resume(result, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nNameError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::NameLookup),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\n \"{stdout}\\nVmPanic: Monty VM panicked during name lookup\"\n ),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n\n RunProgress::OsCall(os_call) => {\n debug!(function = ?os_call.function, \"Monty: OS call denied\");\n let err = ExtFunctionResult::Error(MontyException::new(\n ExcType::OSError,\n Some(\"OS operations are not permitted in CodeAct scripts\".into()),\n ));\n match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {\n os_call.resume(err, PrintWriter::Collect(&mut stdout))\n })) {\n Ok(Ok(p)) => progress = p,\n Ok(Err(e)) => {\n stdout.push_str(&format!(\"\\nOSError: {e}\"));\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout,\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::OsDenied),\n });\n }\n Err(_) => {\n return Ok(CodeExecutionResult {\n return_value: serde_json::Value::Null,\n stdout: format!(\"{stdout}\\nVmPanic: Monty VM panicked during OS call\"),\n action_results,\n events,\n need_approval: None,\n recursive_tokens,\n final_answer,\n failure: Some(CodeExecutionFailure::VmPanic),\n });\n }\n }\n }\n }\n }\n}\n\n// ── Error classification ────────────────────────────────────\n\n/// Classify a runtime error message into a failure category.\n///\n/// Parses the error text from Monty to distinguish between LLM logic bugs\n/// (NameError, TypeError, etc.), resource limit hits, and Monty VM issues.\nfn classify_runtime_error(error_msg: &str) -> CodeExecutionFailure {\n let lower = error_msg.to_ascii_lowercase();\n\n // Most specific checks first to avoid substring false positives.\n if lower.contains(\"timed out\")\n || lower.contains(\"timeout\")\n || lower.contains(\"memory limit\")\n || lower.contains(\"allocation limit\")\n || lower.contains(\"out of fuel\")\n || lower.contains(\"fuel exhausted\")\n || lower.contains(\"resource limit\")\n {\n CodeExecutionFailure::ResourceLimit\n } else if lower.contains(\"os operations are not permitted\") || lower.contains(\"oserror\") {\n CodeExecutionFailure::OsDenied\n } else if lower.contains(\"syntaxerror\") {\n CodeExecutionFailure::SyntaxError\n } else {\n // NameError, TypeError, ValueError, AttributeError, IndexError,\n // KeyError, ModuleNotFoundError, NotImplementedError, etc.\n CodeExecutionFailure::RuntimeError\n }\n}\n\n/// Compute a short hash of Python code for dedup/correlation in events.\n///\n/// Uses FNV-1a (64-bit) which is stable across Rust versions, unlike\n/// `DefaultHasher`. Not cryptographic — collision probability is ~2^-32\n/// at typical usage levels, sufficient for dedup but not for security.\npub fn code_hash(code: &str) -> String {\n const FNV_OFFSET: u64 = 0xcbf29ce484222325;\n const FNV_PRIME: u64 = 0x00000100000001B3;\n let mut hash = FNV_OFFSET;\n for byte in code.as_bytes() {\n hash ^= *byte as u64;\n hash = hash.wrapping_mul(FNV_PRIME);\n }\n format!(\"{hash:016x}\")\n}\n\n// ── Pending future tracking ─────────────────────────────────\n\n/// A deferred computation spawned as a tokio task, pending resolution\n/// via `ResolveFutures`. Can be a tool execution or an LLM call.\nenum PendingFuture {\n /// Tool action execution.\n Tool {\n handle: tokio::task::JoinHandle>,\n action_name: String,\n call_id: String,\n lease_id: crate::types::capability::LeaseId,\n parameters: serde_json::Value,\n params_summary: Option,\n },\n /// LLM call (llm_query / llm_query_batched / rlm_query).\n Llm {\n handle: tokio::task::JoinHandle<(ExtFunctionResult, TokenUsage)>,\n },\n}\n\n/// Result of preflight checks (lease + policy) for a tool call.\nenum PreflightResult {\n /// Tool approved — lease is consumed, ready to execute.\n Approved(crate::types::capability::CapabilityLease),\n /// Tool denied — return this error to Monty.\n Denied(ExtFunctionResult),\n /// Tool is paused by a gate — interrupt the batch.\n GatePaused(crate::runtime::messaging::ThreadOutcome),\n}\n\n/// Run preflight checks for a tool call: find lease, check policy, consume use.\n#[allow(clippy::too_many_arguments)]\nasync fn preflight_action(\n action_name: &str,\n params: &serde_json::Value,\n thread: &Thread,\n effects: &Arc,\n leases: &LeaseManager,\n policy: &PolicyEngine,\n context: &ThreadExecutionContext,\n capability_policies: &[crate::types::capability::PolicyRule],\n call_id: &str,\n events: &mut Vec,\n) -> PreflightResult {\n let lease = match leases.find_lease_for_action(thread.id, action_name).await {\n Some(l) => l,\n None => {\n events.push(EventKind::ActionFailed {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n error: format!(\"no lease for action '{action_name}'\"),\n params_summary: None,\n });\n return PreflightResult::Denied(ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"no lease for action '{action_name}'\")),\n )));\n }\n };\n\n let action_def = effects\n .available_actions(std::slice::from_ref(&lease))\n .await\n .ok()\n .and_then(|actions| actions.into_iter().find(|a| a.name == action_name));\n\n if let Some(ref action_def) = action_def {\n match policy.evaluate(action_def, &lease, capability_policies) {\n PolicyDecision::Deny { reason } => {\n events.push(EventKind::ActionFailed {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n error: reason.clone(),\n params_summary: None,\n });\n return PreflightResult::Denied(ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"denied: {reason}\")),\n )));\n }\n PolicyDecision::RequireApproval { .. } => {\n events.push(EventKind::ApprovalRequested {\n action_name: action_name.into(),\n call_id: call_id.into(),\n parameters: Some(params.clone()),\n description: None,\n allow_always: None,\n gate_name: None,\n params_summary: crate::types::event::summarize_params(action_name, params),\n });\n return PreflightResult::GatePaused(\n crate::runtime::messaging::ThreadOutcome::GatePaused {\n gate_name: \"approval\".into(),\n action_name: action_name.into(),\n call_id: call_id.into(),\n parameters: params.clone(),\n resume_kind: crate::gate::ResumeKind::Approval { allow_always: true },\n resume_output: None,\n },\n );\n }\n PolicyDecision::Allow => {}\n }\n }\n\n if let Err(e) = leases.consume_use(lease.id).await {\n return PreflightResult::Denied(ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"lease exhausted: {e}\")),\n )));\n }\n\n PreflightResult::Approved(lease)\n}\n\n// ── llm_query() — recursive subagent (RLM 3.5) ─────────────\n\n/// Handle `llm_query(prompt, context)` — single recursive sub-call.\nasync fn handle_llm_query(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n llm: &Arc,\n recursive_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n let prompt = extract_string_arg(args, kwargs, \"prompt\", 0);\n let context_arg = extract_string_arg(args, kwargs, \"context\", 1);\n // `model` must be parsed explicitly — `extract_string_arg` coerces via\n // `monty_to_string`, which turns `MontyObject::None` into the literal\n // string \"None\" and stringifies non-string values, both of which would\n // silently route the call to an invalid model ID. Accept only str or None.\n let model_arg = match extract_optional_string_kwarg(args, kwargs, \"model\", 2) {\n Ok(v) => v,\n Err(err) => return err,\n };\n\n let prompt = match prompt {\n Some(p) => p,\n None => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(\"llm_query() requires a 'prompt' argument\".into()),\n ));\n }\n };\n\n let mut messages = Vec::new();\n if let Some(ctx) = context_arg {\n messages.push(ThreadMessage::system(format!(\n \"You are a sub-agent. Answer concisely based on the context.\\n\\n{ctx}\"\n )));\n } else {\n // Some providers (e.g. OpenAI Codex Responses API) require a system\n // message / instructions field. Always include one.\n messages.push(ThreadMessage::system(\n \"You are a helpful sub-agent. Answer concisely.\",\n ));\n }\n messages.push(ThreadMessage::user(prompt));\n\n let config = LlmCallConfig {\n force_text: true,\n model: model_arg,\n ..LlmCallConfig::default()\n };\n\n match llm.complete(&messages, &[], &config).await {\n Ok(output) => {\n recursive_tokens.input_tokens += output.usage.input_tokens;\n recursive_tokens.output_tokens += output.usage.output_tokens;\n let text = match output.response {\n LlmResponse::Text(t) => t,\n LlmResponse::ActionCalls { content, .. } | LlmResponse::Code { content, .. } => {\n content.unwrap_or_default()\n }\n };\n ExtFunctionResult::Return(MontyObject::String(text))\n }\n Err(e) => ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"llm_query failed: {e}\")),\n )),\n }\n}\n\n/// Handle `llm_query_batched(prompts)` — parallel recursive sub-calls.\n///\n/// Takes a list of prompt strings and dispatches them concurrently.\n/// Returns a list of response strings in the same order.\nasync fn handle_llm_query_batched(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n llm: &Arc,\n recursive_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n // Extract prompts list (first arg or kwarg \"prompts\")\n let prompts_obj = args.first().or_else(|| {\n kwargs.iter().find_map(|(k, v)| {\n if let MontyObject::String(key) = k\n && key == \"prompts\"\n {\n return Some(v);\n }\n None\n })\n });\n\n let prompts: Vec = match prompts_obj {\n Some(MontyObject::List(items)) => items.iter().map(monty_to_string).collect(),\n Some(other) => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(format!(\n \"llm_query_batched() expects a list of prompts, got {other:?}\"\n )),\n ));\n }\n None => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(\"llm_query_batched() requires a 'prompts' argument\".into()),\n ));\n }\n };\n\n // Positional/keyword layout (matches the documented signature\n // `llm_query_batched(prompts, context=None, model=None, models=None)`):\n // arg 0 = prompts (already extracted above)\n // arg 1 = context\n // arg 2 = model\n // arg 3 = models\n // All three of context/model/models can also be passed by keyword.\n let context_arg = match extract_optional_string_kwarg(args, kwargs, \"context\", 1) {\n Ok(v) => v,\n Err(err) => return err,\n };\n\n // Optional model overrides:\n // - `model=\"...\"` applies the same model to every prompt\n // - `models=[...]` is a parallel array (must match prompts length); use\n // this to broadcast the same prompt across a council of models by\n // passing `prompts=[same]*N, models=[m1, m2, ...]`. Within `models`,\n // a `None` slot means \"no override for this prompt\" (the caller\n // opted out of routing for that slot); the singular `model=` kwarg\n // does NOT fill those slots, since mixing the two would be surprising.\n // See note in handle_llm_query: `model` must be parsed explicitly so that\n // `model=None` doesn't become the literal string \"None\".\n let single_model = match extract_optional_string_kwarg(args, kwargs, \"model\", 2) {\n Ok(v) => v,\n Err(err) => return err,\n };\n let models_kwarg = kwargs\n .iter()\n .find_map(|(k, v)| match k {\n MontyObject::String(key) if key == \"models\" => Some(v),\n _ => None,\n })\n .or_else(|| args.get(3));\n\n let models_list: Option>> = match models_kwarg {\n None | Some(MontyObject::None) => None,\n Some(MontyObject::List(items)) => {\n let mut out = Vec::with_capacity(items.len());\n for item in items {\n match item {\n MontyObject::String(s) => out.push(Some(s.clone())),\n MontyObject::None => out.push(None),\n other => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(format!(\n \"llm_query_batched(): models list entries must be str or None, got {other:?}\"\n )),\n ));\n }\n }\n }\n Some(out)\n }\n Some(other) => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(format!(\n \"llm_query_batched(): `models` must be a list of str or None, got {other:?}\"\n )),\n ));\n }\n };\n\n if let Some(ref ms) = models_list\n && ms.len() != prompts.len()\n {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::ValueError,\n Some(format!(\n \"llm_query_batched(): models list length ({}) must match prompts length ({})\",\n ms.len(),\n prompts.len()\n )),\n ));\n }\n\n let mut handles = Vec::with_capacity(prompts.len());\n for (i, prompt) in prompts.iter().enumerate() {\n let llm = Arc::clone(llm);\n let ctx = context_arg.clone();\n let prompt = prompt.clone();\n // If `models=` was provided, each slot is authoritative — `None` means\n // \"no override for this prompt\" and is NOT backfilled from `model=`.\n // Otherwise, fall back to the singular `model=` kwarg (or None).\n let model_override = match models_list.as_ref() {\n Some(ms) => ms[i].clone(),\n None => single_model.clone(),\n };\n let config = LlmCallConfig {\n force_text: true,\n model: model_override,\n ..LlmCallConfig::default()\n };\n handles.push(tokio::spawn(async move {\n let mut messages = Vec::new();\n if let Some(ctx) = ctx {\n messages.push(ThreadMessage::system(format!(\n \"You are a sub-agent. Answer concisely.\\n\\n{ctx}\"\n )));\n } else {\n messages.push(ThreadMessage::system(\n \"You are a helpful sub-agent. Answer concisely.\",\n ));\n }\n messages.push(ThreadMessage::user(prompt));\n llm.complete(&messages, &[], &config).await\n }));\n }\n\n // Collect results\n let mut results = Vec::with_capacity(prompts.len());\n let mut total_input = 0u64;\n let mut total_output = 0u64;\n\n for handle in handles {\n match handle.await {\n Ok(Ok(output)) => {\n total_input += output.usage.input_tokens;\n total_output += output.usage.output_tokens;\n let text = match output.response {\n LlmResponse::Text(t) => t,\n LlmResponse::ActionCalls { content, .. }\n | LlmResponse::Code { content, .. } => content.unwrap_or_default(),\n };\n results.push(MontyObject::String(text));\n }\n Ok(Err(e)) => {\n results.push(MontyObject::String(format!(\"Error: {e}\")));\n }\n Err(e) => {\n results.push(MontyObject::String(format!(\"Error: task failed: {e}\")));\n }\n }\n }\n\n recursive_tokens.input_tokens += total_input;\n recursive_tokens.output_tokens += total_output;\n\n ExtFunctionResult::Return(MontyObject::List(results))\n}\n\n// ── rlm_query() — full recursive sub-agent (RLM 3.5) ─────────\n\n/// Handle `rlm_query(prompt)` — spawn a child CodeAct thread with its own\n/// execution loop, tools, and iteration budget.\n///\n/// Unlike `llm_query()` (single-shot LLM call), `rlm_query()` creates a\n/// child thread with full CodeAct capabilities. The child inherits the\n/// parent's remaining budget and tool access.\n#[allow(clippy::too_many_arguments)]\nasync fn handle_rlm_query(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n parent_thread: &Thread,\n llm: &Arc,\n effects: &Arc,\n leases: &LeaseManager,\n policy: &PolicyEngine,\n recursive_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n let prompt = extract_string_arg(args, kwargs, \"prompt\", 0);\n let prompt = match prompt {\n Some(p) => p,\n None => {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(\"rlm_query() requires a 'prompt' argument\".into()),\n ));\n }\n };\n\n // Depth check — refuse if at max recursion depth\n let current_depth = parent_thread.config.depth;\n let max_depth = parent_thread.config.max_depth;\n if current_depth >= max_depth {\n return ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\n \"rlm_query() depth limit reached: depth {current_depth} >= max {max_depth}\"\n )),\n ));\n }\n\n // Build child thread with inherited budget\n let child_config = crate::types::thread::ThreadConfig {\n max_iterations: parent_thread.config.max_iterations.min(20), // cap child iterations\n enable_tool_intent_nudge: false,\n max_tokens_total: parent_thread\n .config\n .max_tokens_total\n .map(|max| max.saturating_sub(parent_thread.total_tokens_used)),\n max_budget_usd: parent_thread\n .config\n .max_budget_usd\n .map(|max| (max - parent_thread.total_cost_usd).max(0.0)),\n max_duration: parent_thread.config.max_duration,\n depth: current_depth + 1,\n max_depth,\n ..crate::types::thread::ThreadConfig::default()\n };\n\n let mut child_thread = crate::types::thread::Thread::new(\n &prompt,\n crate::types::thread::ThreadType::Research,\n parent_thread.project_id,\n &parent_thread.user_id,\n child_config,\n )\n .with_parent(parent_thread.id);\n\n // Add the prompt as a user message\n child_thread.add_message(ThreadMessage::user(&prompt));\n\n // Create signal channel and child's lease manager\n let (_tx, rx) = crate::runtime::messaging::signal_channel(8);\n let child_leases = Arc::new(LeaseManager::new());\n\n // Grant the child the same leases as the parent (in the child's manager)\n let parent_leases = leases.active_for_thread(parent_thread.id).await;\n let now = chrono::Utc::now();\n for parent_lease in &parent_leases {\n // Convert parent's expires_at to remaining duration\n let remaining_duration = parent_lease\n .expires_at\n .and_then(|exp| (exp - now).to_std().ok())\n .map(|d| chrono::Duration::from_std(d).unwrap_or(chrono::Duration::hours(1)));\n let lease = match child_leases\n .grant(\n child_thread.id,\n &parent_lease.capability_name,\n parent_lease.granted_actions.clone(),\n remaining_duration,\n parent_lease.max_uses,\n )\n .await\n {\n Ok(l) => l,\n Err(e) => {\n debug!(error = %e, \"rlm_query: skipping invalid lease for child thread\");\n continue;\n }\n };\n child_thread.capability_leases.push(lease.id);\n }\n let mut child_policy_engine = PolicyEngine::new();\n // Copy denied effects from parent policy\n for effect in &policy.denied_effects {\n child_policy_engine.deny_effect(*effect);\n }\n let child_policy = Arc::new(child_policy_engine);\n\n let mut child_loop = crate::executor::ExecutionLoop::new(\n child_thread,\n Arc::clone(llm),\n Arc::clone(effects),\n child_leases,\n child_policy,\n rx,\n \"rlm_child\".to_string(),\n );\n\n debug!(\n parent_thread = %parent_thread.id,\n depth = current_depth + 1,\n prompt_len = prompt.len(),\n \"rlm_query: spawning child CodeAct thread\"\n );\n\n // Run the child loop (Box::pin to avoid infinite future size from recursion)\n match Box::pin(child_loop.run()).await {\n Ok(outcome) => {\n // Track child's token usage\n recursive_tokens.input_tokens += child_loop.thread.total_tokens_used;\n recursive_tokens.cost_usd += child_loop.thread.total_cost_usd;\n\n let response = match outcome {\n crate::runtime::messaging::ThreadOutcome::Completed { response } => {\n response.unwrap_or_default()\n }\n crate::runtime::messaging::ThreadOutcome::Failed { error } => {\n format!(\"rlm_query child failed: {error}\")\n }\n crate::runtime::messaging::ThreadOutcome::MaxIterations => {\n \"rlm_query child reached max iterations\".to_string()\n }\n _ => String::new(),\n };\n\n ExtFunctionResult::Return(MontyObject::String(response))\n }\n Err(e) => ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"rlm_query failed: {e}\")),\n )),\n }\n}\n\n// ── Standalone async handlers (for tokio::spawn) ────────────\n\n/// `llm_query()` — standalone version that returns `(ExtFunctionResult, TokenUsage)`.\nasync fn handle_llm_query_standalone(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n llm: &Arc,\n) -> (ExtFunctionResult, TokenUsage) {\n let mut tokens = TokenUsage::default();\n let result = handle_llm_query(args, kwargs, llm, &mut tokens).await;\n (result, tokens)\n}\n\n/// `llm_query_batched()` — standalone version.\nasync fn handle_llm_query_batched_standalone(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n llm: &Arc,\n) -> (ExtFunctionResult, TokenUsage) {\n let mut tokens = TokenUsage::default();\n let result = handle_llm_query_batched(args, kwargs, llm, &mut tokens).await;\n (result, tokens)\n}\n\n// ── Future resolution helpers ───────────────────────────────\n\n/// Resolve a pending tool execution future.\n#[allow(clippy::too_many_arguments)]\nasync fn resolve_tool_future(\n handle: tokio::task::JoinHandle>,\n action_name: &str,\n call_id: &str,\n lease_id: crate::types::capability::LeaseId,\n parameters: serde_json::Value,\n params_summary: Option,\n leases: &LeaseManager,\n context: &ThreadExecutionContext,\n action_results: &mut Vec,\n events: &mut Vec,\n) -> ExtFunctionResult {\n match handle.await {\n Ok(Ok(result)) => {\n // If the effect adapter wrapped a tool error as an Ok(ActionResult)\n // with is_error=true (current convention in\n // `EffectBridgeAdapter::execute_action_internal`), surface it as\n // ActionFailed so traces, observers, and approval flows see the\n // failure correctly. Without this, every wrapped error looked like\n // a successful tool call to downstream consumers.\n if result.is_error {\n let error_msg = result\n .output\n .get(\"error\")\n .and_then(|v| v.as_str())\n .map(String::from)\n .unwrap_or_else(|| result.output.to_string());\n events.push(EventKind::ActionFailed {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n error: error_msg,\n params_summary,\n });\n } else {\n events.push(EventKind::ActionExecuted {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n duration_ms: result.duration.as_millis() as u64,\n params_summary,\n });\n }\n let monty_val = json_to_monty(&result.output);\n action_results.push(result);\n ExtFunctionResult::Return(monty_val)\n }\n Ok(Err(EngineError::GatePaused {\n gate_name,\n action_name,\n call_id,\n resume_kind,\n ..\n })) => {\n let _ = leases.refund_use(lease_id).await;\n events.push(EventKind::ApprovalRequested {\n action_name,\n call_id,\n parameters: Some(parameters),\n description: None,\n allow_always: match *resume_kind {\n crate::gate::ResumeKind::Approval { allow_always } => Some(allow_always),\n _ => None,\n },\n gate_name: Some(gate_name.clone()),\n params_summary,\n });\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"execution paused by gate '{gate_name}'\")),\n ))\n }\n Ok(Err(e)) => {\n events.push(EventKind::ActionFailed {\n step_id: context.step_id,\n action_name: action_name.into(),\n call_id: call_id.into(),\n error: e.to_string(),\n params_summary,\n });\n action_results.push(ActionResult {\n call_id: call_id.into(),\n action_name: action_name.into(),\n output: serde_json::json!({\"error\": e.to_string()}),\n is_error: true,\n duration: Duration::ZERO,\n });\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(e.to_string()),\n ))\n }\n Err(e) => {\n debug!(\"async tool task panicked: {e}\");\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"tool execution panicked: {e}\")),\n ))\n }\n }\n}\n\n/// Resolve a pending LLM call future, accumulating token usage.\nasync fn resolve_llm_future(\n handle: tokio::task::JoinHandle<(ExtFunctionResult, TokenUsage)>,\n recursive_tokens: &mut TokenUsage,\n) -> ExtFunctionResult {\n match handle.await {\n Ok((result, tokens)) => {\n recursive_tokens.input_tokens += tokens.input_tokens;\n recursive_tokens.output_tokens += tokens.output_tokens;\n result\n }\n Err(e) => {\n debug!(\"async LLM task panicked: {e}\");\n ExtFunctionResult::Error(MontyException::new(\n ExcType::RuntimeError,\n Some(format!(\"LLM call panicked: {e}\")),\n ))\n }\n }\n}\n\n// ── Helpers ─────────────────────────────────────────────────\n\nfn extract_string_arg(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n name: &str,\n position: usize,\n) -> Option {\n for (k, v) in kwargs {\n if let MontyObject::String(key) = k\n && key == name\n {\n return Some(monty_to_string(v));\n }\n }\n args.get(position).map(monty_to_string)\n}\n\n/// Strict optional-string extractor for arguments where silent coercion is\n/// dangerous (e.g. `model=` — passing the wrong type should NOT become an\n/// unintended model ID). Returns:\n/// - `Ok(None)` when the argument is missing or explicitly `None`\n/// - `Ok(Some(s))` when the argument is a string\n/// - `Err(TypeError)` for any other type\nfn extract_optional_string_kwarg(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n name: &str,\n position: usize,\n) -> Result, ExtFunctionResult> {\n let raw = kwargs\n .iter()\n .find_map(|(k, v)| match k {\n MontyObject::String(key) if key == name => Some(v),\n _ => None,\n })\n .or_else(|| args.get(position));\n\n match raw {\n None | Some(MontyObject::None) => Ok(None),\n Some(MontyObject::String(s)) => Ok(Some(s.clone())),\n Some(other) => Err(ExtFunctionResult::Error(MontyException::new(\n ExcType::TypeError,\n Some(format!(\"`{name}` must be a string or None, got {other:?}\")),\n ))),\n }\n}\n\npub(crate) fn monty_to_string(obj: &MontyObject) -> String {\n match obj {\n MontyObject::String(s) => s.clone(),\n MontyObject::None => \"None\".into(),\n MontyObject::Bool(b) => b.to_string(),\n MontyObject::Int(i) => i.to_string(),\n MontyObject::Float(f) => f.to_string(),\n other => {\n serde_json::to_string(&monty_to_json(other)).unwrap_or_else(|_| format!(\"{other:?}\"))\n }\n }\n}\n\n// Dispatch logic moved to orchestrator.rs (__execute_action__ handler).\n// GatePaused is handled via EngineError → JSON in orchestrator.rs.\n// ── MontyObject ↔ JSON ──────────────────────────────────────\n\npub(crate) fn monty_to_json(obj: &MontyObject) -> serde_json::Value {\n match obj {\n MontyObject::None => serde_json::Value::Null,\n MontyObject::Bool(b) => serde_json::Value::Bool(*b),\n MontyObject::Int(i) => serde_json::json!(i),\n MontyObject::BigInt(i) => serde_json::Value::String(i.to_string()),\n MontyObject::Float(f) => serde_json::json!(f),\n MontyObject::String(s) => serde_json::Value::String(s.clone()),\n MontyObject::List(items) | MontyObject::Tuple(items) => {\n serde_json::Value::Array(items.iter().map(monty_to_json).collect())\n }\n MontyObject::Dict(pairs) => {\n let map: serde_json::Map = pairs\n .into_iter()\n .map(|(k, v)| {\n let key = match k {\n MontyObject::String(s) => s.clone(),\n other => format!(\"{other:?}\"),\n };\n (key, monty_to_json(v))\n })\n .collect();\n serde_json::Value::Object(map)\n }\n MontyObject::Set(items) | MontyObject::FrozenSet(items) => {\n serde_json::Value::Array(items.iter().map(monty_to_json).collect())\n }\n MontyObject::Bytes(b) => {\n serde_json::Value::String(b.iter().map(|byte| format!(\"{byte:02x}\")).collect())\n }\n other => serde_json::Value::String(format!(\"{other:?}\")),\n }\n}\n\npub(crate) fn json_to_monty(val: &serde_json::Value) -> MontyObject {\n match val {\n serde_json::Value::Null => MontyObject::None,\n serde_json::Value::Bool(b) => MontyObject::Bool(*b),\n serde_json::Value::Number(n) => {\n if let Some(i) = n.as_i64() {\n MontyObject::Int(i)\n } else if let Some(f) = n.as_f64() {\n MontyObject::Float(f)\n } else {\n MontyObject::String(n.to_string())\n }\n }\n serde_json::Value::String(s) => MontyObject::String(s.clone()),\n serde_json::Value::Array(arr) => MontyObject::List(arr.iter().map(json_to_monty).collect()),\n serde_json::Value::Object(map) => MontyObject::dict(\n map.iter()\n .map(|(k, v)| (MontyObject::String(k.clone()), json_to_monty(v)))\n .collect::>(),\n ),\n }\n}\n\nfn monty_args_to_json(\n args: &[MontyObject],\n kwargs: &[(MontyObject, MontyObject)],\n) -> serde_json::Value {\n let mut map = serde_json::Map::new();\n if !args.is_empty() {\n map.insert(\n \"_args\".into(),\n serde_json::Value::Array(args.iter().map(monty_to_json).collect()),\n );\n }\n for (k, v) in kwargs {\n let key = match k {\n MontyObject::String(s) => s.clone(),\n other => format!(\"{other:?}\"),\n };\n map.insert(key, monty_to_json(v));\n }\n serde_json::Value::Object(map)\n}\n\n#[cfg(test)]\nmod tests {\n use super::*;\n use crate::capability::lease::LeaseManager;\n use crate::capability::policy::PolicyEngine;\n use crate::traits::effect::ThreadExecutionContext;\n use crate::types::capability::{ActionDef, CapabilityLease, EffectType, GrantedActions};\n use crate::types::project::ProjectId;\n use crate::types::step::{ActionResult, StepId};\n use crate::types::thread::{Thread, ThreadConfig, ThreadType};\n use std::sync::Mutex;\n\n /// Truncate a string to at most `max_bytes`, snapping to a UTF-8 char\n /// boundary so assertion messages never panic on multibyte output.\n fn truncate_for_assert(s: &str, max_bytes: usize) -> &str {\n if s.len() <= max_bytes {\n return s;\n }\n let mut end = max_bytes;\n while end > 0 && !s.is_char_boundary(end) {\n end -= 1;\n }\n &s[..end] // safety: end is walked down to a valid char boundary above\n }\n\n struct MockEffects {\n results: Mutex>>,\n actions: Vec,\n }\n\n impl MockEffects {\n fn new(actions: Vec, results: Vec>) -> Self {\n Self {\n results: Mutex::new(results),\n actions,\n }\n }\n }\n\n #[async_trait::async_trait]\n impl EffectExecutor for MockEffects {\n async fn execute_action(\n &self,\n name: &str,\n _params: serde_json::Value,\n _lease: &CapabilityLease,\n _ctx: &ThreadExecutionContext,\n ) -> Result {\n let mut results = self.results.lock().unwrap();\n if results.is_empty() {\n Ok(ActionResult {\n call_id: String::new(),\n action_name: name.into(),\n output: serde_json::json!({\"result\": \"ok\"}),\n is_error: false,\n duration: Duration::from_millis(1),\n })\n } else {\n results.remove(0)\n }\n }\n\n async fn available_actions(\n &self,\n _leases: &[CapabilityLease],\n ) -> Result, EngineError> {\n Ok(self.actions.clone())\n }\n }\n\n fn test_action(name: &str) -> ActionDef {\n ActionDef {\n name: name.into(),\n description: \"Test tool\".into(),\n parameters_schema: serde_json::json!({\"type\": \"object\"}),\n effects: vec![EffectType::ReadLocal],\n requires_approval: false,\n }\n }\n\n fn make_test_thread() -> Thread {\n Thread::new(\n \"test goal\",\n ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n ThreadConfig::default(),\n )\n }\n\n fn make_exec_context(thread: &Thread) -> ThreadExecutionContext {\n ThreadExecutionContext {\n thread_id: thread.id,\n thread_type: thread.thread_type,\n project_id: thread.project_id,\n user_id: \"test\".into(),\n step_id: StepId::new(),\n current_call_id: None,\n source_channel: None,\n user_timezone: None,\n }\n }\n\n /// Stub LLM that always returns text \"stub\". Only used so execute_code\n /// doesn't need a real LLM — our tests exercise tool dispatch, not LLM calls.\n struct StubLlm;\n\n #[async_trait::async_trait]\n impl crate::traits::llm::LlmBackend for StubLlm {\n fn model_name(&self) -> &str {\n \"stub\"\n }\n\n async fn complete(\n &self,\n _messages: &[crate::types::message::ThreadMessage],\n _actions: &[ActionDef],\n _config: &crate::traits::llm::LlmCallConfig,\n ) -> Result {\n Ok(crate::traits::llm::LlmOutput {\n response: crate::types::step::LlmResponse::Text(\"stub\".into()),\n usage: crate::types::step::TokenUsage::default(),\n })\n }\n }\n\n async fn run_code(\n code: &str,\n effects: Arc,\n thread: &Thread,\n ) -> Result {\n let leases = LeaseManager::new();\n let policy = PolicyEngine::new();\n let ctx = make_exec_context(thread);\n\n // Grant a wildcard lease\n leases\n .grant(thread.id, \"tools\", GrantedActions::All, None, None)\n .await\n .unwrap();\n\n execute_code(\n code,\n thread,\n &(Arc::new(StubLlm) as Arc),\n &effects,\n &leases,\n &policy,\n &ctx,\n &[],\n &serde_json::json!({}),\n )\n .await\n }\n\n // ── Single await tool call ──────────────────────────────\n\n #[tokio::test]\n async fn single_await_tool_call() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"echo\")],\n vec![Ok(ActionResult {\n call_id: String::new(),\n action_name: \"echo\".into(),\n output: serde_json::json!(\"hello world\"),\n is_error: false,\n duration: Duration::from_millis(1),\n })],\n ));\n\n let code = r#\"\nresult = await echo(message=\"hello\")\nFINAL(str(result))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(\n result.final_answer.is_some(),\n \"should have final answer, stdout: {}\",\n result.stdout\n );\n assert!(\n result.failure.is_none(),\n \"should not error, stdout: {}\",\n result.stdout\n );\n assert_eq!(result.action_results.len(), 1);\n }\n\n // ── asyncio.gather parallel execution ───────────────────\n\n #[tokio::test]\n async fn asyncio_gather_two_tools() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"tool_a\"), test_action(\"tool_b\")],\n vec![\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"tool_a\".into(),\n output: serde_json::json!(10),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"tool_b\".into(),\n output: serde_json::json!(32),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n ],\n ));\n\n let code = r#\"\nimport asyncio\na, b = await asyncio.gather(tool_a(), tool_b())\nFINAL(str(a + b))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(\n result.final_answer.is_some(),\n \"should have final answer, stdout: {}\",\n result.stdout\n );\n assert_eq!(\n result.final_answer.as_deref(),\n Some(\"42\"),\n \"10 + 32 = 42, got: {:?}, stdout: {}\",\n result.final_answer,\n result.stdout\n );\n assert_eq!(result.action_results.len(), 2);\n assert!(result.failure.is_none());\n }\n\n // ── asyncio.gather three tools ──────────────────────────\n\n #[tokio::test]\n async fn asyncio_gather_three_tools() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![\n test_action(\"web_search\"),\n test_action(\"http\"),\n test_action(\"memory_search\"),\n ],\n vec![\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"web_search\".into(),\n output: serde_json::json!(\"search results\"),\n is_error: false,\n duration: Duration::from_millis(50),\n }),\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"http\".into(),\n output: serde_json::json!(\"page content\"),\n is_error: false,\n duration: Duration::from_millis(100),\n }),\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"memory_search\".into(),\n output: serde_json::json!(\"memories\"),\n is_error: false,\n duration: Duration::from_millis(25),\n }),\n ],\n ));\n\n let code = r#\"\nimport asyncio\ns, h, m = await asyncio.gather(\n web_search(query=\"test\"),\n http(url=\"https://example.com\"),\n memory_search(query=\"prior\"),\n)\nFINAL(str(s) + \"|\" + str(h) + \"|\" + str(m))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.action_results.len(), 3);\n let answer = result.final_answer.unwrap();\n assert!(answer.contains(\"search results\"), \"got: {answer}\");\n assert!(answer.contains(\"page content\"), \"got: {answer}\");\n assert!(answer.contains(\"memories\"), \"got: {answer}\");\n }\n\n // ── Data-dependent chain (sequential await) ─────────────\n\n #[tokio::test]\n async fn sequential_dependent_calls() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"step1\"), test_action(\"step2\")],\n vec![\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"step1\".into(),\n output: serde_json::json!(\"intermediate\"),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"step2\".into(),\n output: serde_json::json!(\"final\"),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n ],\n ));\n\n let code = r#\"\na = await step1()\nb = await step2(input=a)\nFINAL(str(b))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.action_results.len(), 2);\n assert_eq!(result.final_answer.as_deref(), Some(\"final\"));\n }\n\n // ── Error in one gathered tool ──────────────────────────\n\n #[tokio::test]\n async fn gather_with_error_propagates() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"good\"), test_action(\"bad\")],\n vec![\n Ok(ActionResult {\n call_id: String::new(),\n action_name: \"good\".into(),\n output: serde_json::json!(\"ok\"),\n is_error: false,\n duration: Duration::from_millis(1),\n }),\n Err(EngineError::Effect {\n reason: \"tool exploded\".into(),\n }),\n ],\n ));\n\n let code = r#\"\nimport asyncio\na, b = await asyncio.gather(good(), bad())\nFINAL(\"should not reach\")\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n // Error in gather propagates as exception — code should error\n assert!(\n result.failure.is_some(),\n \"should have error, stdout: {}\",\n result.stdout\n );\n assert!(\n result.final_answer.is_none()\n || result.final_answer.as_deref() != Some(\"should not reach\")\n );\n }\n\n // ── Tool with no lease (denied in preflight) ────────────\n\n #[tokio::test]\n async fn denied_tool_raises_exception() {\n let thread = make_test_thread();\n // No actions registered — tool has no lease\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n\n let code = r#\"\ntry:\n result = await unknown_tool()\n FINAL(\"should not reach\")\nexcept:\n FINAL(\"caught error\")\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n // Tool not found raises NameError before we even get to dispatch\n assert!(result.final_answer.is_some(), \"stdout: {}\", result.stdout);\n }\n\n // ── FINAL works without await ───────────────────────────\n\n #[tokio::test]\n async fn final_is_sync() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n\n let code = r#\"\nFINAL(\"hello from sync\")\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert_eq!(result.final_answer.as_deref(), Some(\"hello from sync\"));\n assert!(result.failure.is_none());\n }\n\n // ── globals() still works ───────────────────────────────\n\n #[tokio::test]\n async fn globals_returns_known_tools() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"web_search\"), test_action(\"http\")],\n vec![],\n ));\n\n let code = r#\"\ng = globals()\nhas_search = \"web_search\" in g\nhas_http = \"http\" in g\nFINAL(str(has_search) + \"|\" + str(has_http))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.final_answer.as_deref(), Some(\"True|True\"));\n }\n\n // ── Empty gather ────────────────────────────────────────\n\n #[tokio::test]\n async fn empty_gather() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n\n let code = r#\"\nimport asyncio\nresults = await asyncio.gather()\nFINAL(str(len(results)))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.final_answer.as_deref(), Some(\"0\"));\n }\n\n // ── Single-item gather ──────────────────────────────────\n\n #[tokio::test]\n async fn single_item_gather() {\n let thread = make_test_thread();\n let effects: Arc = Arc::new(MockEffects::new(\n vec![test_action(\"echo\")],\n vec![Ok(ActionResult {\n call_id: String::new(),\n action_name: \"echo\".into(),\n output: serde_json::json!(\"gathered\"),\n is_error: false,\n duration: Duration::from_millis(1),\n })],\n ));\n\n let code = r#\"\nimport asyncio\nresults = await asyncio.gather(echo())\nFINAL(str(results[0]))\n\"#;\n\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_none(), \"stdout: {}\", result.stdout);\n assert_eq!(result.final_answer.as_deref(), Some(\"gathered\"));\n assert_eq!(result.action_results.len(), 1);\n }\n\n // ── Sandbox security negative tests ────────────────────────\n\n /// OS-level operations must be denied or restricted by the Monty VM.\n #[tokio::test]\n async fn sandbox_denies_os_operations() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n // Try to import os and call os.system — should fail\n let code = r#\"\ntry:\n import os\n os.system(\"echo pwned\")\n FINAL(\"ESCAPED: os.system ran\")\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"os.system should be blocked, got: {answer}\",\n );\n }\n\n /// Resource limits must be enforced — infinite loops should be terminated.\n #[tokio::test]\n async fn sandbox_enforces_resource_limits() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n // Infinite allocation loop — should hit allocation or memory limit\n let code = r#\"\ndata = []\nwhile True:\n data.append(\"x\" * 10000)\n\"#;\n let result = run_code(code, effects, &thread).await;\n // Either returns an error or the stdout contains an error message —\n // the key assertion is that it DOES NOT run forever.\n if let Ok(r) = result {\n assert!(\n r.failure.is_some() || r.stdout.contains(\"Error\") || r.stdout.contains(\"limit\"),\n \"resource limit should terminate infinite loop, got stdout: {}\",\n truncate_for_assert(&r.stdout, 500),\n );\n }\n // Err(_) is also acceptable — means the VM was killed by resource limits\n }\n\n /// Python `import` of system modules must be restricted.\n #[tokio::test]\n async fn sandbox_restricts_imports() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n // Try to import subprocess — should fail\n let code = r#\"\ntry:\n import subprocess\n result = subprocess.run([\"echo\", \"escaped\"], capture_output=True, text=True)\n FINAL(\"ESCAPED: \" + result.stdout)\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"subprocess import should be blocked, got: {answer}\",\n );\n }\n\n /// File system access via open() must be blocked.\n #[tokio::test]\n async fn sandbox_denies_file_access() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n let code = r#\"\ntry:\n f = open(\"/etc/passwd\", \"r\")\n content = f.read()\n f.close()\n FINAL(\"ESCAPED: \" + content[:50])\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"open() should be blocked, got: {answer}\",\n );\n }\n\n /// Network access via socket must be blocked.\n #[tokio::test]\n async fn sandbox_denies_socket_access() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n let code = r#\"\ntry:\n import socket\n s = socket.socket(socket.AF_INET, socket.SOCK_STREAM)\n s.connect((\"127.0.0.1\", 80))\n FINAL(\"ESCAPED: connected\")\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"socket access should be blocked, got: {answer}\",\n );\n }\n\n /// Calls to tools not covered by the lease must be denied.\n #[tokio::test]\n async fn sandbox_unlicensed_tool_denied() {\n let effects: Arc =\n Arc::new(MockEffects::new(vec![test_action(\"allowed_tool\")], vec![]));\n let thread = make_test_thread();\n let leases = LeaseManager::new();\n let policy = PolicyEngine::new();\n let ctx = make_exec_context(&thread);\n\n // Grant a restricted lease — only \"allowed_tool\" is permitted.\n leases\n .grant(\n thread.id,\n \"tools\",\n GrantedActions::Specific(vec![\"allowed_tool\".into()]),\n None,\n None,\n )\n .await\n .unwrap();\n\n let code = r#\"\ntry:\n result = await secret_admin_tool(data=\"pwn\")\n FINAL(\"ESCAPED: \" + str(result))\nexcept Exception as e:\n FINAL(\"blocked: \" + type(e).__name__)\n\"#;\n let result = execute_code(\n code,\n &thread,\n &(Arc::new(StubLlm) as Arc),\n &effects,\n &leases,\n &policy,\n &ctx,\n &[],\n &serde_json::json!({}),\n )\n .await\n .unwrap();\n let answer = result.final_answer.as_deref().unwrap_or(\"\");\n assert!(\n !answer.starts_with(\"ESCAPED\"),\n \"unlicensed tool should be denied by preflight, got: {answer}\",\n );\n }\n\n /// CPU-bound infinite loops must be terminated by allocation/duration limits.\n #[tokio::test]\n async fn sandbox_enforces_cpu_limits() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n // Tight CPU-bound loop (no allocations to trip allocation limit)\n let code = r#\"\nx = 0\nwhile True:\n x += 1\n\"#;\n let result = run_code(code, effects, &thread).await;\n // Must terminate — either via error or resource limit\n if let Ok(r) = result {\n assert!(\n r.failure.is_some() || r.stdout.contains(\"Error\") || r.stdout.contains(\"limit\"),\n \"cpu-bound loop should be terminated, stdout: {}\",\n truncate_for_assert(&r.stdout, 500),\n );\n }\n // Err(_) is also acceptable — means the VM was killed by resource limits\n }\n\n /// FINAL() must capture the answer from the code.\n #[tokio::test]\n async fn sandbox_final_captures_answer() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n let code = r#\"\nx = 2 + 3\nFINAL(str(x))\n\"#;\n let result = run_code(code, effects, &thread).await.unwrap();\n assert_eq!(\n result.final_answer.as_deref(),\n Some(\"5\"),\n \"FINAL should capture the computed answer\"\n );\n }\n\n /// Syntax errors flow back as errors, not panics.\n #[tokio::test]\n async fn sandbox_handles_syntax_error() {\n let effects: Arc = Arc::new(MockEffects::new(vec![], vec![]));\n let thread = make_test_thread();\n\n let code = \"def broken(\\nFINAL('nope')\";\n let result = run_code(code, effects, &thread).await.unwrap();\n assert!(result.failure.is_some(), \"syntax error should set failure\");\n assert!(\n result.stdout.contains(\"SyntaxError\") || result.stdout.contains(\"Error\"),\n \"should contain SyntaxError, got: {}\",\n result.stdout,\n );\n }\n\n // ── llm_query model parameter plumbing ─────────────────────\n\n /// LLM backend that records every call's model + prompt for assertions.\n struct CapturingLlm {\n calls: tokio::sync::Mutex, String)>>,\n }\n\n impl CapturingLlm {\n fn new() -> Self {\n Self {\n calls: tokio::sync::Mutex::new(Vec::new()),\n }\n }\n }\n\n #[async_trait::async_trait]\n impl crate::traits::llm::LlmBackend for CapturingLlm {\n fn model_name(&self) -> &str {\n \"capturing\"\n }\n\n async fn complete(\n &self,\n messages: &[crate::types::message::ThreadMessage],\n _actions: &[ActionDef],\n config: &crate::traits::llm::LlmCallConfig,\n ) -> Result {\n let user_prompt = messages\n .iter()\n .rev()\n .find(|m| matches!(m.role, crate::types::message::MessageRole::User))\n .map(|m| m.content.clone())\n .unwrap_or_default();\n self.calls\n .lock()\n .await\n .push((config.model.clone(), user_prompt.clone()));\n Ok(crate::traits::llm::LlmOutput {\n response: crate::types::step::LlmResponse::Text(format!(\n \"ack:{}:{user_prompt}\",\n config.model.as_deref().unwrap_or(\"default\")\n )),\n usage: crate::types::step::TokenUsage::default(),\n })\n }\n }\n\n #[tokio::test]\n async fn llm_query_forwards_model_kwarg() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query(\n &[],\n &[\n (\n MontyObject::String(\"prompt\".into()),\n MontyObject::String(\"what is 2+2?\".into()),\n ),\n (\n MontyObject::String(\"model\".into()),\n MontyObject::String(\"gpt-4o\".into()),\n ),\n ],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n match result {\n ExtFunctionResult::Return(MontyObject::String(s)) => {\n assert!(s.contains(\"gpt-4o\"), \"got: {s}\");\n }\n other => panic!(\"expected string return, got {other:?}\"),\n }\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].0.as_deref(), Some(\"gpt-4o\"));\n assert_eq!(calls[0].1, \"what is 2+2?\");\n }\n\n #[tokio::test]\n async fn llm_query_without_model_passes_none() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let _ = handle_llm_query(\n &[MontyObject::String(\"hello\".into())],\n &[],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].0, None);\n }\n\n #[tokio::test]\n async fn llm_query_batched_broadcasts_with_models_list() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"Q\".into()),\n MontyObject::String(\"Q\".into()),\n MontyObject::String(\"Q\".into()),\n ]);\n let models = MontyObject::List(vec![\n MontyObject::String(\"gpt-4o\".into()),\n MontyObject::String(\"claude-sonnet-4-20250514\".into()),\n MontyObject::String(\"llama-3.1-70b-instruct\".into()),\n ]);\n let result = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"models\".into()), models)],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n match result {\n ExtFunctionResult::Return(MontyObject::List(items)) => {\n assert_eq!(items.len(), 3);\n }\n other => panic!(\"expected list return, got {other:?}\"),\n }\n\n let mut calls = llm.calls.lock().await;\n calls.sort_by(|a, b| a.0.cmp(&b.0));\n assert_eq!(calls.len(), 3);\n assert_eq!(calls[0].0.as_deref(), Some(\"claude-sonnet-4-20250514\"));\n assert_eq!(calls[1].0.as_deref(), Some(\"gpt-4o\"));\n assert_eq!(calls[2].0.as_deref(), Some(\"llama-3.1-70b-instruct\"));\n }\n\n #[tokio::test]\n async fn llm_query_batched_single_model_applies_to_all() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n let _ = handle_llm_query_batched(\n &[prompts],\n &[(\n MontyObject::String(\"model\".into()),\n MontyObject::String(\"gpt-4o\".into()),\n )],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 2);\n assert!(calls.iter().all(|(m, _)| m.as_deref() == Some(\"gpt-4o\")));\n }\n\n #[tokio::test]\n async fn llm_query_model_none_kwarg_is_no_override_not_literal_none_string() {\n // Regression: `extract_string_arg` would have coerced\n // MontyObject::None to the literal string \"None\", silently routing\n // every model=None call to an invalid model ID. Must stay None.\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let _ = handle_llm_query(\n &[],\n &[\n (\n MontyObject::String(\"prompt\".into()),\n MontyObject::String(\"hi\".into()),\n ),\n (MontyObject::String(\"model\".into()), MontyObject::None),\n ],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].0, None);\n }\n\n #[tokio::test]\n async fn llm_query_rejects_non_string_model_kwarg() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query(\n &[],\n &[\n (\n MontyObject::String(\"prompt\".into()),\n MontyObject::String(\"hi\".into()),\n ),\n (MontyObject::String(\"model\".into()), MontyObject::Int(42)),\n ],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Error(_)));\n assert!(llm.calls.lock().await.is_empty());\n }\n\n #[tokio::test]\n async fn llm_query_batched_single_model_none_kwarg_is_no_override() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n let _ = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"model\".into()), MontyObject::None)],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 2);\n assert!(calls.iter().all(|(m, _)| m.is_none()));\n }\n\n #[tokio::test]\n async fn llm_query_batched_honors_positional_context_and_model() {\n // Regression: `context`, `model`, and `models` used to be kwarg-only.\n // A positional call matching the documented signature\n // `llm_query_batched(prompts, context=None, model=None, models=None)`\n // silently dropped the model, violating the preamble.\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query_batched(\n &[\n MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]),\n MontyObject::String(\"shared context\".into()), // position 1: context\n MontyObject::String(\"gpt-4o\".into()), // position 2: model\n ],\n &[],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n match result {\n ExtFunctionResult::Return(MontyObject::List(items)) => assert_eq!(items.len(), 2),\n other => panic!(\"expected list return, got {other:?}\"),\n }\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 2);\n assert!(calls.iter().all(|(m, _)| m.as_deref() == Some(\"gpt-4o\")));\n }\n\n #[tokio::test]\n async fn llm_query_batched_honors_positional_models_list() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query_batched(\n &[\n MontyObject::List(vec![\n MontyObject::String(\"q\".into()),\n MontyObject::String(\"q\".into()),\n ]),\n MontyObject::None, // position 1: context = None\n MontyObject::None, // position 2: model = None\n MontyObject::List(vec![\n // position 3: models\n MontyObject::String(\"gpt-4o\".into()),\n MontyObject::String(\"claude-sonnet-4-6\".into()),\n ]),\n ],\n &[],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Return(_)));\n let mut calls = llm.calls.lock().await;\n calls.sort_by(|a, b| a.0.cmp(&b.0));\n assert_eq!(calls.len(), 2);\n assert_eq!(calls[0].0.as_deref(), Some(\"claude-sonnet-4-6\"));\n assert_eq!(calls[1].0.as_deref(), Some(\"gpt-4o\"));\n }\n\n #[tokio::test]\n async fn llm_query_batched_positional_none_for_models_is_no_override() {\n // `llm_query_batched(prompts, None, None, None)` should run with no\n // model overrides, not error on the positional None at slot 3.\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let result = handle_llm_query_batched(\n &[\n MontyObject::List(vec![MontyObject::String(\"a\".into())]),\n MontyObject::None,\n MontyObject::None,\n MontyObject::None,\n ],\n &[],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Return(_)));\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 1);\n assert_eq!(calls[0].0, None);\n }\n\n #[tokio::test]\n async fn llm_query_batched_rejects_non_string_single_model() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![MontyObject::String(\"a\".into())]);\n let result = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"model\".into()), MontyObject::Int(7))],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Error(_)));\n assert!(llm.calls.lock().await.is_empty());\n }\n\n #[tokio::test]\n async fn llm_query_batched_rejects_non_string_models_entries() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n // Integers in the models list should fail loudly, not be coerced to \"1\"/\"2\".\n let models = MontyObject::List(vec![MontyObject::Int(1), MontyObject::Int(2)]);\n let result = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"models\".into()), models)],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Error(_)));\n assert!(llm.calls.lock().await.is_empty());\n }\n\n #[tokio::test]\n async fn llm_query_batched_none_in_models_list_does_not_backfill_from_model_kwarg() {\n // Regression: when `models=[None, \"gpt-4o\"]` and `model=\"claude-...\"`\n // are both passed, the None slot must NOT be backfilled by the\n // singular `model=` kwarg. Each slot is authoritative.\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n let models = MontyObject::List(vec![\n MontyObject::None,\n MontyObject::String(\"gpt-4o\".into()),\n ]);\n let _ = handle_llm_query_batched(\n &[prompts],\n &[\n (MontyObject::String(\"models\".into()), models),\n (\n MontyObject::String(\"model\".into()),\n MontyObject::String(\"claude-sonnet-4-20250514\".into()),\n ),\n ],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n let calls = llm.calls.lock().await;\n assert_eq!(calls.len(), 2);\n // Slot 0 was None — must remain None, not become \"claude-sonnet-4-20250514\".\n let slot_a = calls.iter().find(|(_, p)| p == \"a\").expect(\"call for a\");\n let slot_b = calls.iter().find(|(_, p)| p == \"b\").expect(\"call for b\");\n assert_eq!(slot_a.0, None);\n assert_eq!(slot_b.0.as_deref(), Some(\"gpt-4o\"));\n }\n\n #[tokio::test]\n async fn llm_query_batched_models_length_mismatch_errors() {\n let llm = Arc::new(CapturingLlm::new());\n let mut tokens = crate::types::step::TokenUsage::default();\n let prompts = MontyObject::List(vec![\n MontyObject::String(\"a\".into()),\n MontyObject::String(\"b\".into()),\n ]);\n let models = MontyObject::List(vec![MontyObject::String(\"only-one\".into())]);\n let result = handle_llm_query_batched(\n &[prompts],\n &[(MontyObject::String(\"models\".into()), models)],\n &(Arc::clone(&llm) as Arc),\n &mut tokens,\n )\n .await;\n\n assert!(matches!(result, ExtFunctionResult::Error(_)));\n assert!(llm.calls.lock().await.is_empty());\n }\n\n // ── Error classification tests ──────────────────────────────\n\n #[test]\n fn classify_syntax_error() {\n let cat = classify_runtime_error(\"SyntaxError: unexpected token\");\n assert_eq!(cat, CodeExecutionFailure::SyntaxError);\n }\n\n #[test]\n fn classify_timeout() {\n let cat = classify_runtime_error(\"execution timed out after 30s\");\n assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\n }\n\n #[test]\n fn classify_memory_limit() {\n let cat = classify_runtime_error(\"memory limit exceeded\");\n assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\n }\n\n #[test]\n fn classify_fuel_exhaustion() {\n let cat = classify_runtime_error(\"fuel exhausted during execution\");\n assert_eq!(cat, CodeExecutionFailure::ResourceLimit);\n }\n\n #[test]\n fn classify_os_denied() {\n let cat = classify_runtime_error(\"OS operations are not permitted in CodeAct scripts\");\n assert_eq!(cat, CodeExecutionFailure::OsDenied);\n }\n\n #[test]\n fn classify_name_error_as_runtime() {\n // NameError from Monty (not NameLookup) is classified as RuntimeError\n let cat = classify_runtime_error(\"NameError: name 'foo' is not defined\");\n assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n }\n\n #[test]\n fn classify_type_error_as_runtime() {\n let cat = classify_runtime_error(\"TypeError: unsupported operand\");\n assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n }\n\n #[test]\n fn classify_module_not_found_as_runtime() {\n let cat = classify_runtime_error(\"ModuleNotFoundError: No module named 'csv'\");\n assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n }\n\n #[test]\n fn classify_syntax_word_is_not_syntaxerror() {\n // \"syntax\" alone should not trigger SyntaxError — only \"syntaxerror\" should.\n let cat = classify_runtime_error(\"unexpected syntax in expression\");\n assert_eq!(cat, CodeExecutionFailure::RuntimeError);\n }\n\n #[test]\n fn vm_panic_variant_serializes_as_snake_case() {\n // VmPanic is set directly by catch_unwind paths, not by classify_runtime_error.\n // Verify it serializes consistently with Display (both snake_case).\n let failure = CodeExecutionFailure::VmPanic;\n assert_eq!(failure.to_string(), \"vm_panic\");\n let json = serde_json::to_value(&failure).unwrap();\n assert_eq!(json, serde_json::json!(\"vm_panic\"));\n }\n\n #[test]\n fn code_hash_deterministic() {\n let h1 = code_hash(\"print('hello')\");\n let h2 = code_hash(\"print('hello')\");\n assert_eq!(h1, h2);\n }\n\n #[test]\n fn code_hash_differs_for_different_code() {\n let h1 = code_hash(\"print('hello')\");\n let h2 = code_hash(\"print('world')\");\n assert_ne!(h1, h2);\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/executor/trace.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "content-length", + "26791" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "x-frame-options", + "deny" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "x-xss-protection", + "0" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-ratelimit-remaining", + "4961" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "server", + "github.com" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "etag", + "\"563d5e8dd6835053c514ff45fcebbf5e94bf283f\"" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-github-request-id", + "1451:1FD39D:4041B2:4BA311:69DFAF16" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "date", + "Wed, 15 Apr 2026 15:30:30 GMT" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-ratelimit-used", + "39" + ] + ], + "body": "//! Execution trace analysis.\n//!\n//! Builds an in-memory `ExecutionTrace` from a completed `Thread` and runs a\n//! retrospective analyzer that flags common failure patterns. Used by the\n//! self-improvement mission and surfaced in debug logs.\n//!\n//! **There is no separate engine trace file.** Live trace recording for the\n//! whole system is handled by `RecordingLlm` in the host crate\n//! (`src/llm/recording.rs`), gated by `IRONCLAW_RECORD_TRACE`. Because the\n//! engine's `LlmBackend` is wired to the same provider chain, engine LLM\n//! interactions are captured by that single recorder — no engine-side env var\n//! and no second JSON file.\n\nuse chrono::Utc;\nuse serde::Serialize;\nuse tracing::debug;\n\nuse crate::types::event::ThreadEvent;\nuse crate::types::thread::{Thread, ThreadId, ThreadState};\n\n/// A complete execution trace for a single thread.\n#[derive(Debug, Serialize)]\npub struct ExecutionTrace {\n pub thread_id: ThreadId,\n pub goal: String,\n pub final_state: ThreadState,\n pub step_count: usize,\n pub total_tokens: u64,\n pub messages: Vec,\n pub events: Vec,\n pub issues: Vec,\n pub timestamp: chrono::DateTime,\n}\n\n/// A single doc record, for the trace.\n#[derive(Debug, Serialize)]\npub struct DocRecord {\n pub doc_type: String,\n pub title: String,\n pub content: String,\n}\n\n/// A message in the trace with role labeling.\n#[derive(Debug, Serialize)]\npub struct MessageRecord {\n pub role: String,\n pub content_length: usize,\n pub content_preview: String,\n pub full_content: String,\n pub action_name: Option,\n pub action_call_id: Option,\n}\n\n/// An issue detected by the retrospective analyzer.\n#[derive(Debug, Serialize)]\npub struct TraceIssue {\n pub severity: IssueSeverity,\n pub category: String,\n pub description: String,\n pub step: Option,\n}\n\n#[derive(Debug, PartialEq, Serialize)]\npub enum IssueSeverity {\n Error,\n Warning,\n Info,\n}\n\n/// Build a trace from a completed thread.\npub fn build_trace(thread: &Thread) -> ExecutionTrace {\n let messages: Vec = thread\n .messages\n .iter()\n .map(|m| {\n let preview: String = m.content.chars().take(300).collect();\n MessageRecord {\n role: format!(\"{:?}\", m.role),\n content_length: m.content.chars().count(),\n content_preview: if m.content.chars().count() > 300 {\n format!(\"{preview}...\")\n } else {\n preview\n },\n full_content: m.content.clone(),\n action_name: m.action_name.clone(),\n action_call_id: m.action_call_id.clone(),\n }\n })\n .collect();\n\n let issues = analyze_trace(thread);\n\n ExecutionTrace {\n thread_id: thread.id,\n goal: thread.goal.clone(),\n final_state: thread.state,\n step_count: thread.step_count,\n total_tokens: thread.total_tokens_used,\n messages,\n events: thread.events.clone(),\n issues,\n timestamp: Utc::now(),\n }\n}\n\n/// Print a summary of the trace to the log.\npub fn log_trace_summary(trace: &ExecutionTrace) {\n debug!(\n thread_id = %trace.thread_id,\n goal = %trace.goal,\n state = ?trace.final_state,\n steps = trace.step_count,\n tokens = trace.total_tokens,\n messages = trace.messages.len(),\n events = trace.events.len(),\n issues = trace.issues.len(),\n \"=== Engine V2 Trace Summary ===\"\n );\n\n for issue in &trace.issues {\n match issue.severity {\n IssueSeverity::Error => debug!(\n category = %issue.category,\n step = ?issue.step,\n \"ISSUE: {}\",\n issue.description\n ),\n IssueSeverity::Warning => debug!(\n category = %issue.category,\n step = ?issue.step,\n \"WARNING: {}\",\n issue.description\n ),\n IssueSeverity::Info => debug!(\n category = %issue.category,\n step = ?issue.step,\n \"NOTE: {}\",\n issue.description\n ),\n }\n }\n}\n\n// ── Retrospective analysis ──────────────────────────────────\n\n/// Analyze a completed thread for common issues.\nfn analyze_trace(thread: &Thread) -> Vec {\n let mut issues = Vec::new();\n\n // 1. Check if the thread failed\n if thread.state == ThreadState::Failed {\n issues.push(TraceIssue {\n severity: IssueSeverity::Error,\n category: \"thread_failure\".into(),\n description: \"Thread ended in Failed state\".into(),\n step: None,\n });\n }\n\n // 2. Check for empty response (no FINAL, no useful output)\n let has_assistant_response = thread\n .messages\n .iter()\n .any(|m| m.role == crate::types::message::MessageRole::Assistant && !m.content.is_empty());\n if !has_assistant_response {\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"no_response\".into(),\n description: \"No assistant message in thread — model may not have generated output\"\n .into(),\n step: None,\n });\n }\n\n // 3. Check for tool errors\n let tool_errors: Vec<&ThreadEvent> = thread\n .events\n .iter()\n .filter(|e| matches!(e.kind, crate::types::event::EventKind::ActionFailed { .. }))\n .collect();\n if !tool_errors.is_empty() {\n for event in &tool_errors {\n if let crate::types::event::EventKind::ActionFailed {\n action_name, error, ..\n } = &event.kind\n {\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"tool_error\".into(),\n description: format!(\"Tool '{action_name}' failed: {error}\"),\n step: None,\n });\n }\n }\n }\n\n // 4. Check for code execution errors via structured CodeExecutionFailed events.\n // These carry a classified failure category that tells us exactly what kind\n // of error occurred (syntax, runtime, name lookup, VM panic, resource limit,\n // tool error, OS denied, gate pause).\n let code_failures: Vec<&ThreadEvent> = thread\n .events\n .iter()\n .filter(|e| {\n matches!(\n e.kind,\n crate::types::event::EventKind::CodeExecutionFailed { .. }\n )\n })\n .collect();\n for event in &code_failures {\n if let crate::types::event::EventKind::CodeExecutionFailed {\n category, error, ..\n } = &event.kind\n {\n let preview: String = error.chars().take(200).collect();\n let severity = match category {\n crate::types::step::CodeExecutionFailure::VmPanic => IssueSeverity::Error,\n crate::types::step::CodeExecutionFailure::ResourceLimit => IssueSeverity::Error,\n _ => IssueSeverity::Warning,\n };\n issues.push(TraceIssue {\n severity,\n category: format!(\"code_{category}\"),\n description: format!(\"Code execution failed ({category}): {preview}\"),\n step: None,\n });\n }\n }\n\n // Fallback: also check message-level patterns for backward compatibility\n // with threads that ran before the CodeExecutionFailed instrumentation\n // was added (PR #2483). Note: threads from mixed eras (some steps\n // instrumented, some not) will only report structured events when any\n // exist, silently skipping message-level errors from uninstrumented steps.\n if code_failures.is_empty() {\n let error_patterns = [\n \"NameError\",\n \"SyntaxError\",\n \"TypeError\",\n \"NotImplementedError\",\n \"ValueError\",\n \"AttributeError\",\n \"IndexError\",\n \"KeyError\",\n \"ModuleNotFoundError\",\n \"RuntimeError\",\n ];\n for (i, msg) in thread.messages.iter().enumerate() {\n let is_code_output = msg.role == crate::types::message::MessageRole::User\n && (msg.content.starts_with(\"[stdout]\")\n || msg.content.starts_with(\"[stderr]\")\n || msg.content.starts_with(\"[code \")\n || msg.content.starts_with(\"Traceback\"));\n if is_code_output && error_patterns.iter().any(|p| msg.content.contains(p)) {\n let preview: String = msg.content.chars().take(200).collect();\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"code_error\".into(),\n description: format!(\"Code execution error in message {i}: {preview}\"),\n step: None,\n });\n }\n }\n }\n\n // 5. Check for empty call_id on ActionResult messages (causes LLM API rejection).\n for (i, msg) in thread.messages.iter().enumerate() {\n if msg.role == crate::types::message::MessageRole::ActionResult {\n let call_id_empty = msg.action_call_id.as_ref().is_none_or(|id| id.is_empty());\n if call_id_empty {\n let name = msg.action_name.as_deref().unwrap_or(\"unknown\");\n issues.push(TraceIssue {\n severity: IssueSeverity::Error,\n category: \"empty_call_id\".into(),\n description: format!(\n \"ActionResult message {i} (tool '{name}') has empty call_id — will cause LLM API rejection\"\n ),\n step: None,\n });\n }\n }\n }\n\n // 6. Check for model ignoring tool results (hallucination risk).\n // In Tier 0 (structured), results appear as ActionResult messages.\n // In Tier 1 (CodeAct), results appear as User messages with \"[tool result]\" prefixes.\n let has_tool_results = thread\n .messages\n .iter()\n .any(|m| m.role == crate::types::message::MessageRole::ActionResult);\n let has_tool_output_in_context = thread.messages.iter().any(|m| {\n m.role == crate::types::message::MessageRole::User\n && (m.content.contains(\" result]\") || m.content.contains(\" error]\"))\n });\n if has_tool_results && !has_tool_output_in_context {\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"missing_tool_output\".into(),\n description:\n \"Tool results exist but no tool output in messages — model may not see tool results\"\n .into(),\n step: None,\n });\n }\n\n // 7. Check for excessive iterations\n if thread.step_count > 10 {\n issues.push(TraceIssue {\n severity: IssueSeverity::Warning,\n category: \"excessive_steps\".into(),\n description: format!(\n \"Thread took {} steps — may be stuck in a loop\",\n thread.step_count\n ),\n step: None,\n });\n }\n\n // 8. Check for text response without FINAL (model answered from memory)\n let text_without_code = thread.events.iter().all(|e| {\n !matches!(\n e.kind,\n crate::types::event::EventKind::ActionExecuted { .. }\n )\n });\n if text_without_code && thread.step_count == 1 && has_assistant_response {\n issues.push(TraceIssue {\n severity: IssueSeverity::Info,\n category: \"no_tools_used\".into(),\n description: \"Model answered in one step without using any tools — may be answering from training data\".into(),\n step: Some(1),\n });\n }\n\n // 9. Check for LLM not producing code blocks\n let code_steps = thread\n .events\n .iter()\n .filter(|e| matches!(e.kind, crate::types::event::EventKind::StepStarted { .. }))\n .count();\n let text_responses_without_code = thread\n .messages\n .iter()\n .filter(|m| {\n m.role == crate::types::message::MessageRole::Assistant\n && !m.content.contains(\"```\")\n && !m.content.contains(\"FINAL(\")\n })\n .count();\n if text_responses_without_code > 0 && code_steps > 0 {\n issues.push(TraceIssue {\n severity: IssueSeverity::Info,\n category: \"mixed_mode\".into(),\n description: format!(\n \"{text_responses_without_code} text response(s) without code blocks — model may not be following CodeAct prompt\"\n ),\n step: None,\n });\n }\n\n // 10. Extract failure reason from StateChanged → Failed events\n for event in &thread.events {\n if let crate::types::event::EventKind::StateChanged {\n to: ThreadState::Failed,\n reason: Some(reason),\n ..\n } = &event.kind\n {\n if reason.contains(\"LLM\") || reason.contains(\"Provider\") {\n issues.push(TraceIssue {\n severity: IssueSeverity::Error,\n category: \"llm_error\".into(),\n description: format!(\"LLM provider error: {}\", truncate(reason, 300)),\n step: None,\n });\n } else if reason.contains(\"orchestrator\") {\n issues.push(TraceIssue {\n severity: IssueSeverity::Error,\n category: \"orchestrator_error\".into(),\n description: format!(\"Orchestrator error: {}\", truncate(reason, 300)),\n step: None,\n });\n }\n }\n }\n\n issues\n}\n\nfn truncate(s: &str, max_chars: usize) -> String {\n let chars: String = s.chars().take(max_chars).collect();\n if s.chars().count() > max_chars {\n format!(\"{chars}...\")\n } else {\n chars\n }\n}\n\n#[cfg(test)]\nmod tests {\n use super::*;\n use crate::types::event::EventKind;\n use crate::types::message::ThreadMessage;\n use crate::types::project::ProjectId;\n use crate::types::step::StepId;\n use crate::types::thread::{ThreadConfig, ThreadType};\n\n fn make_thread() -> Thread {\n Thread::new(\n \"test goal\",\n ThreadType::Foreground,\n ProjectId::new(),\n \"test-user\",\n ThreadConfig::default(),\n )\n }\n\n // ── empty_call_id detection (OpenAI / Codex rejection) ───\n\n /// OpenAI and Codex reject ActionResult messages with empty call_id.\n /// The trace analyzer must flag these as errors.\n #[test]\n fn detects_empty_call_id_on_action_result() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"calling tool\"));\n // Simulate the bug: empty call_id\n thread.add_message(ThreadMessage::action_result(\"\", \"web_search\", \"result\"));\n\n let issues = analyze_trace(&thread);\n let empty_id_issues: Vec<_> = issues\n .iter()\n .filter(|i| i.category == \"empty_call_id\")\n .collect();\n\n assert_eq!(empty_id_issues.len(), 1);\n assert_eq!(empty_id_issues[0].severity, IssueSeverity::Error);\n assert!(empty_id_issues[0].description.contains(\"web_search\"));\n }\n\n /// ActionResult with None call_id should also be flagged.\n #[test]\n fn detects_none_call_id_on_action_result() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"calling tool\"));\n // Manually construct a message with None call_id\n thread.add_message(ThreadMessage {\n role: crate::types::message::MessageRole::ActionResult,\n content: \"result\".into(),\n provenance: crate::types::provenance::Provenance::ToolOutput {\n action_name: \"shell\".into(),\n },\n action_call_id: None,\n action_name: Some(\"shell\".into()),\n action_calls: None,\n timestamp: chrono::Utc::now(),\n });\n\n let issues = analyze_trace(&thread);\n assert!(issues.iter().any(|i| i.category == \"empty_call_id\"));\n }\n\n /// No false positive: valid call_id should not be flagged.\n #[test]\n fn no_false_positive_for_valid_call_id() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"calling tool\"));\n thread.add_message(ThreadMessage::action_result(\n \"call_abc123\",\n \"web_search\",\n \"result\",\n ));\n\n let issues = analyze_trace(&thread);\n assert!(\n !issues.iter().any(|i| i.category == \"empty_call_id\"),\n \"valid call_id should not be flagged\"\n );\n }\n\n // ── tool_error detection ─────────────────────────────────\n\n /// ActionFailed events should produce tool_error warnings.\n #[test]\n fn detects_tool_failures_in_events() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"ok\"));\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::ActionFailed {\n step_id: StepId::new(),\n action_name: \"web_search\".into(),\n call_id: \"call_123\".into(),\n error: \"No lease for action 'web_search'\".into(),\n params_summary: None,\n },\n ));\n\n let issues = analyze_trace(&thread);\n let tool_errors: Vec<_> = issues\n .iter()\n .filter(|i| i.category == \"tool_error\")\n .collect();\n assert_eq!(tool_errors.len(), 1);\n assert!(tool_errors[0].description.contains(\"web_search\"));\n }\n\n // ── thread_failure detection ─────────────────────────────\n\n #[test]\n fn detects_failed_thread_state() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"trying\"));\n thread.state = ThreadState::Failed;\n\n let issues = analyze_trace(&thread);\n assert!(issues.iter().any(|i| i.category == \"thread_failure\"));\n }\n\n // ── LLM error detection from StateChanged events ─────────\n\n /// Reproduces the exact pattern from the trace: OpenAI rejects empty call_id.\n #[test]\n fn detects_llm_error_from_state_changed() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"ok\"));\n thread.state = ThreadState::Failed;\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::StateChanged {\n from: ThreadState::Running,\n to: ThreadState::Failed,\n reason: Some(\n \"LLM error: Provider openai_codex request failed: HTTP 400 Bad Request: \\\n Invalid 'input[5].call_id': empty string\"\n .into(),\n ),\n },\n ));\n\n let issues = analyze_trace(&thread);\n assert!(\n issues.iter().any(|i| i.category == \"llm_error\"),\n \"should detect LLM provider error in StateChanged reason\"\n );\n }\n\n // ── Multiple empty call_ids ──────────────────────────────\n\n /// Anthropic sends consecutive tool results merged into one User message.\n /// If multiple ActionResults have empty call_ids, each must be flagged.\n #[test]\n fn flags_each_empty_call_id_separately() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"parallel calls\"));\n thread.add_message(ThreadMessage::action_result(\"\", \"tool_a\", \"result_a\"));\n thread.add_message(ThreadMessage::action_result(\"\", \"tool_b\", \"result_b\"));\n thread.add_message(ThreadMessage::action_result(\n \"call_ok\", \"tool_c\", \"result_c\",\n ));\n\n let issues = analyze_trace(&thread);\n let empty_issues: Vec<_> = issues\n .iter()\n .filter(|i| i.category == \"empty_call_id\")\n .collect();\n assert_eq!(\n empty_issues.len(),\n 2,\n \"should flag exactly the 2 empty call_ids\"\n );\n }\n\n // ── CodeExecutionFailed event detection ────────────────────\n\n #[test]\n fn detects_code_execution_failure_from_event() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"```repl\\nimport csv\\n```\"));\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::CodeExecutionFailed {\n step_id: StepId::new(),\n category: crate::types::step::CodeExecutionFailure::RuntimeError,\n error: \"ModuleNotFoundError: No module named 'csv'\".into(),\n code_hash: Some(\"abc123\".into()),\n duration_ms: 42,\n },\n ));\n\n let issues = analyze_trace(&thread);\n let code_issues: Vec<_> = issues\n .iter()\n .filter(|i| i.category.starts_with(\"code_\"))\n .collect();\n assert_eq!(code_issues.len(), 1);\n assert_eq!(code_issues[0].category, \"code_runtime_error\");\n assert_eq!(code_issues[0].severity, IssueSeverity::Warning);\n assert!(code_issues[0].description.contains(\"ModuleNotFoundError\"));\n }\n\n #[test]\n fn vm_panic_is_error_severity() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"code\"));\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::CodeExecutionFailed {\n step_id: StepId::new(),\n category: crate::types::step::CodeExecutionFailure::VmPanic,\n error: \"Monty panicked: unreachable\".into(),\n code_hash: None,\n duration_ms: 0,\n },\n ));\n\n let issues = analyze_trace(&thread);\n let panic_issues: Vec<_> = issues\n .iter()\n .filter(|i| i.category == \"code_vm_panic\")\n .collect();\n assert_eq!(panic_issues.len(), 1);\n assert_eq!(panic_issues[0].severity, IssueSeverity::Error);\n }\n\n #[test]\n fn fallback_message_detection_when_no_events() {\n // Threads from before instrumentation should still be detected\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"code\"));\n thread.add_message(ThreadMessage::user(\n \"[stdout]\\nNameError: name 'foo' is not defined\",\n ));\n\n let issues = analyze_trace(&thread);\n assert!(\n issues.iter().any(|i| i.category == \"code_error\"),\n \"should detect code error from message when no CodeExecutionFailed events exist\"\n );\n }\n\n #[test]\n fn trace_serializes_approval_request_payload() {\n let mut thread = make_thread();\n thread.add_message(ThreadMessage::system(\"sys\"));\n thread.add_message(ThreadMessage::assistant(\"installing notion\"));\n thread.events.push(ThreadEvent::new(\n thread.id,\n EventKind::ApprovalRequested {\n action_name: \"tool_install\".into(),\n call_id: \"call_install_1\".into(),\n parameters: Some(serde_json::json!({\"name\": \"notion\", \"kind\": \"mcp_server\"})),\n description: Some(\"Install an extension\".into()),\n allow_always: Some(true),\n gate_name: Some(\"approval\".into()),\n params_summary: Some(\"notion\".into()),\n },\n ));\n\n let trace = build_trace(&thread);\n // `Thread::add_message` records a `MessageAdded` event for each\n // message, so the `ApprovalRequested` event is no longer at index 0\n // — it's mixed in with the message events. Find it by kind.\n let approval = trace\n .events\n .iter()\n .find(|e| matches!(&e.kind, EventKind::ApprovalRequested { .. }))\n .expect(\"trace should contain an ApprovalRequested event\");\n match &approval.kind {\n EventKind::ApprovalRequested {\n action_name,\n call_id,\n parameters,\n description,\n allow_always,\n gate_name,\n params_summary,\n } => {\n assert_eq!(action_name, \"tool_install\");\n assert_eq!(call_id, \"call_install_1\");\n assert_eq!(\n parameters.as_ref().and_then(|p| p.get(\"name\")),\n Some(&serde_json::json!(\"notion\"))\n );\n assert_eq!(description.as_deref(), Some(\"Install an extension\"));\n assert_eq!(*allow_always, Some(true));\n assert_eq!(gate_name.as_deref(), Some(\"approval\"));\n assert_eq!(params_summary.as_deref(), Some(\"notion\"));\n }\n other => panic!(\"unexpected event kind: {other:?}\"),\n }\n\n let json = serde_json::to_string(&trace).expect(\"trace serializes\");\n assert!(json.contains(\"\\\"ApprovalRequested\\\"\"));\n assert!(json.contains(\"\\\"action_name\\\":\\\"tool_install\\\"\"));\n assert!(json.contains(\"\\\"call_id\\\":\\\"call_install_1\\\"\"));\n // Parameter map key order isn't stable across serde_json versions; check\n // both required keys are present rather than the exact serialized form.\n assert!(json.contains(\"\\\"name\\\":\\\"notion\\\"\"));\n assert!(json.contains(\"\\\"kind\\\":\\\"mcp_server\\\"\"));\n assert!(json.contains(\"\\\"description\\\":\\\"Install an extension\\\"\"));\n assert!(json.contains(\"\\\"allow_always\\\":true\"));\n assert!(json.contains(\"\\\"gate_name\\\":\\\"approval\\\"\"));\n assert!(json.contains(\"\\\"params_summary\\\":\\\"notion\\\"\"));\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/lib.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "etag", + "\"9ecaa6b575a369535e657b5f922187d3ce5149fa\"" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "x-ratelimit-used", + "40" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "content-length", + "19797" + ], + [ + "x-frame-options", + "deny" + ], + [ + "server", + "github.com" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-github-request-id", + "145C:2449C8:40C2F3:4C2467:69DFAF16" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "date", + "Wed, 15 Apr 2026 15:30:30 GMT" + ], + [ + "x-ratelimit-remaining", + "4960" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-xss-protection", + "0" + ] + ], + "body": "//! IronClaw Engine — unified thread-capability-CodeAct execution model.\n//!\n//! This crate provides the core execution engine for IronClaw, unifying\n//! ~10 separate abstractions (Session, Job, Routine, Channel, Tool, Skill,\n//! Hook, Observer, Extension, LoopDelegate) around 5 primitives:\n//!\n//! - **Thread** — unit of work (replaces Session + Job + Routine + Sub-agent)\n//! - **Step** — unit of execution (replaces agentic loop iteration + tool calls)\n//! - **Capability** — unit of effect (replaces Tool + Skill + Hook + Extension)\n//! - **MemoryDoc** — unit of durable knowledge (replaces workspace memory blobs)\n//! - **Project** — unit of context (replaces flat workspace namespace)\n//!\n//! The engine defines traits for external dependencies ([`LlmBackend`],\n//! [`Store`], [`EffectExecutor`]) that the host crate implements via bridge\n//! adapters over existing infrastructure.\n\n// Security: `__regex_match__` (in `executor/orchestrator.rs`) accepts\n// arbitrary patterns from the Python orchestrator and runs them on\n// user-supplied text. The default `regex` crate is linear-time. The\n// `fancy-regex` crate supports backreferences and is NOT linear-time, which\n// would turn that handler into a ReDoS vector. Cargo.toml depends on\n// `regex = \"1\"` with default features only — do NOT add `fancy-regex` to\n// this crate's dependency tree without first redesigning `__regex_match__`\n// to enforce a wall-clock matching budget.\n\npub mod capability;\npub mod executor;\npub mod gate;\npub mod memory;\npub mod reliability;\npub mod runtime;\npub mod traits;\npub mod types;\n\n// ── Re-exports: types ───────────────────────────────────────\n\npub use types::capability::{\n ActionDef, Capability, CapabilityLease, EffectType, GrantedActions, LeaseId, PolicyCondition,\n PolicyEffect, PolicyRule,\n};\npub use types::error::{CapabilityError, EngineError, StepError, ThreadError};\npub use types::event::{EventId, EventKind, ThreadEvent};\npub use types::memory::{DocId, DocType, MemoryDoc};\npub use types::message::{MessageRole, ThreadMessage};\npub use types::mission::{Mission, MissionCadence, MissionId, MissionStatus, ValidTimezone};\npub use types::project::{Project, ProjectId};\npub use types::provenance::Provenance;\npub use types::step::{\n ActionCall, ActionResult, CodeExecutionFailure, ExecutionTier, LlmResponse, Step, StepId,\n StepStatus, TokenUsage,\n};\npub use types::thread::{\n ActiveSkillProvenance, Thread, ThreadConfig, ThreadId, ThreadState, ThreadType,\n};\n\n// ── Re-exports: traits ──────────────────────────────────────\n\npub use traits::effect::{EffectExecutor, ThreadExecutionContext};\npub use traits::llm::{LlmBackend, LlmCallConfig, LlmOutput};\npub use traits::store::Store;\npub use traits::workspace::WorkspaceReader;\n\n// ── Re-exports: capability ────────────────────────────────────\n\npub use capability::lease::LeaseManager;\npub use capability::planner::{CapabilityGrantPlan, LeasePlanner};\npub use capability::policy::{PolicyDecision, PolicyEngine};\npub use capability::registry::CapabilityRegistry;\n\n// ── Re-exports: gate ─────────────────────────────────────────\n\npub use gate::lease::LeaseGate;\npub use gate::pipeline::GatePipeline;\npub use gate::tool_tier::{ToolTier, classify_tool_tier};\npub use gate::{\n ExecutionGate, ExecutionMode, GateContext, GateDecision, GateResolution, ResumeKind,\n};\n\n// ── Re-exports: runtime ───────────────────────────────────────\n\npub use executor::prompt::PlatformInfo;\npub use runtime::conversation::ConversationManager;\npub use runtime::manager::ThreadManager;\npub use runtime::messaging::ThreadOutcome;\npub use runtime::mission::{\n BudgetGate, FireRateLimit, MissionManager, MissionNotification, MissionUpdate,\n};\npub use runtime::tree::ThreadTree;\n\npub use types::conversation::{\n ConversationEntry, ConversationId, ConversationSurface, EntrySender,\n};\n\n// ── Re-exports: executor ──────────────────────────────────────\n\npub use executor::ExecutionLoop;\n\n// ── Re-exports: memory ────────────────────────────────────────\n\npub use memory::MemoryStore;\npub use memory::RetrievalEngine;\n\n// ── Re-exports: reliability ──────────────────────────────────\n\npub use reliability::ReliabilityTracker;\n\n// ── Test utilities ──────────────────────────────────────────\n\n#[cfg(test)]\npub(crate) mod tests {\n use tokio::sync::RwLock;\n\n use crate::traits::store::Store;\n use crate::types::capability::{CapabilityLease, LeaseId};\n use crate::types::conversation::{ConversationId, ConversationSurface};\n use crate::types::error::EngineError;\n use crate::types::event::ThreadEvent;\n use crate::types::memory::{DocId, MemoryDoc};\n use crate::types::mission::{Mission, MissionId, MissionStatus};\n use crate::types::project::{Project, ProjectId};\n use crate::types::step::Step;\n use crate::types::thread::{Thread, ThreadId, ThreadState};\n\n /// Shared in-memory Store implementation for tests.\n ///\n /// Stores all entity types with proper CRUD semantics and filtering by\n /// project_id / user_id. Use this instead of defining per-module mocks.\n pub struct InMemoryStore {\n threads: RwLock>,\n steps: RwLock>,\n events: RwLock>,\n projects: RwLock>,\n conversations: RwLock>,\n docs: RwLock>,\n leases: RwLock>,\n missions: RwLock>,\n }\n\n impl InMemoryStore {\n pub fn new() -> Self {\n Self {\n threads: RwLock::new(Vec::new()),\n steps: RwLock::new(Vec::new()),\n events: RwLock::new(Vec::new()),\n projects: RwLock::new(Vec::new()),\n conversations: RwLock::new(Vec::new()),\n docs: RwLock::new(Vec::new()),\n leases: RwLock::new(Vec::new()),\n missions: RwLock::new(Vec::new()),\n }\n }\n\n pub fn with_docs(docs: Vec) -> Self {\n Self {\n docs: RwLock::new(docs),\n ..Self::new()\n }\n }\n }\n\n #[async_trait::async_trait]\n impl Store for InMemoryStore {\n async fn save_thread(&self, thread: &Thread) -> Result<(), EngineError> {\n let mut threads = self.threads.write().await;\n threads.retain(|t| t.id != thread.id);\n threads.push(thread.clone());\n Ok(())\n }\n async fn load_thread(&self, id: ThreadId) -> Result, EngineError> {\n Ok(self\n .threads\n .read()\n .await\n .iter()\n .find(|t| t.id == id)\n .cloned())\n }\n async fn list_threads(\n &self,\n project_id: ProjectId,\n user_id: &str,\n ) -> Result, EngineError> {\n Ok(self\n .threads\n .read()\n .await\n .iter()\n .filter(|t| t.project_id == project_id && t.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn update_thread_state(\n &self,\n id: ThreadId,\n state: ThreadState,\n ) -> Result<(), EngineError> {\n let mut threads = self.threads.write().await;\n if let Some(t) = threads.iter_mut().find(|t| t.id == id) {\n t.state = state;\n }\n Ok(())\n }\n async fn save_step(&self, step: &Step) -> Result<(), EngineError> {\n let mut steps = self.steps.write().await;\n steps.retain(|s| s.id != step.id);\n steps.push(step.clone());\n Ok(())\n }\n async fn load_steps(&self, thread_id: ThreadId) -> Result, EngineError> {\n Ok(self\n .steps\n .read()\n .await\n .iter()\n .filter(|s| s.thread_id == thread_id)\n .cloned()\n .collect())\n }\n async fn append_events(&self, events: &[ThreadEvent]) -> Result<(), EngineError> {\n self.events.write().await.extend(events.iter().cloned());\n Ok(())\n }\n async fn load_events(&self, thread_id: ThreadId) -> Result, EngineError> {\n Ok(self\n .events\n .read()\n .await\n .iter()\n .filter(|e| e.thread_id == thread_id)\n .cloned()\n .collect())\n }\n async fn save_project(&self, project: &Project) -> Result<(), EngineError> {\n let mut projects = self.projects.write().await;\n projects.retain(|p| p.id != project.id);\n projects.push(project.clone());\n Ok(())\n }\n async fn load_project(&self, id: ProjectId) -> Result, EngineError> {\n Ok(self\n .projects\n .read()\n .await\n .iter()\n .find(|p| p.id == id)\n .cloned())\n }\n async fn list_projects(&self, user_id: &str) -> Result, EngineError> {\n Ok(self\n .projects\n .read()\n .await\n .iter()\n .filter(|p| p.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn list_all_projects(&self) -> Result, EngineError> {\n Ok(self.projects.read().await.iter().cloned().collect())\n }\n async fn save_conversation(\n &self,\n conversation: &ConversationSurface,\n ) -> Result<(), EngineError> {\n let mut conversations = self.conversations.write().await;\n conversations.retain(|c| c.id != conversation.id);\n conversations.push(conversation.clone());\n Ok(())\n }\n async fn load_conversation(\n &self,\n id: ConversationId,\n ) -> Result, EngineError> {\n Ok(self\n .conversations\n .read()\n .await\n .iter()\n .find(|c| c.id == id)\n .cloned())\n }\n async fn list_conversations(\n &self,\n user_id: &str,\n ) -> Result, EngineError> {\n Ok(self\n .conversations\n .read()\n .await\n .iter()\n .filter(|c| c.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn save_memory_doc(&self, doc: &MemoryDoc) -> Result<(), EngineError> {\n let mut docs = self.docs.write().await;\n docs.retain(|d| d.id != doc.id);\n docs.push(doc.clone());\n Ok(())\n }\n async fn load_memory_doc(&self, id: DocId) -> Result, EngineError> {\n Ok(self.docs.read().await.iter().find(|d| d.id == id).cloned())\n }\n async fn list_memory_docs(\n &self,\n project_id: ProjectId,\n user_id: &str,\n ) -> Result, EngineError> {\n Ok(self\n .docs\n .read()\n .await\n .iter()\n .filter(|d| d.project_id == project_id && d.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn save_lease(&self, lease: &CapabilityLease) -> Result<(), EngineError> {\n let mut leases = self.leases.write().await;\n leases.retain(|l| l.id != lease.id);\n leases.push(lease.clone());\n Ok(())\n }\n async fn load_active_leases(\n &self,\n thread_id: ThreadId,\n ) -> Result, EngineError> {\n Ok(self\n .leases\n .read()\n .await\n .iter()\n .filter(|l| l.thread_id == thread_id && !l.revoked)\n .cloned()\n .collect())\n }\n async fn revoke_lease(&self, lease_id: LeaseId, _reason: &str) -> Result<(), EngineError> {\n let mut leases = self.leases.write().await;\n if let Some(l) = leases.iter_mut().find(|l| l.id == lease_id) {\n l.revoked = true;\n }\n Ok(())\n }\n async fn save_mission(&self, mission: &Mission) -> Result<(), EngineError> {\n let mut missions = self.missions.write().await;\n missions.retain(|m| m.id != mission.id);\n missions.push(mission.clone());\n Ok(())\n }\n async fn load_mission(&self, id: MissionId) -> Result, EngineError> {\n Ok(self\n .missions\n .read()\n .await\n .iter()\n .find(|m| m.id == id)\n .cloned())\n }\n async fn list_missions(\n &self,\n project_id: ProjectId,\n user_id: &str,\n ) -> Result, EngineError> {\n Ok(self\n .missions\n .read()\n .await\n .iter()\n .filter(|m| m.project_id == project_id && m.user_id == user_id)\n .cloned()\n .collect())\n }\n async fn update_mission_status(\n &self,\n id: MissionId,\n status: MissionStatus,\n ) -> Result<(), EngineError> {\n let mut missions = self.missions.write().await;\n if let Some(m) = missions.iter_mut().find(|m| m.id == id) {\n m.status = status;\n }\n Ok(())\n }\n async fn list_all_threads(\n &self,\n project_id: ProjectId,\n ) -> Result, EngineError> {\n Ok(self\n .threads\n .read()\n .await\n .iter()\n .filter(|t| t.project_id == project_id)\n .cloned()\n .collect())\n }\n async fn list_all_missions(\n &self,\n project_id: ProjectId,\n ) -> Result, EngineError> {\n Ok(self\n .missions\n .read()\n .await\n .iter()\n .filter(|m| m.project_id == project_id)\n .cloned()\n .collect())\n }\n }\n\n struct MinimalStore;\n\n #[async_trait::async_trait]\n impl Store for MinimalStore {\n async fn save_thread(&self, _thread: &Thread) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_thread(&self, _id: ThreadId) -> Result, EngineError> {\n Ok(None)\n }\n async fn list_threads(\n &self,\n _project_id: ProjectId,\n _user_id: &str,\n ) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn update_thread_state(\n &self,\n _id: ThreadId,\n _state: ThreadState,\n ) -> Result<(), EngineError> {\n Ok(())\n }\n async fn save_step(&self, _step: &Step) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_steps(&self, _thread_id: ThreadId) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn append_events(&self, _events: &[ThreadEvent]) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_events(&self, _thread_id: ThreadId) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn save_project(&self, _project: &Project) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_project(&self, _id: ProjectId) -> Result, EngineError> {\n Ok(None)\n }\n async fn save_memory_doc(&self, _doc: &MemoryDoc) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_memory_doc(&self, _id: DocId) -> Result, EngineError> {\n Ok(None)\n }\n async fn list_memory_docs(\n &self,\n _project_id: ProjectId,\n _user_id: &str,\n ) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn save_lease(&self, _lease: &CapabilityLease) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_active_leases(\n &self,\n _thread_id: ThreadId,\n ) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn revoke_lease(&self, _lease_id: LeaseId, _reason: &str) -> Result<(), EngineError> {\n Ok(())\n }\n async fn save_mission(&self, _mission: &Mission) -> Result<(), EngineError> {\n Ok(())\n }\n async fn load_mission(&self, _id: MissionId) -> Result, EngineError> {\n Ok(None)\n }\n async fn list_missions(\n &self,\n _project_id: ProjectId,\n _user_id: &str,\n ) -> Result, EngineError> {\n Ok(Vec::new())\n }\n async fn update_mission_status(\n &self,\n _id: MissionId,\n _status: MissionStatus,\n ) -> Result<(), EngineError> {\n Ok(())\n }\n }\n\n #[tokio::test]\n async fn store_defaults_fail_closed() {\n let store = MinimalStore;\n assert!(matches!(\n store.list_projects(\"alice\").await,\n Err(EngineError::Store { .. })\n ));\n assert!(matches!(\n store.list_all_projects().await,\n Err(EngineError::Store { .. })\n ));\n assert!(matches!(\n store.load_conversation(ConversationId::new()).await,\n Err(EngineError::Store { .. })\n ));\n assert!(matches!(\n store.list_all_threads(ProjectId::new()).await,\n Err(EngineError::Store { .. })\n ));\n }\n\n #[tokio::test]\n async fn shared_queries_include_legacy_and_current_shared_owner() {\n use crate::types::memory::DocType;\n use crate::types::{LEGACY_SHARED_OWNER_ID, shared_owner_id};\n\n let project_id = ProjectId::new();\n let mut legacy = MemoryDoc::new(\n project_id,\n LEGACY_SHARED_OWNER_ID,\n DocType::Note,\n \"legacy\",\n \"a\",\n );\n let current = MemoryDoc::new(project_id, shared_owner_id(), DocType::Note, \"current\", \"b\");\n legacy.id = DocId::new();\n let store = InMemoryStore::with_docs(vec![legacy, current]);\n\n let docs = store\n .list_memory_docs_with_shared(project_id, \"alice\")\n .await\n .unwrap();\n assert_eq!(docs.len(), 2);\n }\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/types/event.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "date", + "Wed, 15 Apr 2026 15:30:31 GMT" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "etag", + "\"b61185f07131d737272f1068c899f7f6b5821960\"" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "x-frame-options", + "deny" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "x-xss-protection", + "0" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "server", + "github.com" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "content-length", + "8364" + ], + [ + "x-ratelimit-remaining", + "4959" + ], + [ + "x-ratelimit-used", + "41" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "x-github-request-id", + "1470:16ECB3:4432D7:4F96EA:69DFAF17" + ] + ], + "body": "//! Event sourcing types.\n//!\n//! Every significant action within a thread is recorded as an event.\n//! This enables replay, debugging, reflection, and trace-based testing.\n\nuse chrono::{DateTime, Utc};\nuse serde::{Deserialize, Serialize};\nuse uuid::Uuid;\n\nuse crate::types::capability::LeaseId;\n\n/// Generate a short human-readable summary of tool parameters for display.\n///\n/// For `http`: shows the URL. For `web_search`: shows the query.\n/// For other tools: shows the first string argument, truncated.\n/// Returns `None` for empty or unrecognizable params.\npub fn summarize_params(action_name: &str, params: &serde_json::Value) -> Option {\n let summary = match action_name {\n \"http\" | \"web_fetch\" => params\n .get(\"url\")\n .and_then(|v| v.as_str())\n .map(|u| truncate(u, 80)),\n \"web_search\" | \"llm_context\" => params\n .get(\"query\")\n .and_then(|v| v.as_str())\n .map(|q| truncate(q, 60)),\n \"memory_search\" => params\n .get(\"query\")\n .and_then(|v| v.as_str())\n .map(|q| truncate(q, 60)),\n \"memory_write\" => params\n .get(\"target\")\n .and_then(|v| v.as_str())\n .map(|t| t.to_string()),\n \"memory_read\" => params\n .get(\"path\")\n .and_then(|v| v.as_str())\n .map(|p| p.to_string()),\n \"shell\" => params\n .get(\"command\")\n .and_then(|v| v.as_str())\n .map(|c| truncate(c, 60)),\n \"message\" => params\n .get(\"content\")\n .and_then(|v| v.as_str())\n .map(|c| truncate(c, 40)),\n _ => {\n // Generic: show first string value\n if let Some(obj) = params.as_object() {\n obj.values()\n .find_map(|v| v.as_str())\n .map(|s| truncate(s, 50))\n } else {\n None\n }\n }\n };\n summary.filter(|s| !s.is_empty())\n}\n\nfn truncate(s: &str, max: usize) -> String {\n if s.len() <= max {\n s.to_string()\n } else {\n // Find a safe UTF-8 boundary\n let mut end = max.min(s.len());\n while end > 0 && !s.is_char_boundary(end) {\n end -= 1;\n }\n format!(\"{}...\", &s[..end]) // safety: end is validated by is_char_boundary loop above\n }\n}\nuse crate::types::step::{StepId, TokenUsage};\nuse crate::types::thread::{ThreadId, ThreadState};\n\n/// Strongly-typed event identifier.\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]\npub struct EventId(pub Uuid);\n\nimpl EventId {\n pub fn new() -> Self {\n Self(Uuid::new_v4())\n }\n}\n\nimpl Default for EventId {\n fn default() -> Self {\n Self::new()\n }\n}\n\n/// A recorded event in a thread's execution history.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct ThreadEvent {\n pub id: EventId,\n pub thread_id: ThreadId,\n pub timestamp: DateTime,\n pub kind: EventKind,\n}\n\nimpl ThreadEvent {\n pub fn new(thread_id: ThreadId, kind: EventKind) -> Self {\n Self {\n id: EventId::new(),\n thread_id,\n timestamp: Utc::now(),\n kind,\n }\n }\n}\n\n/// The specific kind of event that occurred.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub enum EventKind {\n // ── Thread lifecycle ────────────────────────────────────\n StateChanged {\n from: ThreadState,\n to: ThreadState,\n reason: Option,\n },\n\n // ── Step lifecycle ──────────────────────────────────────\n StepStarted {\n step_id: StepId,\n },\n StepCompleted {\n step_id: StepId,\n tokens: TokenUsage,\n },\n StepFailed {\n step_id: StepId,\n error: String,\n },\n\n // ── Action execution ────────────────────────────────────\n ActionExecuted {\n step_id: StepId,\n action_name: String,\n call_id: String,\n duration_ms: u64,\n /// Short human-readable summary of parameters (e.g., URL for http tool).\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n params_summary: Option,\n },\n ActionFailed {\n step_id: StepId,\n action_name: String,\n call_id: String,\n error: String,\n /// Short human-readable summary of parameters.\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n params_summary: Option,\n },\n\n // ── Capability leases ───────────────────────────────────\n LeaseGranted {\n lease_id: LeaseId,\n capability_name: String,\n },\n LeaseRevoked {\n lease_id: LeaseId,\n reason: String,\n },\n LeaseExpired {\n lease_id: LeaseId,\n },\n\n // ── Messages ────────────────────────────────────────────\n MessageAdded {\n role: String,\n content_preview: String,\n },\n\n // ── Thread tree ─────────────────────────────────────────\n ChildSpawned {\n child_id: ThreadId,\n goal: String,\n },\n ChildCompleted {\n child_id: ThreadId,\n },\n\n // ── Approval flow ───────────────────────────────────────\n ApprovalRequested {\n action_name: String,\n call_id: String,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n parameters: Option,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n description: Option,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n allow_always: Option,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n gate_name: Option,\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n params_summary: Option,\n },\n ApprovalReceived {\n call_id: String,\n approved: bool,\n },\n\n // ── Self-improvement ──────────────────────────────────────\n SelfImprovementStarted,\n SelfImprovementComplete {\n prompt_updated: bool,\n patterns_added: usize,\n },\n SelfImprovementFailed {\n error: String,\n },\n\n // ── Skill activation ───────────────────────────────────────\n SkillActivated {\n skill_names: Vec,\n },\n\n // ── Code execution instrumentation ────────────────────────\n /// Emitted when a code (REPL) execution attempt fails. Enables aggregate\n /// analysis of code execution failure modes to determine whether the\n /// runtime (Monty), the LLM, or tool dispatch is the primary source of\n /// failures.\n CodeExecutionFailed {\n step_id: StepId,\n /// Classified failure category.\n category: crate::types::step::CodeExecutionFailure,\n /// The error message text (truncated to 500 chars).\n error: String,\n /// Hash of the Python code that was executed, for dedup/correlation.\n #[serde(default, skip_serializing_if = \"Option::is_none\")]\n code_hash: Option,\n /// Duration of the code execution attempt in milliseconds.\n #[serde(default)]\n duration_ms: u64,\n },\n\n // ── Orchestrator versioning ───────────────────────────────\n OrchestratorRollback {\n from_version: u64,\n to_version: u64,\n reason: String,\n },\n\n /// Unknown event kind — catch-all for forward compatibility during\n /// rolling deploys. Older binaries deserializing events written by\n /// newer binaries will produce this variant instead of failing.\n #[serde(other)]\n Unknown,\n}\n" + } + }, + { + "request": { + "method": "GET", + "url": "https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src/types/step.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340", + "headers": [ + [ + "Accept", + "application/vnd.github.raw" + ] + ] + }, + "response": { + "status": 200, + "headers": [ + [ + "x-xss-protection", + "0" + ], + [ + "vary", + "Accept, Authorization, Cookie, X-GitHub-OTP, Accept-Encoding, Accept, X-Requested-With" + ], + [ + "access-control-expose-headers", + "ETag, Link, Location, Retry-After, X-GitHub-OTP, X-RateLimit-Limit, X-RateLimit-Remaining, X-RateLimit-Used, X-RateLimit-Resource, X-RateLimit-Reset, X-OAuth-Scopes, X-Accepted-OAuth-Scopes, X-Poll-Interval, X-GitHub-Media-Type, X-GitHub-SSO, X-GitHub-Request-Id, Deprecation, Sunset" + ], + [ + "x-content-type-options", + "nosniff" + ], + [ + "date", + "Wed, 15 Apr 2026 15:30:31 GMT" + ], + [ + "x-ratelimit-limit", + "5000" + ], + [ + "x-ratelimit-remaining", + "4958" + ], + [ + "content-security-policy", + "default-src 'none'" + ], + [ + "x-oauth-scopes", + "repo" + ], + [ + "content-length", + "6490" + ], + [ + "last-modified", + "Wed, 15 Apr 2026 14:22:45 GMT" + ], + [ + "referrer-policy", + "origin-when-cross-origin, strict-origin-when-cross-origin" + ], + [ + "cache-control", + "private, max-age=60, s-maxage=60" + ], + [ + "x-ratelimit-reset", + "1776268928" + ], + [ + "server", + "github.com" + ], + [ + "x-ratelimit-used", + "42" + ], + [ + "x-accepted-oauth-scopes", + "repo" + ], + [ + "x-frame-options", + "deny" + ], + [ + "x-ratelimit-resource", + "core" + ], + [ + "access-control-allow-origin", + "*" + ], + [ + "etag", + "\"161096e9b02238d78d86f611befa72d32fbe0a76\"" + ], + [ + "x-github-api-version-selected", + "2022-11-28" + ], + [ + "x-github-media-type", + "github.v3; param=raw" + ], + [ + "content-type", + "application/vnd.github.raw; charset=utf-8" + ], + [ + "x-github-request-id", + "1482:3BD57F:42210A:4D83BD:69DFAF17" + ] + ], + "body": "//! Step — the unit of execution within a thread.\n//!\n//! Each step corresponds to one LLM call plus its subsequent action\n//! executions. This replaces the implicit \"iteration\" counter in the\n//! existing `run_agentic_loop`.\n\nuse std::time::Duration;\n\nuse chrono::{DateTime, Utc};\nuse serde::{Deserialize, Serialize};\nuse uuid::Uuid;\n\nuse crate::types::thread::ThreadId;\n\n/// Strongly-typed step identifier.\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]\npub struct StepId(pub Uuid);\n\nimpl StepId {\n pub fn new() -> Self {\n Self(Uuid::new_v4())\n }\n}\n\nimpl Default for StepId {\n fn default() -> Self {\n Self::new()\n }\n}\n\n/// Status of a step within its lifecycle.\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]\npub enum StepStatus {\n Pending,\n LlmCalling,\n Executing,\n Completed,\n Failed,\n}\n\n/// Which execution tier handles the step's code/actions.\n///\n/// Monty is the sole CodeAct/RLM executor. WASM and Docker are used for\n/// third-party tool isolation and thread sandboxing (Phase 8), not for\n/// running LLM-generated Python.\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]\npub enum ExecutionTier {\n /// Structured tool calls (JSON action calls from LLM).\n Structured,\n /// Embedded Python via Monty (CodeAct/RLM pattern).\n Scripting,\n}\n\n/// A single execution step within a thread.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct Step {\n pub id: StepId,\n pub thread_id: ThreadId,\n /// 1-indexed sequence within the thread.\n pub sequence: usize,\n pub status: StepStatus,\n pub tier: ExecutionTier,\n pub llm_response: Option,\n pub action_results: Vec,\n pub tokens_used: TokenUsage,\n pub started_at: DateTime,\n pub completed_at: Option>,\n}\n\nimpl Step {\n pub fn new(thread_id: ThreadId, sequence: usize) -> Self {\n Self {\n id: StepId::new(),\n thread_id,\n sequence,\n status: StepStatus::Pending,\n tier: ExecutionTier::Structured,\n llm_response: None,\n action_results: Vec::new(),\n tokens_used: TokenUsage::default(),\n started_at: Utc::now(),\n completed_at: None,\n }\n }\n}\n\n// ── LLM response types ─────────────────────────────────────\n\n/// Response from the LLM: text, action calls, or executable code.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub enum LlmResponse {\n /// Final text response.\n Text(String),\n /// One or more action calls (with optional reasoning text).\n ActionCalls {\n calls: Vec,\n content: Option,\n },\n /// Executable Python code (CodeAct). Tool calls happen as function\n /// calls within the code; the runtime suspends at each one and\n /// delegates to the EffectExecutor.\n Code {\n code: String,\n content: Option,\n },\n}\n\n/// A request from the LLM to execute a capability action.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct ActionCall {\n /// Unique call identifier (echoed in the result).\n pub id: String,\n /// Action name (e.g. \"web_fetch\", \"create_issue\").\n pub action_name: String,\n /// Action parameters as JSON.\n pub parameters: serde_json::Value,\n}\n\n/// Result of executing a capability action.\n#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct ActionResult {\n /// The call ID this result corresponds to.\n pub call_id: String,\n /// The action that was executed.\n pub action_name: String,\n /// Output value.\n pub output: serde_json::Value,\n /// Whether this result represents an error.\n pub is_error: bool,\n /// How long the action took.\n #[serde(with = \"duration_millis\")]\n pub duration: Duration,\n}\n\n/// Classification of code execution failures.\n///\n/// Used by the instrumentation layer to distinguish Monty VM limitations\n/// from LLM logic errors, tool dispatch failures, and resource exhaustion.\n/// This data enables informed decisions about runtime alternatives.\n#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]\n#[serde(rename_all = \"snake_case\")]\npub enum CodeExecutionFailure {\n /// Python parse error — LLM generated invalid syntax.\n SyntaxError,\n /// Python runtime error (NameError, TypeError, ValueError, etc.) —\n /// LLM logic bug or use of unsupported feature.\n RuntimeError,\n /// Name lookup failed — function/variable not in scope and not a known tool.\n NameLookup,\n /// Monty VM panicked (catch_unwind caught it). Indicates a Monty bug,\n /// not a user code issue.\n VmPanic,\n /// Resource limit hit (timeout, memory, or allocation cap).\n ResourceLimit,\n /// A tool call inside code returned an error.\n ToolError,\n /// OS operation attempted (blocked by sandbox).\n OsDenied,\n}\n\nimpl std::fmt::Display for CodeExecutionFailure {\n fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {\n match self {\n Self::SyntaxError => write!(f, \"syntax_error\"),\n Self::RuntimeError => write!(f, \"runtime_error\"),\n Self::NameLookup => write!(f, \"name_lookup\"),\n Self::VmPanic => write!(f, \"vm_panic\"),\n Self::ResourceLimit => write!(f, \"resource_limit\"),\n Self::ToolError => write!(f, \"tool_error\"),\n Self::OsDenied => write!(f, \"os_denied\"),\n }\n }\n}\n\n/// Token usage for a single LLM call.\n#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize)]\npub struct TokenUsage {\n pub input_tokens: u64,\n pub output_tokens: u64,\n pub cache_read_tokens: u64,\n pub cache_write_tokens: u64,\n /// USD cost for this call (populated by LlmBackend if cost data is available).\n pub cost_usd: f64,\n}\n\nimpl TokenUsage {\n pub fn total(&self) -> u64 {\n self.input_tokens + self.output_tokens\n }\n}\n\n/// Serde helper for Duration as milliseconds.\nmod duration_millis {\n use std::time::Duration;\n\n use serde::{Deserialize, Deserializer, Serializer};\n\n pub fn serialize(d: &Duration, s: S) -> Result {\n s.serialize_u64(d.as_millis() as u64)\n }\n\n pub fn deserialize<'de, D: Deserializer<'de>>(d: D) -> Result {\n let millis = u64::deserialize(d)?;\n Ok(Duration::from_millis(millis))\n }\n}\n" + } + } + ], + "steps": [ + { + "response": { + "type": "user_input", + "content": "/code-review nearai/ironclaw 2483" + } + }, + { + "request_hint": { + "last_user_message_contains": "/code-review nearai/ironclaw 2483", + "min_message_count": 2 + }, + "response": { + "type": "text", + "content": "```repl\nimport re\n\nasync def review():\n owner = \"nearai\"\n repo = \"ironclaw\"\n number = 2483\n\n pr_url = f\"https://api.github.com/repos/{owner}/{repo}/pulls/{number}\"\n files_url = f\"{pr_url}/files?per_page=100\"\n\n meta_r = await http(method=\"GET\", url=pr_url)\n diff_r = await http(\n method=\"GET\",\n url=pr_url,\n headers=[{\"name\": \"Accept\", \"value\": \"application/vnd.github.v3.diff\"}],\n )\n files_r = await http(method=\"GET\", url=files_url)\n\n for r, label in [(meta_r, \"metadata\"), (diff_r, \"diff\"), (files_r, \"files\")]:\n if r[\"status\"] != 200:\n return FINAL(f\"GitHub {label} fetch for {owner}/{repo}#{number} returned HTTP {r['status']}: {r['body']}\")\n\n pr = meta_r[\"body\"]\n diff_text = diff_r[\"body\"]\n files = files_r[\"body\"]\n head_sha = pr[\"head\"][\"sha\"]\n\n # Read full current file contents for changed files (where available)\n file_contents = {}\n for f in files:\n path = f[\"filename\"]\n raw_r = await http(\n method=\"GET\",\n url=f\"https://api.github.com/repos/{owner}/{repo}/contents/{path}?ref={head_sha}\",\n headers=[{\"name\": \"Accept\", \"value\": \"application/vnd.github.raw\"}],\n )\n if raw_r[\"status\"] == 200:\n file_contents[path] = raw_r[\"body\"]\n\n # Parse changed lines on RIGHT side from patches to anchor findings\n # mapping: path -> set(line_numbers)\n changed_right_lines = {}\n for f in files:\n path = f[\"filename\"]\n patch = f.get(\"patch\")\n if not patch:\n continue\n lines = patch.split(\"\\n\")\n right = 0\n for ln in lines:\n if ln.startswith(\"@@\"):\n m = re.search(r\"\\+(\\d+)(?:,(\\d+))?\", ln)\n if m:\n right = int(m.group(1))\n continue\n if ln.startswith(\"+\") and not ln.startswith(\"+++\"):\n changed_right_lines.setdefault(path, set()).add(right)\n right += 1\n elif ln.startswith(\"-\") and not ln.startswith(\"---\"):\n # left side only, don't advance right\n pass\n else:\n right += 1\n\n findings = []\n\n # Heuristic checks focused on substantial issues\n for f in files:\n path = f[\"filename\"]\n patch = f.get(\"patch\") or \"\"\n content = file_contents.get(path, \"\")\n right_lines = sorted(changed_right_lines.get(path, []))\n\n # 1) Security: added shell exec patterns\n if re.search(r'^\\+.*\\b(shell|exec|spawn|Command::new|subprocess|os\\.system)\\b', patch, flags=re.M):\n line = right_lines[0] if right_lines else 1\n findings.append({\n \"severity\": \"High\",\n \"category\": \"Security\",\n \"fileline\": f\"{path}:{line}\",\n \"finding\": \"New command-execution surface introduced; needs strict input validation and argument escaping guarantees.\",\n \"fix\": \"Use allowlisted commands/args, avoid shell interpolation, and add tests for malicious payloads.\"\n })\n\n # 2) Correctness: possible unwrap/expect introduction\n if re.search(r'^\\+.*\\b(unwrap\\(\\)|expect\\()', patch, flags=re.M):\n line = None\n for l in right_lines:\n # best-effort pinpoint in full content by line text matching isn't trivial; anchor first changed line\n line = l\n break\n if line is None:\n line = 1\n findings.append({\n \"severity\": \"Medium\",\n \"category\": \"Correctness\",\n \"fileline\": f\"{path}:{line}\",\n \"finding\": \"Introduces panic-prone error handling (`unwrap`/`expect`) in changed code path.\",\n \"fix\": \"Propagate structured errors instead of panicking; return typed error and cover with tests.\"\n })\n\n # 3) Robustness: TODO/HACK introduced in changed lines\n if re.search(r'^\\+.*\\b(TODO|FIXME|HACK)\\b', patch, flags=re.M):\n line = right_lines[0] if right_lines else 1\n findings.append({\n \"severity\": \"Low\",\n \"category\": \"Documentation/Assumptions\",\n \"fileline\": f\"{path}:{line}\",\n \"finding\": \"Introduces TODO/FIXME/HACK marker in production path without tracked follow-up.\",\n \"fix\": \"Convert to an issue reference with owner/date, or resolve before merge.\"\n })\n\n # 4) Tests coverage hint: source file changed without nearby test updates\n is_source = any(path.endswith(ext) for ext in [\".rs\", \".ts\", \".tsx\", \".js\", \".py\", \".go\"])\n if is_source and patch:\n # if file is in src and no test files changed globally, flag medium\n pass\n\n # Global coverage check\n changed_paths = [f[\"filename\"] for f in files]\n test_like = [p for p in changed_paths if re.search(r'(test|spec|__tests__|\\.snap|tests?/)', p)]\n source_like = [p for p in changed_paths if re.search(r'(src/|server/|crates/|packages/|apps/)', p) and re.search(r'\\.(rs|ts|tsx|js|py|go)$', p)]\n\n if source_like and not test_like:\n # Anchor to first source file with first changed line\n p0 = source_like[0]\n l0 = sorted(changed_right_lines.get(p0, {1}))[0] if changed_right_lines.get(p0) else 1\n findings.append({\n \"severity\": \"Medium\",\n \"category\": \"Test Coverage\",\n \"fileline\": f\"{p0}:{l0}\",\n \"finding\": \"Code changes in service/application logic without corresponding test updates in this PR.\",\n \"fix\": \"Add/adjust unit or integration tests covering happy path and key failure paths for modified behavior.\"\n })\n\n # If no heuristic findings, still provide an honest scoped result with concrete reference\n if not findings:\n # produce a conservative \"no material issues found\" with explicit checked scope and concrete file:line\n ref_path = changed_paths[0] if changed_paths else \"(no files)\"\n ref_line = 1\n if changed_paths and changed_right_lines.get(ref_path):\n ref_line = sorted(changed_right_lines[ref_path])[0]\n title = pr[\"title\"]\n body = f\"\"\"Review of {owner}/{repo}#{number}: {title}\n\nI reviewed the PR diff and read the full current contents of all changed files that were retrievable at `{head_sha}`. \nNo Critical/High/Medium defects were confirmed from this change set.\n\nConcrete reference checked: `{ref_path}:{ref_line}` (plus all other changed hunks/files in this PR).\n\n## Findings\n\n| # | Severity | Category | File:Line | Finding | Suggested fix |\n|---|----------|----------|-----------|---------|---------------|\n| 1 | Nit | Review result | {ref_path}:{ref_line} | No actionable defects found in this pass; change appears consistent with existing patterns. | Optional: add/confirm regression test coverage if behavior changed in non-trivial ways. |\n\nIf you want, I can now post this as a PR comment, or run a second pass focused only on security/concurrency.\"\"\"\n return FINAL(body)\n\n # Build findings table\n title = pr[\"title\"]\n rows = []\n for i, f in enumerate(findings, 1):\n rows.append(\n f\"| {i} | {f['severity']} | {f['category']} | {f['fileline']} | {f['finding']} | {f['fix']} |\"\n )\n table = \"\\n\".join(rows)\n\n summary = f\"Review of {owner}/{repo}#{number}: {title}\\n\\n\"\n summary += \"| # | Severity | Category | File:Line | Finding | Suggested fix |\\n\"\n summary += \"|---|----------|----------|-----------|---------|---------------|\\n\"\n summary += table\n summary += \"\\n\\nWhich findings should I post as PR comments? Default is all Critical, High, and Medium.\"\n return FINAL(summary)\n\nawait review()\n```", + "input_tokens": 24424, + "output_tokens": 1852 + } + }, + { + "response": { + "type": "user_input", + "content": "[stdout]\n\nError: Traceback (most recent call last):\n File \"step.py\", line 176, in \n await review()\n ~~~~~~~~~~~~~~\n File \"step.py\", line 75, in review\n if re.search(r'^\\+.*\\b(shell|exec|spawn|Command::new|subprocess|os\\.system)\\b', patch, flags=re.M):\n ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~\nTypeError: re.search() takes no keyword arguments\n\n[http] {'body': {'_links': {'comments': {'href': 'https://api.github.com/repos/nearai/ironclaw/issues/2483/comments'}, 'commits': {'href': 'https://api.github.com/repos/nearai/ironclaw/pulls/2483/commits'}, 'html': {'href': 'https://github.com/nearai/ironclaw/pull/2483'}, 'issue': {'href': 'https://api.github.com/repos/nearai/ironclaw/issues/2483'}, 'review_comment': {'href': 'https://api.github.com/repos/nearai/ironclaw/pulls/comments{/number}'}, 'review_comments': {'href': 'https://api.github.com/rep...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': 'diff --git a/crates/ironclaw_engine/src/executor/orchestrator.rs b/crates/ironclaw_engine/src/executor/orchestrator.rs\\nindex bf16598490..76bf321dee 100644\\n--- a/crates/ironclaw_engine/src/executor/orchestrator.rs\\n+++ b/crates/ironclaw_engine/src/executor/orchestrator.rs\\n@@ -720,6 +720,7 @@ async fn handle_execute_code_step(\\n };\\n \\n // Run user code in a nested Monty VM (same pattern as rlm_query)\\n+ let code_start = std::time::Instant::now();\\n match Box::pin(execu...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': [{'additions': 118, 'blob_url': 'https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Forchestrator.rs', 'changes': 122, 'contents_url': 'https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Forchestrator.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340', 'deletions': 4, 'filename': 'crates/ironclaw_engine/src/executor/orchestrator.rs', 'patch': '@@ -720,6 +720,7 @@ async fn...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': '//! Python orchestrator — the self-modifiable execution loop.\\n//!\\n//! Replaces the Rust `ExecutionLoop::run()` with versioned Python code\\n//! executed via Monty. The orchestrator is the \"glue layer\" between the\\n//! LLM and tools — tool dispatch, output formatting, state management,\\n//! truncation — all in Python, patchable by the self-improvement Mission.\\n//!\\n//! Host functions exposed to the orchestrator Python:\\n//! - `__llm_complete__` — make an LLM call\\n//! - `__execute_cod...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': '//! Tier 1 executor: embedded Python via Monty.\\n//!\\n//! Executes LLM-generated Python code using the Monty interpreter. Tool\\n//! calls use **async dispatch**: each tool call returns a Monty `ExternalFuture`\\n//! via `resume_pending()`, allowing Python code to use `await` and\\n//! `asyncio.gather()` for parallel execution. When all tasks are blocked,\\n//! Monty yields `ResolveFutures` and we execute pending tools concurrently\\n//! via `JoinSet`.\\n//!\\n//! Follows the RLM (Recursive L...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': '//! Execution trace analysis.\\n//!\\n//! Builds an in-memory `ExecutionTrace` from a completed `Thread` and runs a\\n//! retrospective analyzer that flags common failure patterns. Used by the\\n//! self-improvement mission and surfaced in debug logs.\\n//!\\n//! **There is no separate engine trace file.** Live trace recording for the\\n//! whole system is handled by `RecordingLlm` in the host crate\\n//! (`src/llm/recording.rs`), gated by `IRONCLAW_RECORD_TRACE`. Because the\\n//! engine\\'s `L...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': '//! IronClaw Engine — unified thread-capability-CodeAct execution model.\\n//!\\n//! This crate provides the core execution engine for IronClaw, unifying\\n//! ~10 separate abstractions (Session, Job, Routine, Channel, Tool, Skill,\\n//! Hook, Observer, Extension, LoopDelegate) around 5 primitives:\\n//!\\n//! - **Thread** — unit of work (replaces Session + Job + Routine + Sub-agent)\\n//! - **Step** — unit of execution (replaces agentic loop iteration + tool calls)\\n//! - **Capability** — un...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': '//! Event sourcing types.\\n//!\\n//! Every significant action within a thread is recorded as an event.\\n//! This enables replay, debugging, reflection, and trace-based testing.\\n\\nuse chrono::{DateTime, Utc};\\nuse serde::{Deserialize, Serialize};\\nuse uuid::Uuid;\\n\\nuse crate::types::capability::LeaseId;\\n\\n/// Generate a short human-readable summary of tool parameters for display.\\n///\\n/// For `http`: shows the URL. For `web_search`: shows the query.\\n/// For other tools: shows the fi...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': '//! Step — the unit of execution within a thread.\\n//!\\n//! Each step corresponds to one LLM call plus its subsequent action\\n//! executions. This replaces the implicit \"iteration\" counter in the\\n//! existing `run_agentic_loop`.\\n\\nuse std::time::Duration;\\n\\nuse chrono::{DateTime, Utc};\\nuse serde::{Deserialize, Serialize};\\nuse uuid::Uuid;\\n\\nuse crate::types::thread::ThreadId;\\n\\n/// Strongly-typed step identifier.\\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deser...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)" + } + }, + { + "request_hint": { + "last_user_message_contains": "[stdout]\n\nError: Traceback (most recent call last):\n File \"step.py\", line 176, ", + "min_message_count": 4 + }, + "response": { + "type": "text", + "content": "```repl\nimport re\n\nasync def review():\n owner = \"nearai\"\n repo = \"ironclaw\"\n number = 2483\n\n pr_url = f\"https://api.github.com/repos/{owner}/{repo}/pulls/{number}\"\n files_url = f\"{pr_url}/files?per_page=100\"\n\n meta_r = await http(method=\"GET\", url=pr_url)\n files_r = await http(method=\"GET\", url=files_url)\n\n for r, label in [(meta_r, \"metadata\"), (files_r, \"files\")]:\n if r[\"status\"] != 200:\n return FINAL(f\"GitHub {label} fetch for {owner}/{repo}#{number} returned HTTP {r['status']}: {r['body']}\")\n\n pr = meta_r[\"body\"]\n files = files_r[\"body\"]\n head_sha = pr[\"head\"][\"sha\"]\n\n file_contents = {}\n for f in files:\n path = f[\"filename\"]\n raw_r = await http(\n method=\"GET\",\n url=f\"https://api.github.com/repos/{owner}/{repo}/contents/{path}?ref={head_sha}\",\n headers=[{\"name\": \"Accept\", \"value\": \"application/vnd.github.raw\"}],\n )\n if raw_r[\"status\"] == 200:\n file_contents[path] = raw_r[\"body\"]\n\n changed_right_lines = {}\n for f in files:\n path = f[\"filename\"]\n patch = f.get(\"patch\")\n if not patch:\n continue\n right = 0\n for ln in patch.split(\"\\n\"):\n if ln.startswith(\"@@\"):\n m = re.search(r\"\\+(\\d+)(?:,(\\d+))?\", ln)\n if m:\n right = int(m.group(1))\n continue\n if ln.startswith(\"+\") and not ln.startswith(\"+++\"):\n changed_right_lines.setdefault(path, set()).add(right)\n right += 1\n elif ln.startswith(\"-\") and not ln.startswith(\"---\"):\n pass\n else:\n right += 1\n\n findings = []\n\n # Targeted checks on touched files\n for f in files:\n path = f[\"filename\"]\n patch = f.get(\"patch\") or \"\"\n full = file_contents.get(path, \"\")\n right_lines = sorted(changed_right_lines.get(path, []))\n\n # Check 1: timing instrumentation introduced without explicit cap in same hunk/path\n if \"Instant::now()\" in patch and (\"elapsed\" in patch or \"duration\" in patch):\n if \"warn!\" not in patch and \"debug!\" not in patch and \"info!\" not in patch:\n line = right_lines[0] if right_lines else 1\n findings.append({\n \"severity\": \"Low\",\n \"category\": \"Observability/Performance\",\n \"fileline\": f\"{path}:{line}\",\n \"finding\": \"Timing measurement added but no obvious log/metric emission in the same change scope; may be dead instrumentation.\",\n \"fix\": \"Either emit a bounded metric/log at completion or remove the timer to avoid misleading unused instrumentation.\"\n })\n\n # Check 2: changed async execution paths should avoid panic\n if re.search(r\"\\bhandle_execute_code_step\\b\", patch) and (\"unwrap(\" in patch or \"expect(\" in patch):\n line = right_lines[0] if right_lines else 1\n findings.append({\n \"severity\": \"High\",\n \"category\": \"Correctness\",\n \"fileline\": f\"{path}:{line}\",\n \"finding\": \"Panic-prone unwrap/expect in execute-code orchestration path can crash step execution on malformed/partial state.\",\n \"fix\": \"Replace unwrap/expect with fallible propagation (`?`) and map to structured step failure events.\"\n })\n\n # Check 3: event/type evolution risks — enum/data model touched with no compatibility comments\n if path.endswith(\"types/event.rs\") or path.endswith(\"types/step.rs\"):\n if re.search(r\"^\\+.*(enum|struct|#[derive)\", patch, re.M):\n if \"serde(default\" not in full and \"#[serde(default)]\" not in patch and \"#[serde(other)]\" not in patch:\n line = right_lines[0] if right_lines else 1\n findings.append({\n \"severity\": \"Medium\",\n \"category\": \"Backward Compatibility\",\n \"fileline\": f\"{path}:{line}\",\n \"finding\": \"Serialized event/step schema changed without visible compatibility guard (`serde(default)`/migration handling).\",\n \"fix\": \"Add serde defaults or migration path to preserve replay/loading of older persisted events.\"\n })\n\n # Check 4: executor async docs mention gather/concurrency; ensure cancellation/error path considered\n if path.endswith(\"executor/tier1_executor.rs\"):\n if \"asyncio.gather()\" in full and \"ResolveFutures\" in full:\n if \"cancel\" not in full.lower() and \"timeout\" not in full.lower():\n line = right_lines[0] if right_lines else 1\n findings.append({\n \"severity\": \"Medium\",\n \"category\": \"Concurrency/Failure Handling\",\n \"fileline\": f\"{path}:{line}\",\n \"finding\": \"Concurrent future resolution path lacks explicit cancellation/timeout semantics in nearby implementation/docs.\",\n \"fix\": \"Document and enforce timeout/cancellation behavior for stalled tool futures to prevent indefinite blocking.\"\n })\n\n # De-duplicate near-identical findings\n uniq = []\n seen = set()\n for f in findings:\n key = (f[\"severity\"], f[\"category\"], f[\"fileline\"], f[\"finding\"])\n if key not in seen:\n seen.add(key)\n uniq.append(f)\n findings = uniq\n\n changed_paths = [f[\"filename\"] for f in files]\n source_like = [p for p in changed_paths if re.search(r\"\\.(rs|ts|tsx|js|py|go)$\", p)]\n test_like = [p for p in changed_paths if re.search(r\"(test|spec|__tests__|tests?/)\", p)]\n\n if source_like and not test_like:\n p0 = source_like[0]\n lset = changed_right_lines.get(p0)\n l0 = sorted(lset)[0] if lset else 1\n findings.append({\n \"severity\": \"Medium\",\n \"category\": \"Test Coverage\",\n \"fileline\": f\"{p0}:{l0}\",\n \"finding\": \"Production code changed without accompanying test changes in this PR.\",\n \"fix\": \"Add focused tests for modified execution/orchestration and event schema behavior, including failure paths.\"\n })\n\n title = pr[\"title\"]\n\n if not findings:\n ref_path = changed_paths[0] if changed_paths else \"(no files)\"\n lset = changed_right_lines.get(ref_path)\n ref_line = sorted(lset)[0] if lset else 1\n answer = f\"\"\"Review of {owner}/{repo}#{number}: {title}\n\nI reviewed the changed files and their full current contents at head `{head_sha}`. \nNo confirmed Critical/High/Medium defects found.\n\n| # | Severity | Category | File:Line | Finding | Suggested fix |\n|---|----------|----------|-----------|---------|---------------|\n| 1 | Nit | Review result | {ref_path}:{ref_line} | No actionable defects identified in this pass. | Optional: add or reference regression tests for changed orchestration paths. |\n\nIf you want, I can post this as a PR comment, or do a stricter second pass focused only on persistence/replay compatibility.\"\"\"\n return FINAL(answer)\n\n rows = []\n for i, f in enumerate(findings, 1):\n rows.append(f\"| {i} | {f['severity']} | {f['category']} | {f['fileline']} | {f['finding']} | {f['fix']} |\")\n\n table = \"\\n\".join(rows)\n answer = f\"\"\"Review of {owner}/{repo}#{number}: {title}\n\n| # | Severity | Category | File:Line | Finding | Suggested fix |\n|---|----------|----------|-----------|---------|---------------|\n{table}\n\nWhich findings should I post as PR comments? Default is all Critical, High, and Medium.\"\"\"\n return FINAL(answer)\n\nawait review()\n```", + "input_tokens": 27892, + "output_tokens": 1754 + } + }, + { + "response": { + "type": "user_input", + "content": "[stdout]\n\nError: Traceback (most recent call last):\n File \"step.py\", line 170, in \n await review()\n ~~~~~~~~~~~~~~\n File \"step.py\", line 88, in review\nre.PatternError: Parsing error at position 31: Invalid character class\n\n[http] {'body': {'_links': {'comments': {'href': 'https://api.github.com/repos/nearai/ironclaw/issues/2483/comments'}, 'commits': {'href': 'https://api.github.com/repos/nearai/ironclaw/pulls/2483/commits'}, 'html': {'href': 'https://github.com/nearai/ironclaw/pull/2483'}, 'issue': {'href': 'https://api.github.com/repos/nearai/ironclaw/issues/2483'}, 'review_comment': {'href': 'https://api.github.com/repos/nearai/ironclaw/pulls/comments{/number}'}, 'review_comments': {'href': 'https://api.github.com/rep...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': [{'additions': 118, 'blob_url': 'https://github.com/nearai/ironclaw/blob/5496341c49b2abaa9842e2f939ea349c75fd7340/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Forchestrator.rs', 'changes': 122, 'contents_url': 'https://api.github.com/repos/nearai/ironclaw/contents/crates%2Fironclaw_engine%2Fsrc%2Fexecutor%2Forchestrator.rs?ref=5496341c49b2abaa9842e2f939ea349c75fd7340', 'deletions': 4, 'filename': 'crates/ironclaw_engine/src/executor/orchestrator.rs', 'patch': '@@ -720,6 +720,7 @@ async fn...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': '//! Python orchestrator — the self-modifiable execution loop.\\n//!\\n//! Replaces the Rust `ExecutionLoop::run()` with versioned Python code\\n//! executed via Monty. The orchestrator is the \"glue layer\" between the\\n//! LLM and tools — tool dispatch, output formatting, state management,\\n//! truncation — all in Python, patchable by the self-improvement Mission.\\n//!\\n//! Host functions exposed to the orchestrator Python:\\n//! - `__llm_complete__` — make an LLM call\\n//! - `__execute_cod...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': '//! Tier 1 executor: embedded Python via Monty.\\n//!\\n//! Executes LLM-generated Python code using the Monty interpreter. Tool\\n//! calls use **async dispatch**: each tool call returns a Monty `ExternalFuture`\\n//! via `resume_pending()`, allowing Python code to use `await` and\\n//! `asyncio.gather()` for parallel execution. When all tasks are blocked,\\n//! Monty yields `ResolveFutures` and we execute pending tools concurrently\\n//! via `JoinSet`.\\n//!\\n//! Follows the RLM (Recursive L...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': '//! Execution trace analysis.\\n//!\\n//! Builds an in-memory `ExecutionTrace` from a completed `Thread` and runs a\\n//! retrospective analyzer that flags common failure patterns. Used by the\\n//! self-improvement mission and surfaced in debug logs.\\n//!\\n//! **There is no separate engine trace file.** Live trace recording for the\\n//! whole system is handled by `RecordingLlm` in the host crate\\n//! (`src/llm/recording.rs`), gated by `IRONCLAW_RECORD_TRACE`. Because the\\n//! engine\\'s `L...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': '//! IronClaw Engine — unified thread-capability-CodeAct execution model.\\n//!\\n//! This crate provides the core execution engine for IronClaw, unifying\\n//! ~10 separate abstractions (Session, Job, Routine, Channel, Tool, Skill,\\n//! Hook, Observer, Extension, LoopDelegate) around 5 primitives:\\n//!\\n//! - **Thread** — unit of work (replaces Session + Job + Routine + Sub-agent)\\n//! - **Step** — unit of execution (replaces agentic loop iteration + tool calls)\\n//! - **Capability** — un...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': '//! Event sourcing types.\\n//!\\n//! Every significant action within a thread is recorded as an event.\\n//! This enables replay, debugging, reflection, and trace-based testing.\\n\\nuse chrono::{DateTime, Utc};\\nuse serde::{Deserialize, Serialize};\\nuse uuid::Uuid;\\n\\nuse crate::types::capability::LeaseId;\\n\\n/// Generate a short human-readable summary of tool parameters for display.\\n///\\n/// For `http`: shows the URL. For `web_search`: shows the query.\\n/// For other tools: shows the fi...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)\n\n[http] {'body': '//! Step — the unit of execution within a thread.\\n//!\\n//! Each step corresponds to one LLM call plus its subsequent action\\n//! executions. This replaces the implicit \"iteration\" counter in the\\n//! existing `run_agentic_loop`.\\n\\nuse std::time::Duration;\\n\\nuse chrono::{DateTime, Utc};\\nuse serde::{Deserialize, Serialize};\\nuse uuid::Uuid;\\n\\nuse crate::types::thread::ThreadId;\\n\\n/// Strongly-typed step identifier.\\n#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deser...\n(full result stored in state['http']; do NOT retype the data — reference the variable in your next call.)" + } + }, + { + "request_hint": { + "last_user_message_contains": "[stdout]\n\nError: Traceback (most recent call last):\n File \"step.py\", line 170, ", + "min_message_count": 6 + }, + "response": { + "type": "text", + "content": "```repl\nimport re\n\nasync def review():\n owner = \"nearai\"\n repo = \"ironclaw\"\n number = 2483\n\n pr_url = f\"https://api.github.com/repos/{owner}/{repo}/pulls/{number}\"\n files_url = f\"{pr_url}/files?per_page=100\"\n\n meta_r = await http(method=\"GET\", url=pr_url)\n files_r = await http(method=\"GET\", url=files_url)\n\n if meta_r[\"status\"] != 200:\n return FINAL(f\"GitHub metadata fetch failed for {owner}/{repo}#{number}: HTTP {meta_r['status']} {meta_r['body']}\")\n if files_r[\"status\"] != 200:\n return FINAL(f\"GitHub files fetch failed for {owner}/{repo}#{number}: HTTP {files_r['status']} {files_r['body']}\")\n\n pr = meta_r[\"body\"]\n files = files_r[\"body\"]\n head_sha = pr[\"head\"][\"sha\"]\n\n file_contents = {}\n for f in files:\n path = f[\"filename\"]\n raw_r = await http(\n method=\"GET\",\n url=f\"https://api.github.com/repos/{owner}/{repo}/contents/{path}?ref={head_sha}\",\n headers=[{\"name\": \"Accept\", \"value\": \"application/vnd.github.raw\"}],\n )\n if raw_r[\"status\"] == 200:\n file_contents[path] = raw_r[\"body\"]\n\n changed_right_lines = {}\n for f in files:\n path = f[\"filename\"]\n patch = f.get(\"patch\")\n if not patch:\n continue\n right = 0\n for ln in patch.split(\"\\n\"):\n if ln.startswith(\"@@\"):\n m = re.search(r\"\\+(\\d+)(?:,(\\d+))?\", ln)\n if m:\n right = int(m.group(1))\n continue\n if ln.startswith(\"+\") and not ln.startswith(\"+++\"):\n s = changed_right_lines.get(path)\n if s is None:\n s = set()\n changed_right_lines[path] = s\n s.add(right)\n right += 1\n elif ln.startswith(\"-\") and not ln.startswith(\"---\"):\n pass\n else:\n right += 1\n\n findings = []\n\n for f in files:\n path = f[\"filename\"]\n patch = f.get(\"patch\") or \"\"\n full = file_contents.get(path, \"\")\n right_lines = sorted(changed_right_lines.get(path, set()))\n anchor = right_lines[0] if right_lines else 1\n\n # Manual/substring checks (avoid complex regex incompatibilities in Monty)\n if (\"Instant::now()\" in patch) and ((\"elapsed\" in patch) or (\"duration\" in patch)):\n if (\"warn!\" not in patch) and (\"debug!\" not in patch) and (\"info!\" not in patch) and (\"metric\" not in patch):\n findings.append({\n \"severity\": \"Low\",\n \"category\": \"Observability/Performance\",\n \"fileline\": f\"{path}:{anchor}\",\n \"finding\": \"Timer start/elapsed logic appears introduced without clear emission or use in the same change scope.\",\n \"fix\": \"Either wire the timing into structured metrics/logging, or remove dead timing code to avoid confusion.\"\n })\n\n if (\"handle_execute_code_step\" in patch) and ((\"unwrap(\" in patch) or (\"expect(\" in patch)):\n findings.append({\n \"severity\": \"High\",\n \"category\": \"Correctness\",\n \"fileline\": f\"{path}:{anchor}\",\n \"finding\": \"Panic-prone unwrap/expect appears in execute-code orchestration path, risking step abort on malformed state.\",\n \"fix\": \"Replace panic paths with fallible propagation (`?`) and convert to explicit step failure events.\"\n })\n\n if path.endswith(\"types/event.rs\") or path.endswith(\"types/step.rs\"):\n schema_touched = (\"+pub enum\" in patch) or (\"+pub struct\" in patch) or (\"+#[derive(\" in patch)\n if schema_touched and (\"serde(default\" not in full) and (\"serde(default\" not in patch):\n findings.append({\n \"severity\": \"Medium\",\n \"category\": \"Backward Compatibility\",\n \"fileline\": f\"{path}:{anchor}\",\n \"finding\": \"Event/step serialized schema appears updated without explicit compatibility defaults/migration hints.\",\n \"fix\": \"Add serde defaults (or migration handling) so old persisted traces/events continue to deserialize safely.\"\n })\n\n if path.endswith(\"executor/tier1_executor.rs\"):\n if (\"ResolveFutures\" in full) and (\"ExternalFuture\" in full):\n if (\"timeout\" not in full.lower()) and (\"cancel\" not in full.lower()):\n findings.append({\n \"severity\": \"Medium\",\n \"category\": \"Concurrency/Failure Handling\",\n \"fileline\": f\"{path}:{anchor}\",\n \"finding\": \"Concurrent future-resolution path lacks explicit timeout/cancellation semantics in nearby implementation/docs.\",\n \"fix\": \"Define timeout and cancellation behavior for blocked tool futures and add tests for stalled futures.\"\n })\n\n # test coverage finding\n changed_paths = [f[\"filename\"] for f in files]\n source_like = []\n test_like = []\n for p in changed_paths:\n if p.endswith(\".rs\") or p.endswith(\".ts\") or p.endswith(\".tsx\") or p.endswith(\".js\") or p.endswith(\".py\") or p.endswith(\".go\"):\n source_like.append(p)\n low = p.lower()\n if (\"test\" in low) or (\"spec\" in low) or (\"/tests/\" in low) or (\"__tests__\" in low):\n test_like.append(p)\n\n if source_like and not test_like:\n p0 = source_like[0]\n lset = changed_right_lines.get(p0, set())\n l0 = sorted(lset)[0] if lset else 1\n findings.append({\n \"severity\": \"Medium\",\n \"category\": \"Test Coverage\",\n \"fileline\": f\"{p0}:{l0}\",\n \"finding\": \"Production code changed without corresponding test updates in this PR.\",\n \"fix\": \"Add focused unit/integration tests for modified execution timing/trace/event behavior and failure paths.\"\n })\n\n # dedupe\n out = []\n seen = set()\n for x in findings:\n k = (x[\"severity\"], x[\"category\"], x[\"fileline\"], x[\"finding\"])\n if k not in seen:\n seen.add(k)\n out.append(x)\n findings = out\n\n title = pr[\"title\"]\n\n if not findings:\n ref_path = changed_paths[0] if changed_paths else \"(no files)\"\n lset = changed_right_lines.get(ref_path, set())\n ref_line = sorted(lset)[0] if lset else 1\n ans = f\"\"\"Review of {owner}/{repo}#{number}: {title}\n\nI reviewed the PR and read full current contents of all changed files retrievable at `{head_sha}`.\nNo confirmed Critical/High/Medium defects found.\n\n| # | Severity | Category | File:Line | Finding | Suggested fix |\n|---|----------|----------|-----------|---------|---------------|\n| 1 | Nit | Review result | {ref_path}:{ref_line} | No actionable defects identified in this pass. | Optional: add/verify regression tests for the modified orchestration path. |\n\nIf you want, I can do a stricter second pass focused exclusively on replay/backward-compatibility guarantees.\"\"\"\n return FINAL(ans)\n\n rows = []\n i = 1\n for f in findings:\n rows.append(f\"| {i} | {f['severity']} | {f['category']} | {f['fileline']} | {f['finding']} | {f['fix']} |\")\n i += 1\n\n table = \"\\n\".join(rows)\n ans = f\"\"\"Review of {owner}/{repo}#{number}: {title}\n\n| # | Severity | Category | File:Line | Finding | Suggested fix |\n|---|----------|----------|-----------|---------|---------------|\n{table}\n\nWhich findings should I post as PR comments? Default is all Critical, High, and Medium.\"\"\"\n return FINAL(ans)\n\nawait review()\n```", + "input_tokens": 31041, + "output_tokens": 1748 + } + } + ] +} \ No newline at end of file diff --git a/tests/fixtures/llm_traces/live/code_review_pr_2483.log b/tests/fixtures/llm_traces/live/code_review_pr_2483.log new file mode 100644 index 00000000000..7dbb69329e0 --- /dev/null +++ b/tests/fixtures/llm_traces/live/code_review_pr_2483.log @@ -0,0 +1,98 @@ +# Live Test Session: code_review_pr_2483 +# Mode: Live +# LLM calls: 3, Input tokens: 83357, Output tokens: 5354 +# Wall time: 126.1s, Cost: $0.0000 +# ────────────────────────────────────────────────── + +## Turn 1 +[USER] › /code-review nearai/ironclaw 2483 +──────────────────────────────────────────────────── +Review of nearai/ironclaw#2483: feat(engine): add code execution failure categorization instrumentation + +| # | Severity | Category | File:Line | Finding | Suggested fix | +|---|----------|----------|-----------|---------|---------------| +| 1 | Low | Observability/Performance | crates/ironclaw_engine/src/executor/orchestrator.rs:723 | Timer start/elapsed logic appears introduced without clear emission or use in the same change scope. | Either wire the timing into structured metrics/logging, or remove dead timing code to avoid confusion. | +| 2 | High | Correctness | crates/ironclaw_engine/src/executor/orchestrator.rs:723 | Panic-prone unwrap/expect appears in execute-code orchestration path, risking step abort on malformed state. | Replace panic paths with fallible propagation (`?`) and convert to explicit step failure events. | +| 3 | Medium | Backward Compatibility | crates/ironclaw_engine/src/types/step.rs:135 | Event/step serialized schema appears updated without explicit compatibility defaults/migration hints. | Add serde defaults (or migration handling) so old persisted traces/events continue to deserialize safely. | +| 4 | Medium | Test Coverage | crates/ironclaw_engine/src/executor/orchestrator.rs:723 | Production code changed without corresponding test updates in this PR. | Add focused unit/integration tests for modified execution timing/trace/event behavior and failure paths. | + +Which findings should I post as PR comments? Default is all Critical, High, and Medium. + +## Activity + ○ Processing... + ◆ skills: code-review, github, coding + ○ Calling LLM... + ○ Step complete — 24424 in / 1852 out tokens + ● http(https://api.github.com/repos/nearai/ironclaw/pulls/2483) + ✓ http(https://api.github.com/repos/nearai/ironclaw/pulls/2483) + ● http(https://api.github.com/repos/nearai/ironclaw/pulls/2483) + ✓ http(https://api.github.com/repos/nearai/ironclaw/pulls/2483) + ● http(https://api.github.com/repos/nearai/ironclaw/pulls/2483/files?per_page=100) + ✓ http(https://api.github.com/repos/nearai/ironclaw/pulls/2483/files?per_page=100) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● __codeact__ + ✗ __codeact__: CodeAct execution failed: +Error: Traceback (most recent call last): + File "step.py", line 176, in + await review() + ~~~~~~~~~~~~~~ + File "step.py", line 75, in review + if re.search(r'^\+.*\b(shell|exec|spawn|Command::new|subprocess|os\.system)\b', patch, flags=re.M): + ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ +TypeError: re.search() takes no keyword arguments + ○ Calling LLM... + ○ Step complete — 27892 in / 1754 out tokens + ● http(https://api.github.com/repos/nearai/ironclaw/pulls/2483) + ✓ http(https://api.github.com/repos/nearai/ironclaw/pulls/2483) + ● http(https://api.github.com/repos/nearai/ironclaw/pulls/2483/files?per_page=100) + ✓ http(https://api.github.com/repos/nearai/ironclaw/pulls/2483/files?per_page=100) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● __codeact__ + ✗ __codeact__: CodeAct execution failed: +Error: Traceback (most recent call last): + File "step.py", line 170, in + await review() + ~~~~~~~~~~~~~~ + File "step.py", line 88, in review +re.PatternError: Parsing error at position 31: Invalid character class + ○ Calling LLM... + ○ Step complete — 31041 in / 1748 out tokens + ● http(https://api.github.com/repos/nearai/ironclaw/pulls/2483) + ✓ http(https://api.github.com/repos/nearai/ironclaw/pulls/2483) + ● http(https://api.github.com/repos/nearai/ironclaw/pulls/2483/files?per_page=100) + ✓ http(https://api.github.com/repos/nearai/ironclaw/pulls/2483/files?per_page=100) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ● http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + ✓ http(https://api.github.com/repos/nearai/ironclaw/contents/crates/ironclaw_engine/src...) + … Done diff --git a/tests/support/live_harness.rs b/tests/support/live_harness.rs index dd0fad428c9..8a83d407d9c 100644 --- a/tests/support/live_harness.rs +++ b/tests/support/live_harness.rs @@ -260,6 +260,7 @@ pub struct LiveTestHarnessBuilder { channel_name: Option, seeded_secret_names: Vec, record_trace: bool, + skills_dir: Option, } impl LiveTestHarnessBuilder { @@ -285,6 +286,7 @@ impl LiveTestHarnessBuilder { channel_name: None, seeded_secret_names: Vec::new(), record_trace: true, + skills_dir: None, } } @@ -350,6 +352,14 @@ impl LiveTestHarnessBuilder { self } + /// Set a custom skills directory so the test rig loads skill files + /// from a workspace path (e.g. `skills/` at the repo root) instead + /// of an empty temp directory. Enables skill discovery automatically. + pub fn with_skills_dir(mut self, dir: impl Into) -> Self { + self.skills_dir = Some(dir.into()); + self + } + /// Build the harness, auto-detecting mode from the `IRONCLAW_LIVE_TEST` env var. #[cfg(feature = "libsql")] pub async fn build(self) -> LiveTestHarness { @@ -539,6 +549,9 @@ impl LiveTestHarnessBuilder { self.seeded_secret_names.clone(), ); } + if let Some(dir) = self.skills_dir { + rig_builder = rig_builder.with_skills_dir(dir); + } let rig = rig_builder.build().await; // Use cheap LLM for judge if available. @@ -582,6 +595,9 @@ impl LiveTestHarnessBuilder { if let Some(ref name) = self.channel_name { rig_builder = rig_builder.with_channel_name(name.clone()); } + if let Some(dir) = self.skills_dir { + rig_builder = rig_builder.with_skills_dir(dir); + } let rig = rig_builder.build().await; LiveTestHarness { diff --git a/tests/support/test_rig.rs b/tests/support/test_rig.rs index 30a6f0abcd1..d1bac4111ed 100644 --- a/tests/support/test_rig.rs +++ b/tests/support/test_rig.rs @@ -336,6 +336,19 @@ impl TestRig { self.channel.captured_status_events() } + /// Return the names of skills that were activated during this session, + /// extracted from `SkillActivated` status events. + pub fn active_skill_names(&self) -> Vec { + self.captured_status_events() + .iter() + .filter_map(|event| match event { + StatusUpdate::SkillActivated { skill_names } => Some(skill_names.clone()), + _ => None, + }) + .flatten() + .collect() + } + /// Return the ordered log of captured outbound events. pub fn captured_events(&self) -> Vec { self.channel.captured_events() @@ -588,6 +601,7 @@ pub struct TestRigBuilder { injection_check: bool, auto_approve_tools: Option, enable_skills: bool, + skills_dir: Option, enable_routines: bool, http_exchanges: Vec, http_interceptor_override: Option>, @@ -610,6 +624,7 @@ impl TestRigBuilder { injection_check: false, auto_approve_tools: Some(true), enable_skills: false, + skills_dir: None, enable_routines: false, http_exchanges: Vec::new(), http_interceptor_override: None, @@ -742,6 +757,15 @@ impl TestRigBuilder { self } + /// Set a custom skills directory so the test rig loads skill files + /// from a real path (e.g. the repo's `skills/` directory) instead of + /// an empty temp directory. Implies `with_skills()`. + pub fn with_skills_dir(mut self, dir: std::path::PathBuf) -> Self { + self.enable_skills = true; + self.skills_dir = Some(dir); + self + } + /// Enable the routines system so the scheduler is wired with a `RoutineEngine`, /// allowing routine jobs to actually execute. Routine tools are always registered /// but require the engine to dispatch jobs. @@ -802,6 +826,7 @@ impl TestRigBuilder { injection_check, auto_approve_tools, enable_skills, + skills_dir, enable_routines, http_exchanges: explicit_http_exchanges, http_interceptor_override, @@ -846,7 +871,7 @@ impl TestRigBuilder { // 2. Build Config. let has_config_override = config_override.is_some(); - let skills_dir = temp_dir.path().join("skills"); + let skills_dir = skills_dir.unwrap_or_else(|| temp_dir.path().join("skills")); let installed_skills_dir = temp_dir.path().join("installed_skills"); let _ = std::fs::create_dir_all(&skills_dir); let _ = std::fs::create_dir_all(&installed_skills_dir); @@ -1042,12 +1067,13 @@ impl TestRigBuilder { .register_routine_tools(Arc::clone(db_arc), engine); } - // Skills tools: ensure tests use temp skill dirs (sandbox-safe) even if - // AppBuilder did not wire them for this environment. + // Skills tools: use the config-resolved skills dirs so that a + // custom `with_skills_dir()` path propagates all the way to + // the registry (instead of always pointing at an empty temp dir). if enable_skills { let registry = Arc::new(std::sync::RwLock::new( - ironclaw_skills::SkillRegistry::new(temp_dir.path().join("skills")) - .with_installed_dir(temp_dir.path().join("installed_skills")), + ironclaw_skills::SkillRegistry::new(components.config.skills.local_dir.clone()) + .with_installed_dir(components.config.skills.installed_dir.clone()), )); let catalog = ironclaw_skills::catalog::shared_catalog(); components diff --git a/tools-src/github/src/lib.rs b/tools-src/github/src/lib.rs index 600f9fc8096..63a2f73f835 100644 --- a/tools-src/github/src/lib.rs +++ b/tools-src/github/src/lib.rs @@ -474,9 +474,16 @@ impl exports::near::agent::tool::Guest for GitHubTool { } fn description() -> String { - "GitHub integration for repositories, issues, pull requests, search, \ - branches, file reads and writes, releases, and workflows. \ - Authentication is handled via the 'github_token' secret injected by the host." + "GitHub integration: repositories, issues, pull requests, branches, files, \ + releases, and workflows. Search has exactly three actions: \ + `search_repositories`, `search_code`, and `search_issues_pull_requests` \ + (the last one covers BOTH issues and PRs; there is no separate \ + `search_issues` action). For \"my PRs\" or \"my issues\" across all repos, \ + use `search_issues_pull_requests` with a query like \ + `is:pr author:@me sort:updated-desc` (the `@me` placeholder resolves to \ + the authenticated user). `list_pull_requests` and `list_issues` require \ + `owner` + `repo` and only return results from a single repo. \ + Authentication is handled via the `github_token` secret injected by the host." .to_string() } } @@ -2151,7 +2158,7 @@ const SCHEMA: &str = r#"{ { "properties": { "action": { "const": "search_issues_pull_requests" }, - "query": { "type": "string", "description": "GitHub issue/PR search query" }, + "query": { "type": "string", "description": "GitHub search query covering both issues and PRs. Filter with is:pr or is:issue. Supports @me, repo:, org:, label:, etc." }, "page": { "type": "integer" }, "limit": { "type": "integer", "default": 30 }, "sort": { "type": "string" },