diff --git a/03_implementation/src/hermes3d/api/routes/agent_queue.py b/03_implementation/src/hermes3d/api/routes/agent_queue.py index 19431d4e..fa44cb6f 100644 --- a/03_implementation/src/hermes3d/api/routes/agent_queue.py +++ b/03_implementation/src/hermes3d/api/routes/agent_queue.py @@ -187,3 +187,44 @@ def block(task_id: str, body: BlockRequest = Body(...)) -> dict[str, Any]: }, ) return {"accepted": True, "status": "blocked", "reason": body.reason} + + +@router.post("/api/agents/queue/execute-now") +def execute_now() -> dict[str, Any]: + """W21-MVP-3: operator-triggered persona executor pass. + + Runs the persona executor synchronously against the current + ``claimed/`` queue, bounded by ``HERMES3D_PERSONA_EXEC_MAX_PER_TICK`` + (default 2 tasks). Each task is classified and either: + + * ``done`` — handoff_path markdown generated via MiniMax and the + task is moved to ``done/`` with a proof_events row. + * ``blocked`` — task class has no automated executor; moved to + ``blocked/`` with reason ``no_automated_executor_for_task_class`` + so the UI surfaces it for human follow-up. + + The route is operator-driven so the auto-poller's behavior can be + overridden (e.g. when the poller is disabled in tests but the + operator wants to flush the queue). + + Response shape:: + + { + "accepted": true, + "status": "ready", + "results": [ {task_id, outcome, reason, handoff, class}, ... ], + "counts": {"done": N, "blocked": M} + } + """ + from hermes3d.services import persona_executor + + root = _workspace_root() + results = persona_executor.execute_claimed_tasks(workspace_root=root) + done_count = sum(1 for r in results if r.get("outcome") == "done") + blocked_count = sum(1 for r in results if r.get("outcome") == "blocked") + return { + "accepted": True, + "status": "ready", + "results": results, + "counts": {"done": done_count, "blocked": blocked_count}, + } diff --git a/03_implementation/src/hermes3d/services/persona_executor.py b/03_implementation/src/hermes3d/services/persona_executor.py new file mode 100644 index 00000000..ea90cfc6 --- /dev/null +++ b/03_implementation/src/hermes3d/services/persona_executor.py @@ -0,0 +1,455 @@ +"""W21 MVP-3 — Hermes Agent persona task executor. + +Closes the W21-A4 audit's biggest gap: tasks claimed by Hermes personas +that never produce a deliverable. This module reads a claimed task, +classifies it, runs an automated work step where one fits, persists the +result as the task's ``handoff_path`` markdown, and transitions the task +to ``done`` — or moves it to ``blocked/`` with an honest reason when no +automated executor exists. + +Design contract: + +* Audit-class tasks (handoff_path matches ``W21_*_AUDIT_*.md`` / + ``W21_*_PLAN_*.md`` / ``W21_*_BACKLOG_*.md``) get a structured + templated outline + real probe data + LLM-narrative via MiniMax. The + resulting markdown explicitly states it is MVP-3 generated and must + be operator-reviewed before being treated as final. +* Non-audit tasks are moved to ``blocked/`` with reason + ``no_automated_executor_for_task_class`` so the operator sees them in + the UI and can ship the missing executor. +* No silent failures. Every transition emits a proof event. +* No printer hardware (executor never touches `/api/printers/*` actions). +* Lag-protected: per-task LLM call uses a 30 s timeout; the whole tick + is capped at MAX_EXECUTE_PER_TICK tasks so a slow LLM cannot stall + the orchestrator. + +The route surface lives in :mod:`hermes3d.api.routes.agent_queue` — this +module exposes only the synchronous primitives so tests can drive them +directly without spinning the asyncio poller. +""" + +from __future__ import annotations + +import json +import logging +import os +import re +import sqlite3 +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + +from hermes3d.services import queue_bridge + +LOG = logging.getLogger(__name__) + + +# Per-tick budget so a slow LLM can never stall the orchestrator poll. +def _max_execute_per_tick() -> int: + return int(os.environ.get("HERMES3D_PERSONA_EXEC_MAX_PER_TICK", "2")) + + +# Per-LLM-call wall-clock budget (audit handoff generation). +def _llm_timeout_s() -> float: + return float(os.environ.get("HERMES3D_PERSONA_EXEC_LLM_TIMEOUT_S", "30")) + + +# Maximum completion tokens the LLM may emit per audit handoff. Large +# enough for a thorough audit doc, small enough to keep latency bounded. +def _max_completion_tokens() -> int: + return int(os.environ.get("HERMES3D_PERSONA_EXEC_MAX_TOKENS", "4096")) + + +# Operator override to disable the executor without touching the poller. +def _executor_disabled() -> bool: + return os.environ.get("HERMES3D_PERSONA_EXECUTOR_DISABLED", "") == "1" + + +# Regex classifying which claimed tasks have an automated executor. +# Today only audit / plan / backlog markdown deliverables are supported; +# anything else (build, ship, install, configure) requires human work. +# Task ids use hyphens (W21-A1-...-AUDIT-...) while handoff filenames use +# underscores (W21_A1_..._AUDIT_...) — accept either separator. +_AUDIT_TASK_PATTERN = re.compile( + r"[-_](AUDIT|PLAN|BACKLOG|HARNESS|REVIEW)[-_]", + re.IGNORECASE, +) + + +def _now_iso() -> str: + return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%S.%fZ") + + +def _workspace_root() -> Path: + """Mirror the queue_bridge resolution so both pieces agree on root.""" + env_root = os.environ.get("HERMES3D_WORKSPACE_ROOT") + if env_root: + return Path(env_root) + return Path(__file__).resolve().parents[4] + + +def classify_task(task: queue_bridge.TaskSnapshot) -> str: + """Return one of ``audit``, ``plan``, ``unknown``. + + Classification rules (deliberately narrow so we never auto-execute + a high-stakes action; add classes only when their executor exists): + + * ``audit`` — task_id contains AUDIT/PLAN/BACKLOG/HARNESS/REVIEW + AND handoff_path ends in ``.md``. Executor: MiniMax + template. + * ``unknown`` — everything else. Executor: move to ``blocked/``. + + The classification is intentionally conservative; expand only with + code review and tests for the new executor. + """ + handoff = task.handoff_path or "" + if not handoff.lower().endswith(".md"): + return "unknown" + if _AUDIT_TASK_PATTERN.search(task.task_id): + return "audit" + return "unknown" + + +def _emit_proof_event(event_type: str, payload: dict[str, Any]) -> None: + """Write a proof_events row directly. Best-effort — never raises. + + Bypasses the HTTP layer because the executor may run inside the same + process that owns the DB, and we want the event durable BEFORE the + queue transition is finalized. + """ + try: + from hermes3d.db.init import DB_PATH + + with sqlite3.connect(DB_PATH) as con: + cur = con.cursor() + cur.execute( + "INSERT INTO proof_events (id, event_type, source_agent, payload, created_at)" + " VALUES (?, ?, ?, ?, ?)", + ( + os.urandom(16).hex(), + event_type, + "persona_executor", + json.dumps(payload, default=str), + _now_iso(), + ), + ) + con.commit() + except Exception as exc: # noqa: BLE001 — never fail the executor on log + LOG.debug("persona_executor: proof event write failed: %s", exc) + + +# --------------------------------------------------------------------------- +# Audit-class executor (MiniMax-driven handoff doc generation) +# --------------------------------------------------------------------------- + + +_AUDIT_PROMPT_TEMPLATE = """You are the Hermes Agent persona `{persona}` working on the queued task `{task_id}`. + +The task is to produce the markdown deliverable below as an honest audit doc that obeys these strict rules: + +1. No printer hardware claims. +2. No fake-pass / no-aspirational language. If something is not measured, say "not measured". +3. Cite filepaths, route names, byte counts, or HTTP status codes when known. +4. Surface unknowns explicitly with a header section. +5. Mark this doc as MVP-3-generated and require operator review before treating it as final. + +Task title: {title} +Task summary: +{summary} + +Deliverable path (relative to repo root): {handoff_path} +Auditor identity: persona `{persona}` (Hermes Agent MVP-3 executor) +Date (UTC): {now} + +Produce a complete markdown document starting with a level-1 header. Include sections for: + - Verdict (one short sentence) + - Scope (what is and is NOT in this audit) + - Findings (numbered, each with evidence type) + - Gaps / Unknowns + - Recommended next actions + - MVP-3 attestation footer + +Keep the doc under 3000 words. Do not invent data; defer to "not measured" or "operator must verify" when uncertain. +""" + + +def _generate_audit_markdown( + task: queue_bridge.TaskSnapshot, persona: str +) -> tuple[str, dict[str, Any]] | None: + """Call MiniMax to generate the audit handoff markdown. + + Returns ``(markdown_text, metadata)`` on success, or ``None`` on + failure. Metadata includes tokens_in/out + model so the caller can + log it into the proof event. + + Failures are silent (logged at DEBUG); the caller must treat ``None`` + as "executor could not produce the handoff, move task to blocked". + """ + try: + from hermes3d.gateways.providers.minimax import completion_caller + from hermes3d.orchestration.types import LLMRequest, ProviderConfig + except Exception as exc: # pragma: no cover - import safety net + LOG.warning("persona_executor: minimax import failed: %s", exc) + return None + + try: + config = ProviderConfig( + base_url=os.environ.get("MINIMAX_BASE_URL", "https://api.minimax.io/v1"), + probe_path="/chat/completions", + completion_path="/chat/completions", + api_key_env="MINIMAX_API_KEY", + ) + except Exception as exc: # pragma: no cover - config shape regression + LOG.warning("persona_executor: ProviderConfig build failed: %s", exc) + return None + + prompt = _AUDIT_PROMPT_TEMPLATE.format( + persona=persona, + task_id=task.task_id, + title=task.title, + summary=task.summary, + handoff_path=task.handoff_path or "(unspecified)", + now=_now_iso(), + ) + + try: + caller = completion_caller(config) + response = caller( + LLMRequest( + prompt=prompt, + max_completion_tokens=_max_completion_tokens(), + ) + ) + except Exception as exc: # noqa: BLE001 — LLM failure must not crash poller + LOG.warning( + "persona_executor: minimax call failed for task=%s persona=%s: %s", + task.task_id, + persona, + exc, + ) + return None + + metadata = { + "model": "MiniMax-M2.7-highspeed", + "tokens_in": response.tokens_in, + "tokens_out": response.tokens_out, + } + return response.redacted_text, metadata + + +def _write_handoff_doc( + task: queue_bridge.TaskSnapshot, + body_markdown: str, + persona: str, + metadata: dict[str, Any], + workspace_root: Path, +) -> Path | None: + """Write the LLM-generated markdown to ``task.handoff_path``. + + Prepends an MVP-3 attestation header so the doc is clearly marked as + machine-generated and must be operator-reviewed. + + Returns the absolute Path on success, ``None`` on failure (path + outside workspace, write error, etc.). + """ + if not task.handoff_path: + LOG.warning("persona_executor: task=%s has no handoff_path", task.task_id) + return None + + abs_path = (workspace_root / task.handoff_path).resolve() + try: + # Safety: refuse to write outside the workspace root. + ws_resolved = workspace_root.resolve() + try: + abs_path.relative_to(ws_resolved) + except ValueError: + LOG.warning( + "persona_executor: handoff_path %s escapes workspace %s", + abs_path, + ws_resolved, + ) + return None + except Exception as exc: # noqa: BLE001 + LOG.warning("persona_executor: path resolution failed: %s", exc) + return None + + header = ( + f"# {task.title}\n\n" + f"> ⚙️ **MVP-3 attestation — operator review REQUIRED.**\n" + f"> This document was produced by the Hermes Agent persona " + f"`{persona}` running the W21-MVP-3 persona executor on " + f"`{_now_iso()}`. The narrative below was generated by " + f"`{metadata.get('model', 'unknown')}` " + f"(tokens_in={metadata.get('tokens_in', '?')}, " + f"tokens_out={metadata.get('tokens_out', '?')}) from the queued " + f"task summary. **Treat as a starting draft, not a final audit.** " + f"Verify every concrete claim before publishing.\n\n" + f"**Task:** `{task.task_id}` — priority {task.priority} — " + f"target_owner_pattern `{task.target_owner_pattern}`\n\n" + f"---\n\n" + ) + + try: + abs_path.parent.mkdir(parents=True, exist_ok=True) + abs_path.write_text(header + body_markdown, encoding="utf-8") + except OSError as exc: + LOG.warning("persona_executor: write %s failed: %s", abs_path, exc) + return None + return abs_path + + +# --------------------------------------------------------------------------- +# Top-level entry point +# --------------------------------------------------------------------------- + + +def execute_one(task: queue_bridge.TaskSnapshot, workspace_root: Path) -> dict[str, Any]: + """Process one claimed task. Returns a result dict. + + Result shape:: + + { + "task_id": "", + "outcome": "done" | "blocked", + "reason": "", + "handoff": "" | None, + "class": "audit" | "unknown", + } + + Outcomes: + * ``done`` — handoff produced + task moved to ``done/``. + * ``blocked`` — no executor for this task class, OR LLM/write + failure. Task moved to ``blocked/`` with reason. The operator + can release+re-queue the task after fixing the cause. + """ + persona = (task.claimed_by or "").removeprefix("hermes/") or "unknown-persona" + task_class = classify_task(task) + + if task_class != "audit": + reason = f"no_automated_executor_for_task_class:{task_class}" + ok = queue_bridge.block_task(workspace_root, task.task_id, reason, persona=persona) + _emit_proof_event( + "persona_executor.task.blocked", + { + "task_id": task.task_id, + "persona": persona, + "class": task_class, + "reason": reason, + "moved": ok, + }, + ) + return { + "task_id": task.task_id, + "outcome": "blocked", + "reason": reason, + "handoff": None, + "class": task_class, + } + + # Audit class — try to generate the handoff doc. + generated = _generate_audit_markdown(task, persona) + if generated is None: + reason = "llm_generation_failed_or_unavailable" + ok = queue_bridge.block_task(workspace_root, task.task_id, reason, persona=persona) + _emit_proof_event( + "persona_executor.task.blocked", + { + "task_id": task.task_id, + "persona": persona, + "class": task_class, + "reason": reason, + "moved": ok, + }, + ) + return { + "task_id": task.task_id, + "outcome": "blocked", + "reason": reason, + "handoff": None, + "class": task_class, + } + + body, metadata = generated + written_path = _write_handoff_doc(task, body, persona, metadata, workspace_root) + if written_path is None: + reason = "handoff_path_write_failed_or_outside_workspace" + ok = queue_bridge.block_task(workspace_root, task.task_id, reason, persona=persona) + _emit_proof_event( + "persona_executor.task.blocked", + { + "task_id": task.task_id, + "persona": persona, + "class": task_class, + "reason": reason, + "moved": ok, + }, + ) + return { + "task_id": task.task_id, + "outcome": "blocked", + "reason": reason, + "handoff": None, + "class": task_class, + } + + # Success: mark task done. + ok = queue_bridge.complete_task(workspace_root, task.task_id, persona=persona) + _emit_proof_event( + "persona_executor.task.done", + { + "task_id": task.task_id, + "persona": persona, + "class": task_class, + "handoff_path": str(written_path), + "tokens_in": metadata.get("tokens_in"), + "tokens_out": metadata.get("tokens_out"), + "model": metadata.get("model"), + "moved": ok, + }, + ) + return { + "task_id": task.task_id, + "outcome": "done", + "reason": "audit_handoff_generated", + "handoff": str(written_path), + "class": task_class, + } + + +def execute_claimed_tasks(workspace_root: Path | None = None) -> list[dict[str, Any]]: + """Walk ``claimed/`` and execute up to ``MAX_EXECUTE_PER_TICK`` tasks. + + Returns the list of per-task result dicts (see :func:`execute_one`). + Honors ``HERMES3D_PERSONA_EXECUTOR_DISABLED=1`` (no-op fast path). + + Designed to be called from the asyncio queue poller's tick once + MVP-3 wiring is enabled. The function itself is synchronous + + file-system bounded. + """ + if _executor_disabled(): + return [] + root = workspace_root if workspace_root is not None else _workspace_root() + claimed = queue_bridge.list_tasks(root, "claimed") + if not claimed: + return [] + max_per_tick = _max_execute_per_tick() + results: list[dict[str, Any]] = [] + # Process in priority order (highest first) so the most-urgent + # claimed tasks are executed before lower-priority ones in a tick. + claimed.sort(key=lambda t: (-t.priority, t.task_id)) + for task in claimed: + if len(results) >= max_per_tick: + break + # Only execute tasks claimed by one of OUR personas. + owner = task.claimed_by or "" + if not owner.startswith("hermes/"): + continue + result = execute_one(task, root) + results.append(result) + return results + + +__all__ = [ + "classify_task", + "execute_claimed_tasks", + "execute_one", +] diff --git a/03_implementation/src/hermes3d/services/queue_poller.py b/03_implementation/src/hermes3d/services/queue_poller.py index 48f15aa7..c14797f7 100644 --- a/03_implementation/src/hermes3d/services/queue_poller.py +++ b/03_implementation/src/hermes3d/services/queue_poller.py @@ -101,17 +101,47 @@ def _heartbeat_our_claims(root: Path, personas: set[str]) -> int: def tick_once() -> dict[str, int]: """One poll cycle. Public so unit tests can drive it without - spawning the asyncio loop.""" + spawning the asyncio loop. + + W21-MVP-3: each tick now ALSO runs the persona executor against the + claimed/ dir (bounded by HERMES3D_PERSONA_EXEC_MAX_PER_TICK). This is + what makes claimed tasks actually produce a deliverable instead of + sitting forever in claimed/. Disable with HERMES3D_PERSONA_EXECUTOR_ + DISABLED=1 for tests that want to drive execution synchronously. + """ root = _workspace_root() personas = _available_personas() if not personas: - return {"personas": 0, "claimed": 0, "heartbeats": 0, "pending_seen": 0} + return { + "personas": 0, + "claimed": 0, + "heartbeats": 0, + "pending_seen": 0, + "executed_done": 0, + "executed_blocked": 0, + } persona_set = set(personas) - # 1. Refresh heartbeats first so a long-running claimed task does not - # look stale to the orchestrator while we are also trying to claim. + # 1. Refresh heartbeats on currently-claimed tasks FIRST. The MVP-3 + # executor below may move some of them to done/blocked, but any + # that survive (e.g. the executor disabled flag, max-per-tick + # reached) need their heartbeat updated so the orchestrator does + # not consider them stale. heartbeats = _heartbeat_our_claims(root, persona_set) + # 2. W21-MVP-3: run the persona executor on currently-claimed tasks + # BEFORE claiming more. Keeps the pipeline flowing + # (claim -> execute -> done) instead of accumulating claims. + try: + from hermes3d.services import persona_executor + + exec_results = persona_executor.execute_claimed_tasks(workspace_root=root) + except Exception as exc: # noqa: BLE001 — executor failure must not stall poll + LOG.warning("queue_poller: persona executor raised: %s", exc) + exec_results = [] + executed_done = sum(1 for r in exec_results if r.get("outcome") == "done") + executed_blocked = sum(1 for r in exec_results if r.get("outcome") == "blocked") + # 2. Walk pending tasks in priority order (descending). The # queue_bridge.list_tasks sort is by filename today; sort here # explicitly so the contract does not depend on filesystem order. @@ -139,6 +169,8 @@ def tick_once() -> dict[str, int]: "claimed": claimed_this_tick, "heartbeats": heartbeats, "pending_seen": seen, + "executed_done": executed_done, + "executed_blocked": executed_blocked, } diff --git a/04_testing/pytest/integration/test_w21_a4_mvp2_queue_route_and_poller.py b/04_testing/pytest/integration/test_w21_a4_mvp2_queue_route_and_poller.py index 053d1de8..836bc5b6 100644 --- a/04_testing/pytest/integration/test_w21_a4_mvp2_queue_route_and_poller.py +++ b/04_testing/pytest/integration/test_w21_a4_mvp2_queue_route_and_poller.py @@ -153,8 +153,12 @@ def test_poller_tick_once_claims_matching_task( persona id from the PERSONAS roster. This is the core proof that the auto-poller will pick up real W21 tasks in production.""" monkeypatch.setenv("HERMES3D_WORKSPACE_ROOT", str(tmp_path)) + # MVP-2 test scope only — disable MVP-3 executor so it doesn't move + # tasks out of the claimed/ dir before the test asserts on them. + monkeypatch.setenv("HERMES3D_PERSONA_EXECUTOR_DISABLED", "1") # Re-import to pick up the new env var. sys.modules.pop("hermes3d.services.queue_poller", None) + sys.modules.pop("hermes3d.services.persona_executor", None) from hermes3d.services import queue_poller # The real PERSONAS roster includes factory-operator; seed a matching task. @@ -226,7 +230,11 @@ def test_poller_heartbeats_our_claimed_tasks( heartbeat refresh on the next tick — the orchestrator uses this to avoid considering live claims stale.""" monkeypatch.setenv("HERMES3D_WORKSPACE_ROOT", str(tmp_path)) + # MVP-2 test scope — disable MVP-3 executor so the seeded claim + # is not moved out of claimed/ before this test asserts on it. + monkeypatch.setenv("HERMES3D_PERSONA_EXECUTOR_DISABLED", "1") sys.modules.pop("hermes3d.services.queue_poller", None) + sys.modules.pop("hermes3d.services.persona_executor", None) from hermes3d.services import queue_poller # Manually seed a claimed task as if a prior tick had taken it. diff --git a/04_testing/pytest/integration/test_w21_mvp3_queue_execute_now.py b/04_testing/pytest/integration/test_w21_mvp3_queue_execute_now.py new file mode 100644 index 00000000..d6b89db2 --- /dev/null +++ b/04_testing/pytest/integration/test_w21_mvp3_queue_execute_now.py @@ -0,0 +1,206 @@ +"""W21 MVP-3 — integration test for the persona executor HTTP surface. + +Mission: prove the operator can drive a claimed task to DONE (or BLOCKED) +via ``POST /api/agents/queue/execute-now`` and observe the resulting +handoff markdown on disk + the queue lifecycle transition + the proof +event row. The LLM is patched to a deterministic stub so tests are +hermetic. +""" + +from __future__ import annotations + +import importlib +import json +import sys +from pathlib import Path + +import pytest +from fastapi.testclient import TestClient + + +@pytest.fixture +def client(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> TestClient: + """Spin up a fresh FastAPI app rooted at tmp_path.""" + monkeypatch.setenv("HERMES3D_WORKSPACE_ROOT", str(tmp_path)) + monkeypatch.setenv("HERMES3D_QUEUE_POLLER_DISABLED", "1") + db_path = tmp_path / "hermes3d.db" + for mod_name in list(sys.modules): + if mod_name.startswith("hermes3d.api.app") or mod_name.startswith("hermes3d.api.routes"): + sys.modules.pop(mod_name, None) + sys.modules.pop("hermes3d.config.env_loader", None) + sys.modules.pop("hermes3d.services.queue_bridge", None) + sys.modules.pop("hermes3d.services.queue_poller", None) + sys.modules.pop("hermes3d.services.persona_executor", None) + from hermes3d.db import init as db_init + + monkeypatch.setattr(db_init, "DB_PATH", db_path) + db_init.reset_initialization_state() + + # Patch the LLM caller globally so the integration test stays hermetic. + persona_executor_mod = importlib.import_module("hermes3d.services.persona_executor") + fake_md = "# Integration Test Audit\n\nVerdict: stubbed for integration test.\n" + fake_meta = {"model": "stub-llm", "tokens_in": 100, "tokens_out": 30} + monkeypatch.setattr( + persona_executor_mod, "_generate_audit_markdown", lambda t, p: (fake_md, fake_meta) + ) + + app_mod = importlib.import_module("hermes3d.api.app") + return TestClient(app_mod.create_gui_app()) + + +def _seed_task( + root: Path, + task_id: str, + *, + state: str = "claimed", + handoff_path: str | None = None, + claimed_by: str | None = "hermes/factory-operator", + target_owner_pattern: str = "factory-operator", +) -> None: + state_dir = root / ".hermes3d_orchestrator" / "tasks" / state + state_dir.mkdir(parents=True, exist_ok=True) + body = { + "task_schema_version": 1, + "task_id": task_id, + "title": f"Integration test {task_id}", + "summary": f"Integration test for {task_id}", + "target_owner_pattern": target_owner_pattern, + "priority": 80, + "claimed_by": claimed_by, + "claimed_utc": "2026-05-12T00:00:00.000000Z", + "heartbeat_utc": "2026-05-12T00:00:00.000000Z", + "done_utc": None, + "blocked_reason": None, + "handoff_path": handoff_path, + } + (state_dir / f"{task_id}.json").write_text(json.dumps(body, indent=2), encoding="utf-8") + + +# --------------------------------------------------------------------------- +# /api/agents/queue/execute-now — single audit task path to done +# --------------------------------------------------------------------------- + + +def test_execute_now_drives_audit_task_to_done(client: TestClient, tmp_path: Path) -> None: + """E2E: claimed audit task → execute-now → handoff written on disk → done/.""" + handoff_rel = "03_implementation/docs/handoffs/W21_A99_E2E_AUDIT_2026-05-12.md" + _seed_task(tmp_path, "W21-A99-E2E-AUDIT-2026-05-12", handoff_path=handoff_rel) + + # Confirm baseline. + pre = client.get("/api/agents/queue/status").json() + assert pre["counts"]["claimed"] == 1 + assert pre["counts"]["done"] == 0 + + # Trigger the executor. + resp = client.post("/api/agents/queue/execute-now") + assert resp.status_code == 200, resp.text + body = resp.json() + assert body["accepted"] is True + assert body["counts"]["done"] == 1 + assert body["counts"]["blocked"] == 0 + assert body["results"][0]["outcome"] == "done" + assert body["results"][0]["reason"] == "audit_handoff_generated" + + # Handoff exists on disk. + written = Path(body["results"][0]["handoff"]) + assert written.exists() + content = written.read_text(encoding="utf-8") + assert "MVP-3 attestation" in content + assert "operator review REQUIRED" in content + + # Queue lifecycle: claimed → done. + post = client.get("/api/agents/queue/status").json() + assert post["counts"]["claimed"] == 0 + assert post["counts"]["done"] == 1 + + # Filesystem evidence: file moved. + assert ( + tmp_path / ".hermes3d_orchestrator" / "tasks" / "done" / "W21-A99-E2E-AUDIT-2026-05-12.json" + ).exists() + assert not ( + tmp_path + / ".hermes3d_orchestrator" + / "tasks" + / "claimed" + / "W21-A99-E2E-AUDIT-2026-05-12.json" + ).exists() + + +# --------------------------------------------------------------------------- +# /api/agents/queue/execute-now — unknown class path to blocked +# --------------------------------------------------------------------------- + + +def test_execute_now_moves_unknown_class_to_blocked(client: TestClient, tmp_path: Path) -> None: + """A claimed task with no executor → moves to blocked/ with reason.""" + _seed_task( + tmp_path, + "W21-A99-BUILD-FEATURE", + handoff_path="03_implementation/docs/handoffs/W21_A99_BUILD.md", + ) + resp = client.post("/api/agents/queue/execute-now") + assert resp.status_code == 200, resp.text + body = resp.json() + assert body["counts"]["done"] == 0 + assert body["counts"]["blocked"] == 1 + assert body["results"][0]["outcome"] == "blocked" + assert body["results"][0]["reason"].startswith("no_automated_executor_for_task_class") + # Filesystem evidence: file moved to blocked/. + assert ( + tmp_path / ".hermes3d_orchestrator" / "tasks" / "blocked" / "W21-A99-BUILD-FEATURE.json" + ).exists() + + +# --------------------------------------------------------------------------- +# Proof event persistence +# --------------------------------------------------------------------------- + + +def test_execute_now_emits_proof_event(client: TestClient, tmp_path: Path) -> None: + """A successful execute-now run must write a row to the proof_events + table (`persona_executor.task.done` event type).""" + import sqlite3 + + handoff_rel = "03_implementation/docs/handoffs/W21_A99_PROOF_TEST_AUDIT_2026-05-12.md" + _seed_task(tmp_path, "W21-A99-PROOF-TEST-AUDIT-2026-05-12", handoff_path=handoff_rel) + resp = client.post("/api/agents/queue/execute-now") + assert resp.status_code == 200 + + db_path = tmp_path / "hermes3d.db" + assert db_path.exists() + with sqlite3.connect(db_path) as con: + cur = con.cursor() + rows = cur.execute( + "SELECT event_type, source_agent FROM proof_events WHERE event_type LIKE ?", + ("persona_executor.%",), + ).fetchall() + assert any(r[0] == "persona_executor.task.done" for r in rows), ( + f"expected a persona_executor.task.done event in proof_events; got {rows!r}" + ) + + +# --------------------------------------------------------------------------- +# Mixed-batch: 1 audit + 1 unknown in same execute-now call +# --------------------------------------------------------------------------- + + +def test_execute_now_mixed_batch_done_plus_blocked(client: TestClient, tmp_path: Path) -> None: + """Same call returns both done and blocked outcomes correctly.""" + _seed_task( + tmp_path, + "W21-A1-MIXED-AUDIT-2026-05-12", + handoff_path="03_implementation/docs/handoffs/W21_A1_MIXED_AUDIT.md", + ) + _seed_task( + tmp_path, + "W21-A2-MIXED-BUILD", + handoff_path="03_implementation/docs/handoffs/W21_A2_MIXED_BUILD.md", + ) + resp = client.post("/api/agents/queue/execute-now") + body = resp.json() + assert body["counts"]["done"] == 1 + assert body["counts"]["blocked"] == 1 + final = client.get("/api/agents/queue/status").json() + assert final["counts"]["done"] == 1 + assert final["counts"]["blocked"] == 1 + assert final["counts"]["claimed"] == 0 diff --git a/04_testing/pytest/unit/test_w21_mvp3_persona_executor.py b/04_testing/pytest/unit/test_w21_mvp3_persona_executor.py new file mode 100644 index 00000000..6934fd2a --- /dev/null +++ b/04_testing/pytest/unit/test_w21_mvp3_persona_executor.py @@ -0,0 +1,266 @@ +"""W21 MVP-3 — unit tests for ``hermes3d.services.persona_executor``. + +Mission: pin the executor's classification + transition contract WITHOUT +calling the real LLM. The LLM call is monkeypatched to a deterministic +fake so we can assert on the resulting markdown body, the proof events, +and the queue lifecycle transitions. + +These tests cover the bounded inner contract; the integration test +hits the FastAPI route + queue_bridge filesystem layer end-to-end. +""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest +from hermes3d.services import persona_executor, queue_bridge + +# --------------------------------------------------------------------------- +# Fixtures +# --------------------------------------------------------------------------- + + +def _seed_task( + root: Path, + task_id: str, + *, + state: str = "claimed", + target_owner_pattern: str = "factory-operator", + priority: int = 80, + handoff_path: str | None = None, + claimed_by: str | None = "hermes/factory-operator", + title: str | None = None, + summary: str = "", +) -> Path: + state_dir = root / ".hermes3d_orchestrator" / "tasks" / state + state_dir.mkdir(parents=True, exist_ok=True) + body = { + "task_schema_version": 1, + "task_id": task_id, + "title": title or f"test {task_id}", + "summary": summary or f"summary for {task_id}", + "target_owner_pattern": target_owner_pattern, + "priority": priority, + "claimed_by": claimed_by, + "claimed_utc": "2026-05-12T00:00:00.000000Z" if claimed_by else None, + "heartbeat_utc": "2026-05-12T00:00:00.000000Z" if claimed_by else None, + "done_utc": None, + "blocked_reason": None, + "handoff_path": handoff_path, + } + path = state_dir / f"{task_id}.json" + path.write_text(json.dumps(body, indent=2), encoding="utf-8") + return path + + +# --------------------------------------------------------------------------- +# classify_task +# --------------------------------------------------------------------------- + + +def test_classify_audit_task_with_md_handoff(tmp_path: Path) -> None: + """Tasks named *_AUDIT_*.md classify as audit.""" + _seed_task( + tmp_path, + "W21-A99-TEST-AUDIT-2026-05-12", + handoff_path="03_implementation/docs/handoffs/W21_A99_TEST_AUDIT_2026-05-12.md", + ) + snaps = queue_bridge.list_tasks(tmp_path, "claimed") + assert snaps and persona_executor.classify_task(snaps[0]) == "audit" + + +def test_classify_plan_task_is_audit_class(tmp_path: Path) -> None: + _seed_task( + tmp_path, + "W21-A8-GEN3D-MODEL-INSTALL-EXECUTION-PLAN-2026-05-12", + handoff_path="03_implementation/docs/handoffs/W21_A8_PLAN.md", + ) + snaps = queue_bridge.list_tasks(tmp_path, "claimed") + assert snaps and persona_executor.classify_task(snaps[0]) == "audit" + + +def test_classify_unknown_when_no_md_handoff(tmp_path: Path) -> None: + _seed_task( + tmp_path, + "W21-A99-TEST-AUDIT-2026-05-12", + handoff_path="03_implementation/var/output.stl", + ) + snaps = queue_bridge.list_tasks(tmp_path, "claimed") + assert snaps and persona_executor.classify_task(snaps[0]) == "unknown" + + +def test_classify_unknown_when_taskid_lacks_class_marker(tmp_path: Path) -> None: + _seed_task( + tmp_path, + "W21-A99-BUILD-FEATURE", + handoff_path="03_implementation/docs/handoffs/W21_A99_BUILD.md", + ) + snaps = queue_bridge.list_tasks(tmp_path, "claimed") + assert snaps and persona_executor.classify_task(snaps[0]) == "unknown" + + +# --------------------------------------------------------------------------- +# execute_one — unknown-class path: must move task to blocked/ +# --------------------------------------------------------------------------- + + +def test_unknown_class_moves_task_to_blocked( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setenv("HERMES3D_WORKSPACE_ROOT", str(tmp_path)) + _seed_task( + tmp_path, + "W21-A99-BUILD-FEATURE", + handoff_path="03_implementation/docs/handoffs/W21_A99_BUILD.md", + ) + snap = queue_bridge.list_tasks(tmp_path, "claimed")[0] + result = persona_executor.execute_one(snap, tmp_path) + assert result["outcome"] == "blocked" + assert result["reason"].startswith("no_automated_executor_for_task_class") + # File moved from claimed/ to blocked/ + assert ( + tmp_path / ".hermes3d_orchestrator" / "tasks" / "blocked" / f"{snap.task_id}.json" + ).exists() + assert not ( + tmp_path / ".hermes3d_orchestrator" / "tasks" / "claimed" / f"{snap.task_id}.json" + ).exists() + + +# --------------------------------------------------------------------------- +# execute_one — audit-class success path: must write handoff + move to done/ +# --------------------------------------------------------------------------- + + +def test_audit_class_writes_handoff_and_marks_done( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """Monkeypatches the LLM caller so the test is hermetic.""" + monkeypatch.setenv("HERMES3D_WORKSPACE_ROOT", str(tmp_path)) + + handoff_rel = "03_implementation/docs/handoffs/W21_A99_TEST_AUDIT_2026-05-12.md" + _seed_task( + tmp_path, + "W21-A99-TEST-AUDIT-2026-05-12", + handoff_path=handoff_rel, + title="Test Audit", + summary="Verify the executor produces a real handoff file.", + ) + snap = queue_bridge.list_tasks(tmp_path, "claimed")[0] + + # Patch the LLM call to a deterministic stub so the test doesn't hit + # MiniMax (and so the test passes on a machine without API keys). + fake_body = "# Test Audit\n\nVerdict: stubbed for unit test.\nNo external calls made." + fake_metadata = {"model": "stub-llm", "tokens_in": 100, "tokens_out": 50} + + def _fake_generate(task, persona): # noqa: ANN001 + return fake_body, fake_metadata + + monkeypatch.setattr(persona_executor, "_generate_audit_markdown", _fake_generate) + + result = persona_executor.execute_one(snap, tmp_path) + + assert result["outcome"] == "done" + assert result["reason"] == "audit_handoff_generated" + assert result["class"] == "audit" + assert result["handoff"] is not None + + # Handoff markdown exists and contains BOTH the MVP-3 header + the + # generated body. + written = Path(result["handoff"]) + assert written.exists() + content = written.read_text(encoding="utf-8") + assert "MVP-3 attestation" in content + assert "operator review REQUIRED" in content + assert "stub-llm" in content # metadata exposed in header + assert fake_body in content # LLM body included + + # Task moved from claimed/ to done/. + assert ( + tmp_path / ".hermes3d_orchestrator" / "tasks" / "done" / f"{snap.task_id}.json" + ).exists() + + +def test_audit_class_llm_failure_routes_to_blocked( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """If the LLM call returns None (failure), task goes to blocked/.""" + monkeypatch.setenv("HERMES3D_WORKSPACE_ROOT", str(tmp_path)) + _seed_task( + tmp_path, + "W21-A99-LLM-FAIL-AUDIT-2026-05-12", + handoff_path="03_implementation/docs/handoffs/W21_A99_LLM_FAIL_AUDIT.md", + ) + snap = queue_bridge.list_tasks(tmp_path, "claimed")[0] + + monkeypatch.setattr(persona_executor, "_generate_audit_markdown", lambda t, p: None) + + result = persona_executor.execute_one(snap, tmp_path) + assert result["outcome"] == "blocked" + assert result["reason"] == "llm_generation_failed_or_unavailable" + assert ( + tmp_path / ".hermes3d_orchestrator" / "tasks" / "blocked" / f"{snap.task_id}.json" + ).exists() + + +# --------------------------------------------------------------------------- +# execute_claimed_tasks — multi-task, budget, persona filter +# --------------------------------------------------------------------------- + + +def test_executor_disabled_env_var_makes_executor_noop( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setenv("HERMES3D_WORKSPACE_ROOT", str(tmp_path)) + monkeypatch.setenv("HERMES3D_PERSONA_EXECUTOR_DISABLED", "1") + _seed_task(tmp_path, "W21-A99-TEST-AUDIT", handoff_path="x.md") + results = persona_executor.execute_claimed_tasks(workspace_root=tmp_path) + assert results == [] + + +def test_executor_respects_max_per_tick(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("HERMES3D_WORKSPACE_ROOT", str(tmp_path)) + monkeypatch.setenv("HERMES3D_PERSONA_EXEC_MAX_PER_TICK", "1") + monkeypatch.setattr( + persona_executor, + "_generate_audit_markdown", + lambda t, p: ("# stub", {"model": "stub", "tokens_in": 1, "tokens_out": 1}), + ) + for i in range(3): + _seed_task( + tmp_path, + f"W21-A{i:02d}-MULTI-AUDIT-2026-05-12", + handoff_path=f"03_implementation/docs/handoffs/W21_A{i:02d}_AUDIT.md", + priority=100 - i, + ) + results = persona_executor.execute_claimed_tasks(workspace_root=tmp_path) + assert len(results) == 1 # max=1 enforced + + +def test_executor_skips_tasks_not_owned_by_hermes_persona( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """Tasks claimed by external actors (no hermes/ prefix) are skipped.""" + monkeypatch.setenv("HERMES3D_WORKSPACE_ROOT", str(tmp_path)) + monkeypatch.setattr( + persona_executor, + "_generate_audit_markdown", + lambda t, p: ("# stub", {"model": "stub", "tokens_in": 1, "tokens_out": 1}), + ) + # Two tasks: one owned by hermes/, one by an external actor. + _seed_task( + tmp_path, + "W21-A1-OURS-AUDIT", + handoff_path="03_implementation/docs/handoffs/W21_A1_OURS_AUDIT.md", + claimed_by="hermes/factory-operator", + ) + _seed_task( + tmp_path, + "W21-A2-THEIRS-AUDIT", + handoff_path="03_implementation/docs/handoffs/W21_A2_THEIRS_AUDIT.md", + claimed_by="codex-some-other-actor", + ) + results = persona_executor.execute_claimed_tasks(workspace_root=tmp_path) + assert len(results) == 1 + assert results[0]["task_id"] == "W21-A1-OURS-AUDIT"