From 4f7d31637e8ef855f6236dd74a50fd6f1bf707dd Mon Sep 17 00:00:00 2001 From: Ghenghis <6685932+Ghenghis@users.noreply.github.com> Date: Thu, 7 May 2026 01:30:31 -0700 Subject: [PATCH] feat(agents): add provider team assignment lane --- 03_implementation/ROADMAP.md | 4 +- .../src/hermes3d/api/routes/agents.py | 53 ++++ .../src/hermes3d/api/routes/code_operator.py | 55 ++++ .../src/hermes3d/services/code_history.py | 237 ++++++++++++++++++ .../pytest/unit/test_agent_action_catalog.py | 34 ++- 04_testing/pytest/unit/test_code_operator.py | 134 ++++++++++ 6 files changed, 510 insertions(+), 7 deletions(-) diff --git a/03_implementation/ROADMAP.md b/03_implementation/ROADMAP.md index 3e7c1f63..0ce0d687 100644 --- a/03_implementation/ROADMAP.md +++ b/03_implementation/ROADMAP.md @@ -69,8 +69,8 @@ Definition of roadmap completion: | P0 | 3D Generation executor and preview proof | DONE | Live proof job `56954aa084b94a71a3bea4cb38f72f55` generated `calibration_cube_c8f48130af.stl`, `calibration_cube_c8f48130af.preview.svg`, `calibration_cube_c8f48130af.proof.json`, inserted proof event `bff1c6b5608e4384ad2e9c3fe4e5e5b5`, and passed truth status `pass`. Unsupported arbitrary prompts fail closed with provider setup guidance. | | P0 | Jobs repair/rollback transitions | DONE | Live proof job `a936044cad8e4b07b378bd53c0ed1187` created REPAIR_APPROVAL `c998960a23e74b7ea1da4aca6390b5b9`, appended proposal proof `12f27d2e5eae4e0d95caaea28421b5c0`, apply/escalation proof `2421f6f2f80a4abd99b888e5d2eb9d82`, retry proof `cd752dcbed3c4add913c9d2901ed07ab`, and rollback proof `41583d82acb34f4db5df313c9e4b4e10`. UI controls call real routes and show blocked reasons. | | P0 | Source OS + Plugins update execution | PARTIAL | Source OS Verify All, Setup Queue, selected-app Verify, selected-app Setup Plan, Backup, Check Update, no-op/current Update, Rollback routing, and proof events are wired. Runtime proof: Verify All `521e8565cfaf4115b3cd66c4fef66cc6` / backend route proof `e07a6629533c4a4c9333ba657120cecc`, Setup Queue proof `96885b17bdaf47b590b57da4d4515ca1`, PrusaSlicer setup-plan proof `5bce091c12c44656a4f3f1209fc6522d`, Azure Speech SDK setup-plan proof `9a7bd64d01644e8fb81359a196288179`, PrusaSlicer CLI `84e425b7fe514d42896ad16083eed561`, FLSUN Slicer CLI `eb18f6c1b98b4f52b8f955aee4b2503c`, source-only Azure Speech SDK `ae75bb3c53d6418ab7c603c7d7747e90`. Update proof on `blender_mcp_candidates`: backup `20260505T225446Z_blender_mcp_candidates_7636d13bded8`, backup proof `fa26b7f476d24de4b74a46f06b182ae5`, check proof `c89d2ea4965d4d9182c53e8b71e086b9`, update proof `970908d222734767a55401644878fa4a`. Remaining: all-app release watch, registering safe setup runners, and richer post-update app-specific smoke gates for the 60-app registry. | -| P0 | Hermes Agent full OS operator coverage | IN_PROGRESS | `/api/agents/action-catalog` now exposes 71 cataloged OS/code actions with public ready/partial/blocked state, proof requirements, approval/rollback flags, payload requirements, and no leaked internal handler names. Current code action surface includes MCP-locked patch/restore plus git readiness, branch, stage-owned, commit-owned, push, and PR actions; git actions are high-risk and proof/approval/rollback-marked where appropriate. Earlier safe sweep returned 25/25 accepted/completed proofs, protected missing-payload probes failed closed 5/5, and non-physical artifact actions passed 3/3: generation proof `783d06b27087489a8914956a5df559dd`, design proof `1ad2e99d3f9541d58b4a38a0e1ff5b16`, T1 #1 camera evidence proof `4a6143a9ef6a47c993530539d31a0dd1`, Azure voice preview proof `e4bbebf265e34035b3e3f47ecb3c4238` / TTS proof `324cbb9c902643a388ef773d83613c61`, Azure catalog proof `cb64df49be2f4036b1945d88ce2cd861`, and Roadmap truth proof `19f7ecf90f8a4a1ba303bf7fd0ec3142`. Focused Playwright passed 10/10 and screenshot proof is `03_implementation/proof/screenshots/agents-operator-catalog-expanded-2026-05-06.png`. Remaining: complete safe registered runners for the 30 Source OS runner gaps, update-all backup/smoke/rollback orchestration, and blocked idle work kinds before calling Hermes Agents fully e2e complete. | -| P0 | Hermes Agent programming ecosystem | IN_PROGRESS | Source inputs are required, not optional: `https://github.com/NousResearch/hermes-agent.git` and `https://github.com/AtomicBot-ai/atomic-hermes.git`, with local source at `G:/Github/hermes-agent-fresh` and `G:/Github/atomic-hermes`. First safe code-operator slice is live and verified: programming readiness, Hermes MCP lock readiness, write readiness, repo status/tree/search, bounded file read, snapshot/diff/restore, proof-backed patch proposal, MCP-locked patch apply, exact-worktree MCP gate list/run APIs, exact-worktree MCP claim/lock/heartbeat/evidence/release APIs, and proof-gated git branch/stage/commit/push/PR APIs are exposed through `/api/code-operator/*` and mirrored into `/api/agents/action-catalog` with proof-required contracts. Route smoke proves `GET /api/code-operator/gates` returns 11 MCP gates and `POST /api/code-operator/gates/run` runs `git-diff-check` through `hermes_run_gate` as PASS while invalid owners fail closed; evidence `ev_4a3482659b8b91d3`. MCP coordination route smoke claimed task `h3d-agent-lock-route-smoke`, locked/hearted/released `03_implementation/ROADMAP.md`, recorded evidence, and rejected `../G/private/.env`; evidence `ev_77276b2ded845997`. Patch-apply route smoke created proposal `642990660dde4b688f14c35589eaf408`, applied it only under active same-owner MCP lock with pre/post snapshots, recorded chained MCP evidence, and passed `git-diff-check`; evidence `ev_2cd18d787167bc33`. PR #73 hardening proof is `ev_dfe36e14a23d418c`. Git shipping lane is implemented as the next stacked Codex slice: it creates only `codex/` or `hermes-agent/` branches from clean worktrees, stages only changed source files that have same-owner snapshots and active MCP file locks, commits with proof IDs, pushes to origin without force, and opens PRs through authenticated `gh`. Remaining before autonomous source mutation is fully complete: provider-backed coding loop, second-agent review lane, and rollback-gated repair workflow. Acceptance: both Hermes agent teams can take real Hermes3D tasks, claim/lock/heartbeat/gate/evidence/release through Hermes MCP locks, edit code through source-backed tools, run real tests/gates, produce proof artifacts, create branches/PRs, and restore any agent-touched file. | +| P0 | Hermes Agent full OS operator coverage | IN_PROGRESS | `/api/agents/action-catalog` now exposes 74 cataloged OS/code actions with public ready/partial/blocked state, proof requirements, approval/rollback flags, payload requirements, and no leaked internal handler names. Current code action surface includes MCP-locked patch/restore, proof-gated git readiness/branch/stage/commit/push/PR, plus provider-team readiness, assignment, and second-team review-request contracts; git actions are high-risk and proof/approval/rollback-marked where appropriate. Earlier safe sweep returned 25/25 accepted/completed proofs, protected missing-payload probes failed closed 5/5, and non-physical artifact actions passed 3/3: generation proof `783d06b27087489a8914956a5df559dd`, design proof `1ad2e99d3f9541d58b4a38a0e1ff5b16`, T1 #1 camera evidence proof `4a6143a9ef6a47c993530539d31a0dd1`, Azure voice preview proof `e4bbebf265e34035b3e3f47ecb3c4238` / TTS proof `324cbb9c902643a388ef773d83613c61`, Azure catalog proof `cb64df49be2f4036b1945d88ce2cd861`, and Roadmap truth proof `19f7ecf90f8a4a1ba303bf7fd0ec3142`. Focused Playwright passed 10/10 and screenshot proof is `03_implementation/proof/screenshots/agents-operator-catalog-expanded-2026-05-06.png`. Remaining: complete safe registered runners for the 30 Source OS runner gaps, update-all backup/smoke/rollback orchestration, and blocked idle work kinds before calling Hermes Agents fully e2e complete. | +| P0 | Hermes Agent programming ecosystem | IN_PROGRESS | Source inputs are required, not optional: `https://github.com/NousResearch/hermes-agent.git` and `https://github.com/AtomicBot-ai/atomic-hermes.git`, with local source at `G:/Github/hermes-agent-fresh` and `G:/Github/atomic-hermes`. First safe code-operator slice is live and verified: programming readiness, Hermes MCP lock readiness, write readiness, repo status/tree/search, bounded file read, snapshot/diff/restore, proof-backed patch proposal, MCP-locked patch apply, exact-worktree MCP gate list/run APIs, exact-worktree MCP claim/lock/heartbeat/evidence/release APIs, proof-gated git branch/stage/commit/push/PR APIs, and provider-team readiness/assignment/review-request APIs are exposed through `/api/code-operator/*` and mirrored into `/api/agents/action-catalog` with proof-required contracts. Route smoke proves `GET /api/code-operator/gates` returns 11 MCP gates and `POST /api/code-operator/gates/run` runs `git-diff-check` through `hermes_run_gate` as PASS while invalid owners fail closed; evidence `ev_4a3482659b8b91d3`. MCP coordination route smoke claimed task `h3d-agent-lock-route-smoke`, locked/hearted/released `03_implementation/ROADMAP.md`, recorded evidence, and rejected `../G/private/.env`; evidence `ev_77276b2ded845997`. Patch-apply route smoke created proposal `642990660dde4b688f14c35589eaf408`, applied it only under active same-owner MCP lock with pre/post snapshots, recorded chained MCP evidence, and passed `git-diff-check`; evidence `ev_2cd18d787167bc33`. PR #73 hardening proof is `ev_dfe36e14a23d418c`. Git shipping lane is implemented as the next stacked Codex slice: it creates only `codex/` or `hermes-agent/` branches from clean worktrees, stages only changed source files that have same-owner snapshots and active MCP file locks, commits with proof IDs, pushes to origin without force, and opens PRs through authenticated `gh`. The provider-team slice exposes Team A MiniMax builders and Team B DeepSeek reviewers as real readiness/assignment/review boundaries: it records proof-backed tasks only when selected source, provider config, and MCP lock prerequisites pass, and returns exact blocked reasons when they do not. It does not yet pretend to invoke providers to write/review code. Remaining before autonomous source mutation is fully complete: provider-backed coding/review execution loop and rollback-gated repair workflow. Acceptance: both Hermes agent teams can take real Hermes3D tasks, claim/lock/heartbeat/gate/evidence/release through Hermes MCP locks, edit code through source-backed tools, run real tests/gates, produce proof artifacts, create branches/PRs, and restore any agent-touched file. | | P0 | Claude 20-agent completion contract kit | IN_PROGRESS | The first 20-lane contract exists at `03_implementation/docs/handoffs/CLAUDE_20_AGENT_COMPLETION_CONTRACT.md`; the follow-up six-agent polish/audit contract exists at `03_implementation/docs/handoffs/CLAUDE_6_AGENT_POLISH_AUDIT_2026-05-06.md`. Acceptance: every Claude lane has exact files, locks, task id, branch/worktree rule, pass/fail gates, proof output, no-fake rule, S1 lock rule, GitHub push/PR rule, and "fix until pass" instruction. Claude agents may complete Source OS runners, tab polish, tests, docs/proof, visual checks, and missing runtime integrations, but must not touch files owned by the current Codex code-operator lane. | | P0 | 60 source-app runtime completion | IN_PROGRESS | Keep the visible Source OS target at the proven 60 source-backed rows. Acceptance: `SOURCE_REGISTRY_TRUTH_AUDIT.json`, `/api/modules`, `/api/modules/runtime/verifiers`, `/api/modules/runtime/setup-queue`, `/api/modules/runtime/cli-surface`, `SOURCE_APP_60_COMPLETION_AUDIT.json`, `SOURCE_APP_CLI_AGENT_READINESS_AUDIT.json`, `SOURCE_APP_CLI_SURFACE_AUDIT.json`, `SOURCE_APP_RUNTIME_ACTION_PLAN.md`, and Playwright Source OS/Plugins/Settings/Roadmap proof all agree; each row is either runtime-ready through a registered verifier or source-ready with an exact runner gap, every available CLI is exposed as an agent-usable verifier/runner before desktop fallback, and no row remains in an unclassified launch-kind bucket. | | P1 | Voice Azure STT pipeline | DONE | Backend `/api/voice/stt` reads Azure Speech secrets only from runtime/private env, returns transcript or exact blocker, appends proof, and chat mic inserts transcript while preserving the audio artifact. Live Azure TTS-to-STT proof event: `5d9b8a25a1ac454f85ea46e5a59866f4`. | diff --git a/03_implementation/src/hermes3d/api/routes/agents.py b/03_implementation/src/hermes3d/api/routes/agents.py index 0ec3b0ca..289ea819 100644 --- a/03_implementation/src/hermes3d/api/routes/agents.py +++ b/03_implementation/src/hermes3d/api/routes/agents.py @@ -667,6 +667,43 @@ def _execute_catalog_handler(handler: str, actor: str, payload: dict[str, Any]) from hermes3d.services import code_history return code_history.programming_readiness() + if handler == "code.teams.readiness": + from hermes3d.services import code_history + + return code_history.provider_team_readiness() + if handler == "code.teams.assign_task": + from hermes3d.services import code_history + + files = payload.get("files") + if not isinstance(files, list) or not files: + raise ValueError("Payload field files must be a non-empty list.") + return code_history.assign_provider_team_task( + owner=actor, + team_id=_required_payload_text(payload, "team_id"), + task_id=_required_payload_text(payload, "task_id"), + title=_required_payload_text(payload, "title"), + files=files, + objective=_required_payload_text(payload, "objective"), + target_branch=str(payload.get("target_branch") or "") or None, + review_required=bool(payload.get("review_required", True)), + ) + if handler == "code.teams.request_review": + from hermes3d.services import code_history + + files = payload.get("files") + proof_ids = payload.get("proof_ids") + if not isinstance(files, list) or not files: + raise ValueError("Payload field files must be a non-empty list.") + if not isinstance(proof_ids, list) or not proof_ids: + raise ValueError("Payload field proof_ids must be a non-empty list.") + return code_history.request_provider_team_review( + owner=actor, + task_id=_required_payload_text(payload, "task_id"), + summary=_required_payload_text(payload, "summary"), + files=files, + proof_ids=proof_ids, + reviewer_team_id=str(payload.get("reviewer_team_id") or "deepseek-reviewers"), + ) if handler == "code.mcp_locks.readiness": from hermes3d.services import code_history @@ -1176,11 +1213,17 @@ def _agent_action_contracts() -> list[dict[str, Any]]: from hermes3d.services import code_history mcp_locks = code_history.mcp_lock_readiness() + provider_teams = code_history.provider_team_readiness() except Exception: mcp_locks = {"ready": False, "blocked_reason": "Hermes MCP lock readiness could not be evaluated."} + provider_teams = {"ready": False, "status": "blocked", "blocked_reasons": ["Hermes Agent provider-team readiness could not be evaluated."]} + team_blocked_reason = "; ".join(str(item) for item in provider_teams.get("blocked_reasons", [])[:4]) if provider_teams.get("blocked_reasons") else None contracts = [ _contract("agents.health.refresh", "Refresh Hermes Agent runtime health", "agents", "ready" if runtime.get("hermes_agent_runtime") == "ready" else "blocked", "read", "low", "GET /api/agents/health", "agents.health", "Checks the configured local/private agent runtime bridge."), _contract("code.programming_readiness.refresh", "Refresh Hermes Agent programming readiness", "agents", "ready", "read", "low", "GET /api/code-operator/programming-readiness", "code.programming_readiness", "Checks true source inputs from Nous Hermes Agent and Atomic Hermes plus MiniMax/DeepSeek provider readiness."), + _contract("code.teams.readiness.refresh", "Refresh Hermes Agent team readiness", "agents", str(provider_teams.get("status") or "blocked"), "read", "low", "GET /api/code-operator/teams/readiness", "code.teams.readiness", "Checks MiniMax builder and DeepSeek reviewer team readiness without exposing provider secrets.", None if provider_teams.get("ready") else (team_blocked_reason or "Hermes Agent provider teams are not ready.")), + _contract("code.teams.assign_task", "Assign provider-backed code task", "agents", "ready" if provider_teams.get("ready") else "blocked", "proof", "medium", "POST /api/code-operator/teams/assign-task", "code.teams.assign_task", "Records a proof-backed provider-team coding task only after selected source, provider, and MCP lock prerequisites are ready.", None if provider_teams.get("ready") else (team_blocked_reason or "Hermes Agent provider teams are not ready.")), + _contract("code.teams.request_review", "Request second-team code review", "agents", "ready" if provider_teams.get("ready") else "blocked", "proof", "medium", "POST /api/code-operator/teams/request-review", "code.teams.request_review", "Requests a proof-backed DeepSeek/Atomic Hermes review for files and proof ids before PR shipping.", None if provider_teams.get("ready") else (team_blocked_reason or "Hermes Agent reviewer team is not ready.")), _contract("code.mcp_locks.readiness.refresh", "Refresh Hermes MCP lock readiness", "agents", "ready" if mcp_locks.get("ready") else "partial", "read", "low", "GET /api/code-operator/mcp-locks/readiness", "code.mcp_locks.readiness", "Checks that the Hermes Agent runtime has hermes3d-locks source/server access and that MCP_LOCK_WORKSPACE matches the actual edit workspace before write tools can enable.", None if mcp_locks.get("ready") else str(mcp_locks.get("blocked_reason") or "Hermes MCP locks are not ready for code writes.")), _contract("code.write_readiness.refresh", "Refresh Hermes Agent write readiness", "agents", "ready" if mcp_locks.get("ready") else "blocked", "read", "low", "GET /api/code-operator/write/readiness", "code.write.readiness", "Explains whether Hermes Agents may enable patch/apply/command/git coding tools yet. Read-only context stays available; write tools stay blocked until locks, source inputs, providers, snapshots, and proof gates are ready.", None if mcp_locks.get("ready") else str(mcp_locks.get("blocked_reason") or "Hermes MCP locks are not ready for code writes.")), _contract("code.history.files.refresh", "Refresh agent-touched file history", "agents", "ready", "read", "low", "GET /api/code-operator/history/files", "code.history.files", "Lists files already snapshotted by Hermes Agent code operations."), @@ -1316,6 +1359,16 @@ def _contract_payload_schema(action_id: str) -> dict[str, Any]: "voice.catalog.refresh": {"required": [], "optional": {"locale": "Azure voice locale prefix, defaults to en"}}, "voice.preview": {"required": [], "optional": {"agent_id": "agent id", "voice": "Azure short name", "text": "preview text", "rate": "0.5-2.0", "pitch_pct": "-50..50"}}, "voice.stt": {"required": ["artifact_id"], "optional": {"locale": "speech locale, defaults to en-US"}, "safety": "Reads an existing audio artifact; no secret values are returned."}, + "code.teams.assign_task": { + "required": ["team_id", "task_id", "title", "files", "objective"], + "optional": {"target_branch": "safe git ref", "review_required": "defaults true"}, + "safety": "Team id must be minimax-builders, deepseek-reviewers, or dual; selected source/provider/MCP prerequisites must be ready before assignment is accepted.", + }, + "code.teams.request_review": { + "required": ["task_id", "summary", "files", "proof_ids"], + "optional": {"reviewer_team_id": "defaults to deepseek-reviewers"}, + "safety": "Review requests require at least one proof/evidence id and route through the proof ledger; no provider secret values are returned.", + }, "code.history.snapshot": {"required": ["relative_path"], "optional": {"action_id": "agent action identifier", "reason": "why this snapshot is needed"}, "safety": "Project-relative source files only; secrets, binary/generated files, .git, node_modules, and printer config writes are blocked."}, "code.repo.tree.refresh": {"required": [], "optional": {"root": "project-relative directory or file; defaults to repository root", "limit": "1-1200 returned paths"}, "safety": "Secrets, generated output, caches, node_modules, and VCS internals are excluded."}, "code.repo.search": {"required": ["pattern"], "optional": {"root": "project-relative search root", "max_results": "1-200"}, "safety": "Bounded ripgrep only; absolute paths, parent traversal, secrets, and generated folders are blocked."}, diff --git a/03_implementation/src/hermes3d/api/routes/code_operator.py b/03_implementation/src/hermes3d/api/routes/code_operator.py index 23db12d5..964b5eeb 100644 --- a/03_implementation/src/hermes3d/api/routes/code_operator.py +++ b/03_implementation/src/hermes3d/api/routes/code_operator.py @@ -134,11 +134,66 @@ class McpEvidenceRequest(StrictBody): data: dict[str, Any] = Field(default_factory=dict) +class ProviderTeamAssignmentRequest(StrictBody): + team_id: str + task_id: str + title: str = Field(min_length=1, max_length=180) + files: list[str] = Field(min_length=1) + objective: str = Field(min_length=1, max_length=1600) + target_branch: str | None = None + review_required: bool = True + + +class ProviderTeamReviewRequest(StrictBody): + task_id: str + summary: str = Field(min_length=1, max_length=1200) + files: list[str] = Field(min_length=1) + proof_ids: list[str] = Field(min_length=1) + reviewer_team_id: str = "deepseek-reviewers" + + @router.get("/programming-readiness") def programming_readiness() -> dict[str, Any]: return code_history.programming_readiness() +@router.get("/teams/readiness") +def provider_team_readiness() -> dict[str, Any]: + return code_history.provider_team_readiness() + + +@router.post("/teams/assign-task") +def provider_team_assign_task(body: ProviderTeamAssignmentRequest) -> dict[str, Any]: + try: + return code_history.assign_provider_team_task( + owner=CODE_OPERATOR_ACTOR, + team_id=body.team_id, + task_id=body.task_id, + title=body.title, + files=body.files, + objective=body.objective, + target_branch=body.target_branch, + review_required=body.review_required, + ) + except (RuntimeError, ValueError, FileNotFoundError) as exc: + raise HTTPException(status_code=422, detail={"reason": str(exc)}) from exc + + +@router.post("/teams/request-review") +def provider_team_request_review(body: ProviderTeamReviewRequest) -> dict[str, Any]: + try: + return code_history.request_provider_team_review( + owner=CODE_OPERATOR_ACTOR, + task_id=body.task_id, + summary=body.summary, + files=body.files, + proof_ids=body.proof_ids, + reviewer_team_id=body.reviewer_team_id, + ) + except (RuntimeError, ValueError, FileNotFoundError) as exc: + raise HTTPException(status_code=422, detail={"reason": str(exc)}) from exc + + @router.get("/mcp-locks/readiness") def mcp_locks_readiness() -> dict[str, Any]: return code_history.mcp_lock_readiness() diff --git a/03_implementation/src/hermes3d/services/code_history.py b/03_implementation/src/hermes3d/services/code_history.py index c1291f01..92fd6788 100644 --- a/03_implementation/src/hermes3d/services/code_history.py +++ b/03_implementation/src/hermes3d/services/code_history.py @@ -190,6 +190,27 @@ class SourceRepo: "hermes_release_task", ) +PROVIDER_TEAMS = { + "minimax-builders": { + "label": "Team A MiniMax builders", + "provider_id": "minimax", + "role": "builder", + "source_input": "nous_hermes_agent", + "mission": "implement bounded Hermes3D code changes through MCP locks, snapshots, gates, and proof", + }, + "deepseek-reviewers": { + "label": "Team B DeepSeek reviewers", + "provider_id": "deepseek", + "role": "reviewer", + "source_input": "atomic_hermes", + "mission": "review code changes, rollback plans, proofs, and security/architecture risk before PR shipping", + }, +} +PROVIDER_TEAM_GROUPS = { + "dual": ("minimax-builders", "deepseek-reviewers"), + **{team_id: (team_id,) for team_id in PROVIDER_TEAMS}, +} + def programming_readiness() -> dict[str, Any]: private_values = private_env() @@ -229,6 +250,66 @@ def programming_readiness() -> dict[str, Any]: } +def provider_team_readiness() -> dict[str, Any]: + private_values = private_env() + source_by_id = {item["id"]: item for item in (_source_repo_status(source) for source in SOURCE_REPOS)} + provider_by_id = { + "minimax": _provider_status("minimax", private_values), + "deepseek": _provider_status("deepseek", private_values), + } + lock_status = mcp_lock_readiness(private_values) + teams: list[dict[str, Any]] = [] + blocked_reasons: list[str] = [] + for team_id, config in PROVIDER_TEAMS.items(): + source = source_by_id.get(str(config["source_input"]), {}) + provider = provider_by_id.get(str(config["provider_id"]), {}) + team_blockers: list[str] = [] + if source.get("status") == "missing": + team_blockers.append(f"Source input {config['source_input']} is missing.") + if provider.get("status") != "ready": + team_blockers.append(f"Provider {config['provider_id']} is not configured.") + if not lock_status["ready"]: + team_blockers.append(str(lock_status.get("blocked_reason") or "Hermes MCP locks are not ready.")) + ready = not team_blockers + blocked_reasons.extend(f"{team_id}: {reason}" for reason in team_blockers) + teams.append( + { + "id": team_id, + "label": config["label"], + "role": config["role"], + "mission": config["mission"], + "source_input": source, + "provider": provider, + "mcp_locks_ready": lock_status["ready"], + "status": "ready" if ready else "blocked", + "ready": ready, + "blocked_reasons": team_blockers, + } + ) + ready_team_ids = [team["id"] for team in teams if team["ready"]] + return { + "status": "ready" if len(ready_team_ids) == len(teams) else ("partial" if ready_team_ids else "blocked"), + "ready": len(ready_team_ids) == len(teams), + "teams": teams, + "team_groups": { + group_id: list(team_ids) + for group_id, team_ids in PROVIDER_TEAM_GROUPS.items() + }, + "mcp_locks": lock_status, + "required_flow": [ + "select provider team", + "verify source input and provider env without exposing secrets", + "claim Hermes task", + "lock target files before write", + "snapshot every touched file", + "run MCP gates and visual proof", + "request second-team review", + "ship branch/PR only after proof", + ], + "blocked_reasons": blocked_reasons, + } + + def mcp_lock_readiness(private_values: dict[str, str] | None = None) -> dict[str, Any]: values = private_values if private_values is not None else private_env() configured_workspace = ( @@ -821,6 +902,147 @@ def claim_mcp_task(*, owner: str, task_id: str, title: str = "", files: list[str return {"status": "claimed" if result.get("ok") is True else str(result.get("status") or "partial"), "workspace": str(PROJECT_ROOT), "result": result} +def assign_provider_team_task( + *, + owner: str, + team_id: str, + task_id: str, + title: str, + files: list[str], + objective: str, + target_branch: str | None = None, + review_required: bool = True, +) -> dict[str, Any]: + _require_mcp_locks_ready() + _validate_owner(owner) + _validate_task_id(task_id) + selected_teams = _validate_provider_team_selection(team_id) + safe_files = _safe_mcp_files(files, must_exist=False) + if not safe_files: + raise ValueError("At least one project-relative target file is required.") + clean_title = _validate_bounded_text(title, "Title", max_chars=180) + clean_objective = _validate_bounded_text(objective, "Objective", max_chars=1600) + branch = _validate_git_ref(target_branch) if target_branch else None + readiness = provider_team_readiness() + team_map = {team["id"]: team for team in readiness["teams"]} + selected_status = [team_map[item] for item in selected_teams] + blocked = [ + f"{team['id']}: {reason}" + for team in selected_status + for reason in (team.get("blocked_reasons") or []) + ] + evidence_payload = { + "team_id": team_id, + "selected_teams": selected_teams, + "task_id": task_id, + "title": clean_title, + "files": safe_files, + "objective_sha256": hashlib.sha256(clean_objective.encode("utf-8")).hexdigest(), + "target_branch": branch, + "review_required": bool(review_required), + "blocked_reasons": blocked, + } + if blocked: + evidence = append_mcp_evidence( + owner=owner, + task_id=task_id, + kind="code_team", + summary=f"Blocked Hermes Agent team assignment {task_id}", + data=evidence_payload, + ) + return { + "status": "blocked", + "accepted": False, + "team_id": team_id, + "selected_teams": selected_status, + "files": safe_files, + "blocked_reasons": blocked, + "mcp_evidence": evidence, + } + claim = claim_mcp_task( + owner=owner, + task_id=task_id, + title=clean_title, + files=safe_files, + reason=clean_objective, + role="provider-team", + ) + evidence = append_mcp_evidence( + owner=owner, + task_id=task_id, + kind="code_team", + summary=f"Assigned Hermes Agent team task {task_id}", + data=evidence_payload, + ) + return { + "status": "assigned", + "accepted": True, + "team_id": team_id, + "selected_teams": selected_status, + "files": safe_files, + "target_branch": branch, + "review_required": bool(review_required), + "claim": claim, + "mcp_evidence": evidence, + } + + +def request_provider_team_review( + *, + owner: str, + task_id: str, + summary: str, + files: list[str], + proof_ids: list[str], + reviewer_team_id: str = "deepseek-reviewers", +) -> dict[str, Any]: + _require_mcp_locks_ready() + _validate_owner(owner) + _validate_task_id(task_id) + selected_teams = _validate_provider_team_selection(reviewer_team_id) + safe_files = _safe_mcp_files(files, must_exist=False) + if not safe_files: + raise ValueError("At least one project-relative file is required for review.") + clean_summary = _validate_bounded_text(summary, "Review summary", max_chars=1200) + safe_proofs = [_validate_proof_ref(item) for item in proof_ids] + if not safe_proofs: + raise ValueError("At least one proof/evidence id is required for review.") + readiness = provider_team_readiness() + team_map = {team["id"]: team for team in readiness["teams"]} + selected_status = [team_map[item] for item in selected_teams] + blocked = [ + f"{team['id']}: {reason}" + for team in selected_status + for reason in (team.get("blocked_reasons") or []) + ] + payload = { + "reviewer_team_id": reviewer_team_id, + "selected_teams": selected_teams, + "task_id": task_id, + "summary_sha256": hashlib.sha256(clean_summary.encode("utf-8")).hexdigest(), + "files": safe_files, + "proof_ids": safe_proofs, + "blocked_reasons": blocked, + } + evidence = append_mcp_evidence( + owner=owner, + task_id=task_id, + kind="code_review", + summary=(f"Blocked Hermes Agent review {task_id}" if blocked else f"Requested Hermes Agent review {task_id}"), + data=payload, + ) + return { + "status": "blocked" if blocked else "review_requested", + "accepted": not blocked, + "reviewer_team_id": reviewer_team_id, + "selected_teams": selected_status, + "files": safe_files, + "proof_ids": safe_proofs, + "blocked_reasons": blocked, + "mcp_evidence": evidence, + } + + def lock_mcp_files(*, owner: str, files: list[str], task_id: str = "", reason: str = "", role: str = "agent", ttl_minutes: int = 90) -> dict[str, Any]: _require_mcp_locks_ready() _validate_owner(owner) @@ -1258,6 +1480,21 @@ def _validate_task_id(task_id: str) -> None: raise ValueError("Task id must be 2-128 safe characters.") +def _validate_provider_team_selection(team_id: str) -> tuple[str, ...]: + value = str(team_id or "").strip() + selected = PROVIDER_TEAM_GROUPS.get(value) + if not selected: + raise ValueError("Team id must be one of: " + ", ".join(sorted(PROVIDER_TEAM_GROUPS))) + return selected + + +def _validate_bounded_text(value: str, label: str, *, max_chars: int) -> str: + text = str(value or "").strip() + if not text or len(text) > max_chars: + raise ValueError(f"{label} must be 1-{max_chars} characters.") + return text + + def _safe_mcp_files(files: list[str], *, must_exist: bool) -> list[str]: if not files: return [] diff --git a/04_testing/pytest/unit/test_agent_action_catalog.py b/04_testing/pytest/unit/test_agent_action_catalog.py index 80cb5215..f1bcc788 100644 --- a/04_testing/pytest/unit/test_agent_action_catalog.py +++ b/04_testing/pytest/unit/test_agent_action_catalog.py @@ -5,10 +5,14 @@ from hermes3d.api.routes import agents -def test_action_catalog_hides_internal_handlers_and_exposes_code_actions() -> None: +def _reset_action_catalog_cache() -> None: agents._ACTION_CONTRACT_CACHE["contracts"] = None agents._ACTION_CONTRACT_CACHE["ts"] = 0.0 + +def test_action_catalog_hides_internal_handlers_and_exposes_code_actions() -> None: + _reset_action_catalog_cache() + payload = agents.action_catalog() contracts = payload["contracts"] by_id = {contract["id"]: contract for contract in contracts} @@ -16,13 +20,14 @@ def test_action_catalog_hides_internal_handlers_and_exposes_code_actions() -> No assert "code.patch.apply" in by_id assert "code.mcp_locks.lock_files" in by_id assert "code.git.commit_owned" in by_id + assert "code.teams.assign_task" in by_id + assert "code.teams.request_review" in by_id assert all("handler" not in contract for contract in contracts) assert set(by_id["code.mcp_locks.lock_files"]["payload_schema"]["required"]) == {"files", "task_id"} def test_code_patch_apply_declares_high_risk_proof_and_rollback_contract() -> None: - agents._ACTION_CONTRACT_CACHE["contracts"] = None - agents._ACTION_CONTRACT_CACHE["ts"] = 0.0 + _reset_action_catalog_cache() patch_apply = { contract["id"]: contract @@ -39,8 +44,7 @@ def test_code_patch_apply_declares_high_risk_proof_and_rollback_contract() -> No def test_code_git_commit_declares_owned_snapshot_contract() -> None: - agents._ACTION_CONTRACT_CACHE["contracts"] = None - agents._ACTION_CONTRACT_CACHE["ts"] = 0.0 + _reset_action_catalog_cache() git_commit = { contract["id"]: contract @@ -53,3 +57,23 @@ def test_code_git_commit_declares_owned_snapshot_contract() -> None: assert git_commit["rollback_required"] is True assert set(git_commit["payload_schema"]["required"]) == {"task_id", "files", "message"} assert "owned snapshot" in git_commit["payload_schema"]["safety"] + + +def test_code_team_actions_declare_assignment_and_review_contracts() -> None: + _reset_action_catalog_cache() + + by_id = {contract["id"]: contract for contract in agents.action_catalog()["contracts"]} + assignment = by_id["code.teams.assign_task"] + review = by_id["code.teams.request_review"] + + assert assignment["kind"] == "proof" + assert assignment["risk"] == "medium" + assert assignment["proof_required"] is True + assert assignment["approval_required"] is False + assert set(assignment["payload_schema"]["required"]) == {"team_id", "task_id", "title", "files", "objective"} + assert "minimax-builders" in assignment["payload_schema"]["safety"] + + assert review["kind"] == "proof" + assert review["risk"] == "medium" + assert set(review["payload_schema"]["required"]) == {"task_id", "summary", "files", "proof_ids"} + assert "proof/evidence id" in review["payload_schema"]["safety"] diff --git a/04_testing/pytest/unit/test_code_operator.py b/04_testing/pytest/unit/test_code_operator.py index 64f8d19f..a504de71 100644 --- a/04_testing/pytest/unit/test_code_operator.py +++ b/04_testing/pytest/unit/test_code_operator.py @@ -15,6 +15,9 @@ def test_code_operator_routes_are_registered() -> None: paths = {route.path for route in app.routes} assert "/api/code-operator/programming-readiness" in paths + assert "/api/code-operator/teams/readiness" in paths + assert "/api/code-operator/teams/assign-task" in paths + assert "/api/code-operator/teams/request-review" in paths assert "/api/code-operator/patch/apply" in paths assert "/api/code-operator/git/commit-owned" in paths assert "/api/code-operator/history/restore" in paths @@ -84,6 +87,137 @@ def test_git_commit_rejects_spoofed_actor_fields() -> None: assert response.status_code == 422 +@pytest.mark.parametrize( + "path", + [ + "/api/code-operator/teams/assign-task", + "/api/code-operator/teams/request-review", + ], +) +def test_provider_team_routes_reject_spoofed_actor_fields(path: str) -> None: + client = TestClient(create_gui_app()) + payload = { + "team_id": "dual", + "reviewer_team_id": "deepseek-reviewers", + "task_id": "TASK-1", + "title": "Fix a tab", + "summary": "Review a tab fix", + "files": ["README.md"], + "objective": "Make the tab real-backed.", + "proof_ids": ["ev_test_1"], + "owner": "other-agent", + } + + response = client.post(path, json=payload) + + assert response.status_code == 422 + + +def test_provider_team_readiness_uses_mocked_inputs_not_environment(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(code_history, "private_env", lambda: {}) + monkeypatch.setattr( + code_history, + "_source_repo_status", + lambda source: { + "id": source.id, + "status": "ready", + "exists": True, + "missing_required_files": [], + }, + ) + monkeypatch.setattr( + code_history, + "_provider_status", + lambda provider_id, _private_values: { + "id": provider_id, + "status": "ready", + "api_key_configured": True, + "base_url_configured": True, + "model_configured": True, + "base_url_label": "local-or-private", + "model": f"{provider_id}-test", + }, + ) + monkeypatch.setattr( + code_history, + "mcp_lock_readiness", + lambda _private_values=None: { + "status": "ready", + "ready": True, + "blocked_reason": None, + "required_workflow": [], + }, + ) + + response = TestClient(create_gui_app()).get("/api/code-operator/teams/readiness") + payload = response.json() + + assert response.status_code == 200 + assert payload["ready"] is True + assert {team["id"] for team in payload["teams"]} == {"minimax-builders", "deepseek-reviewers"} + assert all("api_key" not in str(team["provider"].get("base_url_label", "")).lower() for team in payload["teams"]) + + +def test_provider_team_assignment_blocks_when_provider_missing(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(code_history, "_require_mcp_locks_ready", lambda: None) + monkeypatch.setattr(code_history, "private_env", lambda: {}) + monkeypatch.setattr( + code_history, + "_source_repo_status", + lambda source: {"id": source.id, "status": "ready", "exists": True, "missing_required_files": []}, + ) + monkeypatch.setattr( + code_history, + "_provider_status", + lambda provider_id, _private_values: { + "id": provider_id, + "status": "ready" if provider_id == "minimax" else "missing_config", + "api_key_configured": provider_id == "minimax", + "base_url_configured": False, + "model_configured": provider_id == "minimax", + "base_url_label": None, + "model": f"{provider_id}-test" if provider_id == "minimax" else None, + }, + ) + monkeypatch.setattr( + code_history, + "mcp_lock_readiness", + lambda _private_values=None: {"status": "ready", "ready": True, "blocked_reason": None, "required_workflow": []}, + ) + monkeypatch.setattr( + code_history, + "append_mcp_evidence", + lambda **kwargs: {"status": "recorded", "evidence_id": "ev_team_blocked", "kwargs": kwargs}, + ) + + result = code_history.assign_provider_team_task( + owner="hermes-agent", + team_id="dual", + task_id="TASK-1", + title="Fix dashboard truth", + files=["README.md"], + objective="Make dashboard proof-backed.", + ) + + assert result["accepted"] is False + assert result["status"] == "blocked" + assert "deepseek-reviewers" in result["blocked_reasons"][0] + assert result["mcp_evidence"]["evidence_id"] == "ev_team_blocked" + + +def test_provider_team_review_requires_proof_id(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(code_history, "_require_mcp_locks_ready", lambda: None) + + with pytest.raises(ValueError, match="proof"): + code_history.request_provider_team_review( + owner="hermes-agent", + task_id="TASK-1", + summary="Review the fix.", + files=["README.md"], + proof_ids=[], + ) + + @pytest.mark.parametrize( "malicious", [