From 2ecda4310f8ffc4b7f9d7946c9fb352bdff8f44e Mon Sep 17 00:00:00 2001 From: "firat.sertgoz" Date: Thu, 2 Jul 2026 11:07:31 +0300 Subject: [PATCH 1/3] test(reborn): add semantic judge fallback to live QA (#5531) * test(reborn): add semantic judge fallback to live QA * test(reborn): isolate live QA semantic judge * test(reborn): harden semantic judge parsing --- .github/workflows/live-canary.yml | 2 + .../reborn_webui_v2_live_qa/run_live_qa.py | 22 +- .../reborn_webui_v2_live_qa/semantic_judge.py | 204 ++++++++++++++++++ .../test_run_live_qa.py | 155 +++++++++++++ 4 files changed, 382 insertions(+), 1 deletion(-) create mode 100644 scripts/reborn_webui_v2_live_qa/semantic_judge.py diff --git a/.github/workflows/live-canary.yml b/.github/workflows/live-canary.yml index f3d76ce45e3..03d33e20f99 100644 --- a/.github/workflows/live-canary.yml +++ b/.github/workflows/live-canary.yml @@ -538,6 +538,8 @@ jobs: REBORN_WEBUI_V2_LIVE_QA_LLM_API_KEY_ENV: NEARAI_API_KEY REBORN_WEBUI_V2_LIVE_QA_LLM_PROVIDER_ID: nearai REBORN_WEBUI_V2_LIVE_QA_LLM_MODEL: ${{ vars.REBORN_WEBUI_V2_LIVE_QA_LLM_MODEL || 'deepseek-ai/DeepSeek-V4-Flash' }} + REBORN_WEBUI_V2_LIVE_QA_LLM_JUDGE: ${{ vars.REBORN_WEBUI_V2_LIVE_QA_LLM_JUDGE || '1' }} + REBORN_WEBUI_V2_LIVE_QA_LLM_JUDGE_MODEL: ${{ vars.REBORN_WEBUI_V2_LIVE_QA_LLM_JUDGE_MODEL || vars.REBORN_WEBUI_V2_LIVE_QA_LLM_MODEL || 'deepseek-ai/DeepSeek-V4-Flash' }} IRONCLAW_REBORN_GOOGLE_CLIENT_ID: ${{ secrets.IRONCLAW_REBORN_GOOGLE_CLIENT_ID || secrets.GOOGLE_CLIENT_ID || secrets.GOOGLE_OAUTH_CLIENT_ID }} IRONCLAW_REBORN_GOOGLE_OAUTH_REDIRECT_URI: ${{ vars.IRONCLAW_REBORN_GOOGLE_OAUTH_REDIRECT_URI || vars.GOOGLE_OAUTH_REDIRECT_URI }} IRONCLAW_REBORN_GOOGLE_HOSTED_DOMAIN_HINT: ${{ vars.IRONCLAW_REBORN_GOOGLE_HOSTED_DOMAIN_HINT || vars.GOOGLE_ALLOWED_HD }} diff --git a/scripts/reborn_webui_v2_live_qa/run_live_qa.py b/scripts/reborn_webui_v2_live_qa/run_live_qa.py index 443216bd056..3da202f9af1 100644 --- a/scripts/reborn_webui_v2_live_qa/run_live_qa.py +++ b/scripts/reborn_webui_v2_live_qa/run_live_qa.py @@ -93,6 +93,11 @@ _root_filesystem_json, _root_filesystem_secret_by_handle, ) +from scripts.reborn_webui_v2_live_qa.semantic_judge import ( # noqa: E402 + _compact_json, + _judge_assistant_reply_completion, + _semantic_judge_passed, +) from scripts.reborn_webui_v2_live_qa.slack_helpers import ( # noqa: E402 _append_slack_channel_route, _append_slack_channel_route_if_configured, @@ -856,6 +861,7 @@ async def action(page: object) -> None: marker=marker, required_text=required_text, timeout=timeout, + semantic_goal=prompt, ) if forbidden_text: text = str(observed["text_excerpt"]).lower() @@ -967,6 +973,7 @@ async def action(page: object) -> None: marker=marker, required_text=required_text, timeout=timeout, + semantic_goal=prompt, ) try: @@ -996,6 +1003,7 @@ async def _wait_for_assistant_reply( marker: str | None, required_text: list[str], timeout: float, + semantic_goal: str | None = None, ) -> str: deadline = time.monotonic() + timeout assistant = page.locator("[data-testid='msg-assistant']").last # type: ignore[attr-defined] @@ -1030,10 +1038,22 @@ async def _wait_for_assistant_reply( main_text = await page.locator("main").inner_text(timeout=1000) # type: ignore[attr-defined] except Exception: pass + semantic_judge: dict[str, object] | None = None + if last_text and (not marker or marker in last_text): + semantic_judge = await _judge_assistant_reply_completion( + marker=marker, + required_text=required_text, + assistant_text=last_text, + main_text=main_text, + semantic_goal=semantic_goal, + ) + if _semantic_judge_passed(semantic_judge): + return last_text[-2000:] raise AssertionError( "assistant reply did not contain required text before timeout. " f"marker={marker!r} required_text={required_text!r} " - f"last_assistant={last_text[-500:]!r} main_excerpt={main_text[-1000:]!r}" + f"last_assistant={last_text[-500:]!r} main_excerpt={main_text[-1000:]!r} " + f"semantic_judge={_compact_json(semantic_judge)}" ) diff --git a/scripts/reborn_webui_v2_live_qa/semantic_judge.py b/scripts/reborn_webui_v2_live_qa/semantic_judge.py new file mode 100644 index 00000000000..e82d09825a6 --- /dev/null +++ b/scripts/reborn_webui_v2_live_qa/semantic_judge.py @@ -0,0 +1,204 @@ +"""Semantic completion judge for Reborn WebUI v2 live QA text checks.""" + +from __future__ import annotations + +import json +import os + +from scripts.live_canary.common import env_secret + + +async def _judge_assistant_reply_completion( + *, + marker: str | None, + required_text: list[str], + assistant_text: str, + main_text: str, + semantic_goal: str | None, +) -> dict[str, object] | None: + if not _env_flag_enabled("REBORN_WEBUI_V2_LIVE_QA_LLM_JUDGE", default=True): + return {"enabled": False, "reason": "disabled"} + api_key_env = _judge_api_key_env() + api_key = env_secret(api_key_env) + if not api_key: + return {"enabled": False, "reason": f"{api_key_env} unset"} + + try: + import httpx + + async with httpx.AsyncClient(timeout=30.0) as client: + response = await client.post( + f"{_judge_base_url()}/chat/completions", + headers={ + "Authorization": f"Bearer {api_key}", + "Content-Type": "application/json", + }, + json=_judge_payload( + marker=marker, + required_text=required_text, + assistant_text=assistant_text, + main_text=main_text, + semantic_goal=semantic_goal, + ), + ) + response.raise_for_status() + body = response.json() + content = _completion_content(body) + except Exception as exc: + return {"enabled": True, "error": str(exc)} + + parsed = _parse_json_object(content) + if not isinstance(parsed, dict): + return { + "enabled": True, + "error": "judge response was not a JSON object", + "response_excerpt": str(content)[-500:], + } + parsed["enabled"] = True + return parsed + + +def _semantic_judge_passed(result: dict[str, object] | None) -> bool: + if not result or result.get("completed") is not True: + return False + try: + confidence = float(result.get("confidence") or 0.0) + except (TypeError, ValueError): + confidence = 0.0 + try: + threshold = float( + os.environ.get("REBORN_WEBUI_V2_LIVE_QA_LLM_JUDGE_MIN_CONFIDENCE", "0.75") + ) + except ValueError: + threshold = 0.75 + return confidence >= threshold + + +def _compact_json(value: object) -> str: + if value is None: + return "null" + try: + encoded = json.dumps(value, sort_keys=True) + except TypeError: + encoded = repr(value) + if len(encoded) > 1000: + return f"{encoded[:1000]}..." + return encoded + + +def _judge_api_key_env() -> str: + return os.environ.get( + "REBORN_WEBUI_V2_LIVE_QA_LLM_JUDGE_API_KEY_ENV", + os.environ.get( + "REBORN_WEBUI_V2_LIVE_QA_LLM_API_KEY_ENV", + "NEARAI_API_KEY" + if os.environ.get("NEARAI_API_KEY") or os.environ.get("NEARAI_API_KEY_PATH") + else "LIVE_OPENAI_COMPATIBLE_API_KEY", + ), + ) + + +def _judge_base_url() -> str: + return os.environ.get( + "REBORN_WEBUI_V2_LIVE_QA_LLM_JUDGE_BASE_URL", + os.environ.get( + "REBORN_WEBUI_V2_LIVE_QA_LLM_BASE_URL", + os.environ.get("LIVE_OPENAI_COMPATIBLE_BASE_URL", "https://cloud-api.near.ai/v1"), + ), + ).rstrip("/") + + +def _judge_model() -> str: + return os.environ.get( + "REBORN_WEBUI_V2_LIVE_QA_LLM_JUDGE_MODEL", + os.environ.get( + "REBORN_WEBUI_V2_LIVE_QA_LLM_MODEL", + os.environ.get("LIVE_OPENAI_COMPATIBLE_MODEL", "deepseek-ai/DeepSeek-V4-Flash"), + ), + ) + + +def _judge_payload( + *, + marker: str | None, + required_text: list[str], + assistant_text: str, + main_text: str, + semantic_goal: str | None, +) -> dict[str, object]: + return { + "model": _judge_model(), + "temperature": 0, + "max_tokens": 500, + "messages": [ + { + "role": "system", + "content": ( + "You are a strict semantic verifier for a live WebUI QA canary. " + "Return JSON only. Judge only whether the visible assistant response " + "semantically satisfies the natural-language expectation. Do not infer " + "external side effects, database writes, tool calls, delivery, auth, or " + "capability execution unless the text explicitly says so. If an exact " + "marker is required and missing, completed must be false." + ), + }, + { + "role": "user", + "content": json.dumps( + { + "task_prompt": semantic_goal or "", + "exact_marker_required": marker, + "literal_required_text": required_text, + "assistant_response": assistant_text[-4000:], + "page_excerpt": main_text[-2000:], + "required_json_schema": { + "completed": "boolean", + "confidence": "number from 0 to 1", + "reason": "short string", + "evidence": "array of short strings", + "missing": "array of short strings", + }, + }, + sort_keys=True, + ), + }, + ], + } + + +def _completion_content(body: object) -> object: + if not isinstance(body, dict): + return "" + choices = body.get("choices") + if not isinstance(choices, list) or not choices: + return "" + choice = choices[0] + if not isinstance(choice, dict): + return "" + message = choice.get("message") + if not isinstance(message, dict): + return "" + return message.get("content") or "" + + +def _parse_json_object(text: object) -> object: + if not isinstance(text, str): + return None + try: + return json.loads(text) + except Exception: + start = text.find("{") + end = text.rfind("}") + if start == -1 or end <= start: + return None + try: + return json.loads(text[start : end + 1]) + except Exception: + return None + + +def _env_flag_enabled(name: str, *, default: bool) -> bool: + raw = os.environ.get(name) + if raw is None: + return default + return raw.strip().lower() not in {"0", "false", "no", "off"} diff --git a/scripts/reborn_webui_v2_live_qa/test_run_live_qa.py b/scripts/reborn_webui_v2_live_qa/test_run_live_qa.py index 27039ff40db..4a5a6d1cfff 100644 --- a/scripts/reborn_webui_v2_live_qa/test_run_live_qa.py +++ b/scripts/reborn_webui_v2_live_qa/test_run_live_qa.py @@ -24,8 +24,10 @@ if __package__: from . import run_live_qa + from . import semantic_judge else: import run_live_qa + import semantic_judge class RebornWebUiV2LiveQaRunnerTests(unittest.TestCase): @@ -37,6 +39,46 @@ def _dummy_ctx(self) -> run_live_qa.LiveQaContext: env={}, ) + def _fake_assistant_reply_page(self, response_text: str): + class FakeApprove: + @property + def last(self): + return self + + async def is_visible(self, **_kwargs): + return False + + class FakeAssistantBlocks: + @property + def last(self): + return self + + async def count(self): + return 1 + + async def inner_text(self, **_kwargs): + return response_text + + async def all_inner_texts(self): + return [response_text] + + class FakeMain: + async def inner_text(self, **_kwargs): + return response_text + + class FakePage: + def locator(self, selector): + if selector == "[data-testid='msg-assistant']": + return FakeAssistantBlocks() + if selector == "main": + return FakeMain() + raise AssertionError(f"unexpected selector: {selector}") + + def get_by_role(self, _role, **_kwargs): + return FakeApprove() + + return FakePage() + def test_dismiss_visible_connect_action_clicks_only_visible_card(self): class FakeDismiss: def __init__(self, *, count: int, visible: bool) -> None: @@ -615,6 +657,119 @@ def get_by_role(self, _role, **_kwargs): self.assertIn("routine", text.lower()) self.assertIn("emails", text.lower()) + def test_wait_for_assistant_reply_uses_semantic_judge_for_text_mismatch(self): + response_text = ( + "Schedule: Every hour\n" + "Delivery: Slack channel CQA123\n" + "Trigger ID: trigger-123" + ) + + captured: dict[str, object] = {} + + async def fake_judge(**kwargs): + captured.update(kwargs) + return { + "completed": True, + "confidence": 0.91, + "reason": "The response confirms an hourly Slack scheduled trigger.", + } + + async def fake_sleep(_seconds): + return None + + with ( + patch.object( + run_live_qa, + "_judge_assistant_reply_completion", + side_effect=fake_judge, + ), + patch.object(run_live_qa.asyncio, "sleep", side_effect=fake_sleep), + ): + text = asyncio.run( + run_live_qa._wait_for_assistant_reply( + self._fake_assistant_reply_page(response_text), + marker=None, + required_text=["routine"], + timeout=0.001, + semantic_goal="Create an hourly Hacker News keyword Slack routine.", + ) + ) + + self.assertIn("Trigger ID", text) + self.assertEqual(captured["required_text"], ["routine"]) + self.assertIn("hourly Hacker News", captured["semantic_goal"]) + + def test_wait_for_assistant_reply_does_not_judge_missing_marker(self): + response_text = "The routine has been created. Trigger ID: trigger-123" + + async def fail_if_called(**_kwargs): + raise AssertionError("semantic judge should not run when marker is missing") + + async def fake_sleep(_seconds): + return None + + with ( + patch.object( + run_live_qa, + "_judge_assistant_reply_completion", + side_effect=fail_if_called, + ), + patch.object(run_live_qa.asyncio, "sleep", side_effect=fake_sleep), + ): + with self.assertRaisesRegex(AssertionError, "marker='REBORN_QA_DONE'"): + asyncio.run( + run_live_qa._wait_for_assistant_reply( + self._fake_assistant_reply_page(response_text), + marker="REBORN_QA_DONE", + required_text=["routine"], + timeout=0.001, + ) + ) + + def test_semantic_judge_passed_respects_confidence_threshold(self): + with patch.dict( + os.environ, + {"REBORN_WEBUI_V2_LIVE_QA_LLM_JUDGE_MIN_CONFIDENCE": "0.9"}, + ): + self.assertTrue( + semantic_judge._semantic_judge_passed( + {"completed": True, "confidence": 0.9} + ) + ) + self.assertFalse( + semantic_judge._semantic_judge_passed( + {"completed": True, "confidence": 0.89} + ) + ) + self.assertFalse( + semantic_judge._semantic_judge_passed( + {"completed": False, "confidence": 1.0} + ) + ) + + def test_semantic_judge_completion_content_handles_unexpected_shapes(self): + self.assertEqual(semantic_judge._completion_content(None), "") + self.assertEqual(semantic_judge._completion_content({"choices": "bad"}), "") + self.assertEqual(semantic_judge._completion_content({"choices": ["bad"]}), "") + self.assertEqual( + semantic_judge._completion_content({"choices": [{"message": "bad"}]}), + "", + ) + self.assertEqual( + semantic_judge._completion_content( + {"choices": [{"message": {"content": '{"completed": true}'}}]} + ), + '{"completed": true}', + ) + + def test_semantic_judge_json_parser_handles_non_string_inputs(self): + self.assertIsNone(semantic_judge._parse_json_object(None)) + self.assertIsNone(semantic_judge._parse_json_object(123)) + self.assertEqual( + semantic_judge._parse_json_object('prefix {"completed": true} suffix'), + {"completed": True}, + ) + def test_slack_delivery_target_dm_detection(self): self.assertTrue(run_live_qa._slack_delivery_target_is_dm("D12345")) self.assertFalse(run_live_qa._slack_delivery_target_is_dm("C12345")) From 0b26d0dea6e081fc666a96a67f6b7a65872b5ca1 Mon Sep 17 00:00:00 2001 From: "firat.sertgoz" Date: Thu, 2 Jul 2026 12:23:17 +0300 Subject: [PATCH 2/3] chore(agents): wire codebase knowledge graph + OpenWiki for agent code discovery (#5532) * chore(agents): wire codebase knowledge graph + OpenWiki for agent code discovery Give every agent (fresh session, teammate, cloud, CI) first-class code discovery so they stop rebuilding the map by grepping 1.27M lines. Two complementary layers, both auto-maintained. Graph (structural, precise -- codebase-memory MCP): - .mcp.json: declare the codebase-memory-mcp server at project scope so any Claude Code session in this repo auto-connects (it was only in personal global config, so only one machine had access) - scripts/dev-setup.sh: install the single static binary (no deps, no keys, local) - .gitignore: treat .codebase-memory/ as a build artifact (it was neither tracked nor ignored -- one `git add .` from a 234MB accidental commit) - scripts/codebase-graph.sh: freshness helper (indexed commit vs HEAD) OpenWiki (narrative, prose -- auto-generated, read-only): - .github/workflows/openwiki-update.yml: weekly + manual regen of openwiki/, Anthropic provider (ANTHROPIC_API_KEY), auto-merged PR opened via the GH_RELEASES_MANAGER app so required checks trigger and it lands on main hands-off Docs: - CLAUDE.md / AGENTS.md: "Code Discovery" section -- query the graph first (with recipes: search_graph / trace_path calls|data_flow|cross_service), read openwiki/ for narrative orientation Behavior: no product code touched. dev-setup/CI gain an install step; agents gain tools they call, not code that runs in the app. Co-Authored-By: Claude Fable 5 * docs(agents): steer new work to the Reborn stack, not the v1 src/ monolith Add a "Reborn-first" directive to AGENTS.md + CLAUDE.md: new features go in crates/ (product_workflow -> composition -> webui_v2 -> runtime/serve -> frontend, entry reborn_cli), not src/main.rs. src/ is v1 in retirement -- maintain existing behavior, don't add new features. Fills the gap where the docs' "Where to Work" pointed almost entirely at src/, steering agents into v1 by default. Co-Authored-By: Claude Fable 5 * ci(openwiki): open a PR for human review, not auto-merge (SOC 2) Auto-merging the generated wiki to main bypasses human change review, which SOC 2 change management (CC8.1 / separation of duties) does not allow. Drop the enable-auto-merge step; the bot only authors the PR and a human reviews + merges. Keep the GitHub App token so required checks still trigger on the bot PR (a GITHUB_TOKEN-opened PR triggers no workflows, leaving the reviewer no checks). Co-Authored-By: Claude Fable 5 --------- Co-authored-by: Claude Fable 5 --- .github/workflows/openwiki-update.yml | 76 +++++++++++++++++++++++++++ .gitignore | 5 ++ .mcp.json | 8 +++ AGENTS.md | 15 ++++++ CLAUDE.md | 29 ++++++++++ scripts/codebase-graph.sh | 59 +++++++++++++++++++++ scripts/dev-setup.sh | 15 ++++++ 7 files changed, 207 insertions(+) create mode 100644 .github/workflows/openwiki-update.yml create mode 100644 .mcp.json create mode 100755 scripts/codebase-graph.sh diff --git a/.github/workflows/openwiki-update.yml b/.github/workflows/openwiki-update.yml new file mode 100644 index 00000000000..8d39f47a9a9 --- /dev/null +++ b/.github/workflows/openwiki-update.yml @@ -0,0 +1,76 @@ +name: OpenWiki Update + +# Regenerates the narrative code wiki (openwiki/) and opens a PR for HUMAN review. +# +# Decisions (see docs/plans/2026-07-02 discovery rollout): +# - Provider: Anthropic via ANTHROPIC_API_KEY (already a repo secret). +# - Publish: open a pull request; a human reviews and merges. NO auto-merge — +# SOC 2 change management (CC8.1 / separation of duties) requires every change +# to main to be human-reviewed and approved. The bot only authors the proposal. +# +# Notes: +# - The PR is opened via the GH_RELEASES_MANAGER GitHub App (not GITHUB_TOKEN) so +# the required-check workflows actually trigger on it — a GITHUB_TOKEN-opened PR +# does not trigger other workflows, so a human reviewer would see no checks and +# the ruleset would block the merge. +# - OPENWIKI_MODEL_ID must match OpenWiki's Anthropic model resolution; if the +# run errors on the model, adjust the string (e.g. an "anthropic/"/"anthropic:" +# prefix, or claude-haiku-4-5-20251001 for a cheaper pass). + +on: + workflow_dispatch: + schedule: + - cron: "0 8 * * 1" # Mondays 08:00 UTC (weekly) + +permissions: + contents: write + pull-requests: write + +jobs: + update: + runs-on: ubuntu-latest + steps: + - name: Mint GitHub App token + id: app + uses: actions/create-github-app-token@v1 + with: + app-id: ${{ secrets.GH_RELEASES_MANAGER_APP_ID }} + private-key: ${{ secrets.GH_RELEASES_MANAGER_APP_PRIVATE_KEY }} + + - name: Check out repository + uses: actions/checkout@v4 + with: + token: ${{ steps.app.outputs.token }} + persist-credentials: true + + - name: Set up Node.js + uses: actions/setup-node@v4 + with: + node-version: "22" + + - name: Install OpenWiki + run: npm install --global openwiki + + - name: Regenerate wiki + run: openwiki --update --print + env: + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + OPENWIKI_MODEL_ID: claude-sonnet-4-6 + + - name: Open review PR + id: cpr + uses: peter-evans/create-pull-request@v7 + with: + token: ${{ steps.app.outputs.token }} + add-paths: openwiki + branch: openwiki/update + delete-branch: true + commit-message: "docs: update OpenWiki wiki" + title: "docs: update OpenWiki wiki" + labels: documentation + body: | + Automated OpenWiki narrative-docs refresh (`openwiki/`). + + Review and merge as usual — this PR is NOT auto-merged; a human + approves it per change-management policy. Precise structural discovery + lives in the codebase-memory graph; this is the prose "what/why" layer. diff --git a/.gitignore b/.gitignore index 9095065b285..b406454c85d 100644 --- a/.gitignore +++ b/.gitignore @@ -82,3 +82,8 @@ tests/fixtures/llm_traces/live/*.log # Local test artifacts .anvil/ .codegraph/ + +# codebase-memory knowledge graph — build artifact, not source. +# Rebuilt from code via the codebase-memory MCP; per-environment, never committed. +# See CLAUDE.md -> "Code Discovery". +.codebase-memory/ diff --git a/.mcp.json b/.mcp.json new file mode 100644 index 00000000000..b03271a9a57 --- /dev/null +++ b/.mcp.json @@ -0,0 +1,8 @@ +{ + "mcpServers": { + "codebase-memory-mcp": { + "command": "codebase-memory-mcp", + "args": [] + } + } +} diff --git a/AGENTS.md b/AGENTS.md index e20b909d1a8..9853c01aac8 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -16,6 +16,18 @@ Start with these deeper docs as needed: - `src/NETWORK_SECURITY.md` - `tests/e2e/CLAUDE.md` +## Code Discovery — Query the Knowledge Graph First + +This repo is indexed into a **codebase knowledge graph** via the `codebase-memory` MCP server (~464k nodes / ~2.5M edges over `src/` + `crates/`). For any *where-is / who-calls / how-does-data-flow / what-does-this-touch* question, **query the graph before `grep`** — text search can't see cross-crate call chains, and a feature here crosses many crates (`product_workflow → composition → webui_v2 → runtime → frontend`). + +- **Location:** `.codebase-memory/graph.db.zst` — a git-ignored build artifact (rebuilt from code, one per environment, never committed). +- **Freshness:** run `bash scripts/codebase-graph.sh status` (compares indexed commit vs `HEAD`). If missing → `index_repository(repo_path=".")`; if stale → `detect_changes(since="")` or re-index. The graph is point-in-time — verify its claims against live code before acting. +- **Recipes:** define → `search_graph(name_pattern=…)` + `get_code_snippet(…)`; callers → `trace_path(mode="calls")`; value flow → `trace_path(mode="data_flow")`; cross-crate flow → `trace_path(mode="cross_service")`; area shape → `get_architecture(…)`; Cypher → `query_graph(…)`. + +Use `grep`/read for text, config, and non-code files, and after the graph for code structure — not before. + +**Narrative docs (auto-generated, read-only):** prose subsystem docs live in `openwiki/`, regenerated by `.github/workflows/openwiki-update.yml`. `Read` them for *what/how* orientation; use the graph for precise *where/who*. Don't hand-edit `openwiki/`. + ## Architecture Mental Model - Channels normalize external input into `IncomingMessage`; `ChannelManager` merges all active channel streams. @@ -25,6 +37,9 @@ Start with these deeper docs as needed: ## Where to Work +**Build new features Reborn-side, in `crates/` — not the v1 `src/` monolith.** A Reborn feature crosses `product_workflow → composition → webui_v2 → runtime/serve → frontend`; the binary entry point is `crates/ironclaw_reborn_cli` (`reborn_cli`), **not** `src/main.rs`. Start from the `reborn-feature` skill, which maps the layers. `src/` is v1, being retired under "Clean up old architecture" — touch it only to maintain existing v1 behavior, never to add new features. + +Existing subsystem locations (mostly v1 `src/`; maintain, don't extend): - Agent/runtime behavior: `src/agent/` - Web gateway/API/SSE/WebSocket: `src/channels/web/` - Persistence and DB abstractions: `src/db/` diff --git a/CLAUDE.md b/CLAUDE.md index 714c6728ed4..f87fe7e8a41 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -2,6 +2,35 @@ **IronClaw** is a secure personal AI assistant — user-first security, self-expanding tools, defense in depth, multi-channel access with proactive background execution. +## Code Discovery — Query the Knowledge Graph First + +This repo is indexed into a **codebase knowledge graph** (the `codebase-memory` MCP server): ~464k nodes / ~2.5M edges spanning all of `src/` and `crates/`. For any *where-is / who-calls / how-does-data-flow / what-does-this-touch* question, **query the graph before reaching for `Grep`** — text search cannot see cross-crate call chains, and this codebase's real cost is cross-crate (a feature crosses `product_workflow → composition → webui_v2 → runtime → frontend`). + +**Where it lives:** `.codebase-memory/graph.db.zst` — a **git-ignored build artifact, not source**. One per environment, rebuilt from code. Never commit it. + +**Freshness (check at the start of a discovery task):** run `bash scripts/codebase-graph.sh status` — it compares the graph's indexed commit against `HEAD`. Then: +- **Missing** → `index_repository(repo_path=".")` once to build it. +- **Stale** → `detect_changes(since="")` for the changed symbols + blast radius, or re-run `index_repository` to fully refresh. +- The graph is a point-in-time index — verify anything it asserts against live code before acting. + +**Discovery recipes (use these instead of `Grep` for code structure):** +- Where a symbol is defined → `search_graph(name_pattern=…)`, then `get_code_snippet(qualified_name=…)` +- Who calls X / what X calls → `trace_path(function_name=…, mode="calls")` +- How a value flows across layers → `trace_path(mode="data_flow")` +- Cross-crate / cross-service path (the reborn 5-layer feature flow) → `trace_path(mode="cross_service")` +- Structure of an area → `get_architecture(…)`; graph-augmented text search → `search_code(pattern=…)` +- Arbitrary structural queries → `query_graph()` + +`Grep`/`Glob`/`Read` remain correct for text, config, and non-code files — and for reading a file the graph pointed you to. For *code structure*, the graph comes first. + +**Narrative orientation (what/why, not where):** prose docs for each subsystem live in `openwiki/` — an auto-generated wiki kept fresh by `.github/workflows/openwiki-update.yml`. For *"what does this subsystem do / how does this flow work"* questions, `Read` the relevant `openwiki/` page; use the graph for precise structure. Do not hand-edit `openwiki/` — it is regenerated. The two layers are complementary: `openwiki/` = prose map, the graph = exact index. + +## Where to Build — Reborn-First + +**New feature work targets the Reborn stack in `crates/`, not the v1 `src/` monolith.** A Reborn feature crosses `product_workflow → composition → webui_v2 → runtime/serve → frontend`; the binary entry point is `crates/ironclaw_reborn_cli` (`reborn_cli`), **not** `src/main.rs`. Start from the `reborn-feature` skill — it maps those layers so you wire a feature in one pass instead of layer-by-layer. + +`src/` is the **v1 monolith**, being retired under the roadmap's "Clean up old architecture." Maintain existing v1 behavior there when a bug requires it, but **do not build new features into `src/`** — they belong Reborn-side. The detailed `src/` layout in "Project Structure" below documents v1 for maintenance, not as the default place to add code. + ## Build & Test ```bash diff --git a/scripts/codebase-graph.sh b/scripts/codebase-graph.sh new file mode 100755 index 00000000000..66a93050f1d --- /dev/null +++ b/scripts/codebase-graph.sh @@ -0,0 +1,59 @@ +#!/usr/bin/env bash +# +# codebase-graph.sh — freshness status for the codebase-memory knowledge graph. +# +# The graph (.codebase-memory/graph.db.zst) is a git-ignored build artifact indexed +# by the codebase-memory MCP server. This script only INSPECTS freshness; the actual +# (re)indexing is done through the MCP tools invoked by an agent: +# - build: index_repository(repo_path=".") +# - delta/impact: detect_changes(since="") +# +# Usage: bash scripts/codebase-graph.sh [status] +# +set -euo pipefail + +repo_root="$(git rev-parse --show-toplevel)" +artifact="$repo_root/.codebase-memory/artifact.json" + +read_field() { python3 -c "import json,sys;print(json.load(open('$artifact')).get('$1','?'))"; } + +status() { + if [ ! -f "$artifact" ]; then + echo "graph: MISSING (no .codebase-memory/artifact.json)" + echo "action: build it once — call index_repository(repo_path=\".\") via the codebase-memory MCP" + return 2 + fi + + local indexed head nodes edges when + indexed="$(read_field commit)" + nodes="$(read_field nodes)" + edges="$(read_field edges)" + when="$(read_field indexed_at)" + head="$(git rev-parse HEAD)" + + echo "graph: indexed @ ${indexed:0:9} (${nodes} nodes / ${edges} edges, ${when})" + echo "HEAD: ${head:0:9}" + + if [ "$indexed" = "$head" ]; then + echo "status: FRESH (matches HEAD)" + return 0 + fi + + if git merge-base --is-ancestor "$indexed" HEAD 2>/dev/null; then + local n + n="$(git rev-list --count "$indexed"..HEAD 2>/dev/null || echo '?')" + echo "status: STALE — ${n} commit(s) behind HEAD" + echo "action: delta — detect_changes(since=\"$indexed\") (changed symbols + blast radius)" + echo " or full refresh — index_repository(repo_path=\".\")" + return 1 + fi + + echo "status: DIVERGED — indexed commit is not in current history (rebase/force-push?)" + echo "action: full re-index — index_repository(repo_path=\".\")" + return 1 +} + +case "${1:-status}" in + status) status ;; + *) echo "usage: $0 [status]" >&2; exit 64 ;; +esac diff --git a/scripts/dev-setup.sh b/scripts/dev-setup.sh index e543183541d..dbd6e81b318 100755 --- a/scripts/dev-setup.sh +++ b/scripts/dev-setup.sh @@ -63,6 +63,21 @@ else echo " Skipped: not a git repository" fi +echo "" +# Codebase knowledge graph (codebase-memory MCP) — powers agent code discovery. +# Single static binary, no deps, no API keys, 100% local. See CLAUDE.md -> "Code Discovery". +if command -v codebase-memory-mcp &>/dev/null; then + echo "[graph] codebase-memory-mcp found: $(command -v codebase-memory-mcp)" +else + echo "[graph] Installing codebase-memory-mcp (agent code-discovery graph)..." + if curl -fsSL https://raw.githubusercontent.com/DeusData/codebase-memory-mcp/main/install.sh | bash; then + echo " installed. The repo's .mcp.json wires it into Claude Code automatically." + else + echo " WARN: install failed — agents will fall back to grep." + echo " Install manually: https://github.com/DeusData/codebase-memory-mcp" + fi +fi + echo "" echo "=== Setup complete ===" echo "" From 7d4014d97c2ecba5cc87e7a7dc562c923083b752 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Thu, 2 Jul 2026 09:56:04 +0000 Subject: [PATCH 3/3] ci: normalize upstream import --- .gitignore | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/.gitignore b/.gitignore index b406454c85d..6412a8af0c8 100644 --- a/.gitignore +++ b/.gitignore @@ -87,3 +87,14 @@ tests/fixtures/llm_traces/live/*.log # Rebuilt from code via the codebase-memory MCP; per-environment, never committed. # See CLAUDE.md -> "Code Discovery". .codebase-memory/ + +# Downstream sync exception: nearai/main currently tracks these +# source/control assets even though broad local-ignore rules also +# match them. Keep the explicit exceptions in generated sync PRs so +# the tracked-ignored-files guard rejects only accidental local output. +!.claude/ +!.claude/** +!.codegraph/ +!.codegraph/.gitignore +!crates/ironclaw_webui_v2_static/static/dist/ +!crates/ironclaw_webui_v2_static/static/dist/**