diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 5b52654d..20cd7429 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -71,7 +71,7 @@ { "name": "nihil", "description": "Evidence-gated maintainability discipline in two layers. (1) A native plugin: four modes \u2014 /nihil:raze (root-authority, write-capable transformation of a repo you own; only secret-leak and catastrophic, unrecoverable commands are blocked) plus /nihil:review, /nihil:implement, /nihil:release with confidence scoring, scope control, and release gating \u2014 backed by five read-only review agents and PreToolUse/Stop hooks (a discipline aid, not a security boundary: heuristic command matching, fail-open; a secret / API-key brake is active in every mode). (2) A summonable pantheon of five first-principles dynamic workflows (/nihil orchestrator + /nihil-maat review, /nihil-odin research, /nihil-shiva deletion, /nihil-athena restructure), installed into .claude/workflows/ via /nihil:summon.", - "version": "0.3.0", + "version": "0.3.1", "source": "./plugins/nihil" }, { @@ -89,7 +89,7 @@ { "name": "safety-nets", "description": "Two deterministic Stop-event safety nets (no LLM) that keep a session honest: overclaim-net blocks a 'done/fixed/verified' claim made without a verifying command this turn; slnx-sync blocks unregistered .csproj in a .slnx, auto-skipping VMR-synced upstream forks (dotnet/dotnet codeflow / dotnet upstream remote) and honoring a .slnx-sync-ignore opt-out.", - "version": "0.1.0", + "version": "0.1.1", "source": "./plugins/safety-nets" }, { diff --git a/.coderabbit.yaml b/.coderabbit.yaml index 96a72d34..0037978f 100644 --- a/.coderabbit.yaml +++ b/.coderabbit.yaml @@ -24,6 +24,7 @@ reviews: changed_files_summary: true sequence_diagrams: false poem: false + enable_prompt_for_ai_agents: true auto_review: enabled: true drafts: false @@ -41,6 +42,15 @@ reviews: - "!**/*.min.css" - "!plugins/cc-plugin-eval/fixtures/**" # test scaffolding, not shipped path_instructions: + - path: "**" + instructions: | + AI reviewer coordination. Comments from Codex, Claude, Copilot, or CodeRabbit are all valid + implementation leads. If a review point is outside the GitHub diff range, do not discard it: + translate it into a concrete task against the current source tree by naming the best target + path plus line, symbol, or search phrase. If no exact anchor exists, name the files that must + be inspected and the behavior to change. Multiple agents may work the same reviewer point in + parallel; overlapping findings are signal, not duplication. Do not transcribe reviewer text + into repo files as the fix: forward the finding to the source and change the root cause. - path: "plugins/*/skills/**/SKILL.md" instructions: | Skill definitions. The `description` frontmatter is the ONLY trigger mechanism — check it diff --git a/.cursor/rules/project-config.mdc b/.cursor/rules/project-config.mdc index defa7f41..ec074ce4 100644 --- a/.cursor/rules/project-config.mdc +++ b/.cursor/rules/project-config.mdc @@ -1,205 +1,107 @@ --- -description: ancplua-claude-plugins agent operating guide (hand-maintained copy of CLAUDE.md) +description: ancplua-claude-plugins contributor/dev guide (hand-maintained copy of CLAUDE.md) globs: - "**/*" alwaysApply: true --- -> [!IMPORTANT] -> **Hand-maintained copy.** The canonical agent operating guide is **`CLAUDE.md`** -> at the repo root; this file is a hand-maintained copy of it — keep it in sync by -> hand when CLAUDE.md changes. `.github/copilot-instructions.md` is a separate -> hand-maintained document. +# ancplua-claude-plugins ---- - -# Agent Operating Guide — ancplua-claude-plugins - -> Sources: Boris Cherny (@bcherny) + Thariq (@trq212), 2026-04-16; Claude Opus 4.7 System Card (Anthropic, 2026-04-16, -> 232 pp.) — the most recent published card. Opus 4.8 builds directly on 4.7, so its safety and behavior findings still apply. -> All System Card citations are to the April 16 2026 (4.7) edition; page numbers are stable. No 4.8 System Card has been published. - ---- - -## Model & standard configuration - -| Setting | Value | Why | -|--------------------------------------------|---------------------|----------------------------------------------------------------------------------| -| Model | Claude Opus 4.8 | — | -| `effortLevel` / `CLAUDE_CODE_EFFORT_LEVEL` | `max` | System Card p.192: "standard configuration: **adaptive thinking at max effort**" | -| `defaultMode` | `bypassPermissions` | Standing-authority repos; see Permission model below | -| `autoCompactEnabled` | `false` | Every `/compact` is intentional — never silent | -| `alwaysThinkingEnabled` | `true` | Security property, not just quality (see Prompt injection below) | -| `agentPushNotifEnabled` + `voiceEnabled` | `true` | Leave long tasks running; recaps tell you what shipped and what's next | +A Claude Code **plugin marketplace** — a git repo that distributes plugins +(commands, agents, skills, hooks, MCP servers) through a `marketplace.json` +manifest, mirrored for Codex CLI. The plugin inventory and install steps live in +[`README.md`](README.md); this file is the **contributor / dev guide**. -**Adaptive thinking** means the model determines per-query reasoning depth dynamically. -`max` sets the ceiling; the model may use much less on a trivial query. -> System Card p.53: "the level of effort is **dynamically determined for each query by the model**" - -**When to drop effort**: only for latency-constrained evals with wall-clock timeouts (the card ran Terminal-Bench 2.0 -with thinking disabled for this reason, p.193). Interactive Claude Code sessions are not latency-constrained. Drop to -`high` for genuinely trivial, well-bounded edits (rename, format, single-file fix). - ---- - -## Multi-agent architecture +## Repo structure ``` -Lead (this session) - ├── continue — same task; every token in window still load-bearing - ├── rewind (esc-esc) — wrong path; keep file reads, drop failed attempt - ├── /compact — mid-task bloat; steer the summary toward next direction - ├── /clear + brief — new task; hand-written context only, zero rot - └── spawn subagent ────→ Task tool: own fresh window, returns conclusion only +.claude-plugin/marketplace.json # Claude Code marketplace manifest (lists every plugin) +.agents/plugins/marketplace.json # Codex CLI marketplace manifest (mirror) +plugins// # one directory per plugin (15 today) +.github/workflows/ci.yml # JSON + manifest validation — the only gate +automation/ # cron cleanup runner — IGNORE for dev (see automation/policy.md) +README.md # user-facing: what each plugin does + how to install ``` -**Spawn a subagent when** the next chunk will produce intermediate noise (file reads, greps, -dead ends) that the Lead will never need again. Only the report returns; exploration noise -is garbage-collected when the subagent exits. - -**Don't spawn when** the intermediate output must be woven into ongoing reasoning — -use `/compact` or continue instead. - -Mental test: *Will I need this tool output again, or just the conclusion?* - -### Subagent patterns for this repo - -| Task | Agent type | -|------------------------------------------------------|-----------------------| -| Exploring a plugin's codebase for structure/patterns | Explore | -| Verifying output against a spec or test suite | general-purpose | -| Writing docs from a git diff | general-purpose | -| Reviewing upstream dependency changes | general-purpose | -| Security review of pending changes | security-review skill | - -Context rot threshold for the 1M window: **~300–400k tokens** — task-dependent, not a -hard rule. File reads are the heavy hitter. Compact proactively, before the cliff edge. - ---- - -## Known failure modes (System Card §6.2.1, p.95) - -These are documented pilot-use findings for Opus 4.7 (which 4.8 builds on) in Claude Code and similar scaffolds. -They are not hypothetical — account for them before claiming a task complete. - -| Failure mode | Mitigation | -|-----------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------------| -| Claims to have succeeded at a task it did not fully complete | Run the thing; don't accept a diff as evidence of working behavior | -| Misreports that test failures it *caused* were preexisting | Baseline tests before touching code; compare results against that baseline | -| Overconfident initial root-cause assessment | Engineering principle #10: understand the "why" before moving on; don't trial-and-error | -| Unnecessary follow-up questions on an already-clear request | Scope explicitly; use a bounded `/go`-style skill to eliminate ambiguity | -| Unexpected file deletion when starting a new technical effort (especially in temp dirs) | Avoid unstructured temp-dir workflows; stage deletions explicitly | - -**Positive calibration** (System Card p.91, p.120): -Opus 4.7 takes destructive actions at a "much lower rate than Opus or Sonnet 4.6." Internal pilot reports describe it -as "significantly more conservative." Undisclosed destructive actions: **3 cases** for Opus 4.7 vs. **24 for Opus 4.6**. -Better, not zero — verification catches the residual. - ---- - -## Prompt injection posture - -`alwaysThinkingEnabled: true` is a **security property** in agentic contexts that read -untrusted content (file reads, web fetches, MCP tool output from external services). - -From System Card §5.2.2 (adaptive Shade attacker): - -| Surface | Thinking | 1-attempt ASR | 200-attempt ASR | -|---------------------------------|---------------|---------------|-----------------| -| Coding (p.85) | on (adaptive) | **2.34%** | 60.0% | -| Coding (p.85) | off | 10.43% | 92.5% | -| Browser use + safeguards (p.88) | on (adaptive) | **0.00%** | 0.00% | -| Browser use + safeguards (p.88) | off | 0.00% | 0.00% | +## Plugin layout -At k=100 on the ART benchmark (p.83): 4.8% with adaptive thinking vs. 6.0% without. +Each plugin under `plugins//`: -Disabling thinking (e.g. for speed) materially increases injection risk when processing -external content. Don't disable it in agentic sessions touching files or the web. - ---- - -## Permission model - -`bypassPermissions` applies to standing-authority repos only: - -- `/Users/ancplua/framework/` — ANcpLua.Agents, ANcpLua.Analyzers, ANcpLua.NET.Sdk, ANcpLua.Roslyn.Utilities, renovate-config -- `/Users/ancplua/qyl/` -- `/Users/ancplua/marketplaces/ancplua-claude-plugins/` - -Outside these trees: ask before push, tag, or destructive git operations. - -**Explicit `deny` overrides survive bypass**: `Git stash pop`, `Git stash` — these still prompt regardless of mode. - -**`/fewer-permission-prompts`** — after any tool-heavy session, run this skill to harvest -bash + MCP commands for the allowlist. The allow list is currently empty; each harvested -entry reduces friction without compromising the deny overrides. - ---- - -## Verification contract - -> "4.7 is 2-3× more capable than 4.6 — verification is what keeps that velocity from compounding into invisible -> regressions." -> — Boris Cherny, 2026-04-16 - -| Task type | Verify via | -|---------------------|-----------------------------------------------------------------------------------| -| Plugin / CLI / .NET | Run the test suite; `dotnet run` and hit endpoints; check output against baseline | -| Frontend | `mcp__claude-in-chrome__*`: navigate, screenshot, read console + network | -| Desktop app | `mcp__computer-use__*`: screenshot and interact within tier rules | - -Pattern: ` /go` — a per-project skill that tests end-to-end, runs `/simplify`, opens a PR. -Without a `/go` skill: explicitly state "cannot verify" rather than claiming the task works. - ---- - -## Engineering principles - -Operative rules of thumb when reasoning about code changes in this repo: - -| Situation | Principle | -|---------------------------------|--------------------------------------------------------------------------| -| Starting a new feature | Solve the problem, not necessarily with code. Work backward from outcome. | -| Evaluating approaches | No "best" — only trade-offs. Document what you're optimizing for. | -| Adding code | Less code is better code. Justify every line. Remove unused now. | -| Adding abstraction | Prefer boring. Abstraction should hide complexity, not create it. | -| Encountering a bug | Fix the root cause. Understand the class of bug, not one instance. | -| Something unexpected | Don't trial-and-error to green. What assumption was wrong? | -| Making a non-obvious decision | Document the *why*. ADR / RFC / design doc, not a commit message. | -| An error occurs | Fail loudly. Silent failures are debugging nightmares. | -| Debugging / optimizing | Change one thing at a time. Measure → change → measure again. | -| Designing an API | Easy to use correctly, hard to misuse. Invalid states unrepresentable. | - ---- - -## Focus vs. verbose - -| Mode | What's visible | Use for | -|-----------------------|------------------------------------|----------------------------------------------------| -| **Verbose** (current) | Every tool call + thinking summary | Architecture, security-adjacent, anything drifting | -| `/focus` | Final result only | Trusted task types: refactors, lint, doc rewrites | +``` +.claude-plugin/plugin.json # required: name, version, description +commands/ agents/ skills/ hooks/ # all optional +README.md # per-plugin docs +``` -`verbose: true` + `showThinkingSummaries: true` catches drift early but costs cognitive load. -A/B `/focus` on tasks where you trust the agent; stay verbose where you don't. +**Manifest path rule:** component/source paths in `plugin.json` and +`marketplace.json` must start with `./` (agent paths also end in `.md`). A bare +path fails recent CLIs with `: Invalid input`. + +## Adding or changing a plugin + +1. Create/edit `plugins//` with a valid `.claude-plugin/plugin.json`. +2. Register it in **both** manifests — `.claude-plugin/marketplace.json` (Claude) + and `.agents/plugins/marketplace.json` (Codex). Each entry needs `name`, + `source` (`./plugins/`), `description`. +3. **Bump the plugin's `version`** on any behavioral change (see next section). +4. Validate, commit, open a PR. + +## Version bumps reach the cache; edits don't + +Installed plugins run from `~/.claude/plugins/cache////`, +pinned to the installed commit — **not** from this working tree. `/plugin update` +re-syncs the cache **only when the version string changes**; on an equal version it +reports *"already at the latest"* and keeps running the old code. A fix committed +**without a version bump never reaches the running plugin.** + +| Rule | Why | +|------|-----| +| Bump `plugin.json` `version` (+ matching `marketplace.json` entries) on any change under `hooks/ scripts/ agents/ commands/ skills/` | The version string is the sole update signal. | +| After it lands: `commit → push → /plugin update → restart Claude Code` | Hooks load only at session start. | +| To see what actually runs: `diff -rq` / `md5` the cache install vs the source | Editing the repo proves nothing about the executing copy. | +| Don't hand-patch the cache as the fix | It's a managed artifact the next reinstall clobbers. Durable = bump + publish. | + +**Hook-symptom triage:** `Not logged in · Please run /login` is **session auth** +(run `/login`), never a hook. "haiku" errors come only from `type: prompt` hooks or +the background model — a `type: command` hook (plain bash/python, no model call) +cannot produce one. + +## Validate + +CI (`.github/workflows/ci.yml`) is the gate; run the same checks locally: + +```bash +# every JSON parses +find . -name '*.json' -not -path '*/.git/*' -print0 | xargs -0 -I{} jq . {} >/dev/null +# every plugin.json has name/version/description +find . -path '*/.claude-plugin/plugin.json' -print0 \ + | xargs -0 -I{} jq -e '.name and .version and .description' {} >/dev/null +``` ---- +PRs are reviewed by three independent bots: Claude, Copilot, CodeRabbit. Any of +them may produce or implement fixes. If a review comment cannot be posted inline +because it is outside the PR diff, translate it into a concrete source task: +target file, line or symbol when possible, plus the behavior to change. Do not +copy reviewer prose into the repo as the fix; apply the root change in source. +Parallel agents may work the same review point. Treat overlapping findings as a +strong signal and converge on the common root cause. -## Automation runner — `automation/` +## Conventions -The `automation/` directory contains the cron-triggered runner that keeps -Alexander's working repos clean (the 5 listed in `automation/policy.md`). +- Less code is better; remove unused now. Prefer boring over clever. +- Fix the root cause, not one instance — understand *why* before changing. +- Fail loudly; silent failures are debugging nightmares. +- Document non-obvious decisions in the PR description, not just the commit. -**Rule for dev work:** ignore `automation/` entirely. The cron output -artifact `automation/RUN-LOG.md` is auto-managed (FIFO cap 10, written by -`automation/scripts/log-entry.sh`) — never hand-edit it. The 8 templates in -`automation/templates/` are pasted into the schedule UI, not invoked from -interactive sessions. +## Keep the mirrors in sync -For policy details, see [`automation/policy.md`](automation/policy.md). -For the runner overview, see [`automation/README.md`](automation/README.md). +This file is a copy. The canonical guide is `CLAUDE.md` at the repo root; its +twin copy is `.windsurf/rules/project-config.md`. Edit all three bodies together +(each keeps its own frontmatter). `.github/copilot-instructions.md` is separate — +update it only if the change is relevant there. diff --git a/.github/copilot-instructions.md b/.github/copilot-instructions.md index b17bdec9..a3c417ef 100644 --- a/.github/copilot-instructions.md +++ b/.github/copilot-instructions.md @@ -352,6 +352,13 @@ AIs coordinate through **shared files**, not real-time communication: Each AI does its own complete review. Overlapping findings indicate high confidence issues. +Outside-diff review comments are still valid work. When GitHub cannot anchor a +comment inline, convert it into a concrete source task: target file, line or +symbol when possible, plus the behavior to change. Do not paste reviewer prose +into repository files as the fix; forward the finding to the source and change +the root cause. Multiple agents may implement the same reviewer point in +parallel; the common denominator between their fixes is the signal to preserve. + ### 9.6 Plugin Review Scope All AIs review the same things in this repo: diff --git a/.windsurf/rules/project-config.md b/.windsurf/rules/project-config.md index b3d2896b..4b2c9f1c 100644 --- a/.windsurf/rules/project-config.md +++ b/.windsurf/rules/project-config.md @@ -3,200 +3,102 @@ trigger: always_on --- -> [!IMPORTANT] -> **Hand-maintained copy.** The canonical agent operating guide is **`CLAUDE.md`** -> at the repo root; this file is a hand-maintained copy of it — keep it in sync by -> hand when CLAUDE.md changes. `.github/copilot-instructions.md` is a separate -> hand-maintained document. +# ancplua-claude-plugins ---- - -# Agent Operating Guide — ancplua-claude-plugins - -> Sources: Boris Cherny (@bcherny) + Thariq (@trq212), 2026-04-16; Claude Opus 4.7 System Card (Anthropic, 2026-04-16, -> 232 pp.) — the most recent published card. Opus 4.8 builds directly on 4.7, so its safety and behavior findings still apply. -> All System Card citations are to the April 16 2026 (4.7) edition; page numbers are stable. No 4.8 System Card has been published. - ---- - -## Model & standard configuration - -| Setting | Value | Why | -|--------------------------------------------|---------------------|----------------------------------------------------------------------------------| -| Model | Claude Opus 4.8 | — | -| `effortLevel` / `CLAUDE_CODE_EFFORT_LEVEL` | `max` | System Card p.192: "standard configuration: **adaptive thinking at max effort**" | -| `defaultMode` | `bypassPermissions` | Standing-authority repos; see Permission model below | -| `autoCompactEnabled` | `false` | Every `/compact` is intentional — never silent | -| `alwaysThinkingEnabled` | `true` | Security property, not just quality (see Prompt injection below) | -| `agentPushNotifEnabled` + `voiceEnabled` | `true` | Leave long tasks running; recaps tell you what shipped and what's next | +A Claude Code **plugin marketplace** — a git repo that distributes plugins +(commands, agents, skills, hooks, MCP servers) through a `marketplace.json` +manifest, mirrored for Codex CLI. The plugin inventory and install steps live in +[`README.md`](README.md); this file is the **contributor / dev guide**. -**Adaptive thinking** means the model determines per-query reasoning depth dynamically. -`max` sets the ceiling; the model may use much less on a trivial query. -> System Card p.53: "the level of effort is **dynamically determined for each query by the model**" - -**When to drop effort**: only for latency-constrained evals with wall-clock timeouts (the card ran Terminal-Bench 2.0 -with thinking disabled for this reason, p.193). Interactive Claude Code sessions are not latency-constrained. Drop to -`high` for genuinely trivial, well-bounded edits (rename, format, single-file fix). - ---- - -## Multi-agent architecture +## Repo structure ``` -Lead (this session) - ├── continue — same task; every token in window still load-bearing - ├── rewind (esc-esc) — wrong path; keep file reads, drop failed attempt - ├── /compact — mid-task bloat; steer the summary toward next direction - ├── /clear + brief — new task; hand-written context only, zero rot - └── spawn subagent ────→ Task tool: own fresh window, returns conclusion only +.claude-plugin/marketplace.json # Claude Code marketplace manifest (lists every plugin) +.agents/plugins/marketplace.json # Codex CLI marketplace manifest (mirror) +plugins// # one directory per plugin (15 today) +.github/workflows/ci.yml # JSON + manifest validation — the only gate +automation/ # cron cleanup runner — IGNORE for dev (see automation/policy.md) +README.md # user-facing: what each plugin does + how to install ``` -**Spawn a subagent when** the next chunk will produce intermediate noise (file reads, greps, -dead ends) that the Lead will never need again. Only the report returns; exploration noise -is garbage-collected when the subagent exits. - -**Don't spawn when** the intermediate output must be woven into ongoing reasoning — -use `/compact` or continue instead. - -Mental test: *Will I need this tool output again, or just the conclusion?* - -### Subagent patterns for this repo - -| Task | Agent type | -|------------------------------------------------------|-----------------------| -| Exploring a plugin's codebase for structure/patterns | Explore | -| Verifying output against a spec or test suite | general-purpose | -| Writing docs from a git diff | general-purpose | -| Reviewing upstream dependency changes | general-purpose | -| Security review of pending changes | security-review skill | - -Context rot threshold for the 1M window: **~300–400k tokens** — task-dependent, not a -hard rule. File reads are the heavy hitter. Compact proactively, before the cliff edge. - ---- - -## Known failure modes (System Card §6.2.1, p.95) - -These are documented pilot-use findings for Opus 4.7 (which 4.8 builds on) in Claude Code and similar scaffolds. -They are not hypothetical — account for them before claiming a task complete. - -| Failure mode | Mitigation | -|-----------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------------| -| Claims to have succeeded at a task it did not fully complete | Run the thing; don't accept a diff as evidence of working behavior | -| Misreports that test failures it *caused* were preexisting | Baseline tests before touching code; compare results against that baseline | -| Overconfident initial root-cause assessment | Engineering principle #10: understand the "why" before moving on; don't trial-and-error | -| Unnecessary follow-up questions on an already-clear request | Scope explicitly; use a bounded `/go`-style skill to eliminate ambiguity | -| Unexpected file deletion when starting a new technical effort (especially in temp dirs) | Avoid unstructured temp-dir workflows; stage deletions explicitly | - -**Positive calibration** (System Card p.91, p.120): -Opus 4.7 takes destructive actions at a "much lower rate than Opus or Sonnet 4.6." Internal pilot reports describe it -as "significantly more conservative." Undisclosed destructive actions: **3 cases** for Opus 4.7 vs. **24 for Opus 4.6**. -Better, not zero — verification catches the residual. - ---- - -## Prompt injection posture - -`alwaysThinkingEnabled: true` is a **security property** in agentic contexts that read -untrusted content (file reads, web fetches, MCP tool output from external services). - -From System Card §5.2.2 (adaptive Shade attacker): - -| Surface | Thinking | 1-attempt ASR | 200-attempt ASR | -|---------------------------------|---------------|---------------|-----------------| -| Coding (p.85) | on (adaptive) | **2.34%** | 60.0% | -| Coding (p.85) | off | 10.43% | 92.5% | -| Browser use + safeguards (p.88) | on (adaptive) | **0.00%** | 0.00% | -| Browser use + safeguards (p.88) | off | 0.00% | 0.00% | +## Plugin layout -At k=100 on the ART benchmark (p.83): 4.8% with adaptive thinking vs. 6.0% without. +Each plugin under `plugins//`: -Disabling thinking (e.g. for speed) materially increases injection risk when processing -external content. Don't disable it in agentic sessions touching files or the web. - ---- - -## Permission model - -`bypassPermissions` applies to standing-authority repos only: - -- `/Users/ancplua/framework/` — ANcpLua.Agents, ANcpLua.Analyzers, ANcpLua.NET.Sdk, ANcpLua.Roslyn.Utilities, renovate-config -- `/Users/ancplua/qyl/` -- `/Users/ancplua/marketplaces/ancplua-claude-plugins/` - -Outside these trees: ask before push, tag, or destructive git operations. - -**Explicit `deny` overrides survive bypass**: `Git stash pop`, `Git stash` — these still prompt regardless of mode. - -**`/fewer-permission-prompts`** — after any tool-heavy session, run this skill to harvest -bash + MCP commands for the allowlist. The allow list is currently empty; each harvested -entry reduces friction without compromising the deny overrides. - ---- - -## Verification contract - -> "4.7 is 2-3× more capable than 4.6 — verification is what keeps that velocity from compounding into invisible -> regressions." -> — Boris Cherny, 2026-04-16 - -| Task type | Verify via | -|---------------------|-----------------------------------------------------------------------------------| -| Plugin / CLI / .NET | Run the test suite; `dotnet run` and hit endpoints; check output against baseline | -| Frontend | `mcp__claude-in-chrome__*`: navigate, screenshot, read console + network | -| Desktop app | `mcp__computer-use__*`: screenshot and interact within tier rules | - -Pattern: ` /go` — a per-project skill that tests end-to-end, runs `/simplify`, opens a PR. -Without a `/go` skill: explicitly state "cannot verify" rather than claiming the task works. - ---- - -## Engineering principles - -Operative rules of thumb when reasoning about code changes in this repo: - -| Situation | Principle | -|---------------------------------|--------------------------------------------------------------------------| -| Starting a new feature | Solve the problem, not necessarily with code. Work backward from outcome. | -| Evaluating approaches | No "best" — only trade-offs. Document what you're optimizing for. | -| Adding code | Less code is better code. Justify every line. Remove unused now. | -| Adding abstraction | Prefer boring. Abstraction should hide complexity, not create it. | -| Encountering a bug | Fix the root cause. Understand the class of bug, not one instance. | -| Something unexpected | Don't trial-and-error to green. What assumption was wrong? | -| Making a non-obvious decision | Document the *why*. ADR / RFC / design doc, not a commit message. | -| An error occurs | Fail loudly. Silent failures are debugging nightmares. | -| Debugging / optimizing | Change one thing at a time. Measure → change → measure again. | -| Designing an API | Easy to use correctly, hard to misuse. Invalid states unrepresentable. | - ---- - -## Focus vs. verbose - -| Mode | What's visible | Use for | -|-----------------------|------------------------------------|----------------------------------------------------| -| **Verbose** (current) | Every tool call + thinking summary | Architecture, security-adjacent, anything drifting | -| `/focus` | Final result only | Trusted task types: refactors, lint, doc rewrites | +``` +.claude-plugin/plugin.json # required: name, version, description +commands/ agents/ skills/ hooks/ # all optional +README.md # per-plugin docs +``` -`verbose: true` + `showThinkingSummaries: true` catches drift early but costs cognitive load. -A/B `/focus` on tasks where you trust the agent; stay verbose where you don't. +**Manifest path rule:** component/source paths in `plugin.json` and +`marketplace.json` must start with `./` (agent paths also end in `.md`). A bare +path fails recent CLIs with `: Invalid input`. + +## Adding or changing a plugin + +1. Create/edit `plugins//` with a valid `.claude-plugin/plugin.json`. +2. Register it in **both** manifests — `.claude-plugin/marketplace.json` (Claude) + and `.agents/plugins/marketplace.json` (Codex). Each entry needs `name`, + `source` (`./plugins/`), `description`. +3. **Bump the plugin's `version`** on any behavioral change (see next section). +4. Validate, commit, open a PR. + +## Version bumps reach the cache; edits don't + +Installed plugins run from `~/.claude/plugins/cache////`, +pinned to the installed commit — **not** from this working tree. `/plugin update` +re-syncs the cache **only when the version string changes**; on an equal version it +reports *"already at the latest"* and keeps running the old code. A fix committed +**without a version bump never reaches the running plugin.** + +| Rule | Why | +|------|-----| +| Bump `plugin.json` `version` (+ matching `marketplace.json` entries) on any change under `hooks/ scripts/ agents/ commands/ skills/` | The version string is the sole update signal. | +| After it lands: `commit → push → /plugin update → restart Claude Code` | Hooks load only at session start. | +| To see what actually runs: `diff -rq` / `md5` the cache install vs the source | Editing the repo proves nothing about the executing copy. | +| Don't hand-patch the cache as the fix | It's a managed artifact the next reinstall clobbers. Durable = bump + publish. | + +**Hook-symptom triage:** `Not logged in · Please run /login` is **session auth** +(run `/login`), never a hook. "haiku" errors come only from `type: prompt` hooks or +the background model — a `type: command` hook (plain bash/python, no model call) +cannot produce one. + +## Validate + +CI (`.github/workflows/ci.yml`) is the gate; run the same checks locally: + +```bash +# every JSON parses +find . -name '*.json' -not -path '*/.git/*' -print0 | xargs -0 -I{} jq . {} >/dev/null +# every plugin.json has name/version/description +find . -path '*/.claude-plugin/plugin.json' -print0 \ + | xargs -0 -I{} jq -e '.name and .version and .description' {} >/dev/null +``` ---- +PRs are reviewed by three independent bots: Claude, Copilot, CodeRabbit. Any of +them may produce or implement fixes. If a review comment cannot be posted inline +because it is outside the PR diff, translate it into a concrete source task: +target file, line or symbol when possible, plus the behavior to change. Do not +copy reviewer prose into the repo as the fix; apply the root change in source. +Parallel agents may work the same review point. Treat overlapping findings as a +strong signal and converge on the common root cause. -## Automation runner — `automation/` +## Conventions -The `automation/` directory contains the cron-triggered runner that keeps -Alexander's working repos clean (the 5 listed in `automation/policy.md`). +- Less code is better; remove unused now. Prefer boring over clever. +- Fix the root cause, not one instance — understand *why* before changing. +- Fail loudly; silent failures are debugging nightmares. +- Document non-obvious decisions in the PR description, not just the commit. -**Rule for dev work:** ignore `automation/` entirely. The cron output -artifact `automation/RUN-LOG.md` is auto-managed (FIFO cap 10, written by -`automation/scripts/log-entry.sh`) — never hand-edit it. The 8 templates in -`automation/templates/` are pasted into the schedule UI, not invoked from -interactive sessions. +## Keep the mirrors in sync -For policy details, see [`automation/policy.md`](automation/policy.md). -For the runner overview, see [`automation/README.md`](automation/README.md). +This file is a copy. The canonical guide is `CLAUDE.md` at the repo root; its +twin copy is `.cursor/rules/project-config.mdc`. Edit all three bodies together +(each keeps its own frontmatter). `.github/copilot-instructions.md` is separate — +update it only if the change is relevant there. diff --git a/AGENTS.md b/AGENTS.md deleted file mode 100644 index d9b327f1..00000000 --- a/AGENTS.md +++ /dev/null @@ -1,91 +0,0 @@ -# Agent Operating Guide - ancplua-claude-plugins - -This repo is a local plugin marketplace. Keep Claude Code source artifacts and -Codex artifacts side by side unless the task explicitly asks to remove one. - -## Canonical Surfaces - -- `CLAUDE.md` remains the Claude Code operating guide. -- `AGENTS.md` is the Codex operating guide. -- `.claude-plugin/marketplace.json` is the Claude marketplace catalog. -- `.agents/plugins/marketplace.json` is the Codex marketplace catalog. -- `plugins/*/.claude-plugin/plugin.json` is the Claude plugin manifest. -- `plugins/*/.codex-plugin/plugin.json` is the Codex plugin manifest. -- `plugins/*/skills/*/SKILL.md` are bundled skill entrypoints. -- `.codex/agents/*.toml` contains repo-local Codex custom agents converted - from plugin-local Claude agents. - -Do not hand-edit generated reports. Re-run the owning migration or validation -command and edit the source that generated the report. - -## Repository Hygiene - -- Inspect `git status --short` before edits. -- Stay on `main` unless the task explicitly needs a branch. -- Do not use `git reset`. -- Do not hand-edit `automation/RUN-LOG.md`; it is auto-managed. -- Ignore `automation/` for ordinary plugin development unless the task names it. - -## Codex Plugin Layout - -Codex plugins in this repo use: - -```text -plugins// - .codex-plugin/plugin.json - skills/ -``` - -The repo marketplace file is: - -```text -.agents/plugins/marketplace.json -``` - -Marketplace entries must keep `source.path` as `./plugins/`, include -`policy.installation`, include `policy.authentication`, and include `category`. - -## Migration Boundaries - -- Codex plugin manifests can package skills and supported companion manifests. -- Current Codex plugin validation rejects Claude-only manifest fields and this - repo does not package Claude hooks or Claude agents into Codex plugin - manifests. -- Claude slash commands migrated for Codex are represented as bundled skills - named `source-command-`. -- Claude subagents migrated for Codex are represented as repo-local custom - agents under `.codex/agents/` with plugin-prefixed names. -- Claude Teams API, `Task` tool, `SendMessage`, `TeamCreate`, `TeamDelete`, - slash-command placeholders, and hook semantics are not Codex equivalents. - Preserve those references as migration guidance unless the task explicitly - asks for a semantic rewrite. - -## Validation - -Use the narrowest validation that covers the change: - -```bash -python3 /Users/ancplua/.codex/skills/migrate-to-codex/scripts/migrate-to-codex.py --validate-target . -python3 /Users/ancplua/.codex/skills/.system/plugin-creator/scripts/validate_plugin.py plugins/ -``` - -For repo-wide plugin manifest changes, validate every plugin: - -```bash -for plugin in plugins/*; do - [ -d "$plugin" ] || continue - python3 /Users/ancplua/.codex/skills/.system/plugin-creator/scripts/validate_plugin.py "$plugin" -done -``` - -For JavaScript tooling in `cc-plugin-eval`, run the plugin's own package -scripts from `plugins/cc-plugin-eval` when those files change. - -## Code Quality - -- Prefer the existing plugin structure over adding another packaging layer. -- Keep Codex migration artifacts explicit instead of hiding behavior behind - compatibility wrappers. -- Tests must catch plausible future regressions. Do not add tests just to prove - a known migration output. -- Fail loudly when validation cannot prove the migrated artifact works. diff --git a/CLAUDE.md b/CLAUDE.md index f70630c2..bb64a20d 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1,201 +1,101 @@ -> [!IMPORTANT] -> **Canonical source.** This file (`CLAUDE.md`) is the hand-maintained agent -> operating guide for this repo. `.cursor/rules/project-config.mdc` and -> `.windsurf/rules/project-config.md` are hand-maintained copies — keep them in -> sync by hand when you change this file. `.github/copilot-instructions.md` is a -> **separate** hand-maintained document, not a copy of this one. +# ancplua-claude-plugins ---- +A Claude Code **plugin marketplace** — a git repo that distributes plugins +(commands, agents, skills, hooks, MCP servers) through a `marketplace.json` +manifest, mirrored for Codex CLI. The plugin inventory and install steps live in +[`README.md`](README.md); this file is the **contributor / dev guide**. -# Agent Operating Guide — ancplua-claude-plugins - -> Sources: Boris Cherny (@bcherny) + Thariq (@trq212), 2026-04-16; Claude Opus 4.7 System Card (Anthropic, 2026-04-16, -> 232 pp.) — the most recent published card. Opus 4.8 builds directly on 4.7, so its safety and behavior findings still apply. -> All System Card citations are to the April 16 2026 (4.7) edition; page numbers are stable. No 4.8 System Card has been published. - ---- - -## Model & standard configuration - -| Setting | Value | Why | -|--------------------------------------------|---------------------|----------------------------------------------------------------------------------| -| Model | Claude Opus 4.8 | — | -| `effortLevel` / `CLAUDE_CODE_EFFORT_LEVEL` | `max` | System Card p.192: "standard configuration: **adaptive thinking at max effort**" | -| `defaultMode` | `bypassPermissions` | Standing-authority repos; see Permission model below | -| `autoCompactEnabled` | `false` | Every `/compact` is intentional — never silent | -| `alwaysThinkingEnabled` | `true` | Security property, not just quality (see Prompt injection below) | -| `agentPushNotifEnabled` + `voiceEnabled` | `true` | Leave long tasks running; recaps tell you what shipped and what's next | - -**Adaptive thinking** means the model determines per-query reasoning depth dynamically. -`max` sets the ceiling; the model may use much less on a trivial query. -> System Card p.53: "the level of effort is **dynamically determined for each query by the model**" - -**When to drop effort**: only for latency-constrained evals with wall-clock timeouts (the card ran Terminal-Bench 2.0 -with thinking disabled for this reason, p.193). Interactive Claude Code sessions are not latency-constrained. Drop to -`high` for genuinely trivial, well-bounded edits (rename, format, single-file fix). - ---- - -## Multi-agent architecture +## Repo structure ``` -Lead (this session) - ├── continue — same task; every token in window still load-bearing - ├── rewind (esc-esc) — wrong path; keep file reads, drop failed attempt - ├── /compact — mid-task bloat; steer the summary toward next direction - ├── /clear + brief — new task; hand-written context only, zero rot - └── spawn subagent ────→ Task tool: own fresh window, returns conclusion only +.claude-plugin/marketplace.json # Claude Code marketplace manifest (lists every plugin) +.agents/plugins/marketplace.json # Codex CLI marketplace manifest (mirror) +plugins// # one directory per plugin (15 today) +.github/workflows/ci.yml # JSON + manifest validation — the only gate +automation/ # cron cleanup runner — IGNORE for dev (see automation/policy.md) +README.md # user-facing: what each plugin does + how to install ``` -**Spawn a subagent when** the next chunk will produce intermediate noise (file reads, greps, -dead ends) that the Lead will never need again. Only the report returns; exploration noise -is garbage-collected when the subagent exits. - -**Don't spawn when** the intermediate output must be woven into ongoing reasoning — -use `/compact` or continue instead. - -Mental test: *Will I need this tool output again, or just the conclusion?* - -### Subagent patterns for this repo - -| Task | Agent type | -|------------------------------------------------------|-----------------------| -| Exploring a plugin's codebase for structure/patterns | Explore | -| Verifying output against a spec or test suite | general-purpose | -| Writing docs from a git diff | general-purpose | -| Reviewing upstream dependency changes | general-purpose | -| Security review of pending changes | security-review skill | - -Context rot threshold for the 1M window: **~300–400k tokens** — task-dependent, not a -hard rule. File reads are the heavy hitter. Compact proactively, before the cliff edge. - ---- - -## Known failure modes (System Card §6.2.1, p.95) - -These are documented pilot-use findings for Opus 4.7 (which 4.8 builds on) in Claude Code and similar scaffolds. -They are not hypothetical — account for them before claiming a task complete. - -| Failure mode | Mitigation | -|-----------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------------| -| Claims to have succeeded at a task it did not fully complete | Run the thing; don't accept a diff as evidence of working behavior | -| Misreports that test failures it *caused* were preexisting | Baseline tests before touching code; compare results against that baseline | -| Overconfident initial root-cause assessment | Engineering principle #10: understand the "why" before moving on; don't trial-and-error | -| Unnecessary follow-up questions on an already-clear request | Scope explicitly; use a bounded `/go`-style skill to eliminate ambiguity | -| Unexpected file deletion when starting a new technical effort (especially in temp dirs) | Avoid unstructured temp-dir workflows; stage deletions explicitly | - -**Positive calibration** (System Card p.91, p.120): -Opus 4.7 takes destructive actions at a "much lower rate than Opus or Sonnet 4.6." Internal pilot reports describe it -as "significantly more conservative." Undisclosed destructive actions: **3 cases** for Opus 4.7 vs. **24 for Opus 4.6**. -Better, not zero — verification catches the residual. - ---- - -## Prompt injection posture - -`alwaysThinkingEnabled: true` is a **security property** in agentic contexts that read -untrusted content (file reads, web fetches, MCP tool output from external services). - -From System Card §5.2.2 (adaptive Shade attacker): - -| Surface | Thinking | 1-attempt ASR | 200-attempt ASR | -|---------------------------------|---------------|---------------|-----------------| -| Coding (p.85) | on (adaptive) | **2.34%** | 60.0% | -| Coding (p.85) | off | 10.43% | 92.5% | -| Browser use + safeguards (p.88) | on (adaptive) | **0.00%** | 0.00% | -| Browser use + safeguards (p.88) | off | 0.00% | 0.00% | +## Plugin layout -At k=100 on the ART benchmark (p.83): 4.8% with adaptive thinking vs. 6.0% without. +Each plugin under `plugins//`: -Disabling thinking (e.g. for speed) materially increases injection risk when processing -external content. Don't disable it in agentic sessions touching files or the web. - ---- - -## Permission model - -`bypassPermissions` applies to standing-authority repos only: - -- `/Users/ancplua/framework/` — ANcpLua.Agents, ANcpLua.Analyzers, ANcpLua.NET.Sdk, ANcpLua.Roslyn.Utilities, renovate-config -- `/Users/ancplua/qyl/` -- `/Users/ancplua/marketplaces/ancplua-claude-plugins/` - -Outside these trees: ask before push, tag, or destructive git operations. - -**Explicit `deny` overrides survive bypass**: `Git stash pop`, `Git stash` — these still prompt regardless of mode. - -**`/fewer-permission-prompts`** — after any tool-heavy session, run this skill to harvest -bash + MCP commands for the allowlist. The allow list is currently empty; each harvested -entry reduces friction without compromising the deny overrides. - ---- - -## Verification contract - -> "4.7 is 2-3× more capable than 4.6 — verification is what keeps that velocity from compounding into invisible -> regressions." -> — Boris Cherny, 2026-04-16 - -| Task type | Verify via | -|---------------------|-----------------------------------------------------------------------------------| -| Plugin / CLI / .NET | Run the test suite; `dotnet run` and hit endpoints; check output against baseline | -| Frontend | `mcp__claude-in-chrome__*`: navigate, screenshot, read console + network | -| Desktop app | `mcp__computer-use__*`: screenshot and interact within tier rules | - -Pattern: ` /go` — a per-project skill that tests end-to-end, runs `/simplify`, opens a PR. -Without a `/go` skill: explicitly state "cannot verify" rather than claiming the task works. - ---- - -## Engineering principles - -Operative rules of thumb when reasoning about code changes in this repo: - -| Situation | Principle | -|---------------------------------|--------------------------------------------------------------------------| -| Starting a new feature | Solve the problem, not necessarily with code. Work backward from outcome. | -| Evaluating approaches | No "best" — only trade-offs. Document what you're optimizing for. | -| Adding code | Less code is better code. Justify every line. Remove unused now. | -| Adding abstraction | Prefer boring. Abstraction should hide complexity, not create it. | -| Encountering a bug | Fix the root cause. Understand the class of bug, not one instance. | -| Something unexpected | Don't trial-and-error to green. What assumption was wrong? | -| Making a non-obvious decision | Document the *why*. ADR / RFC / design doc, not a commit message. | -| An error occurs | Fail loudly. Silent failures are debugging nightmares. | -| Debugging / optimizing | Change one thing at a time. Measure → change → measure again. | -| Designing an API | Easy to use correctly, hard to misuse. Invalid states unrepresentable. | - ---- - -## Focus vs. verbose - -| Mode | What's visible | Use for | -|-----------------------|------------------------------------|----------------------------------------------------| -| **Verbose** (current) | Every tool call + thinking summary | Architecture, security-adjacent, anything drifting | -| `/focus` | Final result only | Trusted task types: refactors, lint, doc rewrites | +``` +.claude-plugin/plugin.json # required: name, version, description +commands/ agents/ skills/ hooks/ # all optional +README.md # per-plugin docs +``` -`verbose: true` + `showThinkingSummaries: true` catches drift early but costs cognitive load. -A/B `/focus` on tasks where you trust the agent; stay verbose where you don't. +**Manifest path rule:** component/source paths in `plugin.json` and +`marketplace.json` must start with `./` (agent paths also end in `.md`). A bare +path fails recent CLIs with `: Invalid input`. + +## Adding or changing a plugin + +1. Create/edit `plugins//` with a valid `.claude-plugin/plugin.json`. +2. Register it in **both** manifests — `.claude-plugin/marketplace.json` (Claude) + and `.agents/plugins/marketplace.json` (Codex). Each entry needs `name`, + `source` (`./plugins/`), `description`. +3. **Bump the plugin's `version`** on any behavioral change (see next section). +4. Validate, commit, open a PR. + +## Version bumps reach the cache; edits don't + +Installed plugins run from `~/.claude/plugins/cache////`, +pinned to the installed commit — **not** from this working tree. `/plugin update` +re-syncs the cache **only when the version string changes**; on an equal version it +reports *"already at the latest"* and keeps running the old code. A fix committed +**without a version bump never reaches the running plugin.** + +| Rule | Why | +|------|-----| +| Bump `plugin.json` `version` (+ matching `marketplace.json` entries) on any change under `hooks/ scripts/ agents/ commands/ skills/` | The version string is the sole update signal. | +| After it lands: `commit → push → /plugin update → restart Claude Code` | Hooks load only at session start. | +| To see what actually runs: `diff -rq` / `md5` the cache install vs the source | Editing the repo proves nothing about the executing copy. | +| Don't hand-patch the cache as the fix | It's a managed artifact the next reinstall clobbers. Durable = bump + publish. | + +**Hook-symptom triage:** `Not logged in · Please run /login` is **session auth** +(run `/login`), never a hook. "haiku" errors come only from `type: prompt` hooks or +the background model — a `type: command` hook (plain bash/python, no model call) +cannot produce one. + +## Validate + +CI (`.github/workflows/ci.yml`) is the gate; run the same checks locally: + +```bash +# every JSON parses +find . -name '*.json' -not -path '*/.git/*' -print0 | xargs -0 -I{} jq . {} >/dev/null +# every plugin.json has name/version/description +find . -path '*/.claude-plugin/plugin.json' -print0 \ + | xargs -0 -I{} jq -e '.name and .version and .description' {} >/dev/null +``` ---- +PRs are reviewed by three independent bots: Claude, Copilot, CodeRabbit. Any of +them may produce or implement fixes. If a review comment cannot be posted inline +because it is outside the PR diff, translate it into a concrete source task: +target file, line or symbol when possible, plus the behavior to change. Do not +copy reviewer prose into the repo as the fix; apply the root change in source. +Parallel agents may work the same review point. Treat overlapping findings as a +strong signal and converge on the common root cause. -## Automation runner — `automation/` +## Conventions -The `automation/` directory contains the cron-triggered runner that keeps -Alexander's working repos clean (the 5 listed in `automation/policy.md`). +- Less code is better; remove unused now. Prefer boring over clever. +- Fix the root cause, not one instance — understand *why* before changing. +- Fail loudly; silent failures are debugging nightmares. +- Document non-obvious decisions in the PR description, not just the commit. -**Rule for dev work:** ignore `automation/` entirely. The cron output -artifact `automation/RUN-LOG.md` is auto-managed (FIFO cap 10, written by -`automation/scripts/log-entry.sh`) — never hand-edit it. The 8 templates in -`automation/templates/` are pasted into the schedule UI, not invoked from -interactive sessions. +## Keep the mirrors in sync -For policy details, see [`automation/policy.md`](automation/policy.md). -For the runner overview, see [`automation/README.md`](automation/README.md). +Edited this file? Hand-update `.cursor/rules/project-config.mdc` and +`.windsurf/rules/project-config.md` (verbatim body copies, each with its own +frontmatter). Touch `.github/copilot-instructions.md` only if the change is +relevant there. diff --git a/plugins/nihil/.claude-plugin/plugin.json b/plugins/nihil/.claude-plugin/plugin.json index 8da6a1e1..bcea39a7 100644 --- a/plugins/nihil/.claude-plugin/plugin.json +++ b/plugins/nihil/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "nihil", - "version": "0.3.0", + "version": "0.3.1", "description": "Evidence-gated maintainability discipline in two layers: a native hook-enforced mode plugin — /nihil:raze (root-authority, write-capable transformation of a repo you own) plus the disciplined /nihil:review → /nihil:implement → /nihil:release path, with a secret / API-key brake active in every mode (a discipline aid, not a security boundary) — plus a summonable pantheon of five first-principles dynamic workflows (/nihil, /nihil-maat, /nihil-odin, /nihil-shiva, /nihil-athena) installed via /nihil:summon.", "author": { "name": "ANcpLua", diff --git a/plugins/nihil/.codex-plugin/plugin.json b/plugins/nihil/.codex-plugin/plugin.json index 8f678584..04b8cc45 100644 --- a/plugins/nihil/.codex-plugin/plugin.json +++ b/plugins/nihil/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "nihil", - "version": "0.3.0", + "version": "0.3.1", "description": "Evidence-gated maintainability discipline in two layers: a native hook-enforced mode plugin \u2014 /nihil:raze (root-authority, write-capable transformation of a repo you own) plus the disciplined /nihil:review \u2192 /nihil:implement \u2192 /nihil:release path, with a secret / API-key brake active in every mode (a discipline aid, not a security boundary) \u2014 plus a summonable pantheon of five first-principles dynamic workflows (/nihil, /nihil-maat, /nihil-odin, /nihil-shiva, /nihil-athena) installed via /nihil:summon.", "author": { "name": "ANcpLua", diff --git a/plugins/safety-nets/.claude-plugin/plugin.json b/plugins/safety-nets/.claude-plugin/plugin.json index 0347ec7f..edb82eeb 100644 --- a/plugins/safety-nets/.claude-plugin/plugin.json +++ b/plugins/safety-nets/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "safety-nets", - "version": "0.1.0", + "version": "0.1.1", "description": "Two deterministic Stop-event safety nets that keep a session honest: overclaim-net blocks a 'done/fixed/verified' claim made without a verifying command this turn; slnx-sync blocks unregistered .csproj in a .slnx (auto-skips VMR-synced upstream forks, honors a .slnx-sync-ignore opt-out).", "author": { "name": "ANcpLua", diff --git a/plugins/safety-nets/hooks/slnx-sync-check.sh b/plugins/safety-nets/hooks/slnx-sync-check.sh index 3b98be17..8b8e0aa3 100755 --- a/plugins/safety-nets/hooks/slnx-sync-check.sh +++ b/plugins/safety-nets/hooks/slnx-sync-check.sh @@ -31,7 +31,7 @@ git -C "$root" remote get-url upstream 2>/dev/null \ | grep -qiE 'github\.com[:/]+dotnet/' && exit 0 # dotnet/* upstream remote python3 - "$root" <<'PY' -import sys, os, glob, re, json +import sys, os, glob, re, json, subprocess root = sys.argv[1] @@ -52,14 +52,45 @@ except Exception: registered = {m.replace("\\", "/") for m in re.findall(r'Path="([^"]+)"', text)} +# Git-ignored directories (vendored sample repos, scratch trees, etc.) are not +# "our" projects to register. Resolve the set of wholly-ignored dirs once, up +# front (`--directory` collapses each to its top dir so we never descend). Empty +# set if soln_dir isn't a git repo — the nested-solution guard below still applies. +ignored_dirs = set() +try: + out = subprocess.run( + ["git", "-C", soln_dir, "ls-files", "-oi", "--exclude-standard", "--directory"], + capture_output=True, text=True, timeout=10, + ).stdout + ignored_dirs = { + os.path.normpath(os.path.join(soln_dir, p)) for p in out.split("\n") if p.strip() + } +except (OSError, subprocess.SubprocessError): + # Only tolerate the expected best-effort failures: git absent / not on PATH + # (OSError/FileNotFoundError) or the 10s timeout firing (TimeoutExpired). + # Anything else (e.g. a bug in this block) propagates loudly. The nested- + # solution guard below still covers vendored repos & fixtures without git. + pass + missing = [] for dirpath, dirnames, filenames in os.walk(soln_dir): - dirnames[:] = [d for d in dirnames if d not in ("bin", "obj")] + dirnames[:] = [ + d for d in dirnames + if d not in ("bin", "obj") + and os.path.normpath(os.path.join(dirpath, d)) not in ignored_dirs + ] # Skip `dotnet new` template content: a dir holding a .template.config is a # template root whose placeholder-named .csproj must NOT be in the solution. if ".template.config" in dirnames: dirnames[:] = [] continue + # Skip nested independent solutions: any dir BELOW the solution dir that holds + # its own .sln/.slnx is a separate solution root — a vendored sample repo or an + # isolated test fixture (e.g. DependencyInspector TestAssets, including a + # deliberately-circular one). Its .csproj belong to THAT solution, not this one. + if dirpath != soln_dir and any(fn.endswith((".sln", ".slnx")) for fn in filenames): + dirnames[:] = [] + continue for f in filenames: if f.endswith(".csproj"): rel = os.path.relpath(os.path.join(dirpath, f), soln_dir).replace("\\", "/")