From c783546e0c8d4b52b1bb5a4302a97f3781e0a380 Mon Sep 17 00:00:00 2001 From: Max Date: Tue, 1 Sep 2026 14:06:22 +0200 Subject: [PATCH] docs(auto-combo): complete the mode pack table and gate what it claims --- .../maintenance/12316-mode-packs-doc-gate.md | 1 + docs/architecture/ARCHITECTURE.md | 12 +- docs/architecture/REPOSITORY_MAP.md | 36 +++--- docs/architecture/RESILIENCE_GUIDE.md | 2 +- docs/getting-started/AUTO-COMBO-GUIDE.md | 61 ++++++----- docs/guides/FEATURES.md | 2 +- docs/routing/AUTO-COMBO.md | 65 ++++++----- scripts/check/check-docs-counts-sync.mjs | 103 +++++++++++++++++- skills/omni-combos-routing/SKILL.md | 4 +- src/lib/combos/intelligentRouting.ts | 7 +- tests/unit/check-docs-counts-sync.test.ts | 77 +++++++++++++ .../intelligent-routing-options.test.ts | 53 +++++++++ 12 files changed, 338 insertions(+), 85 deletions(-) create mode 100644 changelog.d/maintenance/12316-mode-packs-doc-gate.md create mode 100644 tests/unit/dashboard/intelligent-routing-options.test.ts diff --git a/changelog.d/maintenance/12316-mode-packs-doc-gate.md b/changelog.d/maintenance/12316-mode-packs-doc-gate.md new file mode 100644 index 00000000000..beef22f5dad --- /dev/null +++ b/changelog.d/maintenance/12316-mode-packs-doc-gate.md @@ -0,0 +1 @@ +- **docs(auto-combo):** The mode pack table in `docs/routing/AUTO-COMBO.md` now lists all six shipped packs with every weight each one sets, replacing a four-pack table whose numbers had also drifted from the source. It states plainly that no pack sets `quality`, so selecting any pack silences the observed-quality signal. Six more documents that quote the scoring factor count joined the `check:docs-counts` gate, which caught five stale claims — including one naming nine factors that do not exist — and two stale mode pack counts. The dashboard routing panel, which offered four of the six packs and labelled the default strategy "6-Factor Scoring", is now covered by a test; the two packs it was missing are `reliability-first` and `chaos-mode`, the latter labelled as the fault-injection profile it is rather than as one more routing preference ([#12316](https://github.com/diegosouzapw/OmniRoute/pull/12316)) diff --git a/docs/architecture/ARCHITECTURE.md b/docs/architecture/ARCHITECTURE.md index 209074cb455..3c7fee2f0ad 100644 --- a/docs/architecture/ARCHITECTURE.md +++ b/docs/architecture/ARCHITECTURE.md @@ -370,14 +370,18 @@ Key capabilities: **auto**, lkgp, context-optimized, context-relay, **fusion**, plus a fallback path) — auto is the headline addition in v3.8.0; `fusion` (panel fan-out + judge synthesis, `open-sse/services/fusion.ts`) is new in v3.8.36. -- **9-factor scoring**: cost, latency p95, success rate, quota headroom, lockout - proximity, breaker state, recent failures, model availability, and tag affinity. +- **15-factor scoring**: quota, health, inverse cost, inverse latency, task fit and + ten more. The canonical table of factors and their default weights lives in + [`docs/routing/AUTO-COMBO.md`](../routing/AUTO-COMBO.md) — restating it here would + give it a second place to go stale. - **Virtual factory** materializes ephemeral combos when no matching named combo exists, sourcing candidates from healthy active provider connections. - **Auto prefixes**: `auto/coding`, `auto/cheap`, `auto/fast`, `auto/offline`, `auto/smart`, `auto/lkgp` — each backed by a tuned weight profile. -- **4 mode packs**: coding, fast, cheap, smart — shipped as preset weight - configurations callable from the dashboard. +- **6 mode packs**: `ship-fast`, `cost-saver`, `quality-first`, `offline-friendly`, + `reliability-first` and `chaos-mode` — preset weight configurations callable from + the dashboard. (Not to be confused with the `auto/*` prefixes above, which are + request-time variants.) For full algorithmic detail (factor formulas, weight tuning), see [`docs/routing/AUTO-COMBO.md`](../routing/AUTO-COMBO.md). diff --git a/docs/architecture/REPOSITORY_MAP.md b/docs/architecture/REPOSITORY_MAP.md index aac2048e07a..2f7f4806976 100644 --- a/docs/architecture/REPOSITORY_MAP.md +++ b/docs/architecture/REPOSITORY_MAP.md @@ -403,24 +403,24 @@ open-sse/ ### Subsystem deep-dives -| Doc | Purpose | -| -------------------------- | ------------------------------------------------------------------- | -| `MCP-SERVER.md` | MCP server: 110 tools, 3 transports, 33 scopes, REST endpoints | -| `A2A-SERVER.md` | A2A v0.3: JSON-RPC, 6 skills, REST helpers, agent card | -| `AGENT_PROTOCOLS_GUIDE.md` | Unified guide: A2A vs ACP vs Cloud Agents | -| `CLOUD_AGENT.md` | Codex Cloud / Devin / Jules orchestration | -| `SKILLS.md` | Skills framework (built-in + marketplace + SkillsSH + sandbox) | -| `RADAR.md` | Radar free-model catalog overlay (`RADAR_ENABLED`, off by default) | -| `MEMORY.md` | Memory system (SQLite FTS5 + Qdrant) | -| `EVALS.md` | Eval framework (suites, runs, rubrics) | -| `GUARDRAILS.md` | PII masker, prompt injection, vision bridge | -| `COMPLIANCE.md` | Audit log, retention, noLog opt-out | -| `WEBHOOKS.md` | HMAC-signed webhook delivery | -| `REASONING_REPLAY.md` | Hybrid memory/SQLite cache for `reasoning_content` | -| `AUTHZ_GUIDE.md` | Authorization pipeline (`classify` → `policies` → `enforce`) | -| `RESILIENCE_GUIDE.md` | Circuit breaker + cooldown + model lockout | -| `STEALTH_GUIDE.md` | TLS fingerprinting (JA3/JA4), Claude Code CCH, MITM cert | -| `AUTO-COMBO.md` | Auto Combo engine (9-factor scoring, 4 mode packs, virtual factory) | +| Doc | Purpose | +| -------------------------- | -------------------------------------------------------------------- | +| `MCP-SERVER.md` | MCP server: 110 tools, 3 transports, 33 scopes, REST endpoints | +| `A2A-SERVER.md` | A2A v0.3: JSON-RPC, 6 skills, REST helpers, agent card | +| `AGENT_PROTOCOLS_GUIDE.md` | Unified guide: A2A vs ACP vs Cloud Agents | +| `CLOUD_AGENT.md` | Codex Cloud / Devin / Jules orchestration | +| `SKILLS.md` | Skills framework (built-in + marketplace + SkillsSH + sandbox) | +| `RADAR.md` | Radar free-model catalog overlay (`RADAR_ENABLED`, off by default) | +| `MEMORY.md` | Memory system (SQLite FTS5 + Qdrant) | +| `EVALS.md` | Eval framework (suites, runs, rubrics) | +| `GUARDRAILS.md` | PII masker, prompt injection, vision bridge | +| `COMPLIANCE.md` | Audit log, retention, noLog opt-out | +| `WEBHOOKS.md` | HMAC-signed webhook delivery | +| `REASONING_REPLAY.md` | Hybrid memory/SQLite cache for `reasoning_content` | +| `AUTHZ_GUIDE.md` | Authorization pipeline (`classify` → `policies` → `enforce`) | +| `RESILIENCE_GUIDE.md` | Circuit breaker + cooldown + model lockout | +| `STEALTH_GUIDE.md` | TLS fingerprinting (JA3/JA4), Claude Code CCH, MITM cert | +| `AUTO-COMBO.md` | Auto Combo engine (15-factor scoring, 6 mode packs, virtual factory) | ### Compression diff --git a/docs/architecture/RESILIENCE_GUIDE.md b/docs/architecture/RESILIENCE_GUIDE.md index 1d81a2a1f69..d90f8ddd225 100644 --- a/docs/architecture/RESILIENCE_GUIDE.md +++ b/docs/architecture/RESILIENCE_GUIDE.md @@ -652,4 +652,4 @@ default `test:integration`, chaos and heap self-skip (without `RUN_CHAOS_INT`/`- - [Architecture Guide](./ARCHITECTURE.md) — System architecture and internals - [User Guide](../guides/USER_GUIDE.md) — Providers, combos, CLI integration -- [Auto-Combo Engine](../routing/AUTO-COMBO.md) — 13-factor scoring, mode packs +- [Auto-Combo Engine](../routing/AUTO-COMBO.md) — 15-factor scoring, mode packs diff --git a/docs/getting-started/AUTO-COMBO-GUIDE.md b/docs/getting-started/AUTO-COMBO-GUIDE.md index fbba5abbe77..1c1ef8869ee 100644 --- a/docs/getting-started/AUTO-COMBO-GUIDE.md +++ b/docs/getting-started/AUTO-COMBO-GUIDE.md @@ -46,14 +46,14 @@ model: "auto/cheap" # Cheapest option ## Which "auto" Should I Use? -| If you want... | Use this | Best for | How it works | -|----------------|----------|----------|--------------| -| **Best overall** | `auto` | General questions, chat | Balances speed, cost, and quality | -| **Best code** | `auto/coding` | Writing code, debugging | Picks models good at coding tasks | -| **Fastest response** | `auto/fast` | Quick answers, low latency | Prioritizes speed over everything | -| **Cheapest option** | `auto/cheap` | Saving money | Picks the cheapest provider | -| **Smartest model** | `auto/smart` | Complex tasks | Quality-first + explores new models | -| **Most available** | `auto/offline` | When providers are busy | Picks providers with most capacity | +| If you want... | Use this | Best for | How it works | +| -------------------- | -------------- | -------------------------- | ----------------------------------- | +| **Best overall** | `auto` | General questions, chat | Balances speed, cost, and quality | +| **Best code** | `auto/coding` | Writing code, debugging | Picks models good at coding tasks | +| **Fastest response** | `auto/fast` | Quick answers, low latency | Prioritizes speed over everything | +| **Cheapest option** | `auto/cheap` | Saving money | Picks the cheapest provider | +| **Smartest model** | `auto/smart` | Complex tasks | Quality-first + explores new models | +| **Most available** | `auto/offline` | When providers are busy | Picks providers with most capacity | ### Examples @@ -81,7 +81,7 @@ curl http://localhost:20128/v1/chat/completions \ When you send a request with `model: "auto"`, OmniRoute: 1. **Looks at all your connected providers** — Every provider you've added (OpenAI, Anthropic, Google, etc.) -2. **Scores each one** on 5 factors: +2. **Scores each one**, weighing among other things: - Is it working? (health) - Does it have capacity? (quota) - How much does it cost? (price) @@ -94,29 +94,29 @@ When you send a request with `model: "auto"`, OmniRoute: Each provider gets a score from 0 to 1. The higher the score, the better the fit. -| Factor | Weight | What it means | -|--------|--------|---------------| -| Health | 20% | Is the provider working? (circuit breaker state) | -| Quota | 15% | Does it have capacity remaining? | -| Cost | 15% | How expensive is it? (cheaper = higher score) | -| Speed | 12% | How fast is it? (lower latency = higher score) | -| Task Fit | 8% | Is it good at this type of task? | -| Stability | 5% | Is it consistent? (low error rate) | -| Tier | 5% | Account tier (Ultra > Pro > Free) | -| Other | 20% | Context affinity, connection density, etc. | +| Factor | Weight | What it means | +| --------- | ------ | ------------------------------------------------ | +| Health | 20% | Is the provider working? (circuit breaker state) | +| Quota | 15% | Does it have capacity remaining? | +| Cost | 15% | How expensive is it? (cheaper = higher score) | +| Speed | 12% | How fast is it? (lower latency = higher score) | +| Task Fit | 8% | Is it good at this type of task? | +| Stability | 5% | Is it consistent? (low error rate) | +| Tier | 5% | Account tier (Ultra > Pro > Free) | +| Other | 20% | Context affinity, connection density, etc. | ### How Variants Change the Scoring Each variant uses different weights: -| Variant | Prioritizes | Key Weights | -|---------|-------------|-------------| -| `auto` | Balanced | health=20%, quota=15%, cost=15% | -| `auto/coding` | Quality | taskFit=37%, stability=15% | -| `auto/fast` | Speed | latency=32%, health=28% | -| `auto/cheap` | Cost | cost=37% | -| `auto/smart` | Quality + Explore | taskFit=37%, exploration=10% | -| `auto/offline` | Capacity | quota=37%, health=28% | +| Variant | Prioritizes | Key Weights | +| -------------- | ----------------- | ------------------------------- | +| `auto` | Balanced | health=20%, quota=15%, cost=15% | +| `auto/coding` | Quality | taskFit=37%, stability=15% | +| `auto/fast` | Speed | latency=32%, health=28% | +| `auto/cheap` | Cost | cost=37% | +| `auto/smart` | Quality + Explore | taskFit=37%, exploration=10% | +| `auto/offline` | Capacity | quota=37%, health=28% | --- @@ -125,15 +125,19 @@ Each variant uses different weights: OmniRoute has **three layers of protection**: ### 1. Auto-Fallback + If the best provider fails, OmniRoute automatically tries the next one. You don't need to do anything. ### 2. Self-Healing + If a provider keeps failing: + - **Score < 0.2** → Excluded for 5 minutes - **Circuit breaker open** → Auto-excluded - **More than 50% providers down** → Incident mode (no exploration) ### 3. Emergency Fallback + If all providers fail, OmniRoute routes to stable free providers (like Kiro or Qoder) as a last resort. --- @@ -209,7 +213,8 @@ Round-robin cycles through providers in order. Auto-combo **scores each provider ## Learn More For developers and contributors, see the [Auto-Combo Technical Reference](../routing/AUTO-COMBO.md) for: -- Full 13-factor scoring algorithm + +- Full 15-factor scoring algorithm - Mode pack weight tables - Implementation file paths - API endpoints diff --git a/docs/guides/FEATURES.md b/docs/guides/FEATURES.md index 2edcf971278..6f8e2807685 100644 --- a/docs/guides/FEATURES.md +++ b/docs/guides/FEATURES.md @@ -18,7 +18,7 @@ Visual guide to every section of the OmniRoute dashboard. The v3.7.x → v3.8.0 cycle added zero-config auto routing, new providers, OAuth flows, deeper resilience, and a much richer CLI experience. Headline features below — full details further in the document and in linked specs. -- 🤖 **Auto Combo / Zero-config auto-routing** — use prefixes `auto/coding`, `auto/fast`, `auto/cheap`, `auto/offline`, `auto/smart`, `auto/lkgp`, `auto/chaos`. Backed by a 15-factor scoring engine and 6 curated **mode packs** (ship-fast, cost-saver, quality-first, offline-friendly) +- 🤖 **Auto Combo / Zero-config auto-routing** — use prefixes `auto/coding`, `auto/fast`, `auto/cheap`, `auto/offline`, `auto/smart`, `auto/lkgp`, `auto/chaos`. Backed by a 15-factor scoring engine and 6 curated **mode packs** (ship-fast, cost-saver, quality-first, offline-friendly, reliability-first, chaos-mode) - 🆕 **Command Code provider** (#2199) — first-class registration with model catalog and quota tracking - 🆕 **Z.AI provider** — new free-tier provider with quota labels - 🎬 **KIE media expansion** — extended catalog including video generation models diff --git a/docs/routing/AUTO-COMBO.md b/docs/routing/AUTO-COMBO.md index dca58329d82..6f96a0b22eb 100644 --- a/docs/routing/AUTO-COMBO.md +++ b/docs/routing/AUTO-COMBO.md @@ -212,26 +212,35 @@ The Auto-Combo Engine dynamically selects the best provider/model for each reque ## Mode Packs -Six pre-defined weight profiles in `open-sse/services/autoCombo/modePacks.ts` — `ship-fast`, `cost-saver`, `quality-first`, `offline-friendly`, `reliability-first` and `chaos-mode` (fault-injection). Each pack overrides the default weights to bias selection toward a specific goal; the seed weights below are renormalized to sum 1.0 at runtime together with the session/context factors every pack also sets. The table shows the four original packs — see `modePacks.ts` for `reliability-first` and `chaos-mode`. - -| Factor | ship-fast | cost-saver | quality-first | offline-friendly | -| :----------- | :-------- | :--------- | :------------ | :--------------- | -| quota | 0.14 | 0.14 | 0.10 | **0.37** | -| health | 0.28 | 0.19 | 0.18 | 0.28 | -| costInv | 0.05 | **0.37** | 0.05 | 0.10 | -| latencyInv | **0.32** | 0.05 | 0.05 | 0.05 | -| taskFit | 0.10 | 0.10 | **0.37** | 0.00 | -| stability | 0.00 | 0.05 | 0.15 | 0.10 | -| tierPriority | 0.05 | 0.05 | 0.05 | 0.05 | +6 pre-defined weight profiles in `open-sse/services/autoCombo/modePacks.ts`. Each pack replaces the default weights outright to bias selection toward one goal. Every pack already sums to `1.0` (`0.9999` as printed at four decimals), so `normalizeScoringWeights()` has nothing meaningful to correct when a pack is active — the values below are, to rounding, the ones the scorer applies. + +| Factor | ship-fast | cost-saver | quality-first | offline-friendly | reliability-first | chaos-mode | +| :-------------------- | :--------- | :--------- | :------------ | :--------------- | :---------------- | :--------- | +| `quota` | 0.1333 | 0.1333 | 0.0952 | **0.3524** | 0.1333 | 0.0476 | +| `health` | 0.2667 | 0.1810 | 0.1714 | 0.2667 | **0.3524** | **0.4000** | +| `costInv` | 0.0476 | **0.3524** | 0.0476 | 0.0952 | 0.0381 | 0.0190 | +| `latencyInv` | **0.3048** | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0286 | +| `taskFit` | 0.0952 | 0.0952 | **0.3524** | 0.0000 | 0.0952 | 0.1905 | +| `stability` | 0.0000 | 0.0476 | 0.1429 | 0.0952 | 0.1905 | 0.1714 | +| `tierPriority` | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0190 | +| `tierAffinity` | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0000 | +| `specificityMatch` | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0000 | +| `contextAffinity` | 0.0095 | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0286 | +| `sessionAvailability` | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0476 | +| `resetWindowAffinity` | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0000 | 0.0000 | +| `connectionDensity` | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0476 | 0.0476 | Notes: -- `tierAffinity` and `specificityMatch` are explicitly set to `0` in every mode pack. +- **No pack sets `quality`, and a pack replaces the weight map wholesale** (`weights = pack`, not a merge). `quality` carries `0.03` in `DEFAULT_WEIGHTS`, but under any mode pack it normalizes to `0` — selecting a pack silences the observed-quality signal completely. If you want quality feedback to influence routing, leave `modePack` unset and tune the weights directly. (`cacheAffinity` is also unset by every pack, but it defaults to `0` anyway, so nothing changes there.) +- `tierAffinity`, `specificityMatch` and `resetWindowAffinity` are explicitly `0` in every pack. - Each pack's emphasis at a glance: - - **ship-fast** → latencyInv 0.32 + health 0.28 (low-latency, healthy connections) - - **cost-saver** → costInv 0.37 (cheapest tokens win) - - **quality-first** → taskFit 0.37 + stability 0.15 (best model for the task, consistent) - - **offline-friendly** → quota 0.37 + health 0.28 (max headroom regardless of speed/cost) + - **ship-fast** → latencyInv 0.3048 + health 0.2667 (low-latency, healthy connections) + - **cost-saver** → costInv 0.3524 (cheapest tokens win) + - **quality-first** → taskFit 0.3524 + stability 0.1429 (best model for the task, consistent) + - **offline-friendly** → quota 0.3524 + health 0.2667 (max headroom regardless of speed/cost) + - **reliability-first** → health 0.3524 + stability 0.1905 (fewest surprises) + - **chaos-mode** → health 0.4000 + taskFit 0.1905 (fault-injection profile) ### Per-Request Controls (headers) — #6023 / #6024 / #6025 / #3470 @@ -753,15 +762,15 @@ intentionally excluded from CI because they require live credentials and VPS acc ## Files -| File | Purpose | -| :-------------------------------------------------------- | :------------------------------------------------------------------------- | -| `open-sse/services/autoCombo/scoring.ts` | 15-factor scoring function, `DEFAULT_WEIGHTS`, pool norm | -| `open-sse/services/autoCombo/taskFitness.ts` | Model × task fitness lookup | -| `open-sse/services/autoCombo/engine.ts` | Selection logic, bandit, budget cap | -| `open-sse/services/autoCombo/selfHealing.ts` | Exclusion, probes, incident mode | -| `open-sse/services/autoCombo/modePacks.ts` | 4 weight profiles (ship-fast, cost-saver, quality-first, offline-friendly) | -| `open-sse/services/autoCombo/autoPrefix.ts` | `auto/` prefix parser + 6 variants | -| `open-sse/services/autoCombo/virtualFactory.ts` | Builds in-memory `AutoComboConfig` from live connections | -| `open-sse/services/autoCombo/providerRegistryAccessor.ts` | Test hook for mocking provider registry | -| `src/shared/constants/routingStrategies.ts` | `ROUTING_STRATEGY_VALUES` (19 strategies) | -| `src/sse/handlers/chat.ts` | Integration: auto-prefix short-circuit | +| File | Purpose | +| :-------------------------------------------------------- | :-------------------------------------------------------------------------------------------------------- | +| `open-sse/services/autoCombo/scoring.ts` | 15-factor scoring function, `DEFAULT_WEIGHTS`, pool norm | +| `open-sse/services/autoCombo/taskFitness.ts` | Model × task fitness lookup | +| `open-sse/services/autoCombo/engine.ts` | Selection logic, bandit, budget cap | +| `open-sse/services/autoCombo/selfHealing.ts` | Exclusion, probes, incident mode | +| `open-sse/services/autoCombo/modePacks.ts` | 6 weight profiles (ship-fast, cost-saver, quality-first, offline-friendly, reliability-first, chaos-mode) | +| `open-sse/services/autoCombo/autoPrefix.ts` | `auto/` prefix parser + 6 variants | +| `open-sse/services/autoCombo/virtualFactory.ts` | Builds in-memory `AutoComboConfig` from live connections | +| `open-sse/services/autoCombo/providerRegistryAccessor.ts` | Test hook for mocking provider registry | +| `src/shared/constants/routingStrategies.ts` | `ROUTING_STRATEGY_VALUES` (19 strategies) | +| `src/sse/handlers/chat.ts` | Integration: auto-prefix short-circuit | diff --git a/scripts/check/check-docs-counts-sync.mjs b/scripts/check/check-docs-counts-sync.mjs index 38cdcc73163..849c5141c5f 100644 --- a/scripts/check/check-docs-counts-sync.mjs +++ b/scripts/check/check-docs-counts-sync.mjs @@ -85,6 +85,27 @@ function countScoringFactors() { return parseScoringFactors(fs.readFileSync(file, "utf8")); } +// PURE: the reference document must NAME every shipped pack. Reporting which one is +// missing is the point — "6 packs" tells a doc it is stale, "chaos-mode is missing" +// tells it what to write. +export function makeModePackNamesValidator(names) { + return (content) => { + if (!names.length) return { ok: true, detail: "no mode packs found in source — skipping" }; + // Token boundary, not `includes`: "ship-fast" is a substring of + // "ship-fast-v2", so a doc could satisfy the gate while naming a pack that + // does not ship — and a future pack named as a prefix of another would be + // masked by it. + const missing = names.filter( + (name) => + !new RegExp(`(^|[^\\w-])${name.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}([^\\w-]|$)`).test( + content + ) + ); + if (!missing.length) return { ok: true, detail: `all ${names.length} mode packs are named` }; + return { ok: false, detail: `mode pack(s) never named in this file: ${missing.join(", ")}` }; + }; +} + // PURE: parse the canonical provider total out of the auto-generated catalog text. export function parseProviderTotal(referenceText) { if (!referenceText) return 0; @@ -164,6 +185,7 @@ export function tallyDrift(checks, getContent) { function readCodeFacts() { const script = [ 'import {computeFreeModelTotals} from "./open-sse/config/freeModelCatalog.ts";', + 'import {MODE_PACKS} from "./open-sse/services/autoCombo/modePacks.ts";', 'import {ENGINE_IDS} from "./open-sse/services/compression/engineCatalog.ts";', 'import {CLI_TOOLS} from "./src/shared/constants/cliTools.ts";', 'import {countUniqueMcpTools} from "./open-sse/mcp-server/toolCount.ts";', @@ -203,7 +225,8 @@ function readCodeFacts() { 'console.log("@@"+JSON.stringify({freeSteady:t.steadyRecurringTokens,entries:t.perModel.length,', "freeFirst:t.firstMonthRealisticTokens,freePools:t.poolCount,engines:ENGINE_IDS.length,", "cliTotal:cli.length,cliCode:by('code'),cliAgent:by('agent'),", - "mcpTools:countUniqueMcpTools(cols),mcpScopes:sc.size,providers:pids.size,freeForever:ff.size}));", + "mcpTools:countUniqueMcpTools(cols),mcpScopes:sc.size,providers:pids.size,freeForever:ff.size,", + "modePacks:Object.keys(MODE_PACKS)}));", ].join(""); const tmp = fs.mkdtempSync(path.join(os.tmpdir(), "docs-counts-")); try { @@ -317,10 +340,31 @@ export function extractNumberClaims(content, { pattern, skipBefore, skipAfter }) return claims; } +// Three spellings of the same claim are in use across the docs, and all three +// must be watched: "6 curated **mode packs**", "6 pre-defined weight profiles", +// "4 weight profiles". Matching only the first left the other two unguarded. +const MODE_PACK_CLAIM_PATTERN = + /(\d+)\s+(?:curated\s+|pre-defined\s+)?\*{0,2}(?:mode\s+packs?|weight\s+profiles?)\b/gi; + export function makeNumberClaimValidator(expected, opts) { return (content) => { const claims = extractNumberClaims(content, opts); - if (!claims.length) return { ok: true, detail: `no ${opts.what} claim in this file` }; + if (!claims.length) { + // Most files in a check's list legitimately never mention the number, so + // "no claim" is normally a pass. But for a reference document that is + // supposed to state it, silence is the failure mode that matters: reword + // the sentence past the pattern and the gate goes quiet while reporting + // green. `requireClaim` says this file must carry the claim. + if (opts.requireClaim) + return { + ok: false, + detail: + `no ${opts.what} claim found, and this file is required to state one — ` + + `either the sentence was reworded past the pattern, or it was deleted ` + + `(code has ${expected})`, + }; + return { ok: true, detail: `no ${opts.what} claim in this file` }; + } const stale = claims.filter((c) => c.value !== expected); if (!stale.length) return { ok: true, detail: `${claims.length} ${opts.what} claim(s) match the code` }; @@ -477,7 +521,56 @@ export function buildChecks() { files, validate: makeNumberClaimValidator(expected, { what, ...opts }), }); + const packs = Array.isArray(f.modePacks) ? f.modePacks : []; return [ + { + // Two packs shipped after the docs were written and nothing noticed. + // The count and the names are two different gates: a table can carry + // the right number and still describe the wrong four out of six. + label: "Auto-Combo mode packs", + actual: packs.length, + docKey: "mode packs", + strict: true, + files: [ + "README.md", + "llm.txt", + "docs/guides/FEATURES.md", + "docs/architecture/ARCHITECTURE.md", + "docs/architecture/REPOSITORY_MAP.md", + "docs/routing/AUTO-COMBO.md", + ], + validate: makeNumberClaimValidator(packs.length, { + what: "mode packs", + // Three spellings are in use across the docs, and all three are the + // same claim: "6 curated **mode packs**", "6 pre-defined weight + // profiles", "4 weight profiles". Matching only the first left the + // other two unwatched. + pattern: MODE_PACK_CLAIM_PATTERN, + }), + }, + { + // Same claim, but on the one document that MUST carry it. Without + // `requireClaim` the strongest gate in this file is also the easiest to + // silence: reword the sentence and "no claim in this file" reads as a pass. + label: "Auto-Combo mode packs (reference doc must state the count)", + actual: packs.length, + docKey: "mode packs", + strict: true, + files: ["docs/routing/AUTO-COMBO.md"], + validate: makeNumberClaimValidator(packs.length, { + what: "mode packs", + pattern: MODE_PACK_CLAIM_PATTERN, + requireClaim: true, + }), + }, + { + label: "Auto-Combo mode packs (named in the reference doc)", + actual: packs.length, + docKey: "mode packs", + strict: true, + files: ["docs/routing/AUTO-COMBO.md"], + validate: makeModePackNamesValidator(packs), + }, { label: "Provider reference total (doc vs live modules)", actual: f.providers, @@ -600,6 +693,12 @@ export function buildChecks() { "docs/diagrams/strategies-grid.svg", "docs/diagrams/auto-combo-scoring.mmd", "llm.txt", + "docs/architecture/ARCHITECTURE.md", + "docs/architecture/REPOSITORY_MAP.md", + "docs/architecture/RESILIENCE_GUIDE.md", + "docs/frameworks/OPEN_SSE_ARCHITECTURE.md", + "docs/getting-started/AUTO-COMBO-GUIDE.md", + "skills/omni-combos-routing/SKILL.md", "open-sse/services/autoCombo/routerStrategy.ts", "open-sse/services/taskAwareRouter.ts", "tests/unit/lkgp-enabled-context-11181.test.ts", diff --git a/skills/omni-combos-routing/SKILL.md b/skills/omni-combos-routing/SKILL.md index 5b14c50d944..7212701892a 100644 --- a/skills/omni-combos-routing/SKILL.md +++ b/skills/omni-combos-routing/SKILL.md @@ -228,7 +228,7 @@ curl -X POST $OMNIROUTE_URL/api/combos \ | `reset-window` | Order targets by their configured reset window | | `headroom` | Prefer targets with more remaining quota headroom | | `strict-random` | Random without repeating until all targets have been used | -| `auto` | Auto-Combo scoring across 13 factors | +| `auto` | Auto-Combo scoring across 15 factors | | `lkgp` | Last-known-good-provider sticky routing | | `context-optimized` | Pick the best model for the request's context size | | `cache-optimized` | Prefer targets with stronger cache affinity | @@ -237,7 +237,7 @@ curl -X POST $OMNIROUTE_URL/api/combos \ ## Auto-combo (recommended for production) -Auto-combo scores each candidate on 13 factors every request: +Auto-combo scores each candidate on 15 factors every request: ```bash curl -X POST $OMNIROUTE_URL/api/combos \ diff --git a/src/lib/combos/intelligentRouting.ts b/src/lib/combos/intelligentRouting.ts index 0c966fdc842..d6cb8d8b05a 100644 --- a/src/lib/combos/intelligentRouting.ts +++ b/src/lib/combos/intelligentRouting.ts @@ -63,10 +63,15 @@ export const MODE_PACK_OPTIONS = [ { id: "cost-saver", label: "Cost Saver", emoji: "savings" }, { id: "quality-first", label: "Quality First", emoji: "target" }, { id: "offline-friendly", label: "Offline Friendly", emoji: "cloud_off" }, + { id: "reliability-first", label: "Reliability First", emoji: "shield" }, + // Named for what it does: `modePacks.ts` ships it as the fault-injection + // profile behind `auto/chaos`. It belongs in the list — the engine offers it — + // but not under a label that reads like a routing preference. + { id: "chaos-mode", label: "Chaos Mode (fault injection — testing)", emoji: "science" }, ] as const; export const ROUTER_STRATEGY_OPTIONS = [ - { id: "rules", label: "Rules (6-Factor Scoring)" }, + { id: "rules", label: "Rules (Weighted Scoring)" }, { id: "score", label: "Highest Weighted Score" }, { id: "cost", label: "Cost Optimized" }, { id: "latency", label: "Latency Optimized" }, diff --git a/tests/unit/check-docs-counts-sync.test.ts b/tests/unit/check-docs-counts-sync.test.ts index 698c5374113..f2e8e696126 100644 --- a/tests/unit/check-docs-counts-sync.test.ts +++ b/tests/unit/check-docs-counts-sync.test.ts @@ -390,3 +390,80 @@ test("version gate compares README-footer and llm.txt prose against package.json assert.equal(v("no version here").ok, true); assert.equal(makeVersionValidator(null)("anything").ok, false); }); + +// --- Mode packs ------------------------------------------------------------ +// Two packs (`reliability-first`, `chaos-mode`) shipped without ever reaching +// the reference table, and two documents still claimed four. A count alone would +// not have caught the table: it names four packs and says so. So the gate reads +// the pack NAMES from the module itself and asks the reference document to +// mention each one. +import { makeModePackNamesValidator } from "../../scripts/check/check-docs-counts-sync.mjs"; + +const packNames = makeModePackNamesValidator as ( + names: string[] +) => (content: string) => { ok: boolean; detail: string }; + +test("a document that names every pack passes", () => { + const validate = packNames(["ship-fast", "cost-saver"]); + assert.equal(validate("We ship ship-fast and cost-saver profiles.").ok, true); +}); + +test("a document that forgets a pack fails, and says which one", () => { + const validate = packNames(["ship-fast", "chaos-mode"]); + const result = validate("We ship the ship-fast profile."); + assert.equal(result.ok, false); + assert.match(result.detail, /chaos-mode/); +}); + +test("a longer name does not satisfy the gate for a shorter one", () => { + // `includes` would let "ship-fast-v2" stand in for "ship-fast", so a doc could + // pass while naming a pack that does not ship. + assert.equal(packNames(["ship-fast"])("only ship-fast-v2 is documented here").ok, false); + assert.equal(packNames(["chaos"])("we document chaos-mode only").ok, false); +}); + +test("the name gate stays quiet when the source yields nothing", () => { + assert.equal(packNames([])("anything at all").ok, true); +}); + +// The pack names come from the same tsx subprocess that already reads every other +// code-derived count, not from a regex over the source text: a reader that parses +// TypeScript by hand is a gate that can be silently wrong, which is worse than no +// gate at all. +test("the module is the source of the names, so the gate cannot misparse it", async () => { + const { MODE_PACKS } = await import("../../open-sse/services/autoCombo/modePacks.ts"); + const names = Object.keys(MODE_PACKS); + assert.ok(names.length > 0); + const page = names.join(", "); + assert.equal(packNames(names)(page).ok, true); + assert.equal(packNames(names)(names.slice(1).join(", ")).ok, false); +}); + +const MODE_PACK_CLAIM = { + what: "mode packs", + pattern: /(\d+)\s+(?:curated\s+|pre-defined\s+)?\*{0,2}(?:mode\s+packs?|weight\s+profiles?)\b/gi, +}; + +test("a stale mode pack count is rejected, in each spelling the docs use", () => { + const v = makeValidator(6, MODE_PACK_CLAIM); + assert.equal(v("- **4 mode packs**: coding, fast, cheap, smart").ok, false); + assert.equal(v("| modePacks.ts | 4 weight profiles (ship-fast, ...) |").ok, false); + assert.equal(v("4 pre-defined weight profiles").ok, false); +}); + +test("markdown emphasis does not hide a mode pack count", () => { + const v = makeValidator(6, MODE_PACK_CLAIM); + assert.equal(v("Backed by 6 curated **mode packs** (ship-fast, ...)").ok, true); + assert.equal(v("6 pre-defined weight profiles in `modePacks.ts`").ok, true); +}); + +test("a required claim cannot be silenced by rewording it away", () => { + // This is the failure mode the gate exists to prevent: reword the sentence past + // the pattern and "no claim in this file" used to read as a pass. + const optional = makeValidator(6, MODE_PACK_CLAIM); + const required = makeValidator(6, { ...MODE_PACK_CLAIM, requireClaim: true }); + const reworded = "half a dozen curated profiles ship with the engine"; + assert.equal(optional(reworded).ok, true, "a file that need not state it still passes"); + assert.equal(required(reworded).ok, false, "the reference document must state it"); + assert.match(required(reworded).detail, /required to state one/); +}); diff --git a/tests/unit/dashboard/intelligent-routing-options.test.ts b/tests/unit/dashboard/intelligent-routing-options.test.ts new file mode 100644 index 00000000000..a04685bea1e --- /dev/null +++ b/tests/unit/dashboard/intelligent-routing-options.test.ts @@ -0,0 +1,53 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { MODE_PACKS } from "../../../open-sse/services/autoCombo/modePacks.ts"; +import { + MODE_PACK_OPTIONS, + ROUTER_STRATEGY_OPTIONS, +} from "../../../src/lib/combos/intelligentRouting.ts"; + +// Importing the engine from a test is free; importing it from the client module +// would not be, which is why the dashboard keeps its own option list. This test is +// what keeps that list honest. + +// "custom" is the only option that is not a pack: it means "use the sliders". +const NOT_A_PACK = new Set(["custom"]); + +test("every shipped mode pack is offered in the dashboard", () => { + const offered = MODE_PACK_OPTIONS.map((option) => option.id).filter((id) => !NOT_A_PACK.has(id)); + assert.deepEqual( + [...offered].sort(), + Object.keys(MODE_PACKS).sort(), + "a pack the engine ships cannot be selected from the dashboard (or vice versa)" + ); +}); + +test("the only non-pack option is the manual one", () => { + const unknown = MODE_PACK_OPTIONS.map((option) => option.id).filter( + (id) => !NOT_A_PACK.has(id) && !(id in MODE_PACKS) + ); + assert.deepEqual(unknown, [], "an option id matches no mode pack and is not 'custom'"); +}); + +test("the fault-injection pack says so in its label", () => { + // `chaos-mode` is the profile behind `auto/chaos`; offering it next to + // "Ship Fast" and "Cost Saver" without saying what it is would read as one + // more routing preference. + const chaos = MODE_PACK_OPTIONS.find((option) => option.id === "chaos-mode"); + assert.ok(chaos, "chaos-mode must be offered — the engine ships it"); + assert.match( + chaos.label, + /fault injection/i, + "an operator must be able to tell this one apart from a routing preference" + ); +}); + +test("no strategy label states a factor count", () => { + const stale = ROUTER_STRATEGY_OPTIONS.filter((option) => /\d+[- ]factor/i.test(option.label)); + assert.deepEqual( + stale.map((option) => option.label), + [], + "a label repeats a number the engine owns; it will go stale and no gate reads labels" + ); +});