Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/fixes/11492-taskfitness-scores-as.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
- **fix(autoCombo):** effort/alias model variants (`gpt-5.6-sol-xhigh`, `gpt-5.6`, cursor's `claude-4.6-opus-high`) inherit their base model's task fitness instead of falling to the wildcard 0.5 ([#11492](https://github.com/diegosouzapw/OmniRoute/pull/11492))
1 change: 1 addition & 0 deletions changelog.d/fixes/11506-arena-elo-base-rows.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
- **fix(intelligence):** Arena ELO sync stores a synthesized base-model row for every effort/harness variant (`claude-opus-5-max` → `claude-opus-5`, `gpt-5.6-sol-xhigh (codex-harness)` → `gpt-5.6-sol`), so bare model requests reach the synced intelligence layer instead of the static fallback table; the hand-maintained `MODEL_ALIAS_MAP` — which collapsed generations (`gpt-5.5` → `gpt-5`) — is gone ([#11506](https://github.com/diegosouzapw/OmniRoute/pull/11506))
42 changes: 42 additions & 0 deletions open-sse/config/providerModels.ts
Original file line number Diff line number Diff line change
Expand Up @@ -145,6 +145,48 @@ function getGlobalModel(modelId: string): RegistryModel | undefined {
return bestMatch;
}

/**
* Exact-id catalog lookup: the registry entry for `modelId` across every
* provider, or for its basename once a `vendor/` prefix is stripped.
*
* This is `getGlobalModel`'s steps 1–2 only. Step 3 (a `startsWith` substring
* scan) deliberately guesses at a base model, which is exactly what a
* catalog-anchor check must not do — `resolveScoresAs` (#11489) uses this to
* verify that a suffix-stripped base is a REAL routable id before inheriting
* its quality scores.
*/
export function findRegistryModelById(modelId: string): RegistryModel | undefined {
if (typeof modelId !== "string" || modelId.length === 0) return undefined;
for (const models of Object.values(PROVIDER_MODELS)) {
const found = models.find((m) => m.id === modelId);
if (found) return found;
}
const basename = modelId.split("/").pop() || modelId;
if (basename === modelId) return undefined;
for (const models of Object.values(PROVIDER_MODELS)) {
const found = models.find((m) => m.id === basename);
if (found) return found;
}
return undefined;
}

/**
* The `scoresAs` target declared for `modelId`, if any (#11489). Scans every
* provider that ships the id rather than stopping at the first hit, so an
* unannotated duplicate of the same id in another provider's catalog cannot
* shadow the entry that actually declares the relation.
*/
export function findRegistryScoresAs(modelId: string): string | undefined {
if (typeof modelId !== "string" || modelId.length === 0) return undefined;
const basename = modelId.split("/").pop() || modelId;
for (const models of Object.values(PROVIDER_MODELS)) {
for (const m of models) {
if ((m.id === modelId || m.id === basename) && m.scoresAs) return m.scoresAs;
}
}
return undefined;
}

export function getProviderModel(aliasOrId: string, modelId: string): RegistryModel | undefined {
const models = PROVIDER_MODELS[aliasOrId];
if (!models) return getGlobalModel(modelId);
Expand Down
64 changes: 50 additions & 14 deletions open-sse/config/providers/registry/cursor/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -26,10 +26,30 @@ export const cursorProvider: RegistryEntry = {
{ id: "gpt-5.3-codex-spark-preview", name: "GPT 5.3 Codex Spark Preview" },
{ id: "gpt-5.3-codex-spark-preview-high", name: "GPT 5.3 Codex Spark Preview High" },
{ id: "gpt-5.3-codex-spark-preview-xhigh", name: "GPT 5.3 Codex Spark Preview XHigh" },
{ id: "claude-4.6-opus-high-thinking-fast", name: "Claude 4.6 Opus High Thinking Fast" },
{ id: "claude-4.6-opus-max-thinking-fast", name: "Claude 4.6 Opus Max Thinking Fast" },
{ id: "claude-4.6-sonnet-medium", name: "Claude 4.6 Sonnet Medium" },
{ id: "claude-4.6-sonnet-medium-thinking", name: "Claude 4.6 Sonnet Medium Thinking" },
// #11489: cursor/agy spell Claude ids <version>-<family> ("claude-4.6-opus-high");
// the effort splitter strips those to "claude-4.6-opus", which is not a catalog id.
// `scoresAs` points each at the canonical <family>-<version> spelling so quality
// scores are inherited. Operational metadata stays on these entries.
{
id: "claude-4.6-opus-high-thinking-fast",
name: "Claude 4.6 Opus High Thinking Fast",
scoresAs: "claude-opus-4-6",
},
{
id: "claude-4.6-opus-max-thinking-fast",
name: "Claude 4.6 Opus Max Thinking Fast",
scoresAs: "claude-opus-4-6",
},
{
id: "claude-4.6-sonnet-medium",
name: "Claude 4.6 Sonnet Medium",
scoresAs: "claude-sonnet-4-6",
},
{
id: "claude-4.6-sonnet-medium-thinking",
name: "Claude 4.6 Sonnet Medium Thinking",
scoresAs: "claude-sonnet-4-6",
},
{ id: "gemini-3.1-pro", name: "Gemini 3.1 Pro" },
{ id: "gemini-3.7-flash", name: "Gemini 3.7 Flash" },
{ id: "gemini-3-flash", name: "Gemini 3 Flash" },
Expand Down Expand Up @@ -184,12 +204,24 @@ export const cursorProvider: RegistryEntry = {
{ id: "gpt-5.4-high-fast", name: "GPT-5.4 High Fast" },
{ id: "gpt-5.4-xhigh", name: "GPT-5.4 1M Extra High" },
{ id: "gpt-5.4-xhigh-fast", name: "GPT-5.4 Extra High Fast" },
{ id: "claude-4.6-opus-high", name: "Opus 4.6 1M" },
{ id: "claude-4.6-opus-max", name: "Opus 4.6 1M Max" },
{ id: "claude-4.6-opus-high-thinking", name: "Opus 4.6 1M Thinking" },
{ id: "claude-4.6-opus-max-thinking", name: "Opus 4.6 1M Max Thinking" },
{ id: "claude-4.5-opus-high", name: "Opus 4.5" },
{ id: "claude-4.5-opus-high-thinking", name: "Opus 4.5 Thinking" },
// #11489: cursor/agy spell Claude ids <version>-<family> ("claude-4.6-opus-high");
// the effort splitter strips those to "claude-4.6-opus", which is not a catalog id.
// `scoresAs` points each at the canonical <family>-<version> spelling so quality
// scores are inherited. Operational metadata stays on these entries.
{ id: "claude-4.6-opus-high", name: "Opus 4.6 1M", scoresAs: "claude-opus-4-6" },
{ id: "claude-4.6-opus-max", name: "Opus 4.6 1M Max", scoresAs: "claude-opus-4-6" },
{
id: "claude-4.6-opus-high-thinking",
name: "Opus 4.6 1M Thinking",
scoresAs: "claude-opus-4-6",
},
{
id: "claude-4.6-opus-max-thinking",
name: "Opus 4.6 1M Max Thinking",
scoresAs: "claude-opus-4-6",
},
{ id: "claude-4.5-opus-high", name: "Opus 4.5", scoresAs: "claude-opus-4-5" },
{ id: "claude-4.5-opus-high-thinking", name: "Opus 4.5 Thinking", scoresAs: "claude-opus-4-5" },
{ id: "gpt-5.2-low", name: "GPT-5.2 Low" },
{ id: "gpt-5.2-low-fast", name: "GPT-5.2 Low Fast" },
{ id: "gpt-5.2-fast", name: "GPT-5.2 Fast" },
Expand Down Expand Up @@ -223,13 +255,17 @@ export const cursorProvider: RegistryEntry = {
{ id: "gpt-5.4-nano-medium", name: "GPT-5.4 Nano" },
{ id: "gpt-5.4-nano-high", name: "GPT-5.4 Nano High" },
{ id: "gpt-5.4-nano-xhigh", name: "GPT-5.4 Nano Extra High" },
{ id: "claude-4.5-sonnet", name: "Sonnet 4.5" },
{ id: "claude-4.5-sonnet-thinking", name: "Sonnet 4.5 Thinking" },
{ id: "claude-4.5-sonnet", name: "Sonnet 4.5", scoresAs: "claude-sonnet-4-5" },
{
id: "claude-4.5-sonnet-thinking",
name: "Sonnet 4.5 Thinking",
scoresAs: "claude-sonnet-4-5",
},
{ id: "gpt-5.1-low", name: "GPT-5.1 Low" },
{ id: "gpt-5.1", name: "GPT-5.1" },
{ id: "gpt-5.1-high", name: "GPT-5.1 High" },
{ id: "claude-4-sonnet", name: "Sonnet 4" },
{ id: "claude-4-sonnet-thinking", name: "Sonnet 4 Thinking" },
{ id: "claude-4-sonnet", name: "Sonnet 4", scoresAs: "claude-sonnet-4" },
{ id: "claude-4-sonnet-thinking", name: "Sonnet 4 Thinking", scoresAs: "claude-sonnet-4" },
{ id: "gpt-5-mini", name: "GPT-5 Mini" },
{ id: "kimi-k3-low", name: "Kimi K3 Low" },
{ id: "kimi-k3-max", name: "Kimi K3" },
Expand Down
5 changes: 4 additions & 1 deletion open-sse/config/providers/registry/openai/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,10 @@ export const openaiProvider: RegistryEntry = {
authHeader: "bearer",
defaultContextLength: 128000,
models: [
{ id: "gpt-5.6", name: "GPT-5.6", ...GPT_5_6_API_CAPABILITIES },
// #11489: per OpenAI's model reference `gpt-5.6` is an ALIAS of `gpt-5.6-sol`,
// not a distinct model — quality scores point forward, which no suffix
// stripper can express. Siblings `-terra`/`-luna` are their own models.
{ id: "gpt-5.6", name: "GPT-5.6", scoresAs: "gpt-5.6-sol", ...GPT_5_6_API_CAPABILITIES },
{ id: "gpt-5.6-sol", name: "GPT-5.6 Sol", ...GPT_5_6_API_CAPABILITIES },
{ id: "gpt-5.6-terra", name: "GPT-5.6 Terra", ...GPT_5_6_API_CAPABILITIES },
{ id: "gpt-5.6-luna", name: "GPT-5.6 Luna", ...GPT_5_6_API_CAPABILITIES },
Expand Down
11 changes: 11 additions & 0 deletions open-sse/config/providers/shared.ts
Original file line number Diff line number Diff line change
Expand Up @@ -83,6 +83,17 @@ export interface RegistryModel {
/** Per-model upstream header-response timeout override — precedes
* `RegistryEntry.timeoutMs` and the global `FETCH_TIMEOUT_MS` (#6354). */
timeoutMs?: number;
/**
* Id whose QUALITY scores (task fitness / arena / user overrides) this id
* inherits (#11489). Operational fields (timeoutMs, cost, context) stay on
* this entry. One hop only; the target must itself be a catalog id.
*
* Only for relations suffix-stripping cannot express: forward vendor aliases
* (`gpt-5.6` → `gpt-5.6-sol`) and cross-provider spellings of the same model
* (cursor's `claude-4.6-opus-high` → `claude-opus-4-6`). Plain effort/`-free`
* variants are derived by `resolveScoresAs` and need no entry here.
*/
scoresAs?: string;
}

// Reasoning models reject temperature, top_p, penalties, logprobs, n.
Expand Down
17 changes: 11 additions & 6 deletions open-sse/services/autoCombo/__tests__/autoCombo.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -101,10 +101,15 @@ describe("Task Fitness", () => {
// The fix: getTaskFitnessWithSource strips a trailing "-free" suffix
// and re-queries arena_elo with the base id. We seed an arena_elo
// row directly via the DB module, look up the free variant, and
// assert the alias path returns the base score with source
// "arena_elo_free_alias".
const baseId = "alias-base-test-4517";
const freeId = "alias-base-test-4517-free";
// assert the alias path returns the base score, now tagged
// "arena_elo:inherited" by the shared resolver (#11489).
//
// #11489 also made the base-id lookup CATALOG-ANCHORED, so this uses a
// real catalog pair instead of the synthetic ids it was written with:
// an id that resolves to a base no provider actually ships is a ghost,
// and inheriting a score for it was never meaningful.
const baseId = "mimo-v2.5";
const freeId = "mimo-v2.5-free";
const { upsertModelIntelligence, deleteModelIntelligence } =
await import("../../../../src/lib/db/modelIntelligence.ts");
// Seed arena_elo on the base id only — no row exists for the free id.
Expand All @@ -121,9 +126,9 @@ describe("Task Fitness", () => {
try {
const result = getTaskFitnessWithSource(freeId, "coding");
// Without the fix: result.source would be "wildcard_boost" (0.5 default).
// With the fix: result.source is "arena_elo_free_alias" with score 0.42.
// With the fix: result.source is "arena_elo:inherited" with score 0.42.
expect(result.score).toBeCloseTo(0.42, 5);
expect(result.source).toBe("arena_elo_free_alias");
expect(result.source).toBe("arena_elo:inherited");
} finally {
deleteModelIntelligence(baseId, "arena_elo", "coding");
invalidateFitnessCache();
Expand Down
92 changes: 92 additions & 0 deletions open-sse/services/autoCombo/__tests__/scoresAs-11489.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
/**
* TDD regression for #11489: auto-combo task fitness scored every catalog id by
* exact string match, while dispatch already resolves `<model>-<effort>` ids back
* to a base model. A variant like `gpt-5.6-sol-xhigh` missed every DB layer and
* landed on the wildcard 0.5, while its base model was scored properly.
*
* `resolveScoresAs` is the shared, catalog-anchored resolver the fitness chain
* consults on a miss. It resolves in three tiers and NEVER guesses:
* 1. an explicit `scoresAs` declared on the registry entry (one hop only),
* 2. a trailing reasoning-effort suffix stripped by an EXISTING splitter,
* 3. a trailing `-free` tier marker,
* and tiers 2–3 only accept a base that is itself a routable catalog id — which
* is what keeps `qwen3.7-max` (where `-max` is the model, not an effort) and
* `grok-4.6-fast-high` (whose stripped base is not in the catalog) unresolved.
*/
import { describe, it, expect } from "vitest";
import { resolveScoresAs } from "../scoresAs";
import { findRegistryModelById } from "../../../config/providerModels";

describe("#11489 resolveScoresAs", () => {
it("strips a reasoning-effort suffix when the base is a catalog id", () => {
expect(resolveScoresAs("gpt-5.6-sol-xhigh")).toEqual({
base: "gpt-5.6-sol",
via: "effort-suffix",
});
});

it("follows a vendor alias that points FORWARD (gpt-5.6 is an alias of gpt-5.6-sol)", () => {
// No suffix-stripper can produce this direction; it is registry data.
expect(resolveScoresAs("gpt-5.6")).toEqual({ base: "gpt-5.6-sol", via: "explicit" });
});

it("leaves sibling models unresolved (gpt-5.6-luna is its own model)", () => {
const result = resolveScoresAs("gpt-5.6-luna");
expect(result.via).toBeNull();
expect(result.base).toBe("gpt-5.6-luna");
expect(result.base).not.toBe("gpt-5.6");
expect(result.base).not.toBe("gpt-5.6-sol");
});

it("rejects an effort-stripped base that is not itself a catalog id", () => {
// `grok-4.6-fast-high` strips to `grok-4.6-fast`, a ghost id on today's
// catalog. Asserted against the catalog rather than hardcoded so the test
// stays true if a provider ever ships the base as a routable id.
const result = resolveScoresAs("grok-4.6-fast-high");
if (findRegistryModelById("grok-4.6-fast")) {
expect(result).toEqual({ base: "grok-4.6-fast", via: "effort-suffix" });
} else {
expect(result).toEqual({ base: "grok-4.6-fast-high", via: null });
}
});

it("does not treat a trailing '-max' that is part of the model name as an effort", () => {
// `qwen3.7-max` IS the model; `qwen3.7` does not exist.
expect(resolveScoresAs("qwen3.7-max")).toEqual({ base: "qwen3.7-max", via: null });
});

it("never collapses a model onto its family", () => {
expect(resolveScoresAs("claude-sonnet-5")).toEqual({ base: "claude-sonnet-5", via: null });
});

it("resolves the cursor/agy spelling of a Claude model to its canonical id", () => {
// `claude-4.6-opus-high` strips to `claude-4.6-opus`, which is not a catalog
// id — the canonical spelling is `claude-opus-4-6`. Explicit registry data.
expect(resolveScoresAs("claude-4.6-opus-high")).toEqual({
base: "claude-opus-4-6",
via: "explicit",
});
expect(resolveScoresAs("claude-4.6-sonnet-medium")).toEqual({
base: "claude-sonnet-4-6",
via: "explicit",
});
});

it("strips a '-free' tier marker when the paid base is a catalog id", () => {
expect(findRegistryModelById("mimo-v2.5")).toBeTruthy();
expect(resolveScoresAs("mimo-v2.5-free")).toEqual({ base: "mimo-v2.5", via: "free-suffix" });
});

it("leaves a '-free' id unresolved when no paid base exists in the catalog", () => {
expect(findRegistryModelById("ox-alpha")).toBeFalsy();
expect(resolveScoresAs("ox-alpha-free")).toEqual({ base: "ox-alpha-free", via: null });
});

it("returns the id unchanged for junk input", () => {
expect(resolveScoresAs("totally-unknown-model")).toEqual({
base: "totally-unknown-model",
via: null,
});
expect(resolveScoresAs("")).toEqual({ base: "", via: null });
});
});
85 changes: 85 additions & 0 deletions open-sse/services/autoCombo/scoresAs.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,85 @@
/**
* scoresAs — resolve a catalog id to the id whose QUALITY scores it inherits (#11489).
*
* Dispatch already treats `<model>-<effort>` ids as variants of a base model
* (`splitClaudeEffortSuffix`, `splitCodexReasoningSuffix` run on the incoming
* request model). Auto-combo's task fitness did not: it scored every catalog id
* by exact string match, so a variant like `gpt-5.6-sol-xhigh` missed every DB
* layer and landed on the wildcard 0.5 while its base model was scored properly.
*
* This module is the single seam that closes that gap. Three tiers, in order:
*
* 1. `explicit` — the registry entry declares `scoresAs`. For relations
* suffix-stripping cannot express: forward vendor aliases
* (`gpt-5.6` IS an alias of `gpt-5.6-sol`, per OpenAI's
* model reference) and cross-provider spellings of the
* same model (`claude-4.6-opus-high` → `claude-opus-4-6`).
* 2. `effort-suffix` — a trailing reasoning-effort token stripped by one of the
* EXISTING dispatch splitters. No new regex is introduced
* here; a fourth pattern would be a fourth place to get
* the same fact wrong (cf. the #8603 shadowing defect).
* 3. `free-suffix` — a trailing `-free` tier marker, so a free-tier variant
* picks up the benchmark of its paid counterpart (#4517,
* previously a standalone arena_elo-only special case in
* `taskFitness.ts`, now folded in and extended to
* `user_override` too).
*
* Tiers 2 and 3 are CATALOG-ANCHORED: a stripped base is accepted only when it
* is itself a routable catalog id. Without that guard, 57 of the catalog's 201
* strippable effort ids resolve to a base that does not exist — `qwen3.7-max`
* would inherit from a phantom `qwen3.7` (`-max` is part of the model name
* here, not an effort), `grok-4.6-fast-high` from a phantom `grok-4.6-fast`,
* and `extra-high` from `extra`. Resolution never guesses: anything the three
* tiers cannot justify comes back unresolved.
*
* One hop only — a `scoresAs` target is not itself re-resolved.
*/
import { findRegistryModelById, findRegistryScoresAs } from "../../config/providerModels.ts";
import { splitClaudeEffortSuffix } from "../../config/providerModels.ts";
import { splitCodexReasoningSuffix } from "../../executors/codex/reasoningSuffix.ts";

/** How a base id was reached; `null` means "not resolved — score the id as given". */
export type ScoresAsVia = "explicit" | "effort-suffix" | "free-suffix" | null;

export interface ScoresAsResolution {
/** The id whose quality scores apply. Equals the input when `via` is `null`. */
base: string;
via: ScoresAsVia;
}

/** Suffix marking a free-tier variant of a paid model (e.g. `mimo-v2.5-free`). */
const FREE_SUFFIX = "-free";

/** Dispatch-time effort splitters, reused verbatim. Order is not significant:
* both are catalog-anchored below, so a wrong strip is rejected either way. */
const EFFORT_SPLITTERS = [splitClaudeEffortSuffix, splitCodexReasoningSuffix] as const;

export function resolveScoresAs(modelId: string): ScoresAsResolution {
const unresolved: ScoresAsResolution = { base: modelId, via: null };
if (typeof modelId !== "string" || modelId.length === 0) return unresolved;

// 1. Explicit registry declaration. One hop: the target must itself be a
// catalog id, and its own `scoresAs` (if any) is deliberately not followed.
const declared = findRegistryScoresAs(modelId);
if (declared && declared !== modelId && findRegistryModelById(declared)) {
return { base: declared, via: "explicit" };
}

// 2. Reasoning-effort suffix, catalog-anchored.
for (const split of EFFORT_SPLITTERS) {
const base = split(modelId).baseModel;
if (base && base !== modelId && findRegistryModelById(base)) {
return { base, via: "effort-suffix" };
}
}

// 3. Free-tier suffix, catalog-anchored.
if (modelId.endsWith(FREE_SUFFIX)) {
const base = modelId.slice(0, -FREE_SUFFIX.length);
if (base.length > 0 && findRegistryModelById(base)) {
return { base, via: "free-suffix" };
}
}

return unresolved;
}
Loading
Loading