Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
- **fix(routing):** reasoning-routing rules that force `max` / `ultra` now keep GPT-6 Astra (and any other model in the Codex max/ultra alias sets) instead of dropping it from the target combo or rejecting it as a single target; `cx/gpt-6-astra-max` / `-ultra` are read as max / ultra requests, and the rules editor offers those tiers for it ([#14720](https://github.com/diegosouzapw/OmniRoute/pull/14720))
21 changes: 11 additions & 10 deletions src/lib/reasoningRouting/policy.ts
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,10 @@ import {
splitClaudeEffortSuffix,
getProviderModels,
} from "@omniroute/open-sse/config/providerModels.ts";
import {
codexModelFamilySupportsExtendedEffort,
isCodexExtendedEffortBaseModel,
} from "@/shared/reasoning/codexExtendedEffort";

type JsonRecord = Record<string, unknown>;
const EFFORTS = new Set<ReasoningEffort>([
Expand Down Expand Up @@ -79,8 +83,9 @@ function splitGenericEffortSuffix(model: string): {
}

function supportsCodexSuffix(candidate: string, normalizedBase: string): boolean {
if (candidate === "max") return /^gpt-5\.6-(?:sol|terra|luna)$/.test(normalizedBase);
if (candidate === "ultra") return /^gpt-5\.6-(?:sol|terra)$/.test(normalizedBase);
if (candidate === "max" || candidate === "ultra") {
return isCodexExtendedEffortBaseModel(normalizedBase, candidate);
}
return true;
}

Expand Down Expand Up @@ -278,7 +283,8 @@ function capabilityFor(
// 2. For unregistered providers/models, a declared (synced or
// operator-overridden) vocabulary listing the tier is authoritative —
// the sanitizer forwards verbatim there (#8057 trust-the-upstream).
// 3. The gpt-5.6 regex remains the fallback for undeclared models.
// 3. The Codex max/ultra alias sets (`codex/reasoningSuffix.ts`) remain
// the fallback for undeclared models.
// This keeps custom OpenAI-compatible providers whose models accept `max`
// natively (e.g. Merge Gateway `zai/glm-5.3-flash`, accepting
// `low|high|max`) usable with forced-max rules instead of 400ing.
Expand Down Expand Up @@ -308,16 +314,11 @@ function capabilityFor(
}
// An operator-declared vocabulary that excludes the tier is terminal —
// the same lookup the override resolves from must not be overruled by the
// legacy regex below.
// alias-set fallback below.
if (capabilities.reasoningEffortsOverride && Array.isArray(declaredEfforts)) {
return "unsupported" as const;
}
const normalized = model.toLowerCase().replace(/^(?:codex|cx)\//, "");
const supported =
targetEffort === "ultra"
? /^gpt-5\.6-(?:sol|terra)(?:-|$)/.test(normalized)
: /^gpt-5\.6-(?:sol|terra|luna)(?:-|$)/.test(normalized);
if (supported) return "supported" as const;
if (codexModelFamilySupportsExtendedEffort(model, targetEffort)) return "supported" as const;
if (capabilities.supportsThinking === null) return "unknown" as const;
return "unsupported" as const;
}
Expand Down
18 changes: 6 additions & 12 deletions src/shared/components/ReasoningRoutingRules.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
import { useCallback, useEffect, useMemo, useState } from "react";
import { useTranslations } from "next-intl";
import { useRef } from "react";
import { codexModelFamilySupportsExtendedEffort } from "@/shared/reasoning/codexExtendedEffort";
import Button from "./Button";
import Card from "./Card";
import Input from "./Input";
Expand Down Expand Up @@ -91,16 +92,6 @@ function emptyRule(apiKeyId?: string): FormState {
};
}

function supportsExtendedCodexEffort(model: string, effort: "max" | "ultra"): boolean {
const normalized = model
.trim()
.toLowerCase()
.replace(/^(?:codex|cx)\//, "");
return effort === "ultra"
? /^gpt-5\.6-(?:sol|terra)(?:-|$)/.test(normalized)
: /^gpt-5\.6-(?:sol|terra|luna)(?:-|$)/.test(normalized);
}

export default function ReasoningRoutingRules({
initialApiKeyId = "",
}: {
Expand Down Expand Up @@ -216,7 +207,10 @@ export default function ReasoningRoutingRules({
const values = [...STANDARD_EFFORTS];
for (const effort of EXTENDED_EFFORTS) {
if (
supportsExtendedCodexEffort(targetModelForCapability, effort as "max" | "ultra") ||
codexModelFamilySupportsExtendedEffort(
targetModelForCapability,
effort as "max" | "ultra"
) ||
form.targetEffort === effort
) {
values.push(effort);
Expand All @@ -230,7 +224,7 @@ export default function ReasoningRoutingRules({
if (!EXTENDED_EFFORTS.includes(form.targetEffort)) return "";
if (form.targetKind === "combo") return t("extendedComboWarning");
if (!targetModelForCapability.trim()) return t("extendedUnknownWarning");
return supportsExtendedCodexEffort(
return codexModelFamilySupportsExtendedEffort(
targetModelForCapability,
form.targetEffort as "max" | "ultra"
)
Expand Down
44 changes: 44 additions & 0 deletions src/shared/reasoning/codexExtendedEffort.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,44 @@
import {
CODEX_MAX_ALIAS_MODELS,
CODEX_ULTRA_ALIAS_MODELS,
} from "@omniroute/open-sse/executors/codex/reasoningSuffix.ts";

// Which Codex models accept the `max` / `ultra` reasoning tiers, read from the
// alias sets the Codex executor uses so routing rules and the rules editor
// cannot drift from what the executor serves.

export type CodexExtendedEffort = "max" | "ultra";

function modelsFor(effort: CodexExtendedEffort): ReadonlySet<string> {
return effort === "ultra" ? CODEX_ULTRA_ALIAS_MODELS : CODEX_MAX_ALIAS_MODELS;
}

function normalizeCodexModelId(model: string): string {
return model
.trim()
.toLowerCase()
.replace(/^(?:codex|cx)\//, "");
}

/** True when `model` (optionally `codex/` or `cx/` prefixed) is exactly a base model that accepts `effort`. */
export function isCodexExtendedEffortBaseModel(
model: string,
effort: CodexExtendedEffort
): boolean {
return modelsFor(effort).has(normalizeCodexModelId(model));
}

/**
* Like {@link isCodexExtendedEffortBaseModel}, but also accepts the base
* model's variants (`<base>-<suffix>`, e.g. `cx/gpt-6-astra-high`).
*/
export function codexModelFamilySupportsExtendedEffort(
model: string,
effort: CodexExtendedEffort
): boolean {
const normalized = normalizeCodexModelId(model);
for (const base of modelsFor(effort)) {
if (normalized === base || normalized.startsWith(`${base}-`)) return true;
}
return false;
}
177 changes: 177 additions & 0 deletions tests/unit/reasoning-routing-codex-extended-effort.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,177 @@
import test from "node:test";
import assert from "node:assert/strict";
import fs from "node:fs";
import os from "node:os";
import path from "node:path";
import { cleanupTempDataDir } from "../_setup/tempDataDir.ts";

// Reasoning-routing rules decided which Codex models accept `max` / `ultra` with a
// hard-coded gpt-5.6 regex, so GPT-6 models the Codex executor already serves at
// those tiers were read as unsupported: a rule forcing `max` dropped cx/gpt-6-astra
// from its target combo (or rejected it as a single target), and cx/gpt-6-astra-max
// was not read as a max request. The gate now reads the executor's alias sets
// (open-sse/executors/codex/reasoningSuffix.ts).

const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-reasoning-extended-"));
process.env.DATA_DIR = TEST_DATA_DIR;

const core = await import("../../src/lib/db/core.ts");
const combosDb = await import("../../src/lib/db/combos.ts");
const rulesDb = await import("../../src/lib/db/reasoningRoutingRules.ts");
const policy = await import("../../src/lib/reasoningRouting/policy.ts");
const { CODEX_MAX_ALIAS_MODELS, CODEX_ULTRA_ALIAS_MODELS } =
await import("../../open-sse/executors/codex/reasoningSuffix.ts");
const { codexModelFamilySupportsExtendedEffort, isCodexExtendedEffortBaseModel } =
await import("../../src/shared/reasoning/codexExtendedEffort.ts");

const ALIAS_MODELS = new Set([...CODEX_MAX_ALIAS_MODELS, ...CODEX_ULTRA_ALIAS_MODELS]);

function resetStorage() {
core.resetDbInstance();
fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 });
fs.mkdirSync(TEST_DATA_DIR, { recursive: true });
rulesDb.invalidateReasoningRoutingRuleCache();
}

function forcedRule(
targetEffort: "max" | "ultra",
target: { targetKind: "model" | "combo"; targetModel?: string; targetComboId?: string }
): rulesDb.ReasoningRoutingRuleInput {
return {
name: `force ${targetEffort}`,
description: "",
scope: "global",
apiKeyId: null,
comboId: null,
connectionId: null,
modelPattern: "reasoning-source",
sourceEffort: "any",
requestTags: [],
tagMatchMode: "any",
effortMode: "force",
targetEffort,
targetKind: target.targetKind,
targetModel: target.targetModel ?? null,
targetComboId: target.targetComboId ?? null,
budgetAction: "preserve",
budgetTokens: null,
priority: 0,
enabled: true,
};
}

async function decide(rule: rulesDb.ReasoningRoutingRuleInput) {
await rulesDb.createReasoningRoutingRule(rule);
const decision = await policy.resolveReasoningRoutingRule({
sourceModel: "reasoning-source",
sourceEffort: "missing",
hasReasoningSignal: false,
});
assert.ok(decision, "the forced rule must match");
return decision;
}

async function removedFromCombo(models: string[], targetEffort: "max" | "ultra") {
const combo = await combosDb.createCombo({
name: "extended-effort-target",
models,
strategy: "priority",
});
const decision = await decide(
forcedRule(targetEffort, { targetKind: "combo", targetComboId: String(combo.id) })
);
return policy.filterComboForReasoningDecision(decision.targetCombo, decision).removed;
}

test.beforeEach(resetStorage);

test.after(async () => {
resetStorage();
await cleanupTempDataDir(TEST_DATA_DIR);
});

test("forced max keeps GPT-6 Astra in a Codex combo", async () => {
assert.deepEqual(
await removedFromCombo(["cx/gpt-6-astra", "codex/gpt-6-astra-high", "cx/gpt-5.6-sol"], "max"),
[]
);
});

test("forced ultra keeps GPT-6 Astra and still drops max-only models", async () => {
assert.deepEqual(await removedFromCombo(["cx/gpt-6-astra", "cx/gpt-5.6-luna"], "ultra"), [
"cx/gpt-5.6-luna",
]);
});

test("forced max/ultra on a single GPT-6 Astra target is supported", async () => {
for (const effort of ["max", "ultra"] as const) {
resetStorage();
const decision = await decide(
forcedRule(effort, { targetKind: "model", targetModel: "cx/gpt-6-astra" })
);
assert.equal(decision.capability, "supported", effort);
}
});

test("every Codex max/ultra alias model passes the gate for its tiers", async () => {
const maxModels = [...CODEX_MAX_ALIAS_MODELS].map((model) => `cx/${model}`);
assert.deepEqual(await removedFromCombo(maxModels, "max"), []);

resetStorage();
const allModels = [...ALIAS_MODELS];
const notUltra = allModels
.filter((model) => !CODEX_ULTRA_ALIAS_MODELS.has(model))
.map((model) => `cx/${model}`);
assert.deepEqual(
await removedFromCombo(
allModels.map((model) => `cx/${model}`),
"ultra"
),
notUltra
);
});

test("max/ultra suffixes on GPT-6 Astra are read as the requested effort", () => {
for (const effort of ["max", "ultra"] as const) {
const intent = policy.extractReasoningIntent(`cx/gpt-6-astra-${effort}`, {});
assert.equal(intent.model, "cx/gpt-6-astra");
assert.equal(intent.effort, effort);
assert.equal(intent.sourceEffort, effort);
}
});

test("suffix parsing follows the alias sets for every model", () => {
for (const model of ALIAS_MODELS) {
for (const effort of ["max", "ultra"] as const) {
const supported = (
effort === "ultra" ? CODEX_ULTRA_ALIAS_MODELS : CODEX_MAX_ALIAS_MODELS
).has(model);
const intent = policy.extractReasoningIntent(`codex/${model}-${effort}`, {});
assert.equal(intent.effort, supported ? effort : null, `${model}-${effort}`);
assert.equal(
intent.model,
supported ? `codex/${model}` : `codex/${model}-${effort}`,
`${model}-${effort}`
);
}
}
});

test("suffix parsing needs the exact base model", () => {
for (const model of ["cx/gpt-6-astra-high-max", "cx/gpt-6-astral-max"]) {
const intent = policy.extractReasoningIntent(model, {});
assert.equal(intent.model, model);
assert.equal(intent.effort, null);
}
});

test("model helpers match whole Codex model ids", () => {
assert.equal(isCodexExtendedEffortBaseModel("CX/GPT-6-Astra", "ultra"), true);
assert.equal(isCodexExtendedEffortBaseModel("cx/gpt-6-astra-high", "max"), false);
assert.equal(codexModelFamilySupportsExtendedEffort("gpt-6-astra", "max"), true);
assert.equal(codexModelFamilySupportsExtendedEffort(" CX/GPT-6-Astra-High ", "ultra"), true);
assert.equal(codexModelFamilySupportsExtendedEffort("gpt-6-astral", "max"), false);
assert.equal(codexModelFamilySupportsExtendedEffort("openai/gpt-6-astra", "max"), false);
assert.equal(codexModelFamilySupportsExtendedEffort("cx/gpt-5.6-luna", "ultra"), false);
assert.equal(codexModelFamilySupportsExtendedEffort("", "max"), false);
});
12 changes: 6 additions & 6 deletions tests/unit/reasoning-routing.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -344,8 +344,8 @@ test("static registry vocabulary outranks the operator override so the gate matc
);

// Case 2: registry-declared model, operator override NARROWS to exclude
// max. The override is terminal — the legacy gpt-5.6 regex must not
// resurrect the tier (grok ids never matched that regex, but the
// max. The override is terminal — the Codex alias-set fallback must not
// resurrect the tier (grok ids are not in those sets, but the
// precedence guarantee must not depend on the id shape).
setModelCapabilityOverride(registeredModel, "reasoning_efforts", ["low", "high"]);
const narrowed = await policy.resolveReasoningRoutingRule({
Expand All @@ -356,7 +356,7 @@ test("static registry vocabulary outranks the operator override so the gate matc
assert.equal(
narrowed?.capability,
"unsupported",
"a narrowed operator override is terminal and must not fall through to the legacy regex"
"a narrowed operator override is terminal and must not fall through to the Codex alias-set fallback"
);

// Case 3: registry model WITHOUT any declared vocabulary, operator
Expand Down Expand Up @@ -395,8 +395,8 @@ test("static registry vocabulary outranks the operator override so the gate matc
"alias-spelled provider prefix must resolve to the same registry namespace"
);

// Case 5: a narrowing override on a gpt-5.6 id is terminal. The legacy
// regex matches this exact id shape — without the terminal check it would
// Case 5: a narrowing override on a gpt-5.6 id is terminal. The Codex
// alias-set fallback matches this exact id — without the terminal check it would
// resurrect forced max the operator explicitly declared away.
const gpt56Model = "codex/gpt-5.6-sol";
setModelCapabilityOverride(gpt56Model, "reasoning_efforts", ["low", "high"]);
Expand All @@ -409,6 +409,6 @@ test("static registry vocabulary outranks the operator override so the gate matc
assert.equal(
denied56.capability,
"unsupported",
"operator narrowing override on gpt-5.6 must not be overruled by the legacy regex"
"operator narrowing override on gpt-5.6 must not be overruled by the Codex alias-set fallback"
);
});