Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion config/quality/eslint-suppressions.json
Original file line number Diff line number Diff line change
Expand Up @@ -1788,7 +1788,7 @@
},
"tests/unit/chatcore-translation-paths.test.ts": {
"@typescript-eslint/no-explicit-any": {
"count": 34
"count": 31
}
},
"tests/unit/chatgpt-web-tools-5240.test.ts": {
Expand Down
1 change: 1 addition & 0 deletions open-sse/config/constants.ts
Original file line number Diff line number Diff line change
Expand Up @@ -172,6 +172,7 @@ export const HTTP_STATUS = {
NOT_FOUND: 404,
NOT_ACCEPTABLE: 406,
REQUEST_TIMEOUT: 408,
GONE: 410,
RATE_LIMITED: 429,
SERVER_ERROR: 500,
BAD_GATEWAY: 502,
Expand Down
2 changes: 2 additions & 0 deletions open-sse/config/errorConfig.ts
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,7 @@ export const ERROR_TYPES: Record<number, ErrorInfo> = {
403: { type: "permission_error", code: "insufficient_quota" },
404: { type: "invalid_request_error", code: "model_not_found" },
406: { type: "invalid_request_error", code: "model_not_supported" },
410: { type: "invalid_request_error", code: "model_shutdown" },
429: { type: "rate_limit_error", code: "rate_limit_exceeded" },
499: { type: "client_disconnected", code: "client_disconnected" },
500: { type: "server_error", code: "internal_server_error" },
Expand All @@ -44,6 +45,7 @@ export const DEFAULT_ERROR_MESSAGES: Record<number, string> = {
403: "You exceeded your current quota",
404: "Model not found",
406: "Model not supported",
410: "Model has been shut down",
429: "Rate limit exceeded",
499: "Client disconnected",
500: "Internal server error",
Expand Down
34 changes: 17 additions & 17 deletions open-sse/handlers/chatCore.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@ import { extractSystemRoleMessages } from "./chatCore/claudeSystemRole.ts";
export { extractSystemRoleMessages } from "./chatCore/claudeSystemRole.ts";
import { checkIdempotencyCache } from "./chatCore/idempotency.ts";
import { checkSemanticCache } from "./chatCore/semanticCache.ts";
import { checkLifecycle, resolveLifecycle } from "./chatCore/modelLifecyclePolicy.ts";
import {
shouldDefaultAllowClassifier,
buildDefaultAllowClaudeMessage,
Expand Down Expand Up @@ -116,7 +117,6 @@ import {
getStripTypesForProviderModel,
stripIncompatibleMessageContent,
} from "../services/modelStrip.ts";
import { resolveModelAlias } from "../services/modelDeprecation.ts";
import { normalizeMimoThinking } from "../services/mimoThinking.ts";
import {
isOpencodeGoProvider,
Expand Down Expand Up @@ -643,6 +643,9 @@ export async function handleChatCore({
responsesInputItems
);

const requestedLifecycleError = checkLifecycle(provider, model, log);
if (requestedLifecycleError) return requestedLifecycleError;

// Check for bypass patterns (warmup, skip) - return fake response
const bypassResponse = handleBypassRequest(body, model, userAgent);
if (bypassResponse) {
Expand Down Expand Up @@ -707,16 +710,13 @@ export async function handleChatCore({
});
}

// Apply custom model aliases (Settings → Model Aliases → Pattern→Target) before routing (#315, #472)
// Custom aliases take priority over built-in and must be resolved here so the
// downstream getModelTargetFormat() lookup AND the actual provider request use
// the correct, aliased model ID. Without this, aliases only affect format detection.
const resolvedModel = resolveModelAlias(model);
// Use resolvedModel for all downstream operations (routing, provider requests, logging)
let effectiveModel = resolvedModel === model ? model : resolvedModel;
if (resolvedModel !== model) {
log?.info?.("ALIAS", `Model alias applied: ${model} → ${resolvedModel}`);
}
// Custom aliases remain explicit; lifecycle replacements are advisory and never silently routed.
let [resolvedModel, effectiveModel, routedLifecycleError] = resolveLifecycle(
provider,
model,
log
);
if (routedLifecycleError) return routedLifecycleError;

// Effort-variant model ids: the Claude / Claude-Code model picker (e.g. VS Code's
// "Effort" slider) advertises claude-...-{low,medium,high,xhigh,max}. Anthropic has
Expand Down Expand Up @@ -3870,8 +3870,8 @@ export async function handleChatCore({
// Before returning a model-unavailable error upstream, try sibling models
// from the same family. This keeps the request alive on the same account
// instead of failing the entire combo.
if (isModelUnavailableError(statusCode, message)) {
const nextModel = getNextFamilyFallback(currentModel, triedModels);
if (isModelUnavailableError(statusCode, message, provider)) {
const nextModel = getNextFamilyFallback(currentModel, triedModels, provider);
if (nextModel) {
triedModels.add(nextModel);
currentModel = nextModel;
Expand Down Expand Up @@ -3957,12 +3957,12 @@ export async function handleChatCore({
);
}
} else if (isContextOverflowError(statusCode, message)) {
const familyCandidates = getModelFamily(currentModel).filter(
const familyCandidates = getModelFamily(currentModel, provider).filter(
(m) => m !== currentModel && !triedModels.has(m)
);
const nextModel =
findLargerContextModel(currentModel, familyCandidates) ??
getNextFamilyFallback(currentModel, triedModels);
findLargerContextModel(currentModel, familyCandidates, provider) ??
getNextFamilyFallback(currentModel, triedModels, provider);
if (nextModel) {
triedModels.add(nextModel);
currentModel = nextModel;
Expand Down Expand Up @@ -4213,7 +4213,7 @@ export async function handleChatCore({
persistFailureUsage(HTTP_STATUS.BAD_GATEWAY, "empty_content");

// Trigger non-recursive fallback for empty content
const nextModel = getNextFamilyFallback(currentModel, triedModels);
const nextModel = getNextFamilyFallback(currentModel, triedModels, provider);
if (nextModel) {
triedModels.add(nextModel);
currentModel = nextModel;
Expand Down
60 changes: 60 additions & 0 deletions open-sse/handlers/chatCore/modelLifecyclePolicy.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,60 @@
import { HTTP_STATUS } from "../../config/constants.ts";
import {
formatModelLifecycleMessage,
getModelLifecycleDecision,
} from "../../services/modelLifecycle.ts";
import { resolveModelAlias } from "../../services/modelDeprecation.ts";
import { createErrorResult } from "../../utils/error.ts";

type LifecycleLogger = {
info?: (tag: string, message: string) => unknown;
warn?: (tag: string, message: string) => unknown;
} | null;

function getModelLifecycleError({
provider,
model,
log,
warnOnDeprecation = false,
}: {
provider: string;
model: string;
log?: LifecycleLogger;
warnOnDeprecation?: boolean;
}): ReturnType<typeof createErrorResult> | null {
const decision = getModelLifecycleDecision(provider, model);
const message = formatModelLifecycleMessage(decision);
if (message && (decision.action === "reject" || warnOnDeprecation)) {
log?.warn?.("MODEL_LIFECYCLE", message);
}
if (decision.action !== "reject" || !message) return null;
return createErrorResult(
HTTP_STATUS.GONE,
message,
null,
"model_shutdown",
"invalid_request_error"
);
}

export function checkLifecycle(provider: string, model: string, log?: LifecycleLogger) {
return getModelLifecycleError({
provider,
model: resolveModelAlias(model),
log,
});
}

export function resolveLifecycle(provider: string, model: string, log?: LifecycleLogger) {
const resolvedModel = resolveModelAlias(model);
if (resolvedModel !== model) {
log?.info?.("ALIAS", `Model alias applied: ${model} → ${resolvedModel}`);
}
const lifecycleError = getModelLifecycleError({
provider,
model: resolvedModel,
log,
warnOnDeprecation: true,
});
return [resolvedModel, resolvedModel === model ? model : resolvedModel, lifecycleError] as const;
}
11 changes: 1 addition & 10 deletions open-sse/services/modelDeprecation.ts
Original file line number Diff line number Diff line change
Expand Up @@ -32,12 +32,6 @@ const BUILT_IN_ALIASES: Record<string, string> = {
"claude-3-5-sonnet-latest": "claude-sonnet-4-20250514",
"claude-3-5-haiku-latest": "claude-3-5-sonnet-20241022",

// OpenAI legacy → current
"gpt-4-turbo-preview": "gpt-4-turbo",
"gpt-4-0125-preview": "gpt-4-turbo",
"gpt-4-1106-preview": "gpt-4-turbo",
"gpt-3.5-turbo-0125": "gpt-3.5-turbo",

// Kimi/Moonshot — Fireworks long-path aliases (#265)
"accounts/fireworks/models/kimi-k2p5": "moonshotai/Kimi-K2.5",
"fireworks/accounts/fireworks/models/kimi-k2p5": "moonshotai/Kimi-K2.5",
Expand Down Expand Up @@ -70,10 +64,7 @@ const BUILT_IN_ALIASES: Record<string, string> = {
// root cause (both instances read/write one store), mirroring the #5312 pattern already
// applied to thinkingBudget.ts and backgroundTaskDetector.ts (and systemPrompt.ts #2470).
const CUSTOM_ALIASES_GLOBAL_KEY = "__omniroute_customAliases__";
const _aliasStore = globalThis as unknown as Record<
string,
Record<string, string> | undefined
>;
const _aliasStore = globalThis as unknown as Record<string, Record<string, string> | undefined>;

function customAliases(): Record<string, string> {
if (!_aliasStore[CUSTOM_ALIASES_GLOBAL_KEY]) {
Expand Down
114 changes: 114 additions & 0 deletions open-sse/services/modelEndpointPolicy.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,114 @@
/**
* Provider model endpoint policy.
*
* Upstream `/models` responses often omit endpoint/modality metadata. In that
* case, specialty models can otherwise be imported as chat models simply
* because "chat" is OmniRoute's historical default. Keep the exceptional
* provider knowledge here so discovery, import, and catalog projection agree.
*/

export type ModelEndpointKind = "chat" | "image" | "video" | "non-chat" | "unknown";

export type ModelEndpointDecision = {
kind: ModelEndpointKind;
chatSelectable: boolean;
reason: "explicit-endpoints" | "provider-policy" | "unclassified";
};

type EndpointAwareModel = {
id: string;
supportedEndpoints?: readonly string[];
};

const CHAT_ENDPOINTS = new Set([
"chat",
"chat-completions",
"chat/completions",
"messages",
"responses",
]);
const IMAGE_ENDPOINTS = new Set(["image", "images", "images/generations"]);
const VIDEO_ENDPOINTS = new Set(["video", "videos", "videos/generations"]);

function normalizeEndpoint(endpoint: string): string {
return endpoint.trim().toLowerCase().replace(/^\/+/, "").replace(/^v1\//, "");
}

function classifyExplicitEndpoints(
supportedEndpoints: readonly string[] | undefined
): ModelEndpointDecision | null {
if (!supportedEndpoints?.length) return null;

const endpoints = supportedEndpoints.map(normalizeEndpoint).filter(Boolean);
if (endpoints.some((endpoint) => CHAT_ENDPOINTS.has(endpoint))) {
return { kind: "chat", chatSelectable: true, reason: "explicit-endpoints" };
}
if (endpoints.some((endpoint) => IMAGE_ENDPOINTS.has(endpoint))) {
return { kind: "image", chatSelectable: false, reason: "explicit-endpoints" };
}
if (endpoints.some((endpoint) => VIDEO_ENDPOINTS.has(endpoint))) {
return { kind: "video", chatSelectable: false, reason: "explicit-endpoints" };
}
return { kind: "non-chat", chatSelectable: false, reason: "explicit-endpoints" };
}

function normalizeOpenAiModelId(modelId: string): string {
return modelId.startsWith("openai/") ? modelId.slice("openai/".length) : modelId;
}

function classifyOpenAiModel(modelId: string): ModelEndpointDecision | null {
const normalized = normalizeOpenAiModelId(modelId).toLowerCase();
if (
normalized.startsWith("gpt-image-") ||
normalized.startsWith("dall-e-") ||
normalized === "chatgpt-image-latest"
) {
return { kind: "image", chatSelectable: false, reason: "provider-policy" };
}
if (normalized.startsWith("sora-")) {
return { kind: "video", chatSelectable: false, reason: "provider-policy" };
}
return null;
}

export function getModelEndpointDecision(
provider: string | null | undefined,
modelId: string,
supportedEndpoints?: readonly string[]
): ModelEndpointDecision {
const explicit = classifyExplicitEndpoints(supportedEndpoints);
if (provider?.trim().toLowerCase() === "openai") {
const openAiDecision = classifyOpenAiModel(modelId);
if (openAiDecision) {
// Old imported rows were persisted with `["chat"]` as a synthetic default
// even when upstream `/models` supplied no endpoint metadata. Do not let
// that default reclassify a known specialty model. A genuinely
// multi-endpoint model can opt in by explicitly naming both its specialty
// endpoint and a chat/Responses endpoint.
const normalizedEndpoints = supportedEndpoints?.map(normalizeEndpoint) ?? [];
const hasSpecialtyEndpoint =
openAiDecision.kind === "image"
? normalizedEndpoints.some((endpoint) => IMAGE_ENDPOINTS.has(endpoint))
: normalizedEndpoints.some((endpoint) => VIDEO_ENDPOINTS.has(endpoint));
if (explicit?.chatSelectable && hasSpecialtyEndpoint) return explicit;
return openAiDecision;
}
}

if (explicit) return explicit;
return { kind: "unknown", chatSelectable: true, reason: "unclassified" };
}

export function isChatSelectableModel(
provider: string | null | undefined,
model: EndpointAwareModel
): boolean {
return getModelEndpointDecision(provider, model.id, model.supportedEndpoints).chatSelectable;
}

export function filterChatSelectableModels<T extends EndpointAwareModel>(
provider: string | null | undefined,
models: readonly T[]
): T[] {
return models.filter((model) => isChatSelectableModel(provider, model));
}
Loading
Loading