Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
- **fix(combo):** defer the known-context-overflow hard rejection for compressible requests so compression runs before the final context gate, instead of a raw-body estimate 400'ing generic Responses clients targeting a large model before OmniRoute can shrink it ([#10225](https://github.com/diegosouzapw/OmniRoute/issues/10225))
35 changes: 35 additions & 0 deletions open-sse/services/combo.ts
Original file line number Diff line number Diff line change
Expand Up @@ -591,6 +591,11 @@ export async function handleComboChat({
nesting = null,
hiddenModelsByProvider = getHiddenModelsByProvider(),
clientManagedResponsesContext = false,
deferContextOverflowWhenCompressible = false,
compressionExclusions,
sourceFormat = null,
endpointPath = null,
requestHeaders = null,
}: HandleComboChatOptions): Promise<Response> {
const comboCtx = createComboContext({ body, combo, settings, relayOptions, log });
const {
Expand Down Expand Up @@ -651,6 +656,11 @@ export async function handleComboChat({
signal,
apiKeyAllowedConnections,
hiddenModelsByProvider,
deferContextOverflowWhenCompressible,
compressionExclusions,
sourceFormat,
endpointPath,
requestHeaders,
runCombo: handleComboChat,
});
if (fusionDispatch) return fusionDispatch;
Expand Down Expand Up @@ -700,6 +710,11 @@ export async function handleComboChat({
signal,
apiKeyAllowedConnections,
hiddenModelsByProvider,
deferContextOverflowWhenCompressible,
compressionExclusions,
sourceFormat,
endpointPath,
requestHeaders,
runCombo: handleComboChat,
});
if (runtimeUnitDispatch) return runtimeUnitDispatch;
Expand All @@ -723,6 +738,11 @@ export async function handleComboChat({
signal,
hiddenModelsByProvider,
clientManagedResponsesContext,
deferContextOverflowWhenCompressible,
compressionExclusions,
sourceFormat,
endpointPath,
requestHeaders,
relayOptions,
});
}
Expand Down Expand Up @@ -750,6 +770,11 @@ export async function handleComboChat({
buildAutoCandidates,
hiddenModelsByProvider,
clientManagedResponsesContext,
deferContextOverflowWhenCompressible,
compressionExclusions,
sourceFormat,
endpointPath,
requestHeaders,
});
if ("earlyResponse" in targetResolution) return targetResolution.earlyResponse;
const { stickyWeightedLimit, getWeightedStepKeyForTarget, preScreenMap } = targetResolution;
Expand Down Expand Up @@ -2441,6 +2466,11 @@ async function handleRoundRobinCombo({
nesting = null,
hiddenModelsByProvider = getHiddenModelsByProvider(),
clientManagedResponsesContext,
deferContextOverflowWhenCompressible = false,
compressionExclusions,
sourceFormat = null,
endpointPath = null,
requestHeaders = null,
relayOptions,
}: HandleRoundRobinOptions): Promise<Response> {
const config = settings
Expand Down Expand Up @@ -2498,6 +2528,11 @@ async function handleRoundRobinCombo({
const evalRankedTargets = orderTargetsByEvalScores(tagFilteredTargets, config.evalRouting, log);
const knownContextOverflow = getKnownContextOverflow(evalRankedTargets, body, {
clientManagedResponsesContext,
deferContextOverflowWhenCompressible,
compressionExclusions,
sourceFormat,
endpointPath,
requestHeaders,
});
if (knownContextOverflow) {
return errorResponseWithComboDiagnostics(
Expand Down
23 changes: 23 additions & 0 deletions open-sse/services/combo/dispatchPrelude.ts
Original file line number Diff line number Diff line change
Expand Up @@ -76,6 +76,14 @@ type PreludeBaseOptionArgs = {
apiKeyAllowedConnections?: string[] | null;
hiddenModelsByProvider?: HiddenModelsByProvider;
clientManagedResponsesContext?: boolean;
/** #10225 — defer the hard context-overflow preflight when compression is enabled. */
deferContextOverflowWhenCompressible?: boolean;
/** Server-side compression exclusions (#8034). */
compressionExclusions?: import("../compression/exclusions.ts").CompressionExclusions;
/** #10503 — request-shape facts for the target-aware deferral check (see knownContextOverflow.ts). */
sourceFormat?: string | null;
endpointPath?: string | null;
requestHeaders?: Headers | Record<string, unknown> | null;
};

/** Rebuild handleComboChat's option bag verbatim for a recursive dispatch. */
Expand All @@ -93,6 +101,11 @@ function buildBaseOptions(a: PreludeBaseOptionArgs): HandleComboChatOptions {
apiKeyAllowedConnections: a.apiKeyAllowedConnections,
hiddenModelsByProvider: a.hiddenModelsByProvider,
clientManagedResponsesContext: a.clientManagedResponsesContext,
deferContextOverflowWhenCompressible: a.deferContextOverflowWhenCompressible,
compressionExclusions: a.compressionExclusions,
sourceFormat: a.sourceFormat,
endpointPath: a.endpointPath,
requestHeaders: a.requestHeaders,
};
}

Expand Down Expand Up @@ -366,6 +379,11 @@ export async function tryFusionDispatch(args: {
signal?: AbortSignal | null;
apiKeyAllowedConnections?: string[] | null;
hiddenModelsByProvider?: HiddenModelsByProvider;
deferContextOverflowWhenCompressible?: boolean;
compressionExclusions?: import("../compression/exclusions.ts").CompressionExclusions;
sourceFormat?: string | null;
endpointPath?: string | null;
requestHeaders?: Headers | Record<string, unknown> | null;
runCombo: RunCombo;
}): Promise<Response | null> {
const { cfg, combo, config, strategy, log } = args;
Expand Down Expand Up @@ -589,6 +607,11 @@ export async function tryRuntimeUnitDispatch(args: {
signal?: AbortSignal | null;
apiKeyAllowedConnections?: string[] | null;
hiddenModelsByProvider?: HiddenModelsByProvider;
deferContextOverflowWhenCompressible?: boolean;
compressionExclusions?: import("../compression/exclusions.ts").CompressionExclusions;
sourceFormat?: string | null;
endpointPath?: string | null;
requestHeaders?: Headers | Record<string, unknown> | null;
runCombo: RunCombo;
}): Promise<Response | null> {
const { body, combo, config, strategy, allCombos, log, settings } = args;
Expand Down
76 changes: 75 additions & 1 deletion open-sse/services/combo/knownContextOverflow.ts
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,8 @@
*/

import { getResolvedModelCapabilities } from "../modelCapabilities.ts";
import { isCompressionExcluded, type CompressionExclusions } from "../compression/exclusions.ts";
import { shouldUseNativeCodexPassthrough } from "../../handlers/chatCore/passthroughHelpers.ts";
import { deriveRequestCompatibilityRequirements } from "./comboStructure.ts";
import type { ResolvedComboTarget } from "./types.ts";

Expand All @@ -28,6 +30,29 @@ export type KnownContextOverflow = {
targetCount: number;
};

export type KnownContextOverflowOptions = {
clientManagedResponsesContext?: boolean;
/**
* When prompt compression is enabled for this request (global compression switch
* AND not API-key opted-out), defer the hard preflight so chatCore's compression
* pipeline runs before the final context gate — instead of a raw-body estimate
* rejecting a compressible request up front. (#10225)
*/
deferContextOverflowWhenCompressible?: boolean;
/** Server-side compression exclusions (#8034) — targets matching one cannot run compression. */
compressionExclusions?: CompressionExclusions;
/**
* #10503: the exact request-shape facts chatCore.ts uses to decide
* `shouldUseNativeCodexPassthrough` (open-sse/handlers/chatCore/passthroughHelpers.ts) —
* threaded down so the deferral decision below can be target-aware instead of
* relying on the looser `clientManagedResponsesContext` proxy. Reused verbatim
* (not re-derived) so the combo-layer decision can never drift from chatCore's own.
*/
sourceFormat?: string | null;
endpointPath?: string | null;
requestHeaders?: Headers | Record<string, unknown> | null;
};

// #7177: an empty array/object (e.g. a default `messages: []` some combo entrypoints inject
// when the caller sent none) has no real content — counting it would charge a few phantom
// "structural" tokens (JSON.stringify braces/brackets) toward the estimate, which is enough
Expand Down Expand Up @@ -69,7 +94,7 @@ export function getKnownContextLimit(
export function getKnownContextOverflow(
targets: ResolvedComboTarget[],
body: Record<string, unknown>,
options: { clientManagedResponsesContext?: boolean } = {}
options: KnownContextOverflowOptions = {}
): KnownContextOverflow | null {
if (targets.length === 0) return null;
// Native Codex Responses clients compact their own item history. Let the concrete
Expand All @@ -85,6 +110,55 @@ export function getKnownContextOverflow(
) {
return null;
}
// #10225 / #10499-sweep #10503: a conservative raw-body context estimate must not
// be treated as proof that a compression-enabled request cannot fit. When
// compression is available for this request AND at least one target can actually
// run it, defer the hard rejection so handleChatCore runs proactive compression
// (chatCore.ts) and its post-compression enforceOutputTokenBudget becomes the
// final context gate — returning a local `context_length_exceeded` only if the
// compressed body still cannot fit (no upstream dispatch).
//
// Target-awareness is load-bearing here: a target is only a valid reason to defer
// when handleChatCore will ACTUALLY attempt compression for it. Two classes are
// excluded from "can compress" even though `isCompressionExcluded` (operator
// exclusions) says nothing about them:
// - Operator-excluded targets (#8034, existing `isCompressionExcluded` check).
// - Native Codex Responses passthrough targets: chatCore.ts unconditionally sets
// `compressionExcluded = nativeCodexPassthrough || ...` for these, computed via
// `shouldUseNativeCodexPassthrough()` (chatCore/passthroughHelpers.ts) — called
// here with the SAME request-shape facts (sourceFormat/endpointPath/headers)
// chatCore itself uses, reused verbatim rather than re-derived from the looser
// `clientManagedResponsesContext` flag (which always requires a VERIFIED native
// client; chatCore's own gate does NOT for provider==="codex" — see
// shouldUseNativeCodexPassthrough's `provider === "codex" || isVerifiedNativeCodexRequest`
// short-circuit). Deferring on such a target's account would let an oversized
// body sail straight through to `fetch()` uncompressed instead of being caught
// by either preflight — silently defeating the whole point of this feature.
// If NO target can compress, the fast raw-body preflight is kept (unchanged).
if (
options.deferContextOverflowWhenCompressible === true &&
targets.some((target) => {
const isNativeCodexPassthroughTarget = shouldUseNativeCodexPassthrough({
provider: target.provider,
sourceFormat: options.sourceFormat,
endpointPath: options.endpointPath,
body,
headers: options.requestHeaders,
});
if (isNativeCodexPassthroughTarget) return false;
return !isCompressionExcluded(
{
provider: target.provider,
model: target.modelStr.includes("/")
? target.modelStr.split("/").slice(1).join("/")
: target.modelStr,
},
options.compressionExclusions
);
})
) {
return null;
}
const requirements = deriveRequestCompatibilityRequirements(body);
if (requirements.requiredContextTokens <= 0) return null;

Expand Down
13 changes: 13 additions & 0 deletions open-sse/services/combo/targetResolution.ts
Original file line number Diff line number Diff line change
Expand Up @@ -115,6 +115,14 @@ export interface ResolveComboTargetPipelineDeps {
hiddenModelsByProvider?: HiddenModelsByProvider;
/** Native Responses clients (for example Codex CLI/Desktop) manage compaction themselves. */
clientManagedResponsesContext?: boolean;
/** #10225 — defer the hard context-overflow preflight when compression is enabled for this request. */
deferContextOverflowWhenCompressible?: boolean;
/** Server-side compression exclusions (#8034) — which targets can run compression. */
compressionExclusions?: import("../compression/exclusions.ts").CompressionExclusions;
/** #10503 — request-shape facts for the target-aware deferral check (see knownContextOverflow.ts). */
sourceFormat?: string | null;
endpointPath?: string | null;
requestHeaders?: Headers | Record<string, unknown> | null;
}

export interface ResolvedComboTargetPipeline {
Expand Down Expand Up @@ -730,6 +738,11 @@ export async function resolveComboTargetPipeline(

const overflow = getKnownContextOverflow(orderedTargets, body, {
clientManagedResponsesContext: deps.clientManagedResponsesContext,
deferContextOverflowWhenCompressible: deps.deferContextOverflowWhenCompressible,
compressionExclusions: deps.compressionExclusions,
sourceFormat: deps.sourceFormat,
endpointPath: deps.endpointPath,
requestHeaders: deps.requestHeaders,
});
if (overflow) {
return { earlyResponse: buildContextOverflowResponse(overflow, orderedTargets, log) };
Expand Down
20 changes: 20 additions & 0 deletions open-sse/services/combo/types.ts
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
* — logic unchanged, re-exported from combo.ts for backward compatibility.
*/

import type { CompressionExclusions } from "../compression/exclusions.ts";
import type { ProviderCandidate } from "../autoCombo/scoring.ts";

export const RESET_WINDOW_NAMES = ["weekly", "session", "monthly"] as const;
Expand Down Expand Up @@ -112,6 +113,25 @@ export type HandleComboChatOptions = {
hiddenModelsByProvider?: HiddenModelsByProvider;
/** Native Responses clients (for example Codex CLI/Desktop) manage compaction themselves. */
clientManagedResponsesContext?: boolean;
/**
* #10225: request-scoped flag — prompt compression is enabled for this request
* (global compression switch ON and not opted-out by the API key). When set, the
* combo preflight defers its hard context-overflow rejection so chatCore's
* compression runs before the final context gate.
*/
deferContextOverflowWhenCompressible?: boolean;
/** Server-side compression exclusions (#8034) — used to check which targets can run compression. */
compressionExclusions?: CompressionExclusions;
/**
* #10503: request-shape facts (mirroring chatCore.ts's own resolution) threaded
* down to getKnownContextOverflow so the deferral decision can be target-aware —
* a native-Codex-Responses-passthrough target must never count as "compressible"
* (chatCore disables compression for it unconditionally). See
* knownContextOverflow.ts::KnownContextOverflowOptions for the full rationale.
*/
sourceFormat?: string | null;
endpointPath?: string | null;
requestHeaders?: Headers | Record<string, unknown> | null;
};

export type HandleRoundRobinOptions = Omit<HandleComboChatOptions, "apiKeyAllowedConnections">;
Expand Down
51 changes: 50 additions & 1 deletion src/sse/handlers/chat.ts
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,8 @@ import type { SingleModelTarget } from "@omniroute/open-sse/services/combo/types
import { mergeAbortSignals } from "@omniroute/open-sse/executors/base.ts";
import { resolveRequestAutoControls } from "@omniroute/open-sse/services/autoCombo/requestControls.ts";
import { isVerifiedNativeCodexRequest } from "@omniroute/open-sse/config/codexIdentity.ts";
import { resolveCompressionSettings } from "@omniroute/open-sse/handlers/chatCore/compressionSettings.ts";
import type { CompressionExclusions } from "@omniroute/open-sse/services/compression/exclusions.ts";
import { resolveComboConfig } from "@omniroute/open-sse/services/comboConfig.ts";
import { injectHandoffIntoBody } from "@omniroute/open-sse/services/contextHandoff.ts";
import {
Expand Down Expand Up @@ -209,6 +211,31 @@ let combosCacheTs = 0;
let combosCacheVersionSnapshot = -1;
const COMBOS_CACHE_TTL_MS = 10_000;

/**
* #10225 — resolve whether this request's combo preflight should DEFER its hard
* context-overflow rejection so chatCore's compression runs first.
*
* Mirrors handleChatCore's own enablement determination (chatCore.ts): defer only
* when the global compression switch is ON and the API key has not opted out
* (`apiKeyInfo.compressionEnabled !== false`). Per-target applicability (server-side
* exclusions) is checked inside getKnownContextOverflow via the returned exclusions.
* Fail closed (defer=false) on any lookup error — the existing hard preflight stays.
*/
async function resolveComboContextOverflowDeferral(
logger: { warn?: (...args: unknown[]) => void } | null | undefined,
apiKeyInfo: { compressionEnabled?: boolean } | null | undefined
): Promise<{ defer: boolean; exclusions: CompressionExclusions | undefined }> {
try {
const compression = await resolveCompressionSettings(logger);
return {
defer: compression.enabled && apiKeyInfo?.compressionEnabled !== false,
exclusions: compression.settings?.exclusions,
};
} catch {
return { defer: false, exclusions: undefined };
}
}

async function getCombosCachedForChat(): Promise<unknown[]> {
const now = Date.now();
// Explicit non-null check: we intentionally cache and return the Promise
Expand Down Expand Up @@ -824,9 +851,20 @@ async function handleChatImplementation(

// Context-relay keeps generation in combo.ts, but handoff injection lives here
// because only this layer knows which connectionId was actually selected.
const { defer: deferContextOverflowWhenCompressible, exclusions: compressionExclusions } =
await resolveComboContextOverflowDeferral(log, apiKeyInfo);
const response = await (handleComboChat as any)({
body,
combo,
deferContextOverflowWhenCompressible,
compressionExclusions,
// #10503: same request-shape facts chatCore.ts resolves for itself
// (resolveChatCoreRequestFormat), so getKnownContextOverflow's target-aware
// deferral check can never drift from chatCore's own native-codex-passthrough
// decision. See knownContextOverflow.ts::KnownContextOverflowOptions.
sourceFormat,
endpointPath: new URL(request.url).pathname,
requestHeaders: request.headers,
clientManagedResponsesContext:
sourceFormat === "openai-responses" &&
new URL(request.url).pathname.split("/").includes("responses") &&
Expand Down Expand Up @@ -1103,11 +1141,22 @@ async function handleSingleModelChat(
);
log.info("ROUTING", `Auto-combo redirect from handleSingleModelChat for "${modelStr}"`);
log.info("ROUTING", `Auto-combo redirect to combo flow for "${modelStr}"`);
const { defer: sNetDefer, exclusions: sNetExclusions } =
await resolveComboContextOverflowDeferral(log, apiKeyInfo);
// #10503: same request-shape facts chatCore.ts resolves for itself — threaded
// down so getKnownContextOverflow's target-aware deferral check can never drift
// from chatCore's own native-codex-passthrough decision.
const sNetSourceFormat = detectFormatFromEndpoint(body, clientRawRequest?.endpoint || "");
return handleComboChat({
body,
combo: redirectCombo,
deferContextOverflowWhenCompressible: sNetDefer,
compressionExclusions: sNetExclusions,
sourceFormat: sNetSourceFormat,
endpointPath: clientRawRequest?.endpoint || "",
requestHeaders: clientRawRequest?.headers,
clientManagedResponsesContext:
detectFormatFromEndpoint(body, clientRawRequest?.endpoint || "") === "openai-responses" &&
sNetSourceFormat === "openai-responses" &&
String(clientRawRequest?.endpoint || "")
.split("/")
.includes("responses") &&
Expand Down
Loading
Loading