diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index bb9dc03583e4..2552ccfc8cfb 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -523,7 +523,9 @@ "open-sse/executors/codex.ts": 1584, "open-sse/executors/cursor.ts": 1868, "open-sse/executors/muse-spark-web.ts": 1405, - "open-sse/handlers/chatCore.ts": 6444, + "open-sse/handlers/chatCore.ts": 6426, + "open-sse/handlers/chatCore/nonStreamingResponse.ts": 1212, + "open-sse/handlers/chatCore/streamingResponse.ts": 1308, "open-sse/handlers/imageGeneration.ts": 3334, "open-sse/handlers/search.ts": 1789, "open-sse/mcp-server/schemas/tools.ts": 1621, @@ -776,6 +778,7 @@ "_rebaseline_2026_09_19_14162_native_codex_auto_resume": "PR #14162 (re-land of #13180, @mdigitalbh81 via @diegosouzapw): native Codex turn auto-resume. open-sse/services/combo/executeTargetAttempt.ts 1258->1273 (+15). Growth is 100% the PR's own, measured against the clean tip (1258 there, gate green): the pin step now advances the logical turn generation and logs the resumed provider/model when the attempt is an auto-resume dispatch, and the generation is passed into pinNativeCodexTurn — the branch has to sit at the pin site because that is the only place the winning target and effective connection are known. Covered by tests/unit/native-codex-auto-resume.test.ts + native-codex-auto-resume-guards.test.ts (15/15) and #13564's native-codex-turn-pin-model-scoped-fallback.test.ts (7/7).", "_rebaseline_2026_09_22_14405_freetier_observed_tools_retry": "Issue #14405 own growth: open-sse/executors/opencode.ts 1251->1303 (measured after merging release/v3.8.51, whose own drift entry _rebaseline_2026_09_22_opencode_train10c_drift had already moved the ceiling 1247->1251; the extra 5 lines over the original 1247->1294 measurement are the #14148 reconciliation: freeTierRetryCtx now reads the contract attempt back from the request body via attemptFor() instead of the removed shared _contractAttempt field) (+47 irreducible call-site wiring: freeTierRetryCtx builder + direct fast-path retry call + rotation-loop refusal arm delegating to handleLoopFreeTierRefusal + api-typecheck fixes (unknown-cast on stall-guard dispatch, callback-shaped loop handler keeping private finalize methods); the retry logic itself lives in the new module open-sse/executors/opencodeFreeTierRetry.ts (131 lines, under cap) and the merge helper in opencodeFreeTierContract.ts, so no logic was added to the frozen file beyond wiring two existing dispatch arms. Compaction attempts measured and reverted: loop-call wrapper (+22 net), fast-path/loop fusion (fragile: retry-403 vs original-403 share status). Covered by tests/unit/opencode-free-tier-refusal-rotation.test.ts (3 retry tests: union success, original refusal + store untouched, no retry without names) + tests/unit/opencode-free-tier-request-contract.test.ts (2 merge tests).", "_rebaseline_2026_09_24_14659_pool_reselect_per_attempt_growth": "PR #14659 (opt-in OPENCODE_POOL_RESELECT, default off): own growth at existing chokepoints, additive and flag-off inert, re-measured on the release tip post-#14620 merge. open-sse/executors/opencode.ts 1301->1355 (+54: flag/resolver hoist, reselect cell, 429-arm hunk after stop/park exits); open-sse/utils/proxyFetch.ts 1294->1311 (+17: reselectPoolMember field on AppliedProxySink + read-only currentAppliedProxySink getter on the shared store); src/sse/handlers/chat.ts 2563->2564 (+1: reselectPoolMember field on the sink literal); src/sse/handlers/chatHelpers.ts 1258->1303 (+45: publishPoolReselectResolver helper + one call at the capture site, resolver re-invokes pool selection). Covered by tests/unit/opencode-429-pool-reselect.test.ts (6/6 GREEN) + flag guards 58/58.", + "_rebaseline_2026_09_23_13065_chatcore_reextract": "Re-extract of #13065 onto the release/v3.8.51 tip (Diego's ask: each leaf must be an empty-diff lift of its chatCore.ts region, imports/exports aside). The two new leaves inherit the exact size of the blocks they carry out of the barrel: nonStreamingResponse.ts holds the if(!stream) leg plus the runNonStreamingProviderLeg/finalizeToolLoopError wrappers; streamingResponse.ts holds the if(stream) provider pipeline. Shrinking below the cap would mean re-modularizing beyond the approved extraction boundaries, which defeats the empty-diff review criterion. Frozen at current size: any further growth of these two files fails the gate.", "_rebaseline_2026_09_30_15186_stream_output_model_lockout": "Model-only lockout skipped for an isolated 5xx after the stream already relayed output (locked again once 3 such failures arrive in a row with no completed stream in between): open-sse/utils/stream.ts 3297->3298 (return registerStreamTiming(sseStream, timing)), open-sse/handlers/chatCore.ts 6441->6444 (hasEmittedOutput passed to createStreamFailureFinalizers + streak reset on a completed stream + its import) and src/sse/handlers/chatHelpers.ts 1285->1286 (streamOutputEmitted read from the failure payload) -- irreducible call-site wiring. The logic lives in new modules: open-sse/utils/streamTiming.ts (output registry), open-sse/services/accountFallback/postOutputFailureStreak.ts (streak) and src/sse/services/syntheticEmptyStream.ts (isRequestScopedServerFailure); src/sse/services/auth.ts stays under its ceiling because buildExhaustionOptions now reuses the markAccountUnavailable option type. Covered by tests/unit/stream-failure-after-output-no-model-lockout.test.ts.", "_rebaseline_2026_10_03_uniform_provider_regime_mount": "Own growth (uniform provider regime mount): src/app/(dashboard)/dashboard/settings/components/ProxyRegistryManager.tsx 1479->1483 (+4 = the PoolUpstreamRegimeLine import and its three-line conditional mount under the pool members label, next to PoolEgressObservation/PoolMemberEgressLines: rendered only when poolScope is 'provider' with a non-blank poolScopeId, which becomes the line's provider; nothing renders for other scopes or a blank id, so no conclusion without a provider). The regime itself lives outside the frozen file, all under cap: PoolUpstreamRegimeLine.tsx, the dedicated GET /api/settings/proxies/pool/uniform-egress route, readUniformEgressRegime/isUpstreamRegime in src/lib/proxyPoolEgressObservation.ts and getProviderUpstreamSummary in src/lib/db/proxies.ts. Only the mount point is irreducible. Covered by tests/unit/ui/proxy-registry-manager-regime-mount.test.tsx and tests/unit/ui/pool-upstream-regime-line.test.tsx.", "_rebaseline_2026_10_02_admission_rejection_call_log": "PR own growth (admission rejection call-log) after extraction of the rejection-logging logic into the new src/sse/handlers/admissionRejectionLog.ts (under cap): src/sse/handlers/chat.ts 2586->2597 (+11 = one import + 10 call-site lines at the single acquire() rejection site, irreducible — plancher +6 prouve, gate exige +0). Covered by tests/unit/admission-rejection-call-log.test.ts.", diff --git a/config/quality/gate-manifest.json b/config/quality/gate-manifest.json index 47f090976970..3610a6135bcb 100644 --- a/config/quality/gate-manifest.json +++ b/config/quality/gate-manifest.json @@ -480,7 +480,7 @@ }, { "name": "test", - "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", + "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", "disposition": "separately-invoked" }, { @@ -535,7 +535,7 @@ }, { "name": "test:coverage:runner", - "command": "node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=8 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true NODE_OPTIONS=--max-old-space-size=8192 c8 --merge-async --output-dir coverage --exclude=tests/** --exclude=**/*.test.* --reporter=text-summary --reporter=html --reporter=json-summary --reporter=lcov --check-coverage --statements 60 --lines 60 --functions 60 --branches 60 node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=8 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", + "command": "node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=8 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true NODE_OPTIONS=--max-old-space-size=8192 c8 --merge-async --output-dir coverage --exclude=tests/** --exclude=**/*.test.* --reporter=text-summary --reporter=html --reporter=json-summary --reporter=lcov --check-coverage --statements 60 --lines 60 --functions 60 --branches 60 node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=8 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", "disposition": "separately-invoked" }, { @@ -615,22 +615,22 @@ }, { "name": "test:unit", - "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", + "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", "disposition": "separately-invoked" }, { "name": "test:unit:ci", - "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", + "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", "disposition": "separately-invoked" }, { "name": "test:unit:ci:shard", - "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 --test-shard=$TEST_SHARD tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 --test-shard=$TEST_SHARD \"tests/unit/dashboard/**/*.test.ts\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=1 --test-shard=$TEST_SHARD \"tests/unit/serial/**/*.test.ts\"", + "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 --test-shard=$TEST_SHARD tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 --test-shard=$TEST_SHARD \"tests/unit/dashboard/**/*.test.ts\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=1 --test-shard=$TEST_SHARD \"tests/unit/serial/**/*.test.ts\"", "disposition": "separately-invoked" }, { "name": "test:unit:fast", - "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-isolation=none tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-isolation=none \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", + "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-isolation=none tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-isolation=none \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", "disposition": "separately-invoked" }, { @@ -645,12 +645,12 @@ }, { "name": "test:unit:shard:1", - "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=1/2 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=1/2 \"tests/unit/dashboard/**/*.test.ts\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=1 --test-shard=1/2 \"tests/unit/serial/**/*.test.ts\"", + "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=1/2 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=1/2 \"tests/unit/dashboard/**/*.test.ts\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=1 --test-shard=1/2 \"tests/unit/serial/**/*.test.ts\"", "disposition": "separately-invoked" }, { "name": "test:unit:shard:2", - "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=2/2 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=2/2 \"tests/unit/dashboard/**/*.test.ts\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=1 --test-shard=2/2 \"tests/unit/serial/**/*.test.ts\"", + "command": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=2/2 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=2/2 \"tests/unit/dashboard/**/*.test.ts\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=1 --test-shard=2/2 \"tests/unit/serial/**/*.test.ts\"", "disposition": "separately-invoked" }, { diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index 27981208a51c..de1441eaaa2f 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -14,7 +14,6 @@ import { import { buildFailureUsageRecord, readCpaAuthIndex, - projectFailureUsageErrorCode, type FailureUsageAggregate, } from "./chatCore/failureUsage.ts"; import { createTranslationFailureResult } from "./chatCore/translationFailure.ts"; @@ -41,31 +40,17 @@ import { detectClassifierFormat, buildDefaultAllowClaudeMessage, } from "./chatCore/claudeClassifierCompat.ts"; -import { buildPostCallGuardrailContext } from "./chatCore/postCallGuardrailContext.ts"; -import { storeSemanticCacheResponse } from "./chatCore/semanticCacheStore.ts"; -import { buildNonStreamingResponseHeaders } from "./chatCore/nonStreamingResponseHeaders.ts"; -import { maybeWrapForcedNonStreamingResponsesJson } from "./chatCore/responsesJsonToSse.ts"; import { enforceOutputTokenBudget } from "./chatCore/outputTokenBudget.ts"; -import { maybeConvertJsonBodyToSse } from "./chatCore/jsonBodyToSse.ts"; import { withResilienceActionsContext } from "./chatCore/resilienceAttemptContext.ts"; -import { runEmptyTurnRetryLoop } from "./chatCore/emptyTurnRetryLoop.ts"; import { notePreviousResponseResumed } from "./chatCore/resumedResilienceNotes.ts"; -import { assembleStreamingResponseHeaders } from "./chatCore/streamingResponseHeaders.ts"; -import { storeStreamingSemanticCacheResponse } from "./chatCore/streamingSemanticCacheStore.ts"; -import { captureStreamReasoningForReplay } from "./chatCore/streamReasoningCapture.ts"; -import { assembleStreamingPipeline } from "./chatCore/streamingPipeline.ts"; +import { runStreamingResponse } from "./chatCore/streamingResponse.ts"; +import { runStreamingTail } from "./chatCore/streamingTail.ts"; import { sanitizeChatRequestBody } from "./chatCore/sanitization.ts"; import { applyReasoningInputPolicy, resolveIncompatibleReasoningAction, } from "../services/reasoningInputPolicy.ts"; -import { - createRoutingEvent, - emitRoutingEvent, - outcomeFromStatus, -} from "../services/routing/index.ts"; -import { routingFinishReason } from "./chatCore/routingFinishReason.ts"; import { getHeaderValueCaseInsensitive, isNoMemoryRequested, @@ -73,22 +58,13 @@ import { } from "./chatCore/headers.ts"; import { - getCodexClientSessionId, isCodexOriginatedHeaders, isClaudeCodeOriginatedHeaders, } from "../config/codexIdentity.ts"; -import { - noteCodexTurnStateProvenance, - readCodexTurnStateHeader, -} from "../config/codexTurnState.ts"; import { trackDevice, extractIpFromHeaders } from "../services/deviceTracker.ts"; import { getCombosCached } from "./chatCore/comboContextCache.ts"; export { clearCombosCache, clearUpstreamProxyConfigCache } from "./chatCore/comboContextCache.ts"; -import { - resolveAccountSemaphoreKey, - resolveAccountSemaphoreMaxConcurrency, - buildClaudePromptCacheLogMeta, -} from "./chatCore/executorHelpers.ts"; +import { resolveAccountSemaphoreKey } from "./chatCore/executorHelpers.ts"; import { shouldUseNativeCodexPassthrough, shouldUseNativeXaiResponsesPassthrough, @@ -98,29 +74,12 @@ import { stripClaudeRejectedTopLevelFields, isClaudeCodeSemanticPassthroughRequest, } from "./chatCore/passthroughHelpers.ts"; -import { recoverAnthropicThinkingSignature } from "./chatCore/thinkingSignatureRecovery.ts"; -import { maybeFallbackAfterReadiness } from "./chatCore/streamReadinessFallback.ts"; -import { runProviderExecutionPipeline } from "./chatCore/providerExecutionPipeline.ts"; -import { runNonStreamingProviderLeg } from "./chatCore/nonStreamingProviderLeg.ts"; -import type { NonStreamingProviderLegResult } from "@/lib/skills/toolLoopTypes.ts"; -import { - applyServerOwnedToolLoopIfNeeded, - derivePostInjectionRequestIdentity, - followUpLegInput, -} from "./chatCore/serverOwnedToolLoopWire.ts"; -import { finalizeToolLoopError } from "./chatCore/nonStreamingFinalization.ts"; -import { markCodexScopeRateLimited } from "./chatCore/codexFailover.ts"; -import { deleteSessionAccountAffinity } from "@/lib/db/sessionAccountAffinity"; + import { buildStreamingResponseHeaders, - materializeDeduplicatedExecutionResult, - stripNextMiddlewareControlHeaders, stripStaleForwardingHeaders, } from "./chatCore/responseHeaders.ts"; -import { - forwardDashboardEventToLiveWs, - maybeSyncClaudeExtraUsageState, -} from "./chatCore/telemetryHelpers.ts"; +import { forwardDashboardEventToLiveWs } from "./chatCore/telemetryHelpers.ts"; // Re-export the previously inline-defined helpers so existing importers of these // symbols from chatCore.ts (tests, sibling modules) keep resolving after the split. export { @@ -141,11 +100,10 @@ import { stripStore, usesClaudeBridge } from "./chatCore/agentRouterProtocol.ts" import { normalizeClaudeToolsForDispatch } from "./chatCore/claudeToolDefaults.ts"; import { injectCustomSystemPrompt, - injectSystemPromptPostTranslation, injectSystemPromptPreTranslation, } from "../services/systemPrompt.ts"; import { applyProviderSystemTransforms } from "../services/systemTransforms.ts"; -import { translateRequest, needsTranslation } from "../translator/index.ts"; +import { translateRequest } from "../translator/index.ts"; import { applyReasoningRuleDirective } from "@/lib/reasoningRouting/policy"; import { withReasoningRuleContext } from "../utils/reasoningRuleContext.ts"; import { FORMATS } from "../translator/formats.ts"; @@ -153,28 +111,14 @@ import { collectCustomToolNamesForSourceFormat } from "../translator/request/ope import { sanitizeKiroTools } from "../utils/kiroSanitizer.ts"; import { splitMisplacedToolResults } from "../translator/helpers/claudeHelper.ts"; import { ensureCacheControlOnLastUserMessage } from "../services/claudeCodeConstraints.ts"; -import { - createSSETransformStreamWithLogger, - createPassthroughStreamWithLogger, - COLORS, -} from "../utils/stream.ts"; -import { ensureStreamReadiness } from "../utils/streamReadiness.ts"; -import { requestTtftMs, streamEmittedOutput } from "../utils/streamTiming.ts"; -import { resolveSuppressThinkClose, THINKING_MARKER_HEADER } from "../utils/thinkCloseMarker.ts"; -import { resolveStreamReadinessTimeout } from "../utils/streamReadinessPolicy.ts"; +import { THINKING_MARKER_HEADER } from "../utils/thinkCloseMarker.ts"; import { resolveAgentGoalPolicy } from "../utils/agentGoalPolicy.ts"; -import { hasActiveClaudeThinking } from "../utils/thinkingBudget.ts"; import { createStreamController } from "../utils/streamHandler.ts"; import * as streamFailure from "../utils/streamFailureFinalization.ts"; import { normalizeUsage } from "../utils/usageTracking.ts"; -import { - refreshWithRetry, - isUnrecoverableRefreshError, - runWithOnPersist, - runWithCasGuard, -} from "../services/tokenRefresh.ts"; +import { refreshWithRetry, runWithOnPersist, runWithCasGuard } from "../services/tokenRefresh.ts"; import { createRequestLogger } from "../utils/requestLogger.ts"; -import { createPreparedRequestLogger, runWithCapture } from "../utils/providerRequestLogging.ts"; +import { createPreparedRequestLogger } from "../utils/providerRequestLogging.ts"; import { summarizeToolSources } from "../utils/toolSources.ts"; import { applyResponsesPreviousResponseIdPolicy } from "../utils/responsesStatePolicy.ts"; import { applyClaudeEffortVariant } from "./chatCore/claudeEffortVariant.ts"; @@ -186,8 +130,11 @@ import { import { shouldUseMidConversationSystem } from "../executors/claudeIdentity.ts"; import { echoModelInObject } from "../services/responseModelEcho.ts"; import { getUnsupportedParams, REGISTRY } from "../config/providerRegistry.ts"; -import { shouldSkipCredentialRefresh } from "./chatCore/skipCredentialRefresh.ts"; + import { checkToolCallingRequiredButUnsupported } from "./chatCore/toolCallingRequiredCheck.ts"; +import { buildExecutorClientHeaders } from "./chatCore/executorClientHeaders.ts"; +import { executeProviderRequest as executeProviderRequestLeaf } from "./chatCore/executeProviderRequest.ts"; +import { runNonStreamingResponse } from "./chatCore/nonStreamingResponse.ts"; import { supportsMaxTokens, getResolvedModelCapabilities, @@ -205,19 +152,10 @@ import { isServerOwnedToolLoopEnabled, } from "@/shared/utils/featureFlags.ts"; import { resolveNoAuthEchoModel } from "./chatCore/noAuthEchoModel.ts"; -import { - REASONING_BUFFER_MIN_TRIGGER, - buildReasoningProbeTruncatedResponse, - isEmptyContentUpstreamFailure, - isTinyBudgetReasoningProbe, - toPositiveInteger, -} from "../services/reasoningTokenBuffer.ts"; +import { toPositiveInteger } from "../services/reasoningTokenBuffer.ts"; import { buildErrorBody, createErrorResult, - parseUpstreamError, - formatProviderError, - projectPublicErrorIdentifier, sanitizeErrorMessage, sanitizeUpstreamDetails, } from "../utils/error.ts"; @@ -231,41 +169,22 @@ import { COOLDOWN_MS, HTTP_STATUS, PROVIDER_MAX_TOKENS, - STREAM_READINESS_MAX_TIMEOUT_MS, - STREAM_READINESS_TIMEOUT_MS, - ANTIGRAVITY_PRE_RESPONSE_TIMEOUT_CODE, - STREAM_RECOVERY, DEFAULT_MAX_TOKENS, STREAM_DISCONNECT_GRACE_PERIOD_MS, } from "../config/constants.ts"; -import { applyStatusRestatement } from "../config/upstreamStatusRestatement.ts"; -import { createRecoverableStream, makeContinuationBody } from "../services/streamRecovery.ts"; -import { buildContinuationLogHooks } from "./chatCore/recoveryTraceLogging.ts"; -import { - resolveResilienceSettings, - isStreamRecoveryExplicitlyConfigured, -} from "@/lib/resilience/settings"; + +import { resolveResilienceSettings } from "@/lib/resilience/settings"; import { classifyProviderError, PROVIDER_ERROR_TYPES } from "../services/errorClassifier.ts"; import { isOpencodeFreeTierRefusalForProvider } from "../executors/opencodeGeoBlock.ts"; import { noteOpencodeFreeTierSkip } from "../services/opencodeFreeTierSkip.ts"; import { updateProviderConnection, getProviderConnectionById } from "@/lib/db/providers"; -import { wasRefreshTokenRotated } from "@omniroute/open-sse/services/refreshSerializer.ts"; + import { connectionHasExtraKeys } from "../services/apiKeyRotator.ts"; import { recordKeyHealthStatus as recordKeyHealthStatusFor } from "./chatCore/keyHealth.ts"; import { getSkillsModelIdForFormat } from "./chatCore/skillsFormat.ts"; -import { readNonStreamingResponseBody } from "./chatCore/nonStreamingResponseBody.ts"; -import { - createSafeAbortError, - createStreamingErrorResult, - isSemaphoreCapacityError, - formatStreamRecoveryRetryWarning, - getSafeErrorMetadata, - getUpstreamErrorIdentifier, -} from "./chatCore/streamErrorResult.ts"; -import { wrapReadableStreamWithFinalize } from "./chatCore/streamFinalize.ts"; +import { isSemaphoreCapacityError, getSafeErrorMetadata } from "./chatCore/streamErrorResult.ts"; import { buildCacheUsageLogMeta } from "./chatCore/cacheUsageMeta.ts"; -import { buildExecutorClientHeaders } from "./chatCore/executorClientHeaders.ts"; -import { getExecutionConnectionId } from "./chatCore/executionCredentials.ts"; + import { resolveExecutionCredentials as resolveExecutionCredentialsFor } from "./chatCore/executionCredentials.ts"; import { createExecutorResolver } from "./chatCore/executorProxy.ts"; import type { ClaudeMessage } from "./chatCore/claudeMessageTypes.ts"; @@ -276,7 +195,7 @@ import { } from "./chatCore/attemptLogging.ts"; import { stageTrace } from "./chatCore/stageTrace.ts"; import { attachCompressionUsageReceiptAfterAnalytics as attachCompressionUsageReceiptAfterAnalyticsFor } from "./chatCore/compressionUsageReceipt.ts"; -import { prepareUpstreamBody } from "./chatCore/upstreamBody.ts"; + import { getQuotaScopeLabelForProvider } from "../services/antigravityQuotaFamily.ts"; import { excludeConnectionForCooldown } from "./chatCore/connectionCooldown.ts"; import { handleRequestRejectedFailure } from "./chatCore/requestRejectedFailure.ts"; @@ -297,13 +216,9 @@ import { initialPendingBody, updatePendingScope, } from "@/lib/usage/pendingRequestScope"; -import { recordCost, recordChatCallCost, buildCostCtx } from "@/domain/costRules"; -import { meteredBudgetCost } from "@/lib/usage/meteredBudgetPolicy"; +import { recordChatCallCost, buildCostCtx } from "@/domain/costRules"; import { calculateCost } from "@/lib/usage/costCalculator"; -import { - buildClaudePassthroughToolNameMap, - mergeResponseToolNameMap, -} from "./chatCore/passthroughToolNames.ts"; +import { buildClaudePassthroughToolNameMap } from "./chatCore/passthroughToolNames.ts"; import { createDisabledCompressionConfig, resolveCompressionSettings, @@ -326,23 +241,11 @@ import { recordCompressionCacheStats } from "./chatCore/compressionCacheStats.ts import { writeCavemanOutputAnalytics } from "./chatCore/cavemanOutputAnalytics.ts"; import { scheduleQuotaShareConsumption } from "./chatCore/quotaShareConsumption.ts"; import { emitRequestGamificationEvent } from "./chatCore/gamificationEvent.ts"; -import { - runPluginOnResponseHook, - runPluginOnStreamCompleteHook, -} from "./chatCore/pluginOnResponse.ts"; -import { scheduleStreamingQuotaShareConsumption } from "./chatCore/streamingQuotaShare.ts"; -import { recordStreamingUsageStats } from "./chatCore/streamingUsageStats.ts"; -import { recordStreamingCost, buildStreamLedgerDetails } from "./chatCore/streamingCost.ts"; +import { runPluginOnResponseHook } from "./chatCore/pluginOnResponse.ts"; import { isJsonRecord } from "./chatCore/nonStreamingResponseParse.ts"; import { recordNonStreamingUsageStats } from "./chatCore/nonStreamingUsageStats.ts"; -import { - normalizeExecutorResult, - executeWithUpstreamStartTimeout, - getExecutorTimeoutMs, - resolveConnectionTimeoutMs, -} from "./chatCore/upstreamTimeouts.ts"; import { getModelNormalizeToolCallId, getModelPreserveOpenAIDeveloperRole } from "@/lib/db/models"; -import { getProviderCredentials, extractSessionAffinityKey } from "@/sse/services/auth"; +import { extractSessionAffinityKey } from "@/sse/services/auth"; import { assertExclusiveConnectionLeaseFence } from "@/lib/db/exclusiveConnectionLeases"; import { getCacheControlSettings } from "@/lib/cacheControlSettings"; @@ -365,39 +268,22 @@ import { type EffectiveServiceTier, } from "./chatCore/serviceTier.ts"; import { isCompactResponsesEndpoint } from "../executors/codex.ts"; -import { persistCodexChildQuotaResponse } from "../services/codexAccount/index.ts"; -import { invalidateCodexQuotaCache } from "../services/codexQuotaFetcher.ts"; -import { invalidateGenericQuotaCacheOnStatus } from "../services/genericQuotaFetcher.ts"; import { extractUsageFromResponse } from "./usageExtractor.ts"; import { - withRateLimit, updateFromHeaders, updateFromResponseBody, initializeRateLimits, - resolveRequestQueueMaxWaitMs, } from "../services/rateLimitManager.ts"; -import * as localLimiterErrors from "../services/rateLimitManager/errors.ts"; -import { rethrowAdmissionError, remainingQueueBudgetMs } from "./chatCore/queueBudget.ts"; -import { - acquireMany as acquireConcurrencyGates, - markBlocked as markAccountSemaphoreBlocked, -} from "../services/accountSemaphore.ts"; +import { markBlocked as markAccountSemaphoreBlocked } from "../services/accountSemaphore.ts"; import { lockModel, lockModelIfPerModelQuota, recordCoreOwnedAntigravityQuotaState, shouldDeferAntigravityQuotaStateToCaller, } from "../services/accountFallback.ts"; -import { clearPostOutputFailureStreak } from "../services/accountFallback/postOutputFailureStreak.ts"; import { saveIdempotency } from "@/lib/idempotencyLayer"; -import { - isModelUnavailableError, - getNextFamilyFallback, - isContextOverflowError, - findLargerContextModel, - getModelFamily, -} from "../services/modelFamilyFallback.ts"; -import { computeRequestHash, deduplicate, shouldDeduplicate } from "../services/requestDedup.ts"; + +import { computeRequestHash, shouldDeduplicate } from "../services/requestDedup.ts"; import { compressContext, estimateTokens, @@ -420,33 +306,17 @@ import { generateRequestId } from "@/shared/utils/requestId"; import { isLocalStreamLifecycleError } from "@/shared/utils/circuitBreaker"; import { shouldIsolateProbeFailures } from "@/shared/utils/probeOrigin"; import { writeTerminalStatus } from "@/shared/utils/terminalStatus"; -import { extractFacts } from "@/lib/memory/extraction"; import { handleToolCallExecution } from "@/lib/skills/interception"; -import { MEMORY_BUILTIN_TOOL_NAMES } from "@/lib/skills/memoryBuiltins"; -import { resolveProviderId } from "@/shared/constants/providers"; import { getClaudeCodeCompatibleRequestDefaults } from "@/lib/providers/requestDefaults"; import { buildClaudeCodeCompatibleRequest, resolveClaudeCodeCompatibleSessionId, } from "../services/claudeCodeCompatible.ts"; import { setGeminiThoughtSignatureMode } from "../services/geminiThoughtSignatureStore.ts"; -import { - classifyModelScope429, - getModelScopeRetryDelayMs, - isModelScopeProvider, -} from "../services/modelscopePolicy.ts"; -import { - incrementRequestCount, - incrementTokenUsage, - isTpmExhausted, -} from "../services/geminiRateLimitTracker.ts"; +import { classifyModelScope429, isModelScopeProvider } from "../services/modelscopePolicy.ts"; +import { incrementTokenUsage, isTpmExhausted } from "../services/geminiRateLimitTracker.ts"; import { getProactiveCompressionRatio } from "@/lib/db/compression"; -type ChatCoreExecutorResult = ReturnType & { - _executionCredentials?: Record; - _accountSemaphoreRelease?: () => void; -}; - /** * #12150 P1b: shape of handleChatCore's optional `videoBridgeLog` param — see * its destructure default below. `handleChatCore`'s own params object has no @@ -3123,511 +2993,63 @@ async function handleChatCoreInner({ ? computeRequestHash(dedupRequestBody, apiKeyInfo?.id, trustedEffortContext) : null; - const executeProviderRequest = async (modelToCall = effectiveModel, allowDedup = false) => { - const execute = async () => { - // Upstream body preparation extracted to chatCore/upstreamBody.ts (#3501 — first internal - // sub-slice of executeProviderRequest); produces the body sent upstream (payload rules + - // tool-limit truncation + prompt_cache_key injection). - let bodyToSend = await prepareUpstreamBody({ - translatedBody, - modelToCall, - ...trustedEffortContext, - provider, - targetFormat, - credentials: getExecutionCredentials(), - log, - bypassDefaultToolLimit: isOpencodeClient, - isOpencodeClient, - rawBody: body, - clientRawRequest, - }); - - // Global System Prompt — SINGLE injection point (post-translation) for - // carrier-ful targets. The old unconditional pre-translation pass - // (former chatCore injectSystemPrompt call) was removed: it chained - // with this pass to inject prefix/suffix 2-3x and dual-wrote - // body.system + messages[] on the claude path, which strict upstreams - // (HCP-Vision vLLM: "System message must be at the beginning") reject - // with 400. Format-aware via targetFormat: messages[] (openai/codex — - // prefix FIRST system, suffix LAST), claude `system` field, gemini - // `systemInstruction`, responses `instructions`. Carrier-less targets - // (kiro user-fold, antigravity Cloud Code envelope) are covered by the - // gated PRE-translation pass before translateRequest instead. - bodyToSend = injectSystemPromptPostTranslation(bodyToSend, { targetFormat }); - - updatePendingScope(pendingScope, { - providerRequest: bodyToSend, - stage: "payload_prepared", - }); - - let releaseRawResultAccountSemaphore = () => {}; - try { - const rawResult: ChatCoreExecutorResult = await (async () => { - let attempts = 0; - const isModelScopeForRequest = isModelScope(); - const maxAttempts = isModelScopeForRequest ? 3 : provider === "codex" ? 3 : 1; - - while (attempts < maxAttempts) { - trace("pre_executor", { attempt: attempts }); - updatePendingScope(pendingScope, { - stage: "sending_to_provider", - }); - const execCreds = getExecutionCredentials(); - const executionConnectionId = getExecutionConnectionId(execCreds); - const attemptConnectionId = executionConnectionId || connectionId; - const accountSemaphoreMaxConcurrency = resolveAccountSemaphoreMaxConcurrency(execCreds); - const accountSemaphoreKey = resolveAccountSemaphoreKey({ - provider, - model: modelToCall, - connectionId: attemptConnectionId, - credentials: execCreds, - }); - const canonicalProviderKey = resolveProviderId(String(provider).trim().toLowerCase()); - const providerConcurrency = - resilienceSettings.providerQuotaOverrides[canonicalProviderKey] - ?.providerConcurrency ?? 0; - - trace("pre_semaphore", { - semaphoreKey: accountSemaphoreKey, - max: accountSemaphoreMaxConcurrency, - }); - if (accountSemaphoreKey && accountSemaphoreMaxConcurrency != null) { - updatePendingScope(pendingScope, { - stage: "waiting_account_slot", - }); - } - const maxWaitMs = resolveRequestQueueMaxWaitMs( - provider, - undefined, - attemptConnectionId ?? undefined - ); - const gateStartedAt = Date.now(); - const releaseAccountSemaphore = await acquireConcurrencyGates( - [ - { - key: "global", - maxConcurrency: resilienceSettings.requestQueue.globalConcurrentRequests, - }, - { - key: `provider:${canonicalProviderKey}`, - maxConcurrency: providerConcurrency, - }, - { - key: accountSemaphoreKey || "", - maxConcurrency: accountSemaphoreKey ? accountSemaphoreMaxConcurrency : null, - }, - ], - { - timeoutMs: maxWaitMs, - maxQueueSize: resilienceSettings.requestQueue.maxQueueDepth, - signal: streamController.signal, - } - ).catch(rethrowAdmissionError); - const remainingAfterGate = remainingQueueBudgetMs(maxWaitMs, gateStartedAt); - trace("post_semaphore", { maxWaitMs, remainingAfterGate }); - updatePendingScope(pendingScope, { - stage: "waiting_rate_limit", - }); - - try { - trace("pre_rate_limit", { connectionId: attemptConnectionId }); - const rawExecutorResult = await withRateLimit( - provider, - attemptConnectionId, - modelToCall, - async () => { - trace("inside_rate_limit", { connectionId: attemptConnectionId }); - updatePendingScope(pendingScope, { - stage: "rate_limit_slot_acquired", - }); - assertManagedLeaseFence(attemptConnectionId); - return executeWithUpstreamStartTimeout({ - executor, - provider, - model: modelToCall, - connectionTimeoutMs: resolveConnectionTimeoutMs( - execCreds?.providerSpecificData - ), - signal: streamController.signal, - log, - execute: (signal) => - runWithCapture(providerRequestCapture, () => - executor.execute({ - model: modelToCall, - body: bodyToSend, - stream: upstreamStream, - credentials: execCreds, - signal, - log, - extendedContext, - upstreamExtraHeaders: buildUpstreamHeadersForExecute(modelToCall), - clientHeaders: getExecutorClientHeaders(), - clientResponseFormat, - onCredentialsRefreshed, - skipUpstreamRetry, - contextEditing: { enabled: contextEditingEnabled }, - correlationId, - }) - ), - }); - }, - streamController.signal, - remainingAfterGate, - correlationId ?? undefined, - { - executor: executor as unknown as { getTimeoutMs?: () => unknown }, - providerSpecificData: execCreds?.providerSpecificData, - } - ); - const res = normalizeExecutorResult(rawExecutorResult); - trace("post_executor", { status: res?.response?.status }); - - // When a payload override rewrote body.model (custom-model alias → - // real upstream id, e.g. `gemini-3.7-flash-high` → `gemini-3.7-flash`), - // log and track the WIRE model so dashboards/telemetry reflect what - // actually shipped and Gemini rate-limit accounting uses the real id - // (the executor already built its URL from the same rewritten model). - const wireModel = - typeof res.model === "string" && res.model ? res.model : modelToCall; - if (wireModel !== modelToCall) { - log?.debug?.( - "PAYLOAD_RULES", - `Payload rules rewrote model for URL: requested=${modelToCall} wire=${wireModel}` - ); - } - - if ( - provider === "codex" && - attemptConnectionId && - !(await shouldIsolateProbeFailures()) - ) { - try { - const persistedQuota = await persistCodexChildQuotaResponse({ - connectionId: String(attemptConnectionId), - model: modelToCall || model || requestedModel || "", - headers: normalizeHeaders(res.response.headers), - status: res.response.status, - }); - if (persistedQuota) { - execCreds.providerSpecificData = persistedQuota.providerSpecificData; - if (persistedQuota.exhaustionLog) { - log?.debug?.("CODEX", persistedQuota.exhaustionLog); - } - } - if (res.response.status === 429) { - invalidateCodexQuotaCache(String(attemptConnectionId)); - } - } catch (err) { - const errMessage = err instanceof Error ? err.message : String(err); - log?.debug?.("CODEX", `Failed to persist codex quota state: ${errMessage}`); - } - } else if (attemptConnectionId && res.response.status === 429) { - // Dropped generic quota cache after 429 - invalidateGenericQuotaCacheOnStatus({ - provider, - connectionId: String(attemptConnectionId), - status: res.response.status, - isolateProbe: await shouldIsolateProbeFailures(), - }); - } - - // Track Gemini RPM + RPD request counts for 429 classification - if (provider === "gemini") { - incrementRequestCount(wireModel); - } - - updatePendingScope(pendingScope, { - stage: "provider_response_started", - }); - - if ( - stream && - (res.response.ok || - res.response.status === HTTP_STATUS.UNAUTHORIZED || - res.response.status === HTTP_STATUS.FORBIDDEN) && - executionConnectionId && - !(await shouldIsolateProbeFailures()) - ) { - const failureDetail = res.response.ok - ? "" - : await res.response - .clone() - .text() - .catch(() => ""); - recordKeyHealthStatus(res.response.status, execCreds, res.transport, failureDetail); - } - - if (isModelScope() && res.response.status === 429 && attempts < maxAttempts - 1) { - const bodyPeek = await res.response - .clone() - .text() - .catch(() => ""); - const normalizedHeaders = normalizeHeaders(res.response.headers); - const decision = classifyModelScope429(bodyPeek, normalizedHeaders); - if (decision.retryable) { - const delay = getModelScopeRetryDelayMs(normalizedHeaders, attempts); - log?.warn?.( - "MODELSCOPE_RETRY", - `429 ${decision.kind}; retrying in ${delay}ms (model remaining: ${decision.snapshot.modelRemaining ?? "unknown"})` - ); - releaseAccountSemaphore(); - await new Promise((r) => setTimeout(r, delay)); - attempts++; - continue; - } - } - - // For streaming: release the semaphore when the client drains or cancels the stream. - // Non-2xx streams must drop the slot before returning so the pipeline can rotate - // accounts without holding the failed connection's concurrency gate. Do NOT - // cancel() the body here — the pipeline clones it (BYOP 422 / toOutcome). - if (stream) { - const originalBody = res.response.body; - const okStatus = res.response.status >= 200 && res.response.status < 300; - if (!originalBody || !okStatus) { - releaseAccountSemaphore(); - return { - ...res, - _executionCredentials: execCreds, - }; - } - - // Opt-in transparent stream recovery (free-claude-code port, default OFF). - // Only engages for a successful (2xx) stream — an error body must never be - // held or replayed. Setting is read once here from the cached resolved - // resilience settings; the default path is byte-for-byte unchanged. - let streamRecoveryEnabled = false; - let continueMidStreamEnabled = false; - let throughputWatchdog = - resolveResilienceSettings(null).streamRecovery.throughputWatchdog; - if (okStatus) { - try { - // Reuse the request-consolidated settings read (see line ~2076) — no - // second DB/cache hit. Default OFF when the setting is absent. - const sr = resolveResilienceSettings(settings).streamRecovery; - // Fail-closed: the agent-goal-policy heuristic may only ADD recovery - // when the operator has no explicit configuration. If the operator - // explicitly configured stream recovery (env var or DB/settings - // override), that value always wins — the goal policy must never - // re-enable recovery the operator explicitly turned off. - const operatorExplicit = isStreamRecoveryExplicitlyConfigured(settings); - const goalOverride = !operatorExplicit && agentGoalPolicy.streamRecoveryEnabled; - streamRecoveryEnabled = sr.enabled || goalOverride; - continueMidStreamEnabled = sr.continueMidStream === true; - throughputWatchdog = sr.throughputWatchdog; - if (goalOverride && !sr.enabled) { - log?.info?.( - "AGENT_GOAL", - `agentGoalPolicy override: stream recovery enabled for goal request requestId=${traceId} model=${modelToCall || model || requestedModel || "unknown"}` - ); - } - } catch { - streamRecoveryEnabled = false; - continueMidStreamEnabled = false; - throughputWatchdog = - resolveResilienceSettings(null).streamRecovery.throughputWatchdog; - } - } - - let clientBody: ReadableStream; - if (streamRecoveryEnabled || throughputWatchdog.enabled) { - // Run the SAME upstream (same account/creds) with a given body and return - // its 2xx stream, or null. Used both by the early-retry re-open (same body) - // and the mid-stream continuation (assistant-prefilled body). - const runUpstreamStream = async ( - body: unknown - ): Promise | null> => { - try { - assertManagedLeaseFence(attemptConnectionId); - const retryRaw = await executeWithUpstreamStartTimeout({ - executor, - provider, - model: modelToCall, - connectionTimeoutMs: resolveConnectionTimeoutMs( - execCreds?.providerSpecificData - ), - signal: streamController.signal, - log, - execute: (signal) => - runWithCapture(providerRequestCapture, () => - executor.execute({ - model: modelToCall, - body, - stream: upstreamStream, - credentials: execCreds, - signal, - log, - extendedContext, - upstreamExtraHeaders: buildUpstreamHeadersForExecute(modelToCall), - clientHeaders: getExecutorClientHeaders(), - clientResponseFormat, - onCredentialsRefreshed, - skipUpstreamRetry, - contextEditing: { enabled: contextEditingEnabled }, - correlationId, - }) - ), - }); - const retryRes = normalizeExecutorResult(retryRaw); - const retryOk = - retryRes.response.status >= 200 && retryRes.response.status < 300; - if (retryOk && retryRes.response.body) { - return retryRes.response.body as ReadableStream; - } - await retryRes.response.body?.cancel().catch(() => {}); - return null; - } catch { - return null; - } - }; - - // Mid-stream continuation (Fase 4.4): re-request with the partial text as an - // assistant prefill. Gated by its own setting and only for OpenAI-compatible - // request bodies, chat or Responses (makeContinuationBody returns null otherwise). - const continueStream = continueMidStreamEnabled - ? (assistantSoFar: string) => { - const continuationBody = makeContinuationBody( - bodyToSend as Record, - assistantSoFar - ); - return continuationBody - ? runUpstreamStream(continuationBody) - : Promise.resolve(null); - } - : undefined; - - clientBody = createRecoverableStream( - originalBody as ReadableStream, - () => runUpstreamStream(bodyToSend), - { - finalize: releaseAccountSemaphore, - onRetry: (attempt, err) => - log?.warn?.( - "STREAM_RECOVERY", - formatStreamRecoveryRetryWarning( - attempt, - STREAM_RECOVERY.EARLY_RETRY_MAX, - err - ) - ), - continueStream, - ...buildContinuationLogHooks(log, correlationId), - throughputWatchdog, - onWatchdogAbort: () => - log?.warn?.( - "STREAM_WATCHDOG", - "active upstream stream stayed below the configured useful-output rate" - ), - } - ); - } else { - clientBody = wrapReadableStreamWithFinalize( - originalBody, - releaseAccountSemaphore - ); - } - - return { - ...res, - _executionCredentials: execCreds, - response: new Response(clientBody, { - status: res.response.status, - statusText: res.response.statusText, - headers: new Headers(normalizeHeaders(res.response.headers)), - }), - }; - } - - return { - ...res, - _executionCredentials: execCreds, - _accountSemaphoreRelease: releaseAccountSemaphore, - }; - } catch (error) { - releaseAccountSemaphore(); - throw error; - } - } - })(); - - if (stream) { - return rawResult; - } - - // Non-stream: release semaphore immediately after reading full response body. - const status = rawResult.response.status; - - releaseRawResultAccountSemaphore = - typeof rawResult._accountSemaphoreRelease === "function" - ? rawResult._accountSemaphoreRelease - : () => {}; - - const statusText = rawResult.response.statusText; - const headersObj = normalizeHeaders(rawResult.response.headers); - const responseHeaders = new Headers(headersObj); - stripStaleForwardingHeaders(responseHeaders); - stripNextMiddlewareControlHeaders(responseHeaders); - // The upstream headers (turn-state included) are about to be committed - // to the client — record which connection minted the blob so a later - // cross-account echo can be stripped (Codex failover guard). - if (provider === "codex" && readCodexTurnStateHeader(responseHeaders)) { - noteCodexTurnStateProvenance( - getCodexClientSessionId(clientRawRequest?.headers), - rawResult._executionCredentials?.connectionId ?? credentials?.connectionId - ); - } - const contentType = (responseHeaders.get("content-type") || "").toLowerCase(); - const payload = await readNonStreamingResponseBody( - rawResult.response, - contentType, - upstreamStream - ); - // Use the exact execution credential selected for this request. Model capability - // failures stay in routing telemetry; authoritative success only recovers this key. - if ( - rawResult._executionCredentials?.connectionId && - (rawResult._executionCredentials.apiKey || rawResult._executionCredentials.accessToken) - ) { - recordKeyHealthStatus( - status, - rawResult._executionCredentials, - rawResult.transport, - status >= 400 ? payload : "" - ); - } - releaseRawResultAccountSemaphore(); - releaseRawResultAccountSemaphore = () => {}; - - return { - ...rawResult, - response: new Response(payload, { status, statusText, headers: responseHeaders }), - _dedupSnapshot: { - status, - statusText, - headers: (() => { - const arr: [string, string][] = []; - responseHeaders.forEach((v, k) => arr.push([k, v])); - return arr; - })(), - payload, - }, - }; - } catch (error) { - releaseRawResultAccountSemaphore(); - throw error; - } - }; - - if (allowDedup && dedupEnabled && dedupHash) { - const dedupResult = await deduplicate(dedupHash, execute); - if (dedupResult.wasDeduplicated) { - log?.debug?.("DEDUP", `Joined in-flight request hash=${dedupHash}`); - } - return materializeDeduplicatedExecutionResult(dedupResult.result); - } - - return execute(); + const executeProviderRequestDeps = { + agentGoalPolicy, + assertManagedLeaseFence, + buildUpstreamHeadersForExecute, + clientRawRequest, + clientResponseFormat, + connectionId, + contextEditingEnabled, + correlationId, + credentials, + dedupEnabled, + dedupHash, + effectiveModel, + executor, + extendedContext, + getExecutionCredentials, + getExecutorClientHeaders, + isModelScope, + isOpencodeClient, + log, + model, + onCredentialsRefreshed, + pendingScope, + pendingConnId, + pendingRequestId, + provider, + providerRequestCapture, + rawBody: body, + recordKeyHealthStatus, + requestedModel, + resilienceSettings, + settings, + skipUpstreamRetry, + stream, + streamController, + targetFormat, + trace, + traceId, + translatedBody, + trustedEffortContext, + upstreamStream, + userAgent, + }; + // The pre-decomposition inline executeProviderRequest closed over the LIVE + // `translatedBody` binding: every mid-flight rebinding (pipeline wire update, + // tool-loop follow-up, thinking-signature recovery) was visible to the next + // dispatch. The deps object above only holds a snapshot, so the leg leaves + // must push each rebinding back through this setter to keep the wire send + // reading the current body. + const syncExecuteTranslatedBody = (next: unknown) => { + translatedBody = next as typeof translatedBody; + executeProviderRequestDeps.translatedBody = next as Record; }; + const executeProviderRequest = ( + modelToCall: string = effectiveModel, + allowDedup: boolean = false + ) => executeProviderRequestLeaf(executeProviderRequestDeps, modelToCall, allowDedup); const registeredProviderRequest = translatedBody && typeof translatedBody === "object" && !Array.isArray(translatedBody) @@ -4145,2290 +3567,289 @@ async function handleChatCoreInner({ let pipelineRecovered = false; if (stream) { - try { - const pipelineOutcome = await runProviderExecutionPipeline({ - policy: { - allowAccountRotation: !managedLease && comboStrategy !== "context-relay", - allowModelFallback: true, - expectedConnectionId: managedLease - ? String(getCurrentConnectionId() || connectionId || "") || undefined - : undefined, - }, - target: { - provider, - requestedModel: effectiveModel, - sourceFormat, - targetFormat, - stream, - }, - connection: { - initialConnectionId: String(getCurrentConnectionId() || connectionId || ""), - getCurrentConnectionId: () => getCurrentConnectionId() || undefined, - getCredentials: () => (credentials || {}) as Record, - replaceCredentials: (next) => { - Object.assign(credentials, next); - }, - onCredentialsRefreshed: handleCredentialsRefreshed, - refreshCredentials: executeRefreshCredentials, - assertManagedLeaseFence: (id) => { - assertManagedLeaseFence(id); - }, - getProviderCredentials, - }, - wire: { - body: translatedBody as Record, - currentModel, - triedModels, - setBodyAndModel: (body, model) => { - translatedBody = body as typeof translatedBody; - currentModel = model; - triedModels.add(model); - }, - }, - state: { - updatePendingStage: (stage, data) => { - updatePendingScope(pendingScope, { stage, ...(data || {}) }); - }, - recordRateLimitHeaders: updateFromHeaders, - recordRateLimitBody: updateFromResponseBody, - writeTerminalStatus, - persistConnectionPatch: updateProviderConnection, - setConnectionRateLimitedUntil: async (id, untilMs) => { - const { setConnectionRateLimitUntil } = await import("@/lib/db/providers"); - setConnectionRateLimitUntil(id, untilMs); - }, - lockModel, - recordAntigravityQuotaState: recordCoreOwnedAntigravityQuotaState, - markAccountSemaphoreBlocked: (key) => { - markAccountSemaphoreBlocked(key, Date.now() + 60_000); - }, - isolateProbeFailures: () => shouldIsolateProbeFailures(), - onCodexScopeRateLimited: async (params) => { - await markCodexScopeRateLimited({ - failedConnectionId: params.failedConnectionId, - model: params.model, - rateLimitedUntil: params.rateLimitedUntil, - credentials: (params.credentials || credentials) as { - connectionId?: string | null; - providerSpecificData?: unknown; - }, - }); - }, - onClearSessionAffinity: () => { - const key = - sessionAffinityKey || - extractSessionAffinityKey(body, clientRawRequest?.headers) || - null; - if (!key) return; - try { - deleteSessionAccountAffinity(key, "codex"); - } catch { - // best-effort - } - }, - onAuditAccountRotation: (params) => { - logAuditEvent({ - action: params.action, - actor: apiKeyInfo?.name || "system", - target: params.newConnectionId, - details: { - failed_connection_id: params.failedConnectionId, - new_connection_id: params.newConnectionId, - attempt: params.attempt, - retry_after_ms: params.retryAfterMs, - }, - }); - }, - }, - sendProviderAttempt: (modelToCall, allowDedup) => - executeProviderRequest(modelToCall, allowDedup), - }); - - pipelineRecovered = true; - currentModel = pipelineOutcome.model; - if (pipelineOutcome.kind === "error") { - providerResponse = pipelineOutcome.result.response; - providerUrl = ""; - providerHeaders = normalizeHeaders(pipelineOutcome.result.response.headers); - finalBody = translatedBody; - } else { - const result = { - response: pipelineOutcome.response, - url: pipelineOutcome.url, - headers: pipelineOutcome.headers, - transformedBody: pipelineOutcome.transformedBody, - }; - providerResponse = result.response; - providerUrl = result.url; - providerHeaders = result.headers; - finalBody = providerRequestCapture.body(result.transformedBody); - } - const responseConnectionId = getCurrentConnectionId(); - effectiveServiceTier = resolveEffectiveServiceTier(finalBody); - claudePromptCacheLogMeta = buildClaudePromptCacheLogMeta( - targetFormat, - finalBody, - providerHeaders, - clientRawRequest?.headers - ); - - // Log target request (final request to provider) - reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody); - updatePendingScope(pendingScope, { - providerRequest: finalBody, - providerUrl, - stage: "provider_response_started", - }); - // Update rate limiter from response headers (learn limits dynamically) - updateFromHeaders( - provider, - responseConnectionId, - providerResponse.headers, - providerResponse.status, - model - ); - - // Store rate-limit headers for quota saturation signals - try { - const { storeRateLimitHeaders } = await import("@/lib/quota/saturationSignals"); - storeRateLimitHeaders( - responseConnectionId, - provider, - providerResponse.headers as Record - ); - } catch { - // fail-open: saturation signal is best-effort - } - } catch (error) { - trackPendingRequest(model, provider, pendingConnId, false, undefined, pendingRequestId); - const errorMetadata = getSafeErrorMetadata(error); - const managedLeaseFenceCode = getManagedLeaseFenceErrorCode(errorMetadata.code); - if (managedLeaseFenceCode) return managedLeaseFenceErrorResult(managedLeaseFenceCode); - // isSemaphoreCapacityError already reads the code through getSafeErrorMetadata, - // so a hostile rejection cannot escape this classification. - if (isSemaphoreCapacityError(error)) { - const semaphoreCode = errorMetadata.code as string; - appendRequestLog({ - model, - provider, - connectionId, - status: `FAILED ${semaphoreCode}`, - }).catch(() => {}); - const failureMessage = sanitizeErrorMessage(errorMetadata.message) || "Semaphore timeout"; - persistAttemptLogs({ - status: HTTP_STATUS.RATE_LIMITED, - error: failureMessage, - providerRequest: finalBody || translatedBody, - clientResponse: buildErrorBody(HTTP_STATUS.RATE_LIMITED, failureMessage), - claudeCacheMeta: claudePromptCacheLogMeta, - cacheSource: "upstream", - }); - persistFailureUsage(HTTP_STATUS.RATE_LIMITED, semaphoreCode); - const result = stream - ? createStreamingErrorResult(HTTP_STATUS.RATE_LIMITED, failureMessage, semaphoreCode) - : createErrorResult(HTTP_STATUS.RATE_LIMITED, failureMessage); - return { - ...result, - errorType: "account_semaphore_capacity", - errorCode: semaphoreCode, - }; - } - // abort(reason) can reject with a raw string lacking `name`/`status`; classify - // it through isLocalStreamLifecycleError so it maps to 499 rather than the - // 502 provider-failure default. - let isRequestAborted = errorMetadata.name === "AbortError"; - if (!isRequestAborted) { - try { - isRequestAborted = isLocalStreamLifecycleError(error); - } catch { - // A hostile Proxy must not escape the provider-error boundary during classification. - } - } - // #8376: proxyFetch tags unreachable transport failures so they remain - // distinguishable from ordinary provider 5xx responses. - const isProxyUnreachableFailure = - !isRequestAborted && errorMetadata.errorCode === "proxy_unreachable"; - const errorCode = errorMetadata.code; - const localRateLimitFailure = localLimiterErrors.getClientSafeLocalRateLimitError(error); - const failureStatus = isRequestAborted - ? 499 - : isProxyUnreachableFailure - ? HTTP_STATUS.BAD_GATEWAY - : localRateLimitFailure - ? localRateLimitFailure.status - : errorMetadata.name === "TimeoutError" || errorMetadata.name === "BodyTimeoutError" - ? HTTP_STATUS.GATEWAY_TIMEOUT - : errorMetadata.status - ? errorMetadata.status - : HTTP_STATUS.BAD_GATEWAY; - const failureMessage = isRequestAborted - ? "Request aborted" - : (() => { - try { - return formatProviderError( - localRateLimitFailure ?? error, - provider, - model, - failureStatus - ); - } catch { - // Formatting is diagnostic only; hostile rejection metadata falls back safely. - return errorMetadata.message || "Upstream provider error"; - } - })(); - const safeFailureMessage = sanitizeErrorMessage(failureMessage) || "Upstream provider error"; - const upstreamErrorCode = - localRateLimitFailure?.code ?? - (isProxyUnreachableFailure ? "proxy_unreachable" : errorCode); - // Tag our own deadline timeouts (fetch-start TimeoutError / body BodyTimeoutError, - // both surfaced as a 504) as "upstream_timeout" so the cooldown layer can tell a - // slow-but-not-failed request apart from a real provider 5xx. (Antigravity already - // tags its pre-response timeout via the code below.) - const isOwnDeadlineTimeout = - failureStatus === HTTP_STATUS.GATEWAY_TIMEOUT && - (errorMetadata.name === "TimeoutError" || errorMetadata.name === "BodyTimeoutError"); - const upstreamErrorType = - upstreamErrorCode === ANTIGRAVITY_PRE_RESPONSE_TIMEOUT_CODE || isOwnDeadlineTimeout - ? "upstream_timeout" - : failureStatus === 401 - ? "authentication_error" - : undefined; - appendRequestLog({ - model, - provider, - connectionId, - status: `FAILED ${failureStatus}`, - }).catch(() => {}); - persistAttemptLogs({ - status: failureStatus, - error: safeFailureMessage, - providerRequest: finalBody || translatedBody, - // On a client-abort (AbortError), the client already disconnected before - // we ever got here — this body is what we WOULD have sent, not what was - // actually delivered. Logging it as `clientResponse` is misleading (the - // dashboard reads that field as "what the client received"), so omit it - // for this case; `error` above already records the failure reason. - clientResponse: - errorMetadata.name === "AbortError" - ? undefined - : buildErrorBody(failureStatus, failureMessage), - claudeCacheMeta: claudePromptCacheLogMeta, - cacheSource: "upstream", - }); - if (isRequestAborted) { - streamController.handleError(createSafeAbortError()); - return createErrorResult(499, "Request aborted"); - } - const persistentErrorCode = projectFailureUsageErrorCode({ - statusCode: failureStatus, - message: failureMessage, - errorCode: projectPublicErrorIdentifier( - upstreamErrorCode || errorMetadata.name, - "upstream_error" - ), - errorType: upstreamErrorType, - }); - persistFailureUsage(failureStatus, persistentErrorCode); - console.log(`${COLORS.red}[ERROR] ${safeFailureMessage}${COLORS.reset}`); - if (stream && upstreamErrorCode) { - const result = createStreamingErrorResult( - failureStatus, - failureMessage, - upstreamErrorCode, - upstreamErrorType - ); - localLimiterErrors.markTrustedLocalRateLimitResponse(result.response, error); - return { - ...result, - errorType: upstreamErrorType, - errorCode: upstreamErrorCode, - }; - } - const result = createErrorResult( - failureStatus, - failureMessage, - null, - upstreamErrorCode, - upstreamErrorType - ); - localLimiterErrors.markTrustedLocalRateLimitResponse(result.response, error); - return result; - } - let upstreamErrorParsed = false; - let parsedStatusCode = providerResponse.status; - let parsedMessage = ""; - let parsedRetryAfterMs: number | null = null; - let upstreamErrorBody: unknown = null; - - // Track whether stream_options was present and stripped — if so, 401/403 after - // that may be from the modification rather than a genuine auth failure, so we - // skip the credential refresh attempt in that case. - const hadStreamOptions = - targetFormat === FORMATS.OPENAI_RESPONSES && "stream_options" in translatedBody; - if (hadStreamOptions) { - delete translatedBody.stream_options; + const streamingOutcome = await runStreamingResponse({ + apiKeyInfo, + persistAttemptLogs, + buildUpstreamHeadersForExecute, + resolveEffectiveServiceTier, + clientResponseFormat, + correlationId, + extendedContext, + executeProviderRequest, + applyProviderFailureClassification, + assertManagedLeaseFence, + body, + clientRawRequest, + comboStrategy, + connectionId, + contextEditingEnabled, + credentials, + effectiveModel, + executeRefreshCredentials, + executor, + getCurrentConnectionId, + getExecutionCredentials, + getExecutorClientHeaders, + getManagedLeaseFenceErrorCode, + handleCredentialsRefreshed, + isCombo, + isOpencodeClient, + log, + managedLease, + managedLeaseFenceErrorResult, + model, + onCredentialsRefreshed, + pendingScope, + pendingConnId, + pendingRequestId, + persistFailureUsage, + provider, + providerRequestCapture, + reqLogger, + sessionAffinityKey, + skillRequestId, + sourceFormat, + stream, + streamController, + syncExecuteTranslatedBody, + targetFormat, + triedModels, + trustedEffortContext, + upstreamStream, + userAgent, + claudePromptCacheLogMeta, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + translatedBody, + }); + if (streamingOutcome) { + translatedBody = streamingOutcome.carry.translatedBody; + currentModel = streamingOutcome.carry.currentModel; + effectiveServiceTier = streamingOutcome.carry.effectiveServiceTier; + finalBody = streamingOutcome.carry.finalBody; + pipelineRecovered = streamingOutcome.carry.pipelineRecovered; + providerHeaders = streamingOutcome.carry.providerHeaders; + providerResponse = streamingOutcome.carry.providerResponse; + providerUrl = streamingOutcome.carry.providerUrl; } + if (streamingOutcome && streamingOutcome.response) return streamingOutcome.response; + } - // Handle 401/403 - try token refresh using executor - // T-PROBE: probe-origin failures never attempt the refresh — a probe must - // not consume a rotating refresh token nor persist an "expired" - // deactivation on refresh failure (#9817). The 401/403 then flows into - // the normal providerFailure classification (record-only in probe mode). - if ( - (providerResponse.status === HTTP_STATUS.UNAUTHORIZED || - providerResponse.status === HTTP_STATUS.FORBIDDEN) && - !hadStreamOptions && // Skip refresh if failure may be from stream_options removal, not auth - !(await shouldIsolateProbeFailures()) && - !(await shouldSkipCredentialRefresh(provider, providerResponse)) - ) { - // Fix A: wrap refreshCredentials in runWithOnPersist so the persist callback - // executes INSIDE the per-connection mutex held by getAccessToken. This makes - // [network refresh + DB write + outer-state mutation] one atomic step and - // prevents concurrent requests from reading a stale refreshToken before the - // DB has been updated (refresh_token_reused on Codex/OpenAI). - // - // Not every executor routes refresh through getAccessToken (e.g. github.ts - // calls refreshCopilotToken directly). When the persistFn doesn't fire from - // inside getAccessToken, we still need to do the credentials mutation + user - // callback after refreshCredentials returns. The `persistFnRan` flag tracks - // which path executed so we don't double-fire (race-prone) or skip (regression). - // Front 3: remember the refresh_token we are about to present so that, if the - // refresh fails as unrecoverable, we can tell a genuine death apart from a - // stale-token reuse that a concurrent/sibling refresh already rotated past. - const attemptedRefreshToken = - typeof credentials?.refreshToken === "string" ? credentials.refreshToken : null; - let persistFnRan = false; - const persistFn = onCredentialsRefreshed - ? async (refreshResult: Record) => { - persistFnRan = true; - // Mutate the shared credentials object so subsequent executor calls - // in this request see the new tokens. Runs INSIDE the mutex. - Object.assign(credentials, refreshResult); - await onCredentialsRefreshed(refreshResult); - } - : undefined; - - // #4038: build a compare-and-swap reread so getAccessToken can skip the persist if a - // concurrent writer (sibling request / HealthCheck / replica) already rotated this - // connection's refresh_token past the one we presented — overwriting would revert it - // and revoke the token family. No connectionId ⇒ no guard (behavior unchanged). - const casConnectionId = - typeof credentials?.connectionId === "string" ? credentials.connectionId.trim() : ""; - const casReread = casConnectionId - ? async () => { - const latest = await getProviderConnectionById(casConnectionId); - return typeof latest?.refreshToken === "string" ? latest.refreshToken : null; - } - : null; - - const newCredentials = (await refreshWithRetry( - () => - runWithCasGuard( - casReread ? { expectedRefreshToken: attemptedRefreshToken, reread: casReread } : null, - () => runWithOnPersist(persistFn, () => executor.refreshCredentials(credentials, log)) - ), - 3, - log, - provider // Explicitly pass the provider to avoid universally tripping the "unknown" circuit breaker - )) as null | { - accessToken?: string; - copilotToken?: string; - }; - - if (newCredentials?.accessToken || newCredentials?.copilotToken) { - log?.info?.("TOKEN", `${provider?.toUpperCase()} | refreshed`); - - // Fall back to post-mutex mutation only for executors that don't route - // through getAccessToken (and therefore never fire onPersist). For - // executors that DO route through it (Codex, Claude, Gemini, etc.) the - // mutation already happened atomically inside the mutex. - if (!persistFnRan) { - Object.assign(credentials, newCredentials); - if (onCredentialsRefreshed) { - await onCredentialsRefreshed(newCredentials); - } - } + // Non-streaming response + if (!stream) { + const nonStreamingOutcome = await runNonStreamingResponse({ + apiKeyInfo, + appendRequestLog, + applyProviderFailureClassification, + assertManagedLeaseFence, + attachCompressionUsageReceiptAfterAnalytics, + body, + bodyForCacheWrite, + buildCacheUsageLogMeta, + buildCostCtx, + buildErrorBody, + calculateCost, + claudePromptCacheLogMeta, + clientRawRequest, + clientRequestedResponsesStream, + clientResponseFormat, + comboStrategy, + compressionResponseMeta, + connectionId, + contextEditingEnabled, + copilotCompatibleReasoning, + createErrorResult, + credentials, + currentModel, + customToolNames, + describeMalformedNonStream, + detectMalformedNonStream, + echoModel, + echoModelInObject, + effectiveModel, + effectiveServiceTier, + emitRequestGamificationEvent, + endpointPath, + executeProviderRequest, + executeRefreshCredentials, + extractSessionAffinityKey, + extractUsageFromResponse, + fallbackAttempts, + finalBody, + finalizePendingScope, + getCurrentConnectionId, + getManagedLeaseFenceErrorCode, + getModelNormalizeToolCallId, + getModelPreserveOpenAIDeveloperRole, + getSafeErrorMetadata, + getSkillsModelIdForFormat, + guardrailRegistry, + handleCredentialsRefreshed, + handleToolCallExecution, + idempotencyKey, + incrementTokenUsage, + injectionResult, + isClaudeCodeCompatible, + isCombo, + isJsonRecord, + isLocalStreamLifecycleError, + isResponsesEndpoint, + isSemaphoreCapacityError, + isServerOwnedToolLoopEnabled, + log, + logAuditEvent, + managedLease, + managedLeaseFenceErrorResult, + markAccountSemaphoreBlocked, + memoryOwnerId, + memorySettings, + model, + normalizeHeaders, + normalizeUsage, + onRequestSuccess, + pendingConnId, + pendingRequestId, + pendingScope, + persistAttemptLogs, + persistFailureUsage, + pipelineRecovered, + pipelineSessionId, + preserveCacheControl, + provider, + providerHeaders, + providerRequestCapture, + providerResponse, + reasoningCacheScope, + reasoningReplayHistory, + recordChatCallCost, + recordContextEditingTelemetryHook, + recordCoreOwnedAntigravityQuotaState, + recordNonStreamingUsageStats, + reportMalformed200, + reqLogger, + requestToolIdentityMap, + resolveReportedServiceTier, + runMemoryExtractionGate, + runPluginOnResponseHook, + sanitizeErrorMessage, + sanitizeUpstreamDetails, + saveIdempotency, + scheduleQuotaShareConsumption, + semanticCacheEnabled, + sessionAffinityKey, + shouldIsolateProbeFailures, + skillRequestId, + sourceFormat, + startTime, + stream, + targetFormat, + toolNameMap, + traceEnabled, + traceId, + trackPendingRequest, + translateRequest, + translatedBody, + triedModels, + updateFromHeaders, + updateFromResponseBody, + updatePendingScope, + updateProviderConnection, + videoBridgeObserved, + webFetchFallbackPlan, + webSearchFallbackPlan, + syncExecuteTranslatedBody, + }); + if (nonStreamingOutcome) { + translatedBody = nonStreamingOutcome.carry.translatedBody; + currentModel = nonStreamingOutcome.carry.currentModel; + finalBody = nonStreamingOutcome.carry.finalBody; + providerResponse = nonStreamingOutcome.carry.providerResponse; + providerHeaders = nonStreamingOutcome.carry.providerHeaders; + effectiveServiceTier = nonStreamingOutcome.carry.effectiveServiceTier; + claudePromptCacheLogMeta = nonStreamingOutcome.carry.claudePromptCacheLogMeta; + reasoningReplayHistory = nonStreamingOutcome.carry.reasoningReplayHistory; + pipelineRecovered = nonStreamingOutcome.carry.pipelineRecovered; + return nonStreamingOutcome.response; + } + } - // Retry with new credentials — model + extra headers follow translatedBody.model so they - // stay aligned if this block ever runs after a path that mutates body.model (e.g. fallback). - try { - const retryModelId = String(translatedBody.model || effectiveModel); - const retryBody = await prepareUpstreamBody({ - translatedBody, - modelToCall: retryModelId, - ...trustedEffortContext, - provider, - targetFormat, - credentials: getExecutionCredentials(), - log, - bypassDefaultToolLimit: isOpencodeClient, - isOpencodeClient, - rawBody: body, - clientRawRequest, - }); - assertManagedLeaseFence(getExecutionConnectionId(getExecutionCredentials())); - const retryResult = normalizeExecutorResult( - await runWithCapture(providerRequestCapture, () => - executor.execute({ - model: retryModelId, - body: retryBody, - stream: upstreamStream, - credentials: getExecutionCredentials(), - signal: streamController.signal, - log, - extendedContext, - upstreamExtraHeaders: buildUpstreamHeadersForExecute(retryModelId), - clientHeaders: getExecutorClientHeaders(), - clientResponseFormat, - onCredentialsRefreshed, - skipUpstreamRetry: isCombo, - contextEditing: { enabled: contextEditingEnabled }, - correlationId, - }) - ) - ); - - if (retryResult.response.ok) { - providerResponse = retryResult.response; - providerUrl = retryResult.url; - providerHeaders = new Headers(retryResult.headers || {}); - finalBody = providerRequestCapture.body(retryResult.transformedBody); - reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody); - updatePendingScope(pendingScope, { - providerRequest: finalBody, - providerUrl, - stage: "provider_response_started", - }); - upstreamErrorParsed = false; // Reset since new response is OK - } else { - providerResponse = retryResult.response; - upstreamErrorParsed = false; // Let it be parsed downstream - } - } catch (retryErr) { - const retryLeaseFenceCode = getManagedLeaseFenceErrorCode( - getUpstreamErrorIdentifier(retryErr) - ); - if (retryLeaseFenceCode) return managedLeaseFenceErrorResult(retryLeaseFenceCode); - // Refresh succeeded but the retry leg failed (network blip, AbortError, - // executor throw). Don't swallow — the operator-visible signal "the user - // saw 401 even though auth was actually fixed" is much more confusing - // than the original 401 alone. Surface at error level with sanitization. - log?.error?.( - "TOKEN", - `${provider?.toUpperCase()} | retry after refresh failed: ${sanitizeErrorMessage(retryErr)}` - ); - } - } else { - log?.warn?.("TOKEN", `${provider?.toUpperCase()} | refresh failed`); - if (isUnrecoverableRefreshError(newCredentials) && onCredentialsRefreshed) { - // Front 3 (reuse-race tolerance): before deactivating, re-read the DB. - // If a sibling/concurrent refresh already rotated this connection's - // refresh_token (common for Codex/OpenAI under one shared Auth0 client), - // the failure we saw was a stale-token reuse — the account is healthy - // with the newer token, so keep it active instead of killing it. - let alreadyRotated = false; - if (typeof connectionId === "string" && connectionId && attemptedRefreshToken) { - try { - const latest = await getProviderConnectionById(connectionId); - if (wasRefreshTokenRotated(attemptedRefreshToken, latest?.refreshToken)) { - alreadyRotated = true; - log?.warn?.( - "TOKEN", - `${provider.toUpperCase()} | refresh_token already rotated by a concurrent refresh — keeping connection active` - ); - } - } catch { - // DB read failed — fall through to the safe default (deactivate). - } - } - if (!alreadyRotated) { - await onCredentialsRefreshed({ testStatus: "expired", isActive: false }); - } - } - } - } - - // Check provider response - return error info for fallback handling - providerFailure: if (!providerResponse.ok) { - trackPendingRequest(model, provider, pendingConnId, false, undefined, pendingRequestId); - - let statusCode = providerResponse.status; - let message = ""; - let retryAfterMs: number | null = null; - let upstreamErrorCode: string | undefined; - let upstreamErrorType: string | undefined; - - if (upstreamErrorParsed) { - statusCode = parsedStatusCode; - message = parsedMessage; - retryAfterMs = parsedRetryAfterMs; - } else { - const details = await parseUpstreamError(providerResponse, provider); - statusCode = details.statusCode; - message = details.message; - retryAfterMs = details.retryAfterMs; - upstreamErrorBody = details.responseBody; - upstreamErrorCode = typeof details.errorCode === "string" ? details.errorCode : undefined; - upstreamErrorType = typeof details.errorType === "string" ? details.errorType : undefined; - } - - // Gateways like agentrouter misstate temporary quota exhaustion as 403/400, - // which downstream classification treats as AUTH_ERROR and clients like - // Claude Code treat as permanent. Restate to 429 (+ synthetic Retry-After) - // BEFORE any classification so both the fallback engine and the surfaced - // client status see a retryable error. Registry-scoped per provider. - const restatement = applyStatusRestatement({ - provider, - status: statusCode, - message, - body: upstreamErrorBody, - retryAfterMs, - }); - if (restatement.ruleId) { - statusCode = restatement.status; - retryAfterMs = restatement.retryAfterMs; - log?.info?.( - "STATUS_RESTATE", - `${provider} ${restatement.fromStatus}→${statusCode} (${restatement.ruleId})` - ); - } - - const signatureRecovery = pipelineRecovered - ? { attempted: false, succeeded: false, execution: null, error: null, recoveryBody: null } - : await recoverAnthropicThinkingSignature({ - provider, - statusCode, - message, - body: translatedBody, - execute: async (recoveryBody) => { - translatedBody = recoveryBody as typeof translatedBody; - return executeProviderRequest(currentModel, false); - }, - parseError: (response) => parseUpstreamError(response, provider), - }); - if (!pipelineRecovered && signatureRecovery.attempted && signatureRecovery.execution) { - providerResponse = signatureRecovery.execution.response; - if (signatureRecovery.succeeded) { - providerUrl = signatureRecovery.execution.url; - providerHeaders = signatureRecovery.execution.headers; - finalBody = providerRequestCapture.body(signatureRecovery.execution.transformedBody); - reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody); - updatePendingScope(pendingScope, { - providerRequest: finalBody, - providerUrl, - stage: "provider_response_started", - }); - log?.info?.( - "THINKING_SIGNATURE", - `Recovered ${provider}/${currentModel} after one historical-thinking retry` - ); - } else if (signatureRecovery.error) { - statusCode = signatureRecovery.error.statusCode; - message = signatureRecovery.error.message; - retryAfterMs = signatureRecovery.error.retryAfterMs; - upstreamErrorBody = signatureRecovery.error.responseBody; - upstreamErrorCode = - typeof signatureRecovery.error.errorCode === "string" - ? signatureRecovery.error.errorCode - : undefined; - upstreamErrorType = - typeof signatureRecovery.error.errorType === "string" - ? signatureRecovery.error.errorType - : undefined; - } - } - - if (signatureRecovery.succeeded) break providerFailure; - - // #10281 — tiny-budget reasoning probes (e.g. Claude Code's `/model` check - // sends `max_tokens: 1`): the model burns the whole budget on thinking, and - // some upstreams (e.g. api.cline.bot for deepseek-v4-flash) answer the empty - // outcome with a 5xx ("empty response content") instead of a truncated 200. - // Answer such probes with a valid truncated response rather than relaying the - // upstream failure — which would also mark the connection unavailable and - // poison fallback/cooldown bookkeeping for a request that is only a probe. - if ( - !stream && - isTinyBudgetReasoningProbe({ model: currentModel, body: finalBody || translatedBody }) && - isEmptyContentUpstreamFailure(statusCode, message) - ) { - providerResponse = buildReasoningProbeTruncatedResponse({ - model: currentModel, - maxTokens: toPositiveInteger( - (finalBody || translatedBody)?.max_tokens ?? - (finalBody || translatedBody)?.max_completion_tokens - ), - requestId: skillRequestId, - }); - log?.warn?.( - "PROBE", - `Reasoning probe (max_tokens < ${REASONING_BUFFER_MIN_TRIGGER}) answered with truncated 200 — upstream reported "${message}"` - ); - break providerFailure; - } - - const errorConnectionId = getCurrentConnectionId() || connectionId; - await applyProviderFailureClassification({ - statusCode, - message, - headers: providerResponse.headers, - upstreamErrorBody, - retryAfterMs, - targetModel: currentModel, - }); - - appendRequestLog({ - model, - provider, - connectionId: errorConnectionId, - status: `FAILED ${statusCode}`, - }).catch(() => {}); - - const errMsg = formatProviderError(new Error(message), provider, model, statusCode); - const safeErrMsg = sanitizeErrorMessage(errMsg) || "Upstream provider error"; - const safeUpstreamErrorBody = sanitizeUpstreamDetails(upstreamErrorBody); - console.log(`${COLORS.red}[ERROR] ${safeErrMsg}${COLORS.reset}`); - - // Log Antigravity retry time if available - if (retryAfterMs && provider === "antigravity") { - const retrySeconds = Math.ceil(retryAfterMs / 1000); - log?.debug?.("RETRY", `Antigravity quota reset in ${retrySeconds}s (${retryAfterMs}ms)`); - } - - // Log error with full request body for debugging - reqLogger.logError(new Error(message), finalBody || translatedBody); - reqLogger.logProviderResponse( - providerResponse.status, - providerResponse.statusText, - providerResponse.headers, - safeUpstreamErrorBody - ); - - // ── T5: Intra-family model fallback ────────────────────────────────────── - // Before returning a model-unavailable error upstream, try sibling models - // from the same family. This keeps the request alive on the same account - // instead of failing the entire combo. - if (!pipelineRecovered && isModelUnavailableError(statusCode, message, provider)) { - const nextModel = getNextFamilyFallback(currentModel, triedModels, provider); - if (nextModel) { - triedModels.add(nextModel); - currentModel = nextModel; - translatedBody.model = nextModel; - log?.info?.( - "MODEL_FALLBACK", - `${model} unavailable (${statusCode}) → trying ${nextModel}` - ); - // Re-execute with the fallback model - try { - const fallbackResult = await executeProviderRequest(nextModel, false); - if (fallbackResult.response.ok) { - providerResponse = fallbackResult.response; - providerUrl = fallbackResult.url; - providerHeaders = fallbackResult.headers; - finalBody = providerRequestCapture.body(fallbackResult.transformedBody); - reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody); - updatePendingScope(pendingScope, { - providerRequest: finalBody, - providerUrl, - stage: "provider_response_started", - }); - // Continue processing with the fallback response — skip error return - log?.info?.("MODEL_FALLBACK", `Serving ${nextModel} as fallback for ${model}`); - // Jump to streaming/non-streaming handling below - // We fall through by NOT returning here - } else { - // Fallback also failed — return original error - persistAttemptLogs({ - status: statusCode, - error: safeErrMsg, - providerRequest: finalBody || translatedBody, - providerResponse: safeUpstreamErrorBody, - clientResponse: buildErrorBody(statusCode, errMsg), - cacheSource: "upstream", - }); - persistFailureUsage(statusCode, "model_unavailable"); - return createErrorResult( - statusCode, - errMsg, - retryAfterMs, - upstreamErrorCode, - upstreamErrorType, - upstreamErrorBody, - { passthrough: sourceFormat === FORMATS.CLAUDE } - ); - } - } catch { - persistAttemptLogs({ - status: statusCode, - error: safeErrMsg, - providerRequest: finalBody || translatedBody, - providerResponse: safeUpstreamErrorBody, - clientResponse: buildErrorBody(statusCode, errMsg), - cacheSource: "upstream", - }); - persistFailureUsage(statusCode, "model_unavailable"); - return createErrorResult( - statusCode, - errMsg, - retryAfterMs, - upstreamErrorCode, - upstreamErrorType, - upstreamErrorBody, - { passthrough: sourceFormat === FORMATS.CLAUDE } - ); - } - } else { - persistAttemptLogs({ - status: statusCode, - error: safeErrMsg, - providerRequest: finalBody || translatedBody, - providerResponse: safeUpstreamErrorBody, - clientResponse: buildErrorBody(statusCode, errMsg), - cacheSource: "upstream", - }); - persistFailureUsage(statusCode, "model_unavailable"); - return createErrorResult( - statusCode, - errMsg, - retryAfterMs, - upstreamErrorCode, - upstreamErrorType, - upstreamErrorBody, - { passthrough: sourceFormat === FORMATS.CLAUDE } - ); - } - } else if (isContextOverflowError(statusCode, message)) { - const familyCandidates = getModelFamily(currentModel, provider).filter( - (m) => m !== currentModel && !triedModels.has(m) - ); - const nextModel = - findLargerContextModel(currentModel, familyCandidates, provider) ?? - getNextFamilyFallback(currentModel, triedModels, provider); - if (nextModel) { - triedModels.add(nextModel); - currentModel = nextModel; - translatedBody.model = nextModel; - log?.info?.( - "CONTEXT_OVERFLOW_FALLBACK", - `${model} context overflow → trying ${nextModel}` - ); - try { - const fallbackResult = await executeProviderRequest(nextModel, false); - if (fallbackResult.response.ok) { - providerResponse = fallbackResult.response; - providerUrl = fallbackResult.url; - providerHeaders = fallbackResult.headers; - finalBody = providerRequestCapture.body(fallbackResult.transformedBody); - reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody); - updatePendingScope(pendingScope, { - providerRequest: finalBody, - providerUrl, - stage: "provider_response_started", - }); - log?.info?.( - "CONTEXT_OVERFLOW_FALLBACK", - `Serving ${nextModel} as fallback for ${model}` - ); - } else { - persistAttemptLogs({ - status: statusCode, - error: safeErrMsg, - providerRequest: finalBody || translatedBody, - providerResponse: safeUpstreamErrorBody, - clientResponse: buildErrorBody(statusCode, errMsg), - cacheSource: "upstream", - }); - persistFailureUsage(statusCode, "context_overflow"); - return createErrorResult( - statusCode, - errMsg, - retryAfterMs, - upstreamErrorCode, - upstreamErrorType, - upstreamErrorBody, - { passthrough: sourceFormat === FORMATS.CLAUDE } - ); - } - } catch { - persistAttemptLogs({ - status: statusCode, - error: safeErrMsg, - providerRequest: finalBody || translatedBody, - providerResponse: safeUpstreamErrorBody, - clientResponse: buildErrorBody(statusCode, errMsg), - cacheSource: "upstream", - }); - persistFailureUsage(statusCode, "context_overflow"); - return createErrorResult( - statusCode, - errMsg, - retryAfterMs, - upstreamErrorCode, - upstreamErrorType, - upstreamErrorBody, - { passthrough: sourceFormat === FORMATS.CLAUDE } - ); - } - } else { - persistAttemptLogs({ - status: statusCode, - error: safeErrMsg, - providerRequest: finalBody || translatedBody, - providerResponse: safeUpstreamErrorBody, - clientResponse: buildErrorBody(statusCode, errMsg), - cacheSource: "upstream", - }); - persistFailureUsage(statusCode, "context_overflow"); - return createErrorResult( - statusCode, - errMsg, - retryAfterMs, - upstreamErrorCode, - upstreamErrorType, - upstreamErrorBody, - { passthrough: sourceFormat === FORMATS.CLAUDE } - ); - } - } else { - persistAttemptLogs({ - status: statusCode, - error: safeErrMsg, - providerRequest: finalBody || translatedBody, - providerResponse: safeUpstreamErrorBody, - clientResponse: buildErrorBody(statusCode, errMsg), - cacheSource: "upstream", - }); - persistFailureUsage(statusCode, `upstream_${statusCode}`); - - // Emergency budget fallback is orchestrated exclusively by the routing layer - // (src/sse/handlers/chat.ts), which resolves credentials FOR the emergency - // provider through account selection. The executor-level hop that used to - // live here re-sent the FAILING provider's credentials to the emergency - // provider's endpoint (e.g. the OpenAI API key to integrate.api.nvidia.com) - // — a cross-provider credential leak that also never succeeded upstream. - return createErrorResult( - statusCode, - errMsg, - retryAfterMs, - upstreamErrorCode, - upstreamErrorType, - upstreamErrorBody, - { passthrough: sourceFormat === FORMATS.CLAUDE } - ); - } - // ── End T5 ─────────────────────────────────────────────────────────────── - } - } - - // Non-streaming response - if (!stream) { - try { - const runNonStreamingPipeline = async ({ - policy, - model: pipelineModel, - translatedBody: wireBody, - }) => { - translatedBody = wireBody as typeof translatedBody; - currentModel = pipelineModel; - triedModels.add(pipelineModel); - return runProviderExecutionPipeline({ - policy, - target: { - provider, - requestedModel: pipelineModel, - sourceFormat, - targetFormat, - stream: false, - }, - connection: { - initialConnectionId: String(getCurrentConnectionId() || connectionId || ""), - getCurrentConnectionId: () => getCurrentConnectionId() || undefined, - getCredentials: () => (credentials || {}) as Record, - replaceCredentials: (next) => { - Object.assign(credentials, next); - }, - onCredentialsRefreshed: handleCredentialsRefreshed, - refreshCredentials: executeRefreshCredentials, - assertManagedLeaseFence: (id) => { - assertManagedLeaseFence(id); - }, - getProviderCredentials, - }, - wire: { - body: translatedBody as Record, - currentModel, - triedModels, - setBodyAndModel: (nextBody, nextModel) => { - translatedBody = nextBody as typeof translatedBody; - currentModel = nextModel; - triedModels.add(nextModel); - }, - }, - state: { - updatePendingStage: (stage, data) => { - updatePendingScope(pendingScope, { stage, ...(data || {}) }); - }, - recordRateLimitHeaders: updateFromHeaders, - recordRateLimitBody: updateFromResponseBody, - writeTerminalStatus, - persistConnectionPatch: updateProviderConnection, - setConnectionRateLimitedUntil: async (id, untilMs) => { - const { setConnectionRateLimitUntil } = await import("@/lib/db/providers"); - setConnectionRateLimitUntil(id, untilMs); - }, - lockModel, - recordAntigravityQuotaState: recordCoreOwnedAntigravityQuotaState, - markAccountSemaphoreBlocked: (key) => { - markAccountSemaphoreBlocked(key, Date.now() + 60_000); - }, - isolateProbeFailures: () => shouldIsolateProbeFailures(), - onCodexScopeRateLimited: async (params) => { - await markCodexScopeRateLimited({ - failedConnectionId: params.failedConnectionId, - model: params.model, - rateLimitedUntil: params.rateLimitedUntil, - credentials: (params.credentials || credentials) as { - connectionId?: string | null; - providerSpecificData?: unknown; - }, - }); - }, - onClearSessionAffinity: () => { - const key = - sessionAffinityKey || - extractSessionAffinityKey(body, clientRawRequest?.headers) || - null; - if (!key) return; - try { - deleteSessionAccountAffinity(key, "codex"); - } catch { - // best-effort - } - }, - onAuditAccountRotation: (params) => { - logAuditEvent({ - action: params.action, - actor: apiKeyInfo?.name || "system", - target: params.newConnectionId, - details: { - failed_connection_id: params.failedConnectionId, - new_connection_id: params.newConnectionId, - attempt: params.attempt, - retry_after_ms: params.retryAfterMs, - }, - }); - }, - }, - sendProviderAttempt: (modelToCall, allowDedup) => - executeProviderRequest(modelToCall, allowDedup), - }); - }; - - let toolLoopRan = false; - let toolLoopUsage = null; - let legResult = await runNonStreamingProviderLeg({ - phase: "initial", - sourceBody: (body || {}) as Record, - expectedConnectionId: managedLease - ? String(getCurrentConnectionId() || connectionId || "") || undefined - : undefined, - allowAccountRotation: !managedLease && comboStrategy !== "context-relay", - allowModelFallback: true, - executeProviderRequest: (modelToCall, allowDedup) => - executeProviderRequest(modelToCall, allowDedup), - runProviderExecution: runNonStreamingPipeline, - setRequestWireState: ({ translatedBody: nextBody, effectiveModel: nextModel }) => { - translatedBody = nextBody as typeof translatedBody; - currentModel = nextModel; - triedModels.add(nextModel); - }, - sourceFormat, - targetFormat, - clientResponseFormat, - provider, - model: effectiveModel, - connectionId: String(getCurrentConnectionId() || connectionId || ""), - getCurrentConnectionId: () => getCurrentConnectionId() || undefined, - effectiveModel: currentModel, - translatedBody: translatedBody as Record, - toolNameMap, - customToolNames, - requestToolIdentityMap, - reasoningCacheScope, - reasoningReplayHistory, - videoTranscriptSensitive: videoBridgeObserved, - clientHeaders: clientRawRequest?.headers ?? null, - isClaudeCodeCompatible, - log, - }); - - if (legResult.kind === "error") { - const err = legResult.result; - const errMessage = - err?.rawMessage || - (err?.originalError instanceof Error ? err.originalError.message : err?.error) || - ""; - const errHeaders = err?.upstreamHeaders || err?.response?.headers; - const errUpstreamBody = err?.upstreamErrorBody; - if (err) { - await applyProviderFailureClassification({ - statusCode: err.status, - message: errMessage, - headers: errHeaders, - upstreamErrorBody: errUpstreamBody, - retryAfterMs: err.retryAfterMs ?? null, - targetModel: currentModel, - }); - } - - const captured = providerRequestCapture.latest?.() ?? null; - finalBody = captured?.body ?? finalBody ?? translatedBody; - if (captured) { - reqLogger.logTargetRequest(captured.url, captured.headers, captured.body); - } - reqLogger.logError(new Error(err.error || "Provider request failed"), finalBody); - const isNetworkThrow = Boolean(err.originalError); - if (err.response && !isNetworkThrow) { - reqLogger.logProviderResponse( - err.status, - err.response.statusText || "Error", - err.response.headers, - err.response - ); - } - appendRequestLog({ - model, - provider, - connectionId, - status: `FAILED ${err.status}`, - }).catch(() => {}); - persistAttemptLogs({ - status: err.status, - error: err.error || "Provider request failed", - providerRequest: finalBody || translatedBody, - providerResponse: isNetworkThrow ? undefined : err.response, - // On a client abort the client already disconnected before we got here, so this - // body is what we WOULD have sent, not what was delivered. The dashboard reads - // `clientResponse` as "what the client received", so logging it misleads — - // `error` above already records the reason. The pre-#12867 path omitted it here; - // the leg-based path must keep doing so. - clientResponse: isLocalStreamLifecycleError(err.originalError) - ? undefined - : buildErrorBody(err.status, err.error || "Provider request failed"), - cacheSource: "upstream", - }); - persistFailureUsage(err.status, err.errorCode || `upstream_${err.status}`); - trackPendingRequest(model, provider, pendingConnId, false, undefined, pendingRequestId); - return err; - } - - pipelineRecovered = true; - const expectedConn = managedLease - ? String(getCurrentConnectionId() || connectionId || "") || undefined - : undefined; - // The identity is the tool loop's execution fence key, and deriveToolRequestIdentity - // canonicalizes the body — which by design rejects Dates, Maps and class instances. - // It was computed eagerly, so a body carrying any of those threw on EVERY - // non-streaming request even with SERVER_OWNED_TOOL_LOOP_ENABLED off (the default). - // Derive it only when the loop can run, and fail closed rather than crash: no - // identity means no fence, and without a fence the loop must not run. - let toolLoopEnabled = isServerOwnedToolLoopEnabled(); - let postInjectionRequestIdentity = ""; - if (toolLoopEnabled) { - try { - postInjectionRequestIdentity = derivePostInjectionRequestIdentity({ - apiKeyId: memoryOwnerId || "local", - headers: clientRawRequest?.headers ?? null, - skillRequestId, - postInjectionBody: (body || {}) as Record, - }); - } catch (identityError) { - log?.warn?.( - "SERVER_OWNED_TOOL_LOOP", - `request body is not canonicalizable, skipping the loop: ${ - identityError instanceof Error ? identityError.message : "unknown" - }` - ); - toolLoopEnabled = false; - } - } - const loopApply = await applyServerOwnedToolLoopIfNeeded({ - enabled: toolLoopEnabled, - stream, - isResponsesEndpoint, - sourceFormat, - initialLeg: legResult, - sourceBody: (body || {}) as Record, - skillsModelId: getSkillsModelIdForFormat(sourceFormat), - executionContext: { - apiKeyId: memoryOwnerId || "local", - sessionId: pipelineSessionId, - requestId: skillRequestId, - requestIdentity: postInjectionRequestIdentity, - builtinToolNames: injectionResult.builtinToolNames, - injectedCustomSkillNames: injectionResult.injectedCustomSkillNames, - customSkillExecutionEnabled: - Boolean(memoryOwnerId) && memorySettings?.skillsEnabled === true, - executionFenceEnabled: true, - provider, - model: effectiveModel, - }, - abortSignal: clientRawRequest?.signal, - expectedConnectionId: expectedConn, - followUpLeg: async (nextSourceBody) => { - translatedBody = translateRequest( - sourceFormat, - targetFormat, - model, - { ...nextSourceBody }, - false, - credentials, - provider, - reqLogger, - { - normalizeToolCallId: getModelNormalizeToolCallId( - provider || "", - model || "", - sourceFormat - ), - preserveDeveloperRole: getModelPreserveOpenAIDeveloperRole( - provider || "", - model || "", - sourceFormat - ), - preserveCacheControl, - signatureNamespace: connectionId, - copilotClient: copilotCompatibleReasoning, - reasoningCacheScope, - onReasoningReplayHistory: (messages) => { - reasoningReplayHistory = messages; - }, - } - ); - return runNonStreamingProviderLeg( - followUpLegInput( - { - executeProviderRequest: (modelToCall, allowDedup) => - executeProviderRequest(modelToCall, allowDedup), - runProviderExecution: runNonStreamingPipeline, - setRequestWireState: ({ translatedBody: nextBody, effectiveModel: nextModel }) => { - translatedBody = nextBody as typeof translatedBody; - currentModel = nextModel; - triedModels.add(nextModel); - }, - sourceFormat, - targetFormat, - clientResponseFormat, - provider, - model: effectiveModel, - connectionId: String(getCurrentConnectionId() || connectionId || ""), - getCurrentConnectionId: () => getCurrentConnectionId() || undefined, - effectiveModel: currentModel, - translatedBody: translatedBody as Record, - toolNameMap, - customToolNames, - requestToolIdentityMap, - reasoningCacheScope, - reasoningReplayHistory, - videoTranscriptSensitive: videoBridgeObserved, - clientHeaders: clientRawRequest?.headers ?? null, - isClaudeCodeCompatible, - log, - }, - nextSourceBody, - expectedConn - ) - ); - }, - logReceipt: (receipt) => reqLogger.logToolLoopReceipt(receipt), - }); - if (loopApply.kind === "error") { - return await finalizeToolLoopError({ - loop: loopApply.loop, - model, - provider, - connectionId: pendingConnId, - providerRequest: loopApply.loop.finalProviderRequest || finalBody || translatedBody, - persistFailureUsage, - persistAttemptLogs, - trackPendingRequest, - pendingRequestId, - }); - } - // `legResult` is declared as the full NonStreamingProviderLegResult union. The - // `kind === "error"` guard above narrows it to the ok variant, but the conditional - // reassignment below widens it back to the declared type, so every field read past - // this point lost the narrowing — 13 TS2339 diagnostics under - // tsconfig.typecheck-api.json, which pulls chatCore.ts in through the route while - // tsconfig.typecheck-core.json does not. Pin the ok variant in its own binding: - // `loopApply.leg` is already `NonStreamingProviderLegResult & { kind: "ok" }`, - // so no cast is involved. - let okLeg: NonStreamingProviderLegResult & { kind: "ok" } = legResult; - if (loopApply.kind === "ok") { - toolLoopRan = true; - toolLoopUsage = loopApply.usage; - okLeg = loopApply.leg; - } - - if (okLeg.upstreamResponse) { - providerResponse = okLeg.upstreamResponse; - providerHeaders = normalizeHeaders(okLeg.upstreamResponse.headers); - } else { - providerResponse = new Response(null, { - status: 200, - headers: okLeg.headers, - }); - providerHeaders = normalizeHeaders(okLeg.headers); - } - finalBody = providerRequestCapture.body(okLeg.providerRequest || translatedBody); - // Built inside executeProviderRequest on the pre-#12867 path. The leg now owns the - // first non-streaming send, so that assignment never runs here and the meta stayed - // null — `_omniroute.claudePromptCache` silently vanished from every call log on - // this path. Same inputs, same helper, at the point where they are available. - claudePromptCacheLogMeta = buildClaudePromptCacheLogMeta( - targetFormat, - finalBody, - providerHeaders, - clientRawRequest?.headers - ); - const capturedOk = providerRequestCapture.latest?.(); - reqLogger.logTargetRequest( - okLeg.requestUrl || capturedOk?.url || "", - okLeg.requestHeaders || capturedOk?.headers || {}, - capturedOk?.body ?? finalBody - ); - const responseBody = okLeg.providerBody; - const responsePayloadFormat = okLeg.responsePayloadFormat; - const looksLikeSSE = okLeg.looksLikeSSE; - let translatedResponse = okLeg.response; - const memoryExtractionResponse = okLeg.responseForMemoryExtraction; - reqLogger.logProviderResponse( - 200, - "OK", - providerResponse.headers, - looksLikeSSE - ? { _streamed: true, _format: "sse-json", summary: responseBody } - : responseBody - ); - effectiveServiceTier = resolveReportedServiceTier(responseBody) ?? effectiveServiceTier; - if (onRequestSuccess) { - await onRequestSuccess(); - } - const successConnectionId = getCurrentConnectionId(); - await maybeSyncClaudeExtraUsageState({ - provider, - connectionId: successConnectionId, - providerSpecificData: credentials?.providerSpecificData, - log, - }); - const usage = toolLoopUsage ?? extractUsageFromResponse(responseBody, provider); - const cacheUsageLogMeta = buildCacheUsageLogMeta(usage); - if (usage && typeof usage === "object") { - attachCompressionUsageReceiptAfterAnalytics(usage as Record, "provider"); - if (provider === "gemini") { - const promptTokens = - typeof (usage as Record).prompt_tokens === "number" - ? ((usage as Record).prompt_tokens as number) - : 0; - if (promptTokens > 0) incrementTokenUsage(model, promptTokens); - } - } - recordContextEditingTelemetryHook({ - contextEditingEnabled, - provider, - responseBody, - skillRequestId, - log, - }); - appendRequestLog({ - model, - provider, - connectionId: successConnectionId, - tokens: usage, - status: "200 OK", - }).catch(() => {}); - recordNonStreamingUsageStats(usage, { - traceEnabled, - provider, - connectionId: successConnectionId, - model, - startTime, - apiKeyInfo, - effectiveServiceTier, - isCombo, - comboStrategy, - endpoint: endpointPath, cpaAuthIndex: readCpaAuthIndex(providerResponse), - }); - - // #12150 P1b surface 3 (fix round 1): a video-bridge-observed request's - // request- AND response-derived text both carry the full transcript (the - // flattened description on the request side, the model's own reply on - // the response side) — neither may populate durable Memory. See - // runMemoryExtractionGate for the shared gate + extraction wiring, unit - // tested directly in tests/unit/video-bridge-memory-suppression.test.ts. - runMemoryExtractionGate({ - memoryOwnerId, - memorySettings, - videoBridgeObserved, - pipelineSessionId, - requestBody: body as Record, - responseBody: memoryExtractionResponse as Record | null, - extractFacts, - log, - }); - - const customSkillExecutionEnabled = - Boolean(memoryOwnerId) && memorySettings?.skillsEnabled === true; - const builtinToolNames = [ - webSearchFallbackPlan.toolName, - webFetchFallbackPlan.toolName, - ...(memoryOwnerId && memorySettings?.enabled ? MEMORY_BUILTIN_TOOL_NAMES : []), - ].filter((name): name is string => Boolean(name)); - if (!toolLoopRan && (customSkillExecutionEnabled || builtinToolNames.length > 0)) { - const skillSessionId = pipelineSessionId; - - translatedResponse = await handleToolCallExecution( - translatedResponse, - getSkillsModelIdForFormat(sourceFormat), - { - apiKeyId: memoryOwnerId || "local", - sessionId: skillSessionId, - requestId: skillRequestId, - builtinToolNames, - customSkillExecutionEnabled, - provider, - model: effectiveModel, - } - ); - } - - const guardrailContext = buildPostCallGuardrailContext({ - apiKeyInfo, - body, - clientRawRequest, - log, - model, - provider, - responsePayloadFormat, - clientResponseFormat, - }); - const postCallGuardrails = await guardrailRegistry.runPostCallHooks( - translatedResponse, - guardrailContext - ); - translatedResponse = postCallGuardrails.response; - - const responseUsage = isJsonRecord(usage) - ? usage - : isJsonRecord(translatedResponse.usage) - ? translatedResponse.usage - : null; - const costUsage = normalizeUsage(responseUsage); - const estimatedCost = costUsage - ? await calculateCost(provider, model, costUsage, { serviceTier: effectiveServiceTier }) - : 0; - const chatCostCtx = buildCostCtx(provider, model, usage, effectiveServiceTier, traceId); - - if (postCallGuardrails.blocked) { - const guardrailMessage = postCallGuardrails.message || "Response blocked by guardrail"; - persistAttemptLogs({ - status: HTTP_STATUS.BAD_REQUEST, - tokens: usage, - responseBody, - providerRequest: finalBody || translatedBody, - providerResponse: looksLikeSSE - ? { - _streamed: true, - _format: "sse-json", - summary: responseBody, - } - : responseBody, - clientResponse: buildErrorBody(HTTP_STATUS.BAD_REQUEST, guardrailMessage), - claudeCacheMeta: claudePromptCacheLogMeta, - claudeCacheUsageMeta: cacheUsageLogMeta, - cacheSource: "upstream", - }); - recordChatCallCost(apiKeyInfo, meteredBudgetCost(provider, estimatedCost), chatCostCtx, false); - log?.warn?.( - "GUARDRAIL", - `Response blocked by ${postCallGuardrails.guardrail || "guardrail"}: ${guardrailMessage}` - ); - finalizePendingScope(pendingScope, { - providerResponse: responseBody, - clientResponse: translatedResponse, - }); - return createErrorResult(HTTP_STATUS.BAD_REQUEST, guardrailMessage); - } - - // Validate the *translated* response actually carries client-usable output. - // isEmptyContentResponse (above) runs on the raw responseBody before translation; - // this check runs after translation + sanitization + tool-call execution to catch - // cases where a provider returns a structurally valid raw body that translates into - // choices:[] or output:[] with no usable content (Responses API shape included). - const malformedTranslatedReason = detectMalformedNonStream(translatedResponse, provider); - if (malformedTranslatedReason) { - const totalLatency = Date.now() - startTime; - const rawBytes = (() => { - try { - return JSON.stringify(responseBody || {}).length; - } catch { - return -1; - } - })(); - reportMalformed200({ - mode: "nonstream", - provider, - model, - connectionId, - reason: malformedTranslatedReason, - recvBytes: rawBytes, - recvLines: -1, - emitted: -1, - events: {}, - ttftMs: totalLatency, - elapsedMs: totalLatency, - }); - appendRequestLog({ - model, - provider, - connectionId, - status: `FAILED ${HTTP_STATUS.BAD_GATEWAY}`, - }).catch(() => {}); - const malformed = describeMalformedNonStream(translatedResponse, malformedTranslatedReason); - const malformedMessage = `[${provider}/${model}] ${malformed.message}`; - const malformedClientBody = buildErrorBody( - HTTP_STATUS.BAD_GATEWAY, - malformedMessage, - undefined, - { code: malformed.code, type: malformed.type } - ); - const sanitizedMalformedResponse = sanitizeUpstreamDetails(responseBody); - const sanitizedMalformedProviderResponse = looksLikeSSE - ? { _streamed: true, _format: "sse-json", summary: sanitizedMalformedResponse } - : sanitizedMalformedResponse; - persistAttemptLogs({ - status: HTTP_STATUS.BAD_GATEWAY, - tokens: usage, - responseBody: sanitizedMalformedResponse, - providerRequest: finalBody || translatedBody, - providerResponse: sanitizedMalformedProviderResponse, - clientResponse: malformedClientBody, - claudeCacheMeta: claudePromptCacheLogMeta, - claudeCacheUsageMeta: cacheUsageLogMeta, - cacheSource: "upstream", - }); - persistFailureUsage(HTTP_STATUS.BAD_GATEWAY, "malformed_translated_response"); - trackPendingRequest(model, provider, pendingConnId, false, undefined, pendingRequestId); - // Routing event (feedback foundation) — record the malformed outcome so - // the quality tracker de-prioritizes this model over time. - void emitRoutingEvent( - createRoutingEvent({ - requestId: traceId || pendingRequestId || "unknown", - provider: provider || "unknown", - model: model || "unknown", - strategy: isCombo ? (comboStrategy ?? "combo") : "direct", - latencyMs: Date.now() - startTime, - ttftMs: null, - inputTokens: null, - outputTokens: null, - cost: null, - retries: 0, - fallbackUsed: false, // combo-level fallback tracked by decisionTrace - outcome: "malformed", - status: HTTP_STATUS.BAD_GATEWAY, - finishReason: routingFinishReason(translatedResponse), - connectionId: credentials?.connectionId ?? null, - }) - ); - return createErrorResult( - HTTP_STATUS.BAD_GATEWAY, - malformedMessage, - null, - malformed.code, - malformed.type - ); - } - - // ── Phase 9.1: Cache store (non-streaming, temp=0) ── - storeSemanticCacheResponse({ - enabled: semanticCacheEnabled, - body: bodyForCacheWrite, - headers: clientRawRequest?.headers, - translatedResponse, - model, - // The dual-layer manager scopes entries per provider (cacheByProvider); - // lookup passes the resolved provider, so the write must too (#14159). - provider, - apiKeyId: apiKeyInfo?.id ?? undefined, - usage, - log, - videoTranscriptSensitive: videoBridgeObserved, - }); - - // ── Phase 9.2: Save for idempotency ── - // Reuse the key resolved by checkIdempotencyCache() above (single derivation per - // request). (#3821-review LEDGER-6) - saveIdempotency(idempotencyKey, translatedResponse, 200); - reqLogger.logConvertedResponse(translatedResponse); - persistAttemptLogs({ - status: 200, - tokens: usage, - responseBody, - providerRequest: finalBody || translatedBody, - providerResponse: looksLikeSSE - ? { - _streamed: true, - _format: "sse-json", - summary: responseBody, - } - : responseBody, - clientResponse: translatedResponse, - claudeCacheMeta: claudePromptCacheLogMeta, - claudeCacheUsageMeta: cacheUsageLogMeta, - cacheSource: "upstream", - }); - recordChatCallCost(apiKeyInfo, meteredBudgetCost(provider, estimatedCost), chatCostCtx, true); - - // === Quota Share POST-hook (B/F7) — fire-and-forget, fail-open === - await scheduleQuotaShareConsumption({ - apiKeyId: apiKeyInfo?.id, - connectionId: credentials?.connectionId, - provider, - model, - usage, - estimatedCost, - log, - }); - // === /Quota Share POST-hook === - - // ── Gamification event (fire-and-forget) ── - await emitRequestGamificationEvent({ apiKeyId: apiKeyInfo?.id, model, provider }); - - finalizePendingScope(pendingScope, { - providerResponse: responseBody, - clientResponse: translatedResponse, - }); - const responseHeaders = buildNonStreamingResponseHeaders({ - provider, - model, - startTime, - responseUsage, - estimatedCost, - requestId: skillRequestId, - compressionResponseMeta, - comboStrategy, - fallbackAttempts, - }); - // #6426: align response body `model` with the `X-OmniRoute-Model` header - // (both must be the resolved backend model). Some upstreams (notably legacy - // /v1/completions text-completion path) return a body `model` field that - // differs from the resolved backend id we advertised in the header, leaving - // strict clients unable to reconcile the two. Rewrite body.model to `model` - // FIRST, then let #1311 echo override it when the opt-in setting is on. - if (typeof model === "string" && model) echoModelInObject(translatedResponse, model); - // #1311: echo the requested alias/combo name in the non-streaming response model. - if (echoModel) echoModelInObject(translatedResponse, echoModel); - - // ── Plugin onResponse hook (fire-and-forget) ── - // #8395: the streaming branch below already calls this; the non-streaming - // (stream:false) branch returned without it, so onResponse never fired for - // non-streaming requests at all. - await runPluginOnResponseHook({ - requestId: traceId, - body, - model, - provider, - apiKeyInfo, - headers: clientRawRequest?.headers, - response: { status: 200, data: translatedResponse }, - }); - - // Routing event (feedback foundation) — fire-and-forget, cheap. - void emitRoutingEvent( - createRoutingEvent({ - requestId: traceId || pendingRequestId || "unknown", - provider: provider || "unknown", - model: model || "unknown", - strategy: isCombo ? (comboStrategy ?? "combo") : "direct", - latencyMs: Date.now() - startTime, - ttftMs: null, - inputTokens: - usage && typeof usage === "object" - ? (() => { - const promptTokens = (usage as Record).prompt_tokens; - return typeof promptTokens === "number" && Number.isFinite(promptTokens) - ? promptTokens - : null; - })() - : null, - outputTokens: - usage && typeof usage === "object" - ? (() => { - const completionTokens = (usage as Record).completion_tokens; - return typeof completionTokens === "number" && Number.isFinite(completionTokens) - ? completionTokens - : null; - })() - : null, - cost: Number.isFinite(estimatedCost) ? estimatedCost : null, - retries: 0, - fallbackUsed: false, // combo-level fallback tracked by decisionTrace - outcome: "success", - status: 200, - finishReason: routingFinishReason(translatedResponse), - connectionId: credentials?.connectionId ?? null, - }) - ); - - return { - success: true, - response: maybeWrapForcedNonStreamingResponsesJson({ - clientRequestedResponsesStream, - body: translatedResponse, - headers: responseHeaders, - }), - }; - } catch (error) { - trackPendingRequest(model, provider, pendingConnId, false, undefined, pendingRequestId); - const errorMetadata = getSafeErrorMetadata(error); - const managedLeaseFenceCode = getManagedLeaseFenceErrorCode(errorMetadata.code); - if (managedLeaseFenceCode) return managedLeaseFenceErrorResult(managedLeaseFenceCode); - // isSemaphoreCapacityError already reads the code through getSafeErrorMetadata, - // so a hostile rejection cannot escape this classification. - if (isSemaphoreCapacityError(error)) { - const semaphoreCode = errorMetadata.code as string; - appendRequestLog({ - model, - provider, - connectionId, - status: `FAILED ${semaphoreCode}`, - }).catch(() => {}); - const failureMessage = sanitizeErrorMessage(errorMetadata.message) || "Semaphore timeout"; - persistAttemptLogs({ - status: HTTP_STATUS.RATE_LIMITED, - error: failureMessage, - providerRequest: finalBody || translatedBody, - clientResponse: buildErrorBody(HTTP_STATUS.RATE_LIMITED, failureMessage), - claudeCacheMeta: claudePromptCacheLogMeta, - cacheSource: "upstream", - }); - persistFailureUsage(HTTP_STATUS.RATE_LIMITED, semaphoreCode); - const result = createErrorResult(HTTP_STATUS.RATE_LIMITED, failureMessage); - return { - ...result, - errorType: "account_semaphore_capacity", - errorCode: semaphoreCode, - }; - } - throw error; - } - } - - // Streaming response - // #3089 — some "reasoning" openai-compatible upstreams ignore a stream:true - // request and return a complete application/json chat-completion body instead - // of an SSE stream. The readiness check below only recognizes SSE `data:` - // frames, so that body produced a spurious STREAM_EARLY_EOF / HTTP 502 even - // though it carried valid content/reasoning_content. Detect a JSON (non-SSE) - // upstream body and synthesize an equivalent OpenAI SSE stream so the - // streaming pipeline (and the client) get a valid stream. - providerResponse = await maybeConvertJsonBodyToSse(providerResponse, { log, provider, model }); - const streamReadinessPolicy = resolveStreamReadinessTimeout({ - baseTimeoutMs: STREAM_READINESS_TIMEOUT_MS, - provider, - model, - body: (finalBody || translatedBody) as Record | null | undefined, - sourceBody: body as Record | null | undefined, - maxTimeoutMs: agentGoalPolicy.detected - ? Math.max(STREAM_READINESS_MAX_TIMEOUT_MS, agentGoalPolicy.readinessMaxTimeoutMs) - : STREAM_READINESS_MAX_TIMEOUT_MS, - cascadeTimeoutMs: getExecutorTimeoutMs( - executor, - provider, - model, - resolveConnectionTimeoutMs(credentials?.providerSpecificData) - ), - }); - if (streamReadinessPolicy.timeoutMs !== streamReadinessPolicy.baseTimeoutMs) { - log?.debug?.( - "STREAM", - `adaptive readiness timeout=${streamReadinessPolicy.timeoutMs}ms base=${streamReadinessPolicy.baseTimeoutMs}ms reason=${streamReadinessPolicy.reasons.join(",")}` - ); - } - - let streamReadiness = await ensureStreamReadiness(providerResponse, { - timeoutMs: streamReadinessPolicy.timeoutMs, - maxTimeoutMs: streamReadinessPolicy.maxTimeoutMs, - provider, - model, - log, - }); - // A stall is an upstream issue, not an account fault — the executor loop - // already ended at headers, so this bounded retry is the only recovery left. - const fallback = await maybeFallbackAfterReadiness({ - streamReadiness, - clientAborted: streamController.signal.aborted, - failedConnectionId: getCurrentConnectionId(), - failedBody: providerResponse, + const streamingTailOutcome = await runStreamingTail({ + agentGoalPolicy, + modelInfo, + forcedConnectionId, + apiKeyInfo, + attachCompressionUsageReceiptAfterAnalytics, + body, + bodyForCacheWrite, + claudePromptCacheLogMeta, + clientRawRequest, + clientResponseFormat, + comboStrategy, + compressionResponseMeta, + connectionId, + contextEditingEnabled, + copilotCompatibleReasoning, + correlationId, + createPiiTransform, + credentials, currentModel, - streamReadinessPolicy, - provider, - model, - log, - reqLogger, - providerUrl, - providerHeaders, - finalBody, - translatedBody, + customToolNames, + echoModel, + effectiveServiceTier, + endpointPath, executeProviderRequest, - providerRequestCapture, - }); - streamReadiness = fallback.readiness; - providerResponse = fallback.providerResponse; - finalBody = fallback.finalBody; - if (streamReadiness.ok === false) { - const { response: failureResponse, reason } = streamReadiness; - const { classificationReason, upstreamDiagnostic } = streamReadiness; - trackPendingRequest(model, provider, pendingConnId, false, undefined, pendingRequestId); - appendRequestLog({ - model, - provider, - connectionId, - status: `FAILED ${failureResponse.status}`, - }).catch(() => {}); - persistAttemptLogs({ - status: failureResponse.status, - error: reason, - providerRequest: finalBody || translatedBody, - clientResponse: buildErrorBody( - failureResponse.status, - classificationReason, - upstreamDiagnostic ? { error: { message: upstreamDiagnostic } } : undefined - ), - claudeCacheMeta: claudePromptCacheLogMeta, - cacheSource: "upstream", - }); - persistFailureUsage(failureResponse.status, streamReadiness.code); - // Do NOT call onStreamFailure — a stream stall is an upstream issue, - // not an account/quota failure. Marking the account unavailable here - // would lock out legitimate accounts when the upstream hangs. - return { - success: false, - status: failureResponse.status, - error: reason, - classificationError: classificationReason, - errorType: streamReadiness.type, - errorCode: streamReadiness.code, - response: failureResponse, - }; - } - providerResponse = streamReadiness.response; - - // Flush-empty retry (opt-in `FLUSH_EMPTY_RETRY_ENABLED`, default off): when the - // upstream turn carries no usable content (reasoning-only 200, or a - // zero-valuable-chunk turn that the empty-stream guard would turn into a 502), - // issue bounded retries through the normal credential path BEFORE anything is - // exposed to the client — in particular before `onRequestSuccess` below. - // Empty turns are stochastic upstream misses, not account faults, so no - // cooldown: the retry prefers another allowed connection, a single slot - // replays itself, and a leased or pinned connection never rotates (#14715). - // Budget: `STREAM_RECOVERY.EMPTY_TURN_RETRY_MAX` retries, then fall - // back to the current behavior. Translate-path streams only (mirror of the - // empty-stream guard); flag off = byte-for-byte unchanged. Bounded reader - // (abandon past the cap, never a full `text()` read); the original - // reconstructed response is piped, only the bounded copy is classified. - // Known TTFT cost when armed: a small valid turn under the cap is fully - // buffered before the first client byte (flag off by default, so the - // streaming path is untouched unless opted in). - if (stream && providerResponse.ok && providerResponse.body) { - let flushEmptyRetryArmed = false; - try { - flushEmptyRetryArmed = isFeatureFlagEnabled("FLUSH_EMPTY_RETRY_ENABLED"); - } catch { - flushEmptyRetryArmed = false; - } - const isTranslatePath = - targetFormat === FORMATS.OPENAI_RESPONSES || - needsTranslation(targetFormat, clientResponseFormat); - if (flushEmptyRetryArmed && isTranslatePath) { - const retried = await runEmptyTurnRetryLoop({ - providerResponse, - credentials, - provider, - currentModel, - model, - targetFormat, - clientResponseFormat, - isAborted: () => clientRawRequest?.signal?.aborted === true, - timeoutMs: streamReadinessPolicy.timeoutMs, - maxTimeoutMs: streamReadinessPolicy.maxTimeoutMs, - maxRetries: STREAM_RECOVERY.EMPTY_TURN_RETRY_MAX, - translatedBody, - finalBody, - providerUrl, - providerHeaders, - correlationId, - traceId, - log, - getProviderCredentials, - routing: { leased: Boolean(managedLease), forcedConnectionId, apiKey: apiKeyInfo }, - executeProviderRequest, - logTargetRequest: (url, headers, body) => reqLogger.logTargetRequest(url, headers, body), - captureBody: (body) => providerRequestCapture.body(body), - }); - providerResponse = retried.providerResponse; - if (retried.adopted) finalBody = retried.finalBody; - } - } - - // Notify success - caller can clear error status if needed - if (onRequestSuccess) { - await onRequestSuccess(); - } - - const responseHeaders = assembleStreamingResponseHeaders({ - providerHeaders: providerResponse.headers, - provider, + executor, + fallbackAttempts, + finalBody, + getCurrentConnectionId, + isCombo, + isDroidCLI, + isResponsesEndpoint, + log, + managedLease, + memoryOwnerId, + memorySettings, model, + onRequestSuccess, + onStreamFailure, + pendingConnId, pendingRequestId, - compressionResponseMeta, - comboStrategy, - fallbackAttempts, - isCombo, // #14116: foreign-account quota-header strip (only meaningful when true) - requestedConnectionId: forcedConnectionId || null, - selectedConnectionId: credentials?.connectionId ?? null, - }); - - // The streaming headers (turn-state included, when present) are committed to - // the client from here on — record which connection minted the blob so a - // later cross-account echo can be stripped (Codex failover guard). The - // in-place failover update means `credentials` is the winning account. - if (provider === "codex" && readCodexTurnStateHeader(providerResponse.headers)) { - noteCodexTurnStateProvenance( - getCodexClientSessionId(clientRawRequest?.headers), - credentials?.connectionId - ); - } - - // Create transform stream with logger for streaming response - let transformStream; - const responseToolNameMap = mergeResponseToolNameMap( - toolNameMap, - (finalBody as Record | null | undefined) ?? null - ); - - let streamCompletionRecorded = false; - let streamFailureCompletionRecorded = false; - - // Callback to save call log when stream completes (include responseBody when provided by stream) - let streamTimingOriginOffsetMs: number | null = null; // startTime → StreamTiming start - const onStreamComplete = ({ - status: streamStatus, - usage: streamUsage, - responseBody: streamResponseBody, - providerPayload, - clientPayload, - reasoningMeta: streamReasoningMeta, - error: streamError, - errorCode: streamErrorCode, - firstOutputMs, - itlMs: streamItlMs, - interrupted: _streamInterrupted, - }) => { - const ttft = requestTtftMs(streamTimingOriginOffsetMs, firstOutputMs); - const normalizedStreamStatus = streamStatus || 200; - if (streamCompletionRecorded) return; - streamCompletionRecorded = true; - if (normalizedStreamStatus !== 200) { - if (streamFailureCompletionRecorded) return; - streamFailureCompletionRecorded = true; - } - const cacheUsageLogMeta = buildCacheUsageLogMeta(streamUsage); - const streamConnectionId = getCurrentConnectionId(); - - if (normalizedStreamStatus === 200) { - clearPostOutputFailureStreak(provider, streamConnectionId, modelInfo.model); - void maybeSyncClaudeExtraUsageState({ - provider, - connectionId: streamConnectionId, - providerSpecificData: credentials?.providerSpecificData, - log, - }); - } - - if (normalizedStreamStatus === 200 && streamResponseBody) { - captureStreamReasoningForReplay({ - streamResponseBody, - clientResponseFormat, - responseToolNameMap, - providerRequestBody: finalBody || translatedBody || body, - translatedBody, - reasoningReplayHistory, - provider, - model, - reasoningCacheScope, - videoTranscriptSensitive: videoBridgeObserved, - }); - } - effectiveServiceTier = resolveReportedServiceTier(streamResponseBody) ?? effectiveServiceTier; - - // Context Editing telemetry (streaming): the reconstructed stream body now carries - // context_management.applied_edits from the final message_delta snapshot. Mirror the - // non-streaming hook so streaming context-clear savings also surface under engine - // "context-editing" in compression analytics. Best-effort, Claude-only. - if (normalizedStreamStatus === 200) { - recordContextEditingTelemetryHook({ - contextEditingEnabled, - provider, - responseBody: streamResponseBody, - skillRequestId, - log, - }); - } - - streamFailure.finalizeStreamRequestLog({ - pendingRequestId, - model, - provider, - connectionId: streamConnectionId, - providerResponse: providerPayload ?? streamResponseBody ?? undefined, - clientResponse: clientPayload ?? streamResponseBody ?? undefined, - status: normalizedStreamStatus, - error: streamError, - errorCode: streamErrorCode, - }); - - // Track cache token metrics for streaming responses - if (streamUsage && typeof streamUsage === "object") { - attachCompressionUsageReceiptAfterAnalytics(streamUsage as Record, "stream"); - // Track Gemini token consumption for TPM rate-limit pre-check - if (provider === "gemini") { - const promptTokens = - typeof (streamUsage as Record).prompt_tokens === "number" - ? ((streamUsage as Record).prompt_tokens as number) - : 0; - if (promptTokens > 0) incrementTokenUsage(model, promptTokens); - } - } - recordStreamingUsageStats(streamUsage, { - provider, - model, - streamStatus: normalizedStreamStatus, - startTime, - ttft, - streamErrorCode, - connectionId: streamConnectionId, - apiKeyInfo, - effectiveServiceTier, - isCombo, - comboStrategy, - endpoint: endpointPath, cpaAuthIndex: readCpaAuthIndex(providerResponse), - }); - - // Routing event (feedback foundation) — fire-and-forget, cheap, never blocks - // the stream. Feeds the quality tracker + optional OTel exporter. - void emitRoutingEvent( - createRoutingEvent({ - requestId: traceId || pendingRequestId || "unknown", - provider: provider || "unknown", - model: model || "unknown", - strategy: isCombo ? (comboStrategy ?? "combo") : "direct", - latencyMs: Date.now() - startTime, - ttftMs: typeof ttft === "number" && Number.isFinite(ttft) && ttft >= 0 ? ttft : null, - itlMs: - typeof streamItlMs === "number" && Number.isFinite(streamItlMs) && streamItlMs >= 0 - ? streamItlMs - : null, - inputTokens: - streamUsage && typeof streamUsage === "object" - ? (() => { - const promptTokens = (streamUsage as Record).prompt_tokens; - return typeof promptTokens === "number" && Number.isFinite(promptTokens) - ? promptTokens - : null; - })() - : null, - outputTokens: - streamUsage && typeof streamUsage === "object" - ? (() => { - const completionTokens = (streamUsage as Record).completion_tokens; - return typeof completionTokens === "number" && Number.isFinite(completionTokens) - ? completionTokens - : null; - })() - : null, - cost: null, - retries: 0, - fallbackUsed: false, // combo-level fallback tracked by decisionTrace - outcome: - normalizedStreamStatus === 200 - ? "success" - : streamErrorCode === "stream_interrupted" || streamErrorCode === "aborted" - ? "stream_interrupted" - : outcomeFromStatus(normalizedStreamStatus), - status: normalizedStreamStatus, - finishReason: routingFinishReason(streamResponseBody), - connectionId: streamConnectionId ?? credentials?.connectionId ?? null, - }) - ); - - persistAttemptLogs({ - status: normalizedStreamStatus, - error: streamError || undefined, - tokens: streamUsage || {}, - responseBody: streamResponseBody ?? undefined, - providerRequest: finalBody || translatedBody, - providerResponse: providerPayload, - clientResponse: clientPayload ?? streamResponseBody ?? undefined, - claudeCacheMeta: claudePromptCacheLogMeta, - claudeCacheUsageMeta: cacheUsageLogMeta, - cacheSource: "upstream", - // #13130: persist TTFT so call_logs.ttft_ms lets the dashboard compute - // generation-time TPS instead of wall-clock TPS. - ttft, - reasoningMeta: streamReasoningMeta ?? null, - }); - - recordStreamingCost({ - apiKeyId: apiKeyInfo?.id, - provider, - model, - streamUsage, - serviceTier: effectiveServiceTier, - calculateCost, - // Only the budget-consumable share may draw down the allowance. - recordCost: (apiKeyId, cost, details) => { - const budgetCost = meteredBudgetCost(provider, cost); - if (budgetCost > 0) recordCost(apiKeyId, budgetCost, details); - }, - ledger: buildStreamLedgerDetails(effectiveServiceTier, normalizedStreamStatus < 400, traceId), - }); - - // === Quota Share POST-hook streaming (B/F7) — fire-and-forget, fail-open === - // Resolve the real per-request cost (calculateCost) so USD-unit pools accrue - // on streaming traffic too; this previously recorded usd:0 hardcoded, which - // meant DeepSeek-style `usd/monthly` shared pools never blocked on streams. - scheduleStreamingQuotaShareConsumption({ - apiKeyId: apiKeyInfo?.id, - connectionId: credentials?.connectionId, - provider, - model, - streamUsage, - streamStatus: normalizedStreamStatus, - serviceTier: effectiveServiceTier, - calculateCost, - log, - }); - // === /Quota Share POST-hook streaming === - - if (streamStatus === 200) { - // #12150 P1b surface 3 (fix round 1): see the matching non-streaming - // gate above — an observed request populates NO durable memory from - // either the request-derived text or this streamed response. - runMemoryExtractionGate({ - memoryOwnerId, - memorySettings, - videoBridgeObserved, - pipelineSessionId, - requestBody: body as Record, - responseBody: (streamResponseBody ?? null) as Record | null, - extractFacts, - log, - }); - } - - // Semantic cache: store assembled streaming response for future cache hits - storeStreamingSemanticCacheResponse({ - enabled: semanticCacheEnabled, - streamStatus, - streamResponseBody, - body: bodyForCacheWrite, - headers: clientRawRequest?.headers, - model, - provider, - apiKeyId: apiKeyInfo?.id ?? undefined, - streamUsage, - log, - videoTranscriptSensitive: videoBridgeObserved, - }); - - // Plugin onStreamComplete hook — fire-and-forget, fail-open (#9571) - // Pass traceId as requestId so plugins can correlate the stream-completion event - // with the originating request (the same id used for onRequest/onResponse). (#11825) - runPluginOnStreamCompleteHook({ - status: normalizedStreamStatus, - usage: streamUsage as Record | undefined, - ttft, - model, - provider, - errorCode: streamErrorCode, - startTime, - requestId: traceId, - }); - }; - - const streamFailureFinalizers = streamFailure.createStreamFailureFinalizers({ - isFailureCompletionRecorded: () => streamFailureCompletionRecorded, - isStreamCompletionRecorded: () => streamCompletionRecorded, - onStreamComplete, + persistAttemptLogs, persistFailureUsage, - onStreamFailure, - hasEmittedOutput: () => streamEmittedOutput(transformStream), - }); - const handleStreamFailure = streamFailureFinalizers.handleStreamFailure; - onPipelineStreamError = streamFailureFinalizers.onPipelineStreamError; - // #9653: gives a genuine, race-delayed completion a chance to land (see - // createClientDisconnectGraceHandler's doc comment) before persisting a false - // 499/0-tokens for a request that actually delivered its full response. - onClientDisconnectFinalize = streamFailure.createClientDisconnectGraceHandler({ - isStreamCompletionRecorded: () => streamCompletionRecorded, - gracePeriodMs: STREAM_DISCONNECT_GRACE_PERIOD_MS, - finalize: (event) => - handleStreamFailure({ - status: 499, - message: `Client disconnected: ${event.reason}`, - code: "client_disconnected", - type: "client_disconnected", - }), - }); - - // For providers using Responses API format, translate stream back to openai (Chat Completions) format - // UNLESS client is Droid CLI which expects openai-responses format back - const needsResponsesTranslation = - targetFormat === FORMATS.OPENAI_RESPONSES && - clientResponseFormat === FORMATS.OPENAI && - !isResponsesEndpoint && - !isDroidCLI; - const streamStateBody = finalBody || body; - - // Client's explicit thinking intent (Anthropic Messages shape). Claude Code - // sends `{type:"enabled"}` or `{type:"adaptive"}` to opt into relaying - // upstream reasoning_content as Claude thinking blocks; `{type:"disabled"}` - // or an omitted `thinking` field opts out. Kept false for every other - // client schema (OpenAI / Responses), which never express intent through - // `body.thinking`. Mirrors hasActiveClaudeThinking() so the request and - // response sides agree on what counts as "thinking requested" — a prior - // inline `=== "enabled"` check silently suppressed `adaptive` (the intent - // Claude Code actually sends), leaking the mismatch as a broken tool-call - // turn (call log 1787566395384-bab9ab: reasoning dropped → model emitted - // DSML tool-call markers as plain text → incomplete `stop` finish). - const requestedThinking = hasActiveClaudeThinking((body ?? {}) as Record); - - streamTimingOriginOffsetMs = Date.now() - startTime; - if (needsResponsesTranslation) { - // Provider returns openai-responses, translate to openai (Chat Completions) that clients expect - log?.debug?.("STREAM", `Responses translation mode: openai-responses → openai`); - transformStream = createSSETransformStreamWithLogger( - "openai-responses", - "openai", - provider, - reqLogger, - responseToolNameMap, - model, - connectionId, - streamStateBody, - onStreamComplete, - apiKeyInfo, - handleStreamFailure, - copilotCompatibleReasoning, - false, - requestedThinking, - customToolNames, - // openai-responses → openai translation still wants the namespace identity - // map for #7936-style round-trip closure when the client also speaks - // Responses (Codex CLI). - requestToolIdentityMap - ); - } else if (needsTranslation(targetFormat, clientResponseFormat)) { - // Standard translation for other providers - log?.debug?.("STREAM", `Translation mode: ${targetFormat} → ${clientResponseFormat}`); - transformStream = createSSETransformStreamWithLogger( - targetFormat, - clientResponseFormat, - provider, - reqLogger, - responseToolNameMap, - model, - connectionId, - streamStateBody, - onStreamComplete, - apiKeyInfo, - handleStreamFailure, - copilotCompatibleReasoning, - // Suppress the `` close marker for clients that render it verbatim - // (e.g. OpenCode by UA; any client via `x-omniroute-thinking-marker: off`); - // preserved for Claude Code / Cursor and unknown clients by default (#5245 / - // #5312). Responses API clients always suppress it (structured reasoning - // items make the marker meaningless); otherwise the header wins over the - // UA allowlist. - resolveSuppressThinkClose({ - userAgent: streamUserAgent, - thinkingMarkerHeader, - clientResponseFormat, - }), - requestedThinking, - customToolNames, - requestToolIdentityMap - ); - } else { - log?.debug?.("STREAM", `Standard passthrough mode`); - transformStream = createPassthroughStreamWithLogger( - provider, - reqLogger, - responseToolNameMap, - model, - connectionId, - streamStateBody, - onStreamComplete, - apiKeyInfo, - handleStreamFailure, - clientResponseFormat, - requestToolIdentityMap - ); - } - - const finalStream = assembleStreamingPipeline({ - providerResponse, - transformStream, - streamController, - createPiiTransform, - clientRawRequestHeaders: clientRawRequest?.headers, - clientResponseFormat, - echoModel, - responseHeaders, - // Same adaptive budget the pre-handoff readiness gate above just used — - // reasoning models that legitimately take a while to say anything keep - // that same patience for their first REAL content, not just their first - // lifecycle frame. See pipeWithDisconnect's own doc comment. - contentStallTimeoutMs: streamReadinessPolicy.timeoutMs, - }); - const clientFacingStream = wrapReadableStreamWithFinalize( - finalStream, - releaseTurnExecution - ); - - // ── Gamification event (fire-and-forget) ── - await emitRequestGamificationEvent({ apiKeyId: apiKeyInfo?.id, model, provider }); - - // ── Plugin onResponse hook (fire-and-forget) ── - await runPluginOnResponseHook({ - requestId: traceId, - body, - model, + pipelineSessionId, provider, - apiKeyInfo, - headers: clientRawRequest?.headers, - response: { status: 200, streamed: true }, + providerHeaders, + providerRequestCapture, + providerResponse, + providerUrl, + reasoningCacheScope, + reasoningReplayHistory, + releaseTurnExecution, + reqLogger, + requestToolIdentityMap, + resolveReportedServiceTier, + semanticCacheEnabled, + skillRequestId, + startTime, + stream, + streamController, + streamUserAgent, + targetFormat, + thinkingMarkerHeader, + toolNameMap, + traceId, + translatedBody, + videoBridgeObserved, }); - - const response = new Response(clientFacingStream, { - headers: responseHeaders, - }); - turnExecutionHandedOffToStream = true; - return { - success: true, - response, - }; + onPipelineStreamError = streamingTailOutcome.carry.onPipelineStreamError; + onClientDisconnectFinalize = streamingTailOutcome.carry.onClientDisconnectFinalize; + turnExecutionHandedOffToStream = streamingTailOutcome.carry.turnExecutionHandedOffToStream; + return streamingTailOutcome.result; } finally { if (!turnExecutionHandedOffToStream) { releaseTurnExecution(); diff --git a/open-sse/handlers/chatCore/executeProviderRequest.ts b/open-sse/handlers/chatCore/executeProviderRequest.ts new file mode 100644 index 000000000000..fa7c536cde99 --- /dev/null +++ b/open-sse/handlers/chatCore/executeProviderRequest.ts @@ -0,0 +1,688 @@ +/** + * One provider send. Lifted out of handleChatCore. The body is the tip + * implementation; the only change is that closed-over request state arrives + * through deps. + */ + +import type { BaseExecutor } from "../../executors/base.ts"; +import type { AgentGoalPolicy } from "../../utils/agentGoalPolicy.ts"; +import { prepareUpstreamBody } from "./upstreamBody.ts"; +import { getExecutionConnectionId } from "./executionCredentials.ts"; +import { + resolveAccountSemaphoreKey, + resolveAccountSemaphoreMaxConcurrency, +} from "./executorHelpers.ts"; +import { + materializeDeduplicatedExecutionResult, + stripNextMiddlewareControlHeaders, + stripStaleForwardingHeaders, +} from "./responseHeaders.ts"; +import { readNonStreamingResponseBody } from "./nonStreamingResponseBody.ts"; +import { wrapReadableStreamWithFinalize } from "./streamFinalize.ts"; +import { + normalizeExecutorResult, + executeWithUpstreamStartTimeout, + resolveConnectionTimeoutMs, +} from "./upstreamTimeouts.ts"; +import { getCodexClientSessionId } from "../../config/codexIdentity.ts"; +import { + noteCodexTurnStateProvenance, + readCodexTurnStateHeader, +} from "../../config/codexTurnState.ts"; +import { HTTP_STATUS, STREAM_RECOVERY } from "../../config/constants.ts"; +import { createRecoverableStream, makeContinuationBody } from "../../services/streamRecovery.ts"; +import { persistCodexChildQuotaResponse } from "../../services/codexAccount/index.ts"; +import { invalidateCodexQuotaCache } from "../../services/codexQuotaFetcher.ts"; +import { invalidateGenericQuotaCacheOnStatus } from "../../services/genericQuotaFetcher.ts"; +import { withRateLimit, resolveRequestQueueMaxWaitMs } from "../../services/rateLimitManager.ts"; +import { acquireMany as acquireConcurrencyGates } from "../../services/accountSemaphore.ts"; +import { rethrowAdmissionError, remainingQueueBudgetMs } from "./queueBudget.ts"; +import { deduplicate } from "../../services/requestDedup.ts"; +import { + classifyModelScope429, + getModelScopeRetryDelayMs, +} from "../../services/modelscopePolicy.ts"; +import { incrementRequestCount } from "../../services/geminiRateLimitTracker.ts"; +import { normalizeHeaders } from "../../utils/headers.ts"; +import { runWithCapture } from "../../utils/providerRequestLogging.ts"; +import type { Capture } from "../../utils/providerRequestLogging.ts"; +import { + isStreamRecoveryExplicitlyConfigured, + resolveResilienceSettings, + type ResilienceSettings, +} from "@/lib/resilience/settings"; +import { updatePendingScope, type PendingRequestScope } from "@/lib/usage/pendingRequestScope"; +import { shouldIsolateProbeFailures } from "@/shared/utils/probeOrigin"; +import { resolveProviderId } from "@/shared/constants/providers"; +import { buildContinuationLogHooks } from "./recoveryTraceLogging.ts"; +import { formatStreamRecoveryRetryWarning } from "./streamErrorResult.ts"; +import { injectSystemPromptPostTranslation } from "../../services/systemPrompt.ts"; + +export type ChatCoreExecutorResult = { + response: Response; + url?: string; + transport?: string; + headers?: Record; + body?: unknown; + _accountSemaphoreRelease?: () => void; + _executionCredentials?: Record; + _dedupSnapshot?: { + status: number; + statusText: string; + headers: [string, string][]; + payload: string; + }; + [key: string]: unknown; +}; + +export type TrustedEffortContext = { + originModel?: string; + resolvedThinkingEffort?: string | null; + defaultThinkingEffort?: string | null; +}; + +type ChatLog = + | { + debug?: (...args: unknown[]) => void; + info?: (...args: unknown[]) => void; + warn?: (...args: unknown[]) => void; + error?: (...args: unknown[]) => void; + } + | null + | undefined; + +export type ExecuteProviderRequestDeps = { + agentGoalPolicy: AgentGoalPolicy; + assertManagedLeaseFence: (attemptConnectionId: string | null | undefined) => void; + buildUpstreamHeadersForExecute: (modelToCall: string) => Record; + clientRawRequest: { + headers?: Headers | Record | null; + signal?: AbortSignal | null; + }; + clientResponseFormat: string | null | undefined; + connectionId: string | null | undefined; + contextEditingEnabled?: boolean; + correlationId: string | null | undefined; + credentials: Record | null | undefined; + dedupEnabled: boolean; + dedupHash: string | null; + effectiveModel: string; + executor: Pick; + extendedContext?: boolean; + getExecutionCredentials: () => Record; + getExecutorClientHeaders: () => Record; + isModelScope: () => boolean; + isOpencodeClient: boolean; + log: ChatLog; + model: string; + onCredentialsRefreshed: + ((next: Record) => void | Promise) | null | undefined; + pendingScope: PendingRequestScope; + provider: string; + providerRequestCapture: Capture; + rawBody: Record; + recordKeyHealthStatus: ( + status: number, + creds: Record | null | undefined, + transport?: string, + failureDetail?: string + ) => void; + requestedModel: string; + resilienceSettings: ResilienceSettings; + settings: Record | null | undefined; + skipUpstreamRetry: boolean; + stream: boolean; + streamController: { signal: AbortSignal }; + targetFormat: string; + trace: (label: string, extra?: Record) => void; + traceId: string; + translatedBody: Record; + trustedEffortContext: TrustedEffortContext; + upstreamStream: boolean; + userAgent: string | undefined; +}; + +export async function executeProviderRequest( + deps: ExecuteProviderRequestDeps, + modelToCall: string = deps.effectiveModel, + allowDedup: boolean = false +): Promise { + const { + agentGoalPolicy, + assertManagedLeaseFence, + buildUpstreamHeadersForExecute, + clientRawRequest, + clientResponseFormat, + connectionId, + contextEditingEnabled, + correlationId, + credentials, + dedupEnabled, + dedupHash, + executor, + extendedContext, + getExecutionCredentials, + getExecutorClientHeaders, + isModelScope, + isOpencodeClient, + log, + model, + onCredentialsRefreshed, + pendingScope, + provider, + providerRequestCapture, + rawBody, + recordKeyHealthStatus, + requestedModel, + resilienceSettings, + settings, + skipUpstreamRetry, + stream, + streamController, + targetFormat, + trace, + traceId, + translatedBody, + trustedEffortContext, + upstreamStream, + userAgent: _userAgent, + } = deps; + const body = rawBody; + const execute = async () => { + // Upstream body preparation extracted to chatCore/upstreamBody.ts (#3501 — first internal + // sub-slice of executeProviderRequest); produces the body sent upstream (payload rules + + // tool-limit truncation + prompt_cache_key injection). + let bodyToSend = await prepareUpstreamBody({ + translatedBody, + modelToCall, + ...trustedEffortContext, + provider, + targetFormat, + credentials: getExecutionCredentials(), + log, + bypassDefaultToolLimit: isOpencodeClient, + isOpencodeClient, + rawBody: body, + clientRawRequest, + }); + + // Global System Prompt — SINGLE injection point (post-translation) for + // carrier-ful targets. The old unconditional pre-translation pass + // (former chatCore injectSystemPrompt call) was removed: it chained + // with this pass to inject prefix/suffix 2-3x and dual-wrote + // body.system + messages[] on the claude path, which strict upstreams + // (HCP-Vision vLLM: "System message must be at the beginning") reject + // with 400. Format-aware via targetFormat: messages[] (openai/codex — + // prefix FIRST system, suffix LAST), claude `system` field, gemini + // `systemInstruction`, responses `instructions`. Carrier-less targets + // (kiro user-fold, antigravity Cloud Code envelope) are covered by the + // gated PRE-translation pass before translateRequest instead. + bodyToSend = injectSystemPromptPostTranslation(bodyToSend, { targetFormat }); + + updatePendingScope(pendingScope, { + providerRequest: bodyToSend, + stage: "payload_prepared", + }); + + let releaseRawResultAccountSemaphore = () => {}; + try { + const rawResult: ChatCoreExecutorResult = await (async () => { + let attempts = 0; + const isModelScopeForRequest = isModelScope(); + const maxAttempts = isModelScopeForRequest ? 3 : provider === "codex" ? 3 : 1; + + while (attempts < maxAttempts) { + trace("pre_executor", { attempt: attempts }); + updatePendingScope(pendingScope, { + stage: "sending_to_provider", + }); + const execCreds = getExecutionCredentials(); + const executionConnectionId = getExecutionConnectionId(execCreds); + const attemptConnectionId = executionConnectionId || connectionId; + const accountSemaphoreMaxConcurrency = resolveAccountSemaphoreMaxConcurrency(execCreds); + const accountSemaphoreKey = resolveAccountSemaphoreKey({ + provider, + model: modelToCall, + connectionId: attemptConnectionId, + credentials: execCreds, + }); + const canonicalProviderKey = resolveProviderId(String(provider).trim().toLowerCase()); + const providerConcurrency = + resilienceSettings.providerQuotaOverrides[canonicalProviderKey]?.providerConcurrency ?? + 0; + + trace("pre_semaphore", { + semaphoreKey: accountSemaphoreKey, + max: accountSemaphoreMaxConcurrency, + }); + if (accountSemaphoreKey && accountSemaphoreMaxConcurrency != null) { + updatePendingScope(pendingScope, { + stage: "waiting_account_slot", + }); + } + const maxWaitMs = resolveRequestQueueMaxWaitMs( + provider, + undefined, + attemptConnectionId ?? undefined + ); + const gateStartedAt = Date.now(); + const releaseAccountSemaphore = await acquireConcurrencyGates( + [ + { + key: "global", + maxConcurrency: resilienceSettings.requestQueue.globalConcurrentRequests, + }, + { + key: `provider:${canonicalProviderKey}`, + maxConcurrency: providerConcurrency, + }, + { + key: accountSemaphoreKey || "", + maxConcurrency: accountSemaphoreKey ? accountSemaphoreMaxConcurrency : null, + }, + ], + { + timeoutMs: maxWaitMs, + maxQueueSize: resilienceSettings.requestQueue.maxQueueDepth, + signal: streamController.signal, + } + ).catch(rethrowAdmissionError); + const remainingAfterGate = remainingQueueBudgetMs(maxWaitMs, gateStartedAt); + trace("post_semaphore", { maxWaitMs, remainingAfterGate }); + updatePendingScope(pendingScope, { + stage: "waiting_rate_limit", + }); + + try { + trace("pre_rate_limit", { connectionId: attemptConnectionId }); + const rawExecutorResult = await withRateLimit( + provider, + attemptConnectionId, + modelToCall, + async () => { + trace("inside_rate_limit", { connectionId: attemptConnectionId }); + updatePendingScope(pendingScope, { + stage: "rate_limit_slot_acquired", + }); + assertManagedLeaseFence(attemptConnectionId); + return executeWithUpstreamStartTimeout({ + executor, + provider, + model: modelToCall, + connectionTimeoutMs: resolveConnectionTimeoutMs(execCreds?.providerSpecificData), + signal: streamController.signal, + log, + execute: (signal) => + runWithCapture(providerRequestCapture, () => + executor.execute({ + model: modelToCall, + body: bodyToSend, + stream: upstreamStream, + credentials: execCreds, + signal, + log, + extendedContext, + upstreamExtraHeaders: buildUpstreamHeadersForExecute(modelToCall), + clientHeaders: getExecutorClientHeaders(), + clientResponseFormat, + onCredentialsRefreshed, + skipUpstreamRetry, + contextEditing: { enabled: contextEditingEnabled }, + correlationId, + }) + ), + }); + }, + streamController.signal, + remainingAfterGate, + correlationId ?? undefined, + { + executor: executor as unknown as { getTimeoutMs?: () => unknown }, + providerSpecificData: execCreds?.providerSpecificData, + } + ); + const res = normalizeExecutorResult(rawExecutorResult); + trace("post_executor", { status: res?.response?.status }); + + // When a payload override rewrote body.model (custom-model alias → + // real upstream id, e.g. `gemini-3.7-flash-high` → `gemini-3.7-flash`), + // log and track the WIRE model so dashboards/telemetry reflect what + // actually shipped and Gemini rate-limit accounting uses the real id + // (the executor already built its URL from the same rewritten model). + const wireModel = typeof res.model === "string" && res.model ? res.model : modelToCall; + if (wireModel !== modelToCall) { + log?.debug?.( + "PAYLOAD_RULES", + `Payload rules rewrote model for URL: requested=${modelToCall} wire=${wireModel}` + ); + } + + if ( + provider === "codex" && + attemptConnectionId && + !(await shouldIsolateProbeFailures()) + ) { + try { + const persistedQuota = await persistCodexChildQuotaResponse({ + connectionId: String(attemptConnectionId), + model: modelToCall || model || requestedModel || "", + headers: normalizeHeaders(res.response.headers), + status: res.response.status, + }); + if (persistedQuota) { + execCreds.providerSpecificData = persistedQuota.providerSpecificData; + if (persistedQuota.exhaustionLog) { + log?.debug?.("CODEX", persistedQuota.exhaustionLog); + } + } + if (res.response.status === 429) { + invalidateCodexQuotaCache(String(attemptConnectionId)); + } + } catch (err) { + const errMessage = err instanceof Error ? err.message : String(err); + log?.debug?.("CODEX", `Failed to persist codex quota state: ${errMessage}`); + } + } else if (attemptConnectionId && res.response.status === 429) { + // Dropped generic quota cache after 429 + invalidateGenericQuotaCacheOnStatus({ + provider, + connectionId: String(attemptConnectionId), + status: res.response.status, + isolateProbe: await shouldIsolateProbeFailures(), + }); + } + + // Track Gemini RPM + RPD request counts for 429 classification + if (provider === "gemini") { + incrementRequestCount(wireModel); + } + + updatePendingScope(pendingScope, { + stage: "provider_response_started", + }); + + if ( + stream && + (res.response.ok || + res.response.status === HTTP_STATUS.UNAUTHORIZED || + res.response.status === HTTP_STATUS.FORBIDDEN) && + executionConnectionId && + !(await shouldIsolateProbeFailures()) + ) { + const failureDetail = res.response.ok + ? "" + : await res.response + .clone() + .text() + .catch(() => ""); + recordKeyHealthStatus(res.response.status, execCreds, res.transport, failureDetail); + } + + if (isModelScope() && res.response.status === 429 && attempts < maxAttempts - 1) { + const bodyPeek = await res.response + .clone() + .text() + .catch(() => ""); + const normalizedHeaders = normalizeHeaders(res.response.headers); + const decision = classifyModelScope429(bodyPeek, normalizedHeaders); + if (decision.retryable) { + const delay = getModelScopeRetryDelayMs(normalizedHeaders, attempts); + log?.warn?.( + "MODELSCOPE_RETRY", + `429 ${decision.kind}; retrying in ${delay}ms (model remaining: ${decision.snapshot.modelRemaining ?? "unknown"})` + ); + releaseAccountSemaphore(); + await new Promise((r) => setTimeout(r, delay)); + attempts++; + continue; + } + } + + // For streaming: release the semaphore when the client drains or cancels the stream. + // Non-2xx streams must drop the slot before returning so the pipeline can rotate + // accounts without holding the failed connection's concurrency gate. Do NOT + // cancel() the body here — the pipeline clones it (BYOP 422 / toOutcome). + if (stream) { + const originalBody = res.response.body; + const okStatus = res.response.status >= 200 && res.response.status < 300; + if (!originalBody || !okStatus) { + releaseAccountSemaphore(); + return { + ...res, + _executionCredentials: execCreds, + }; + } + + // Opt-in transparent stream recovery (free-claude-code port, default OFF). + // Only engages for a successful (2xx) stream — an error body must never be + // held or replayed. Setting is read once here from the cached resolved + // resilience settings; the default path is byte-for-byte unchanged. + let streamRecoveryEnabled = false; + let continueMidStreamEnabled = false; + let throughputWatchdog = + resolveResilienceSettings(null).streamRecovery.throughputWatchdog; + if (okStatus) { + try { + // Reuse the request-consolidated settings read (see line ~2076) — no + // second DB/cache hit. Default OFF when the setting is absent. + const sr = resolveResilienceSettings(settings).streamRecovery; + // Fail-closed: the agent-goal-policy heuristic may only ADD recovery + // when the operator has no explicit configuration. If the operator + // explicitly configured stream recovery (env var or DB/settings + // override), that value always wins — the goal policy must never + // re-enable recovery the operator explicitly turned off. + const operatorExplicit = isStreamRecoveryExplicitlyConfigured(settings); + const goalOverride = !operatorExplicit && agentGoalPolicy.streamRecoveryEnabled; + streamRecoveryEnabled = sr.enabled || goalOverride; + continueMidStreamEnabled = sr.continueMidStream === true; + throughputWatchdog = sr.throughputWatchdog; + if (goalOverride && !sr.enabled) { + log?.info?.( + "AGENT_GOAL", + `agentGoalPolicy override: stream recovery enabled for goal request requestId=${traceId} model=${modelToCall || model || requestedModel || "unknown"}` + ); + } + } catch { + streamRecoveryEnabled = false; + continueMidStreamEnabled = false; + throughputWatchdog = + resolveResilienceSettings(null).streamRecovery.throughputWatchdog; + } + } + + let clientBody: ReadableStream; + if (streamRecoveryEnabled || throughputWatchdog.enabled) { + // Run the SAME upstream (same account/creds) with a given body and return + // its 2xx stream, or null. Used both by the early-retry re-open (same body) + // and the mid-stream continuation (assistant-prefilled body). + const runUpstreamStream = async ( + body: unknown + ): Promise | null> => { + try { + assertManagedLeaseFence(attemptConnectionId); + const retryRaw = await executeWithUpstreamStartTimeout({ + executor, + provider, + model: modelToCall, + connectionTimeoutMs: resolveConnectionTimeoutMs( + execCreds?.providerSpecificData + ), + signal: streamController.signal, + log, + execute: (signal) => + runWithCapture(providerRequestCapture, () => + executor.execute({ + model: modelToCall, + body, + stream: upstreamStream, + credentials: execCreds, + signal, + log, + extendedContext, + upstreamExtraHeaders: buildUpstreamHeadersForExecute(modelToCall), + clientHeaders: getExecutorClientHeaders(), + clientResponseFormat, + onCredentialsRefreshed, + skipUpstreamRetry, + contextEditing: { enabled: contextEditingEnabled }, + correlationId, + }) + ), + }); + const retryRes = normalizeExecutorResult(retryRaw); + const retryOk = + retryRes.response.status >= 200 && retryRes.response.status < 300; + if (retryOk && retryRes.response.body) { + return retryRes.response.body as ReadableStream; + } + await retryRes.response.body?.cancel().catch(() => {}); + return null; + } catch { + return null; + } + }; + + // Mid-stream continuation (Fase 4.4): re-request with the partial text as an + // assistant prefill. Gated by its own setting and only for OpenAI-compatible + // bodies (makeContinuationBody returns null otherwise). + const continueStream = continueMidStreamEnabled + ? (assistantSoFar: string) => { + const continuationBody = makeContinuationBody( + bodyToSend as Record, + assistantSoFar + ); + return continuationBody + ? runUpstreamStream(continuationBody) + : Promise.resolve(null); + } + : undefined; + + clientBody = createRecoverableStream( + originalBody as ReadableStream, + () => runUpstreamStream(bodyToSend), + { + finalize: releaseAccountSemaphore, + onRetry: (attempt, err) => + log?.warn?.( + "STREAM_RECOVERY", + formatStreamRecoveryRetryWarning( + attempt, + STREAM_RECOVERY.EARLY_RETRY_MAX, + err + ) + ), + continueStream, + ...buildContinuationLogHooks(log, correlationId), + throughputWatchdog, + onWatchdogAbort: () => + log?.warn?.( + "STREAM_WATCHDOG", + "active upstream stream stayed below the configured useful-output rate" + ), + } + ); + } else { + clientBody = wrapReadableStreamWithFinalize(originalBody, releaseAccountSemaphore); + } + + return { + ...res, + _executionCredentials: execCreds, + response: new Response(clientBody, { + status: res.response.status, + statusText: res.response.statusText, + headers: new Headers(normalizeHeaders(res.response.headers)), + }), + }; + } + + return { + ...res, + _executionCredentials: execCreds, + _accountSemaphoreRelease: releaseAccountSemaphore, + }; + } catch (error) { + releaseAccountSemaphore(); + throw error; + } + } + })(); + + if (stream) { + return rawResult; + } + + // Non-stream: release semaphore immediately after reading full response body. + const status = rawResult.response.status; + + releaseRawResultAccountSemaphore = + typeof rawResult._accountSemaphoreRelease === "function" + ? rawResult._accountSemaphoreRelease + : () => {}; + + const statusText = rawResult.response.statusText; + const headersObj = normalizeHeaders(rawResult.response.headers); + const responseHeaders = new Headers(headersObj); + stripStaleForwardingHeaders(responseHeaders); + stripNextMiddlewareControlHeaders(responseHeaders); + // The upstream headers (turn-state included) are about to be committed + // to the client — record which connection minted the blob so a later + // cross-account echo can be stripped (Codex failover guard). + if (provider === "codex" && readCodexTurnStateHeader(responseHeaders)) { + noteCodexTurnStateProvenance( + getCodexClientSessionId(clientRawRequest?.headers), + rawResult._executionCredentials?.connectionId ?? credentials?.connectionId + ); + } + const contentType = (responseHeaders.get("content-type") || "").toLowerCase(); + const payload = await readNonStreamingResponseBody( + rawResult.response, + contentType, + upstreamStream + ); + // Use the exact execution credential selected for this request. Model capability + // failures stay in routing telemetry; authoritative success only recovers this key. + if ( + rawResult._executionCredentials?.connectionId && + (rawResult._executionCredentials.apiKey || rawResult._executionCredentials.accessToken) + ) { + recordKeyHealthStatus( + status, + rawResult._executionCredentials, + rawResult.transport, + status >= 400 ? payload : "" + ); + } + releaseRawResultAccountSemaphore(); + releaseRawResultAccountSemaphore = () => {}; + + return { + ...rawResult, + response: new Response(payload, { status, statusText, headers: responseHeaders }), + _dedupSnapshot: { + status, + statusText, + headers: (() => { + const arr: [string, string][] = []; + responseHeaders.forEach((v, k) => arr.push([k, v])); + return arr; + })(), + payload, + }, + }; + } catch (error) { + releaseRawResultAccountSemaphore(); + throw error; + } + }; + + if (allowDedup && dedupEnabled && dedupHash) { + const dedupResult = await deduplicate(dedupHash, execute); + if (dedupResult.wasDeduplicated) { + log?.debug?.("DEDUP", `Joined in-flight request hash=${dedupHash}`); + } + return materializeDeduplicatedExecutionResult(dedupResult.result); + } + + return execute(); +} diff --git a/open-sse/handlers/chatCore/nonStreamingResponse.ts b/open-sse/handlers/chatCore/nonStreamingResponse.ts new file mode 100644 index 000000000000..7484202a45f2 --- /dev/null +++ b/open-sse/handlers/chatCore/nonStreamingResponse.ts @@ -0,0 +1,1107 @@ +/** + * Non-streaming response path, lifted out of handleChatCore. + * The body matches the barrel. Closed-over request state arrives through + * deps, and the values the barrel reads afterwards leave through carry. + */ + +import { meteredBudgetCost } from "@/lib/usage/meteredBudgetPolicy"; +import { readCpaAuthIndex } from "../chatCore/failureUsage.ts"; + +import { createRoutingEvent, emitRoutingEvent } from "../../services/routing/index.ts"; + +import { buildClaudePromptCacheLogMeta } from "../chatCore/executorHelpers.ts"; + +import { + shouldUseNativeCodexPassthrough, + shouldUseNativeXaiResponsesPassthrough, + redactPassthroughThinkingSignatures, + isClaudeCodeSemanticPassthroughRequest, +} from "../chatCore/passthroughHelpers.ts"; + +import { + applyServerOwnedToolLoopIfNeeded, + derivePostInjectionRequestIdentity, + followUpLegInput, +} from "../chatCore/serverOwnedToolLoopWire.ts"; + +import { + buildStreamingResponseHeaders, + stripStaleForwardingHeaders, +} from "../chatCore/responseHeaders.ts"; + +import { maybeSyncClaudeExtraUsageState } from "../chatCore/telemetryHelpers.ts"; + +export { + shouldUseNativeCodexPassthrough, + shouldUseNativeXaiResponsesPassthrough, + redactPassthroughThinkingSignatures, + isClaudeCodeSemanticPassthroughRequest, + buildStreamingResponseHeaders, + stripStaleForwardingHeaders, +}; + +import { HTTP_STATUS } from "../../config/constants.ts"; + +import { lockModel } from "../../services/accountFallback.ts"; + +import { buildPostCallGuardrailContext } from "../chatCore/postCallGuardrailContext.ts"; +import { buildNonStreamingResponseHeaders } from "../chatCore/nonStreamingResponseHeaders.ts"; +import { maybeWrapForcedNonStreamingResponsesJson } from "../chatCore/responsesJsonToSse.ts"; +import { runProviderExecutionPipeline } from "../chatCore/providerExecutionPipeline.ts"; +import { runNonStreamingProviderLeg } from "../chatCore/nonStreamingProviderLeg.ts"; +import type { NonStreamingProviderLegResult } from "@/lib/skills/toolLoopTypes.ts"; +import { finalizeToolLoopError } from "../chatCore/nonStreamingFinalization.ts"; +import { markCodexScopeRateLimited } from "../chatCore/codexFailover.ts"; +import { deleteSessionAccountAffinity } from "@/lib/db/sessionAccountAffinity"; +import { writeTerminalStatus } from "@/shared/utils/terminalStatus"; +import { MEMORY_BUILTIN_TOOL_NAMES } from "@/lib/skills/memoryBuiltins"; + +import { storeSemanticCacheResponse } from "../chatCore/semanticCacheStore.ts"; +import { routingFinishReason } from "../chatCore/routingFinishReason.ts"; +import { getProviderCredentials } from "@/sse/services/auth"; +import { extractFacts } from "@/lib/memory/extraction"; +// The leaf body is unchanged from the barrel, so its closed-over values keep +// the barrel's types. This alias only exists so the deps bag type-checks. +// eslint-disable-next-line @typescript-eslint/no-explicit-any +type Loose = any; +export type NonStreamingDeps = Record & { [k: string]: Loose }; + +export async function runNonStreamingResponse(deps: NonStreamingDeps) { + const { + apiKeyInfo, + appendRequestLog, + applyProviderFailureClassification, + assertManagedLeaseFence, + attachCompressionUsageReceiptAfterAnalytics, + body, + bodyForCacheWrite, + buildCacheUsageLogMeta, + buildCostCtx, + buildErrorBody, + calculateCost, + claudePromptCacheLogMeta: _claudePromptCacheLogMeta, + clientRawRequest, + comboStrategy, + connectionId, + copilotCompatibleReasoning, + createErrorResult, + credentials, + currentModel: _currentModel, + describeMalformedNonStream, + detectMalformedNonStream, + echoModel, + echoModelInObject, + effectiveModel, + effectiveServiceTier: _effectiveServiceTier, + emitRequestGamificationEvent, + endpointPath, + executeProviderRequest, + executeRefreshCredentials, + extractSessionAffinityKey, + extractUsageFromResponse, + finalBody: _finalBody, + finalizePendingScope, + getCurrentConnectionId, + getManagedLeaseFenceErrorCode, + getModelNormalizeToolCallId, + getModelPreserveOpenAIDeveloperRole, + getSafeErrorMetadata, + getSkillsModelIdForFormat, + guardrailRegistry, + handleCredentialsRefreshed, + handleToolCallExecution, + idempotencyKey, + incrementTokenUsage, + injectionResult, + isCombo, + isJsonRecord, + isLocalStreamLifecycleError, + isSemaphoreCapacityError, + isServerOwnedToolLoopEnabled, + log, + logAuditEvent, + managedLease, + managedLeaseFenceErrorResult, + markAccountSemaphoreBlocked, + memoryOwnerId, + memorySettings, + model, + normalizeHeaders, + normalizeUsage, + onRequestSuccess, + pendingConnId, + pendingRequestId, + pendingScope, + persistAttemptLogs, + persistFailureUsage, + pipelineSessionId, + provider, + providerHeaders: _providerHeaders, + providerRequestCapture, + providerResponse: _providerResponse, + reasoningReplayHistory: _reasoningReplayHistory, + recordChatCallCost, + recordContextEditingTelemetryHook, + recordCoreOwnedAntigravityQuotaState, + recordNonStreamingUsageStats, + reportMalformed200, + reqLogger, + resolveReportedServiceTier, + runMemoryExtractionGate, + runPluginOnResponseHook, + sanitizeErrorMessage, + sanitizeUpstreamDetails, + saveIdempotency, + scheduleQuotaShareConsumption, + semanticCacheEnabled, + sessionAffinityKey, + shouldIsolateProbeFailures, + skillRequestId, + sourceFormat, + startTime, + targetFormat, + traceId, + syncExecuteTranslatedBody, + trackPendingRequest, + translateRequest, + translatedBody: _translatedBody, + triedModels, + updateFromHeaders, + updateFromResponseBody, + updatePendingScope, + updateProviderConnection, + webFetchFallbackPlan, + webSearchFallbackPlan, + clientRequestedResponsesStream, + clientResponseFormat, + compressionResponseMeta, + contextEditingEnabled, + customToolNames, + fallbackAttempts, + isClaudeCodeCompatible, + isResponsesEndpoint, + preserveCacheControl, + reasoningCacheScope, + requestToolIdentityMap, + stream, + toolNameMap, + traceEnabled, + videoBridgeObserved, + } = deps; + + let claudePromptCacheLogMeta, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + reasoningReplayHistory, + translatedBody; + claudePromptCacheLogMeta = _claudePromptCacheLogMeta; + currentModel = _currentModel; + effectiveServiceTier = _effectiveServiceTier; + finalBody = _finalBody; + providerHeaders = _providerHeaders; + providerResponse = _providerResponse; + reasoningReplayHistory = _reasoningReplayHistory; + translatedBody = _translatedBody; + pipelineRecovered = false; + + try { + const runNonStreamingPipeline = async ({ + policy, + model: pipelineModel, + translatedBody: wireBody, + }) => { + translatedBody = wireBody as typeof translatedBody; + syncExecuteTranslatedBody(translatedBody); + currentModel = pipelineModel; + triedModels.add(pipelineModel); + return runProviderExecutionPipeline({ + policy, + target: { + provider, + requestedModel: pipelineModel, + sourceFormat, + targetFormat, + stream: false, + }, + connection: { + initialConnectionId: String(getCurrentConnectionId() || connectionId || ""), + getCurrentConnectionId: () => getCurrentConnectionId() || undefined, + getCredentials: () => (credentials || {}) as Record, + replaceCredentials: (next) => { + Object.assign(credentials, next); + }, + onCredentialsRefreshed: handleCredentialsRefreshed, + refreshCredentials: executeRefreshCredentials, + assertManagedLeaseFence: (id) => { + assertManagedLeaseFence(id); + }, + getProviderCredentials, + }, + wire: { + body: translatedBody as Record, + currentModel, + triedModels, + setBodyAndModel: (nextBody, nextModel) => { + translatedBody = nextBody as typeof translatedBody; + syncExecuteTranslatedBody(translatedBody); + currentModel = nextModel; + triedModels.add(nextModel); + }, + }, + state: { + updatePendingStage: (stage, data) => { + updatePendingScope(pendingScope, { stage, ...(data || {}) }); + }, + recordRateLimitHeaders: updateFromHeaders, + recordRateLimitBody: updateFromResponseBody, + writeTerminalStatus, + persistConnectionPatch: updateProviderConnection, + setConnectionRateLimitedUntil: async (id, untilMs) => { + const { setConnectionRateLimitUntil } = await import("@/lib/db/providers"); + setConnectionRateLimitUntil(id, untilMs); + }, + lockModel, + recordAntigravityQuotaState: recordCoreOwnedAntigravityQuotaState, + markAccountSemaphoreBlocked: (key) => { + markAccountSemaphoreBlocked(key, Date.now() + 60_000); + }, + isolateProbeFailures: () => shouldIsolateProbeFailures(), + onCodexScopeRateLimited: async (params) => { + await markCodexScopeRateLimited({ + failedConnectionId: params.failedConnectionId, + model: params.model, + rateLimitedUntil: params.rateLimitedUntil, + credentials: (params.credentials || credentials) as { + connectionId?: string | null; + providerSpecificData?: unknown; + }, + }); + }, + onClearSessionAffinity: () => { + const key = + sessionAffinityKey || + extractSessionAffinityKey(body, clientRawRequest?.headers) || + null; + if (!key) return; + try { + deleteSessionAccountAffinity(key, "codex"); + } catch { + // best-effort + } + }, + onAuditAccountRotation: (params) => { + logAuditEvent({ + action: params.action, + actor: apiKeyInfo?.name || "system", + target: params.newConnectionId, + details: { + failed_connection_id: params.failedConnectionId, + new_connection_id: params.newConnectionId, + attempt: params.attempt, + retry_after_ms: params.retryAfterMs, + }, + }); + }, + }, + sendProviderAttempt: (modelToCall, allowDedup) => + executeProviderRequest(modelToCall, allowDedup), + }); + }; + + let toolLoopRan = false; + let toolLoopUsage = null; + let legResult = await runNonStreamingProviderLeg({ + phase: "initial", + sourceBody: (body || {}) as Record, + expectedConnectionId: managedLease + ? String(getCurrentConnectionId() || connectionId || "") || undefined + : undefined, + allowAccountRotation: !managedLease && comboStrategy !== "context-relay", + allowModelFallback: true, + executeProviderRequest: (modelToCall, allowDedup) => + executeProviderRequest(modelToCall, allowDedup), + runProviderExecution: runNonStreamingPipeline, + setRequestWireState: ({ translatedBody: nextBody, effectiveModel: nextModel }) => { + translatedBody = nextBody as typeof translatedBody; + syncExecuteTranslatedBody(translatedBody); + currentModel = nextModel; + triedModels.add(nextModel); + }, + sourceFormat, + targetFormat, + clientResponseFormat, + provider, + model: effectiveModel, + connectionId: String(getCurrentConnectionId() || connectionId || ""), + getCurrentConnectionId: () => getCurrentConnectionId() || undefined, + effectiveModel: currentModel, + translatedBody: translatedBody as Record, + toolNameMap, + customToolNames, + requestToolIdentityMap, + reasoningCacheScope, + reasoningReplayHistory, + videoTranscriptSensitive: videoBridgeObserved, + clientHeaders: clientRawRequest?.headers ?? null, + isClaudeCodeCompatible, + log, + }); + + if (legResult.kind === "error") { + const err = legResult.result; + const errMessage = + err?.rawMessage || + (err?.originalError instanceof Error ? err.originalError.message : err?.error) || + ""; + const errHeaders = err?.upstreamHeaders || err?.response?.headers; + const errUpstreamBody = err?.upstreamErrorBody; + if (err) { + await applyProviderFailureClassification({ + statusCode: err.status, + message: errMessage, + headers: errHeaders, + upstreamErrorBody: errUpstreamBody, + retryAfterMs: err.retryAfterMs ?? null, + targetModel: currentModel, + }); + } + + const captured = providerRequestCapture.latest?.() ?? null; + finalBody = captured?.body ?? finalBody ?? translatedBody; + if (captured) { + reqLogger.logTargetRequest(captured.url, captured.headers, captured.body); + } + reqLogger.logError(new Error(err.error || "Provider request failed"), finalBody); + const isNetworkThrow = Boolean(err.originalError); + if (err.response && !isNetworkThrow) { + reqLogger.logProviderResponse( + err.status, + err.response.statusText || "Error", + err.response.headers, + err.response + ); + } + appendRequestLog({ + model, + provider, + connectionId, + status: `FAILED ${err.status}`, + }).catch(() => {}); + persistAttemptLogs({ + status: err.status, + error: err.error || "Provider request failed", + providerRequest: finalBody || translatedBody, + providerResponse: isNetworkThrow ? undefined : err.response, + // On a client abort the client already disconnected before we got here, so this + // body is what we WOULD have sent, not what was delivered. The dashboard reads + // `clientResponse` as "what the client received", so logging it misleads — + // `error` above already records the reason. The pre-#12867 path omitted it here; + // the leg-based path must keep doing so. + clientResponse: isLocalStreamLifecycleError(err.originalError) + ? undefined + : buildErrorBody(err.status, err.error || "Provider request failed"), + cacheSource: "upstream", + }); + persistFailureUsage(err.status, err.errorCode || `upstream_${err.status}`); + trackPendingRequest(model, provider, pendingConnId, false, undefined, pendingRequestId); + return { + response: err, + carry: { + translatedBody, + currentModel, + finalBody, + providerResponse, + providerHeaders, + effectiveServiceTier, + claudePromptCacheLogMeta, + reasoningReplayHistory, + pipelineRecovered, + }, + }; + } + + pipelineRecovered = true; + const expectedConn = managedLease + ? String(getCurrentConnectionId() || connectionId || "") || undefined + : undefined; + // The identity is the tool loop's execution fence key, and deriveToolRequestIdentity + // canonicalizes the body — which by design rejects Dates, Maps and class instances. + // It was computed eagerly, so a body carrying any of those threw on EVERY + // non-streaming request even with SERVER_OWNED_TOOL_LOOP_ENABLED off (the default). + // Derive it only when the loop can run, and fail closed rather than crash: no + // identity means no fence, and without a fence the loop must not run. + let toolLoopEnabled = isServerOwnedToolLoopEnabled(); + let postInjectionRequestIdentity = ""; + if (toolLoopEnabled) { + try { + postInjectionRequestIdentity = derivePostInjectionRequestIdentity({ + apiKeyId: memoryOwnerId || "local", + headers: clientRawRequest?.headers ?? null, + skillRequestId, + postInjectionBody: (body || {}) as Record, + }); + } catch (identityError) { + log?.warn?.( + "SERVER_OWNED_TOOL_LOOP", + `request body is not canonicalizable, skipping the loop: ${ + identityError instanceof Error ? identityError.message : "unknown" + }` + ); + toolLoopEnabled = false; + } + } + const loopApply = await applyServerOwnedToolLoopIfNeeded({ + enabled: toolLoopEnabled, + stream, + isResponsesEndpoint, + sourceFormat, + initialLeg: legResult, + sourceBody: (body || {}) as Record, + skillsModelId: getSkillsModelIdForFormat(sourceFormat), + executionContext: { + apiKeyId: memoryOwnerId || "local", + sessionId: pipelineSessionId, + requestId: skillRequestId, + requestIdentity: postInjectionRequestIdentity, + builtinToolNames: injectionResult.builtinToolNames, + injectedCustomSkillNames: injectionResult.injectedCustomSkillNames, + customSkillExecutionEnabled: + Boolean(memoryOwnerId) && memorySettings?.skillsEnabled === true, + executionFenceEnabled: true, + provider, + model: effectiveModel, + }, + abortSignal: clientRawRequest?.signal, + expectedConnectionId: expectedConn, + followUpLeg: async (nextSourceBody) => { + translatedBody = translateRequest( + sourceFormat, + targetFormat, + model, + { ...nextSourceBody }, + false, + credentials, + provider, + reqLogger, + { + normalizeToolCallId: getModelNormalizeToolCallId( + provider || "", + model || "", + sourceFormat + ), + preserveDeveloperRole: getModelPreserveOpenAIDeveloperRole( + provider || "", + model || "", + sourceFormat + ), + preserveCacheControl, + signatureNamespace: connectionId, + copilotClient: copilotCompatibleReasoning, + reasoningCacheScope, + onReasoningReplayHistory: (messages) => { + reasoningReplayHistory = messages; + }, + } + ); + syncExecuteTranslatedBody(translatedBody); + return runNonStreamingProviderLeg( + followUpLegInput( + { + executeProviderRequest: (modelToCall, allowDedup) => + executeProviderRequest(modelToCall, allowDedup), + runProviderExecution: runNonStreamingPipeline, + setRequestWireState: ({ translatedBody: nextBody, effectiveModel: nextModel }) => { + translatedBody = nextBody as typeof translatedBody; + syncExecuteTranslatedBody(translatedBody); + currentModel = nextModel; + triedModels.add(nextModel); + }, + sourceFormat, + targetFormat, + clientResponseFormat, + provider, + model: effectiveModel, + connectionId: String(getCurrentConnectionId() || connectionId || ""), + getCurrentConnectionId: () => getCurrentConnectionId() || undefined, + effectiveModel: currentModel, + translatedBody: translatedBody as Record, + toolNameMap, + customToolNames, + requestToolIdentityMap, + reasoningCacheScope, + reasoningReplayHistory, + videoTranscriptSensitive: videoBridgeObserved, + clientHeaders: clientRawRequest?.headers ?? null, + isClaudeCodeCompatible, + log, + }, + nextSourceBody, + expectedConn + ) + ); + }, + logReceipt: (receipt) => reqLogger.logToolLoopReceipt(receipt), + }); + if (loopApply.kind === "error") { + return { + response: await finalizeToolLoopError({ + loop: loopApply.loop, + model, + provider, + connectionId: pendingConnId, + providerRequest: loopApply.loop.finalProviderRequest || finalBody || translatedBody, + persistFailureUsage, + persistAttemptLogs, + trackPendingRequest, + pendingRequestId, + }), + carry: { + translatedBody, + currentModel, + finalBody, + providerResponse, + providerHeaders, + effectiveServiceTier, + claudePromptCacheLogMeta, + reasoningReplayHistory, + pipelineRecovered, + }, + }; + } + // `legResult` is declared as the full NonStreamingProviderLegResult union. The + // `kind === "error"` guard above narrows it to the ok variant, but the conditional + // reassignment below widens it back to the declared type, so every field read past + // this point lost the narrowing — 13 TS2339 diagnostics under + // tsconfig.typecheck-api.json, which pulls chatCore.ts in through the route while + // tsconfig.typecheck-core.json does not. Pin the ok variant in its own binding: + // `loopApply.leg` is already `NonStreamingProviderLegResult & { kind: "ok" }`, + // so no cast is involved. + let okLeg: NonStreamingProviderLegResult & { kind: "ok" } = legResult; + if (loopApply.kind === "ok") { + toolLoopRan = true; + toolLoopUsage = loopApply.usage; + okLeg = loopApply.leg; + } + + if (okLeg.upstreamResponse) { + providerResponse = okLeg.upstreamResponse; + providerHeaders = normalizeHeaders(okLeg.upstreamResponse.headers); + } else { + providerResponse = new Response(null, { + status: 200, + headers: okLeg.headers, + }); + providerHeaders = normalizeHeaders(okLeg.headers); + } + finalBody = providerRequestCapture.body(okLeg.providerRequest || translatedBody); + // Built inside executeProviderRequest on the pre-#12867 path. The leg now owns the + // first non-streaming send, so that assignment never runs here and the meta stayed + // null — `_omniroute.claudePromptCache` silently vanished from every call log on + // this path. Same inputs, same helper, at the point where they are available. + claudePromptCacheLogMeta = buildClaudePromptCacheLogMeta( + targetFormat, + finalBody, + providerHeaders, + clientRawRequest?.headers + ); + const capturedOk = providerRequestCapture.latest?.(); + reqLogger.logTargetRequest( + okLeg.requestUrl || capturedOk?.url || "", + okLeg.requestHeaders || capturedOk?.headers || {}, + capturedOk?.body ?? finalBody + ); + const responseBody = okLeg.providerBody; + const responsePayloadFormat = okLeg.responsePayloadFormat; + const looksLikeSSE = okLeg.looksLikeSSE; + let translatedResponse = okLeg.response; + const memoryExtractionResponse = okLeg.responseForMemoryExtraction; + reqLogger.logProviderResponse( + 200, + "OK", + providerResponse.headers, + looksLikeSSE ? { _streamed: true, _format: "sse-json", summary: responseBody } : responseBody + ); + effectiveServiceTier = resolveReportedServiceTier(responseBody) ?? effectiveServiceTier; + if (onRequestSuccess) { + await onRequestSuccess(); + } + const successConnectionId = getCurrentConnectionId(); + await maybeSyncClaudeExtraUsageState({ + provider, + connectionId: successConnectionId, + providerSpecificData: credentials?.providerSpecificData, + log, + }); + const usage = toolLoopUsage ?? extractUsageFromResponse(responseBody, provider); + const cacheUsageLogMeta = buildCacheUsageLogMeta(usage); + if (usage && typeof usage === "object") { + attachCompressionUsageReceiptAfterAnalytics(usage as Record, "provider"); + if (provider === "gemini") { + const promptTokens = + typeof (usage as Record).prompt_tokens === "number" + ? ((usage as Record).prompt_tokens as number) + : 0; + if (promptTokens > 0) incrementTokenUsage(model, promptTokens); + } + } + recordContextEditingTelemetryHook({ + contextEditingEnabled, + provider, + responseBody, + skillRequestId, + log, + }); + appendRequestLog({ + model, + provider, + connectionId: successConnectionId, + tokens: usage, + status: "200 OK", + }).catch(() => {}); + recordNonStreamingUsageStats(usage, { + traceEnabled, + provider, + connectionId: successConnectionId, + model, + startTime, + apiKeyInfo, + effectiveServiceTier, + isCombo, + comboStrategy, + endpoint: endpointPath, + cpaAuthIndex: readCpaAuthIndex(providerResponse), + }); + + // #12150 P1b surface 3 (fix round 1): a video-bridge-observed request's + // request- AND response-derived text both carry the full transcript (the + // flattened description on the request side, the model's own reply on + // the response side) — neither may populate durable Memory. See + // runMemoryExtractionGate for the shared gate + extraction wiring, unit + // tested directly in tests/unit/video-bridge-memory-suppression.test.ts. + runMemoryExtractionGate({ + memoryOwnerId, + memorySettings, + videoBridgeObserved, + pipelineSessionId, + requestBody: body as Record, + responseBody: memoryExtractionResponse as Record | null, + extractFacts, + log, + }); + + const customSkillExecutionEnabled = + Boolean(memoryOwnerId) && memorySettings?.skillsEnabled === true; + const builtinToolNames = [ + webSearchFallbackPlan.toolName, + webFetchFallbackPlan.toolName, + ...(memoryOwnerId && memorySettings?.enabled ? MEMORY_BUILTIN_TOOL_NAMES : []), + ].filter((name): name is string => Boolean(name)); + if (!toolLoopRan && (customSkillExecutionEnabled || builtinToolNames.length > 0)) { + const skillSessionId = pipelineSessionId; + + translatedResponse = await handleToolCallExecution( + translatedResponse, + getSkillsModelIdForFormat(sourceFormat), + { + apiKeyId: memoryOwnerId || "local", + sessionId: skillSessionId, + requestId: skillRequestId, + builtinToolNames, + customSkillExecutionEnabled, + provider, + model: effectiveModel, + } + ); + } + + const guardrailContext = buildPostCallGuardrailContext({ + apiKeyInfo, + body, + clientRawRequest, + log, + model, + provider, + responsePayloadFormat, + clientResponseFormat, + }); + const postCallGuardrails = await guardrailRegistry.runPostCallHooks( + translatedResponse, + guardrailContext + ); + translatedResponse = postCallGuardrails.response; + + const responseUsage = isJsonRecord(usage) + ? usage + : isJsonRecord(translatedResponse.usage) + ? translatedResponse.usage + : null; + const costUsage = normalizeUsage(responseUsage); + const estimatedCost = costUsage + ? await calculateCost(provider, model, costUsage, { serviceTier: effectiveServiceTier }) + : 0; + const chatCostCtx = buildCostCtx(provider, model, usage, effectiveServiceTier, traceId); + + if (postCallGuardrails.blocked) { + const guardrailMessage = postCallGuardrails.message || "Response blocked by guardrail"; + persistAttemptLogs({ + status: HTTP_STATUS.BAD_REQUEST, + tokens: usage, + responseBody, + providerRequest: finalBody || translatedBody, + providerResponse: looksLikeSSE + ? { + _streamed: true, + _format: "sse-json", + summary: responseBody, + } + : responseBody, + clientResponse: buildErrorBody(HTTP_STATUS.BAD_REQUEST, guardrailMessage), + claudeCacheMeta: claudePromptCacheLogMeta, + claudeCacheUsageMeta: cacheUsageLogMeta, + cacheSource: "upstream", + }); + recordChatCallCost(apiKeyInfo, meteredBudgetCost(provider, estimatedCost), chatCostCtx, false); + log?.warn?.( + "GUARDRAIL", + `Response blocked by ${postCallGuardrails.guardrail || "guardrail"}: ${guardrailMessage}` + ); + finalizePendingScope(pendingScope, { + providerResponse: responseBody, + clientResponse: translatedResponse, + }); + return { + response: createErrorResult(HTTP_STATUS.BAD_REQUEST, guardrailMessage), + carry: { + translatedBody, + currentModel, + finalBody, + providerResponse, + providerHeaders, + effectiveServiceTier, + claudePromptCacheLogMeta, + reasoningReplayHistory, + pipelineRecovered, + }, + }; + } + + // Validate the *translated* response actually carries client-usable output. + // isEmptyContentResponse (above) runs on the raw responseBody before translation; + // this check runs after translation + sanitization + tool-call execution to catch + // cases where a provider returns a structurally valid raw body that translates into + // choices:[] or output:[] with no usable content (Responses API shape included). + const malformedTranslatedReason = detectMalformedNonStream(translatedResponse, provider); + if (malformedTranslatedReason) { + const totalLatency = Date.now() - startTime; + const rawBytes = (() => { + try { + return JSON.stringify(responseBody || {}).length; + } catch { + return -1; + } + })(); + reportMalformed200({ + mode: "nonstream", + provider, + model, + connectionId, + reason: malformedTranslatedReason, + recvBytes: rawBytes, + recvLines: -1, + emitted: -1, + events: {}, + ttftMs: totalLatency, + elapsedMs: totalLatency, + }); + appendRequestLog({ + model, + provider, + connectionId, + status: `FAILED ${HTTP_STATUS.BAD_GATEWAY}`, + }).catch(() => {}); + const malformed = describeMalformedNonStream(translatedResponse, malformedTranslatedReason); + const malformedMessage = `[${provider}/${model}] ${malformed.message}`; + const malformedClientBody = buildErrorBody( + HTTP_STATUS.BAD_GATEWAY, + malformedMessage, + undefined, + { code: malformed.code, type: malformed.type } + ); + const sanitizedMalformedResponse = sanitizeUpstreamDetails(responseBody); + const sanitizedMalformedProviderResponse = looksLikeSSE + ? { _streamed: true, _format: "sse-json", summary: sanitizedMalformedResponse } + : sanitizedMalformedResponse; + persistAttemptLogs({ + status: HTTP_STATUS.BAD_GATEWAY, + tokens: usage, + responseBody: sanitizedMalformedResponse, + providerRequest: finalBody || translatedBody, + providerResponse: sanitizedMalformedProviderResponse, + clientResponse: malformedClientBody, + claudeCacheMeta: claudePromptCacheLogMeta, + claudeCacheUsageMeta: cacheUsageLogMeta, + cacheSource: "upstream", + }); + persistFailureUsage(HTTP_STATUS.BAD_GATEWAY, "malformed_translated_response"); + trackPendingRequest(model, provider, pendingConnId, false, undefined, pendingRequestId); + // Routing event (feedback foundation) — record the malformed outcome so + // the quality tracker de-prioritizes this model over time. + void emitRoutingEvent( + createRoutingEvent({ + requestId: traceId || pendingRequestId || "unknown", + provider: provider || "unknown", + model: model || "unknown", + strategy: isCombo ? (comboStrategy ?? "combo") : "direct", + latencyMs: Date.now() - startTime, + ttftMs: null, + inputTokens: null, + outputTokens: null, + cost: null, + retries: 0, + fallbackUsed: false, // combo-level fallback tracked by decisionTrace + outcome: "malformed", + status: HTTP_STATUS.BAD_GATEWAY, + finishReason: routingFinishReason(translatedResponse), + connectionId: credentials?.connectionId ?? null, + }) + ); + return { + response: createErrorResult( + HTTP_STATUS.BAD_GATEWAY, + malformedMessage, + null, + malformed.code, + malformed.type + ), + carry: { + translatedBody, + currentModel, + finalBody, + providerResponse, + providerHeaders, + effectiveServiceTier, + claudePromptCacheLogMeta, + reasoningReplayHistory, + pipelineRecovered, + }, + }; + } + + // ── Phase 9.1: Cache store (non-streaming, temp=0) ── + storeSemanticCacheResponse({ + enabled: semanticCacheEnabled, + body: bodyForCacheWrite, + headers: clientRawRequest?.headers, + translatedResponse, + model, + // The dual-layer manager scopes entries per provider (cacheByProvider); + // lookup passes the resolved provider, so the write must too (#14159). + provider, + apiKeyId: apiKeyInfo?.id ?? undefined, + usage, + log, + videoTranscriptSensitive: videoBridgeObserved, + }); + + // ── Phase 9.2: Save for idempotency ── + // Reuse the key resolved by checkIdempotencyCache() above (single derivation per + // request). (#3821-review LEDGER-6) + saveIdempotency(idempotencyKey, translatedResponse, 200); + reqLogger.logConvertedResponse(translatedResponse); + persistAttemptLogs({ + status: 200, + tokens: usage, + responseBody, + providerRequest: finalBody || translatedBody, + providerResponse: looksLikeSSE + ? { + _streamed: true, + _format: "sse-json", + summary: responseBody, + } + : responseBody, + clientResponse: translatedResponse, + claudeCacheMeta: claudePromptCacheLogMeta, + claudeCacheUsageMeta: cacheUsageLogMeta, + cacheSource: "upstream", + }); + recordChatCallCost(apiKeyInfo, meteredBudgetCost(provider, estimatedCost), chatCostCtx, true); + + // === Quota Share POST-hook (B/F7) — fire-and-forget, fail-open === + await scheduleQuotaShareConsumption({ + apiKeyId: apiKeyInfo?.id, + connectionId: credentials?.connectionId, + provider, + model, + usage, + estimatedCost, + log, + }); + // === /Quota Share POST-hook === + + // ── Gamification event (fire-and-forget) ── + await emitRequestGamificationEvent({ apiKeyId: apiKeyInfo?.id, model, provider }); + + finalizePendingScope(pendingScope, { + providerResponse: responseBody, + clientResponse: translatedResponse, + }); + const responseHeaders = buildNonStreamingResponseHeaders({ + provider, + model, + startTime, + responseUsage, + estimatedCost, + requestId: skillRequestId, + compressionResponseMeta, + comboStrategy, + fallbackAttempts, + }); + // #6426: align response body `model` with the `X-OmniRoute-Model` header + // (both must be the resolved backend model). Some upstreams (notably legacy + // /v1/completions text-completion path) return a body `model` field that + // differs from the resolved backend id we advertised in the header, leaving + // strict clients unable to reconcile the two. Rewrite body.model to `model` + // FIRST, then let #1311 echo override it when the opt-in setting is on. + if (typeof model === "string" && model) echoModelInObject(translatedResponse, model); + // #1311: echo the requested alias/combo name in the non-streaming response model. + if (echoModel) echoModelInObject(translatedResponse, echoModel); + + // ── Plugin onResponse hook (fire-and-forget) ── + // #8395: the streaming branch below already calls this; the non-streaming + // (stream:false) branch returned without it, so onResponse never fired for + // non-streaming requests at all. + await runPluginOnResponseHook({ + requestId: traceId, + body, + model, + provider, + apiKeyInfo, + headers: clientRawRequest?.headers, + response: { status: 200, data: translatedResponse }, + }); + + // Routing event (feedback foundation) — fire-and-forget, cheap. + void emitRoutingEvent( + createRoutingEvent({ + requestId: traceId || pendingRequestId || "unknown", + provider: provider || "unknown", + model: model || "unknown", + strategy: isCombo ? (comboStrategy ?? "combo") : "direct", + latencyMs: Date.now() - startTime, + ttftMs: null, + inputTokens: + usage && typeof usage === "object" + ? (() => { + const promptTokens = (usage as Record).prompt_tokens; + return typeof promptTokens === "number" && Number.isFinite(promptTokens) + ? promptTokens + : null; + })() + : null, + outputTokens: + usage && typeof usage === "object" + ? (() => { + const completionTokens = (usage as Record).completion_tokens; + return typeof completionTokens === "number" && Number.isFinite(completionTokens) + ? completionTokens + : null; + })() + : null, + cost: Number.isFinite(estimatedCost) ? estimatedCost : null, + retries: 0, + fallbackUsed: false, // combo-level fallback tracked by decisionTrace + outcome: "success", + status: 200, + finishReason: routingFinishReason(translatedResponse), + connectionId: credentials?.connectionId ?? null, + }) + ); + + return { + response: { + success: true, + response: maybeWrapForcedNonStreamingResponsesJson({ + clientRequestedResponsesStream, + body: translatedResponse, + headers: responseHeaders, + }), + }, + carry: { + translatedBody, + currentModel, + finalBody, + providerResponse, + providerHeaders, + effectiveServiceTier, + claudePromptCacheLogMeta, + reasoningReplayHistory, + pipelineRecovered, + }, + }; + } catch (error) { + trackPendingRequest(model, provider, pendingConnId, false, undefined, pendingRequestId); + const errorMetadata = getSafeErrorMetadata(error); + const managedLeaseFenceCode = getManagedLeaseFenceErrorCode(errorMetadata.code); + if (managedLeaseFenceCode) + return { + response: managedLeaseFenceErrorResult(managedLeaseFenceCode), + carry: { + translatedBody, + currentModel, + finalBody, + providerResponse, + providerHeaders, + effectiveServiceTier, + claudePromptCacheLogMeta, + reasoningReplayHistory, + pipelineRecovered, + }, + }; + // isSemaphoreCapacityError already reads the code through getSafeErrorMetadata, + // so a hostile rejection cannot escape this classification. + if (isSemaphoreCapacityError(error)) { + const semaphoreCode = errorMetadata.code as string; + appendRequestLog({ + model, + provider, + connectionId, + status: `FAILED ${semaphoreCode}`, + }).catch(() => {}); + const failureMessage = sanitizeErrorMessage(errorMetadata.message) || "Semaphore timeout"; + persistAttemptLogs({ + status: HTTP_STATUS.RATE_LIMITED, + error: failureMessage, + providerRequest: finalBody || translatedBody, + clientResponse: buildErrorBody(HTTP_STATUS.RATE_LIMITED, failureMessage), + claudeCacheMeta: claudePromptCacheLogMeta, + cacheSource: "upstream", + }); + persistFailureUsage(HTTP_STATUS.RATE_LIMITED, semaphoreCode); + const result = createErrorResult(HTTP_STATUS.RATE_LIMITED, failureMessage); + return { + response: { + ...result, + errorType: "account_semaphore_capacity", + errorCode: semaphoreCode, + }, + carry: { + translatedBody, + currentModel, + finalBody, + providerResponse, + providerHeaders, + effectiveServiceTier, + claudePromptCacheLogMeta, + reasoningReplayHistory, + pipelineRecovered, + }, + }; + } + throw error; + } +} diff --git a/open-sse/handlers/chatCore/streamingResponse.ts b/open-sse/handlers/chatCore/streamingResponse.ts new file mode 100644 index 000000000000..eda35f12eb8f --- /dev/null +++ b/open-sse/handlers/chatCore/streamingResponse.ts @@ -0,0 +1,1246 @@ +import { projectFailureUsageErrorCode } from "./failureUsage.ts"; + +export { extractSystemRoleMessages, relocateDirectiveOnlyMessages } from "./claudeSystemRole.ts"; + +export { clearCombosCache, clearUpstreamProxyConfigCache } from "./comboContextCache.ts"; +import { buildClaudePromptCacheLogMeta } from "./executorHelpers.ts"; +import { + shouldUseNativeCodexPassthrough, + shouldUseNativeXaiResponsesPassthrough, + redactPassthroughThinkingSignatures, + isClaudeCodeSemanticPassthroughRequest, +} from "./passthroughHelpers.ts"; +import { recoverAnthropicThinkingSignature } from "../chatCore/thinkingSignatureRecovery.ts"; +import { runProviderExecutionPipeline } from "../chatCore/providerExecutionPipeline.ts"; +import { markCodexScopeRateLimited } from "../chatCore/codexFailover.ts"; +import { deleteSessionAccountAffinity } from "@/lib/db/sessionAccountAffinity"; +import { buildStreamingResponseHeaders, stripStaleForwardingHeaders } from "./responseHeaders.ts"; + +// Re-export the previously inline-defined helpers so existing importers of these +// symbols from chatCore.ts (tests, sibling modules) keep resolving after the split. +export { + shouldUseNativeCodexPassthrough, + shouldUseNativeXaiResponsesPassthrough, + redactPassthroughThinkingSignatures, + isClaudeCodeSemanticPassthroughRequest, + buildStreamingResponseHeaders, + stripStaleForwardingHeaders, +}; + +import { normalizeHeaders } from "../../utils/headers.ts"; + +import { FORMATS } from "../../translator/formats.ts"; + +import { COLORS } from "../../utils/stream.ts"; + +import { + refreshWithRetry, + isUnrecoverableRefreshError, + runWithOnPersist, + runWithCasGuard, +} from "../../services/tokenRefresh.ts"; + +import { runWithCapture } from "../../utils/providerRequestLogging.ts"; + +import { shouldSkipCredentialRefresh } from "../chatCore/skipCredentialRefresh.ts"; + +import { + REASONING_BUFFER_MIN_TRIGGER, + buildReasoningProbeTruncatedResponse, + isEmptyContentUpstreamFailure, + isTinyBudgetReasoningProbe, + toPositiveInteger, +} from "../../services/reasoningTokenBuffer.ts"; +import { + buildErrorBody, + createErrorResult, + parseUpstreamError, + formatProviderError, + projectPublicErrorIdentifier, + sanitizeErrorMessage, + sanitizeUpstreamDetails, +} from "../../utils/error.ts"; + +import { HTTP_STATUS, ANTIGRAVITY_PRE_RESPONSE_TIMEOUT_CODE } from "../../config/constants.ts"; +import { applyStatusRestatement } from "../../config/upstreamStatusRestatement.ts"; + +import { updateProviderConnection, getProviderConnectionById } from "@/lib/db/providers"; +import { wasRefreshTokenRotated } from "@omniroute/open-sse/services/refreshSerializer.ts"; + +import { + createSafeAbortError, + createStreamingErrorResult, + isSemaphoreCapacityError, + getSafeErrorMetadata, + getUpstreamErrorIdentifier, +} from "./streamErrorResult.ts"; + +import { getExecutionConnectionId } from "../chatCore/executionCredentials.ts"; + +import { prepareUpstreamBody } from "../chatCore/upstreamBody.ts"; + +import { logAuditEvent } from "@/lib/compliance"; + +import { trackPendingRequest, appendRequestLog } from "@/lib/usageDb"; +import { updatePendingScope } from "@/lib/usage/pendingRequestScope"; + +import { normalizeExecutorResult } from "./upstreamTimeouts.ts"; + +import { getProviderCredentials, extractSessionAffinityKey } from "@/sse/services/auth"; + +import { updateFromHeaders, updateFromResponseBody } from "../../services/rateLimitManager.ts"; +import * as localLimiterErrors from "../../services/rateLimitManager/errors.ts"; +import { markBlocked as markAccountSemaphoreBlocked } from "../../services/accountSemaphore.ts"; +import { lockModel, recordCoreOwnedAntigravityQuotaState } from "../../services/accountFallback.ts"; + +import { + isModelUnavailableError, + getNextFamilyFallback, + isContextOverflowError, + findLargerContextModel, + getModelFamily, +} from "../../services/modelFamilyFallback.ts"; + +import { isLocalStreamLifecycleError } from "@/shared/utils/circuitBreaker"; +import { shouldIsolateProbeFailures } from "@/shared/utils/probeOrigin"; +import { writeTerminalStatus } from "@/shared/utils/terminalStatus"; + +/** + * Core chat handler - shared between SSE and Worker + * Returns { success, response, status, error } for caller to handle fallback + * @param {object} options + * @param {object} options.body - Request body + * @param {object} options.modelInfo - { provider, model } + * @param {object} options.credentials - Provider credentials + * @param {object} options.log - Logger instance (optional) + * @param {function} options.onCredentialsRefreshed - Callback when credentials are refreshed + * @param {function} options.onRequestSuccess - Callback when request succeeds (to clear error status) + * @param {function} options.onDisconnect - Callback when client disconnects + * @param {string} options.connectionId - Connection ID for usage tracking + * @param {object} options.apiKeyInfo - API key metadata for usage attribution + * @param {string} options.comboName - Combo name if this is a combo request + * @param {string} options.comboStrategy - Combo routing strategy (e.g., 'priority', 'cost-optimized') + * @param {boolean} options.isCombo - Whether this request is from a combo + * @param {string} options.connectionId - Connection ID for settings lookup + */ +// extractSystemRoleMessages extracted to chatCore/claudeSystemRole.ts (#3501); re-exported above so +// existing importers (e.g. tests/unit/system-role-extraction.test.ts) keep resolving it from here. + +// eslint-disable-next-line @typescript-eslint/no-explicit-any -- deps bag mirrors the barrel closure +type Loose = any; + +export type StreamingDeps = Record; + +export async function runStreamingResponse(deps: StreamingDeps) { + const { + apiKeyInfo, + persistAttemptLogs, + buildUpstreamHeadersForExecute, + resolveEffectiveServiceTier, + clientResponseFormat, + correlationId, + extendedContext, + executeProviderRequest, + applyProviderFailureClassification, + assertManagedLeaseFence, + body, + clientRawRequest, + comboStrategy, + connectionId, + contextEditingEnabled, + credentials, + effectiveModel, + executeRefreshCredentials, + executor, + getCurrentConnectionId, + getExecutionCredentials, + getExecutorClientHeaders, + getManagedLeaseFenceErrorCode, + handleCredentialsRefreshed, + isCombo, + isOpencodeClient, + log, + managedLease, + managedLeaseFenceErrorResult, + model, + onCredentialsRefreshed, + pendingConnId, + pendingRequestId, + pendingScope, + persistFailureUsage, + provider, + providerRequestCapture, + reqLogger, + sessionAffinityKey, + skillRequestId, + sourceFormat, + stream, + streamController, + syncExecuteTranslatedBody, + targetFormat, + triedModels, + trustedEffortContext, + upstreamStream, + } = deps; + + let claudePromptCacheLogMeta, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + translatedBody; + + claudePromptCacheLogMeta = deps.claudePromptCacheLogMeta; + currentModel = deps.currentModel; + effectiveServiceTier = deps.effectiveServiceTier; + finalBody = deps.finalBody; + pipelineRecovered = deps.pipelineRecovered; + void effectiveServiceTier; + providerHeaders = deps.providerHeaders; + providerResponse = deps.providerResponse; + providerUrl = deps.providerUrl; + translatedBody = deps.translatedBody; + try { + const pipelineOutcome = await runProviderExecutionPipeline({ + policy: { + allowAccountRotation: !managedLease && comboStrategy !== "context-relay", + allowModelFallback: true, + expectedConnectionId: managedLease + ? String(getCurrentConnectionId() || connectionId || "") || undefined + : undefined, + }, + target: { + provider, + requestedModel: effectiveModel, + sourceFormat, + targetFormat, + stream, + }, + connection: { + initialConnectionId: String(getCurrentConnectionId() || connectionId || ""), + getCurrentConnectionId: () => getCurrentConnectionId() || undefined, + getCredentials: () => (credentials || {}) as Record, + replaceCredentials: (next) => { + Object.assign(credentials, next); + }, + onCredentialsRefreshed: handleCredentialsRefreshed, + refreshCredentials: executeRefreshCredentials, + assertManagedLeaseFence: (id) => { + assertManagedLeaseFence(id); + }, + getProviderCredentials, + }, + wire: { + body: translatedBody as Record, + currentModel, + triedModels, + setBodyAndModel: (body, model) => { + translatedBody = body as typeof translatedBody; + syncExecuteTranslatedBody(translatedBody); + currentModel = model; + triedModels.add(model); + }, + }, + state: { + updatePendingStage: (stage, data) => { + updatePendingScope(pendingScope, { stage, ...(data || {}) }); + }, + recordRateLimitHeaders: updateFromHeaders, + recordRateLimitBody: updateFromResponseBody, + writeTerminalStatus, + persistConnectionPatch: updateProviderConnection, + setConnectionRateLimitedUntil: async (id, untilMs) => { + const { setConnectionRateLimitUntil } = await import("@/lib/db/providers"); + setConnectionRateLimitUntil(id, untilMs); + }, + lockModel, + recordAntigravityQuotaState: recordCoreOwnedAntigravityQuotaState, + markAccountSemaphoreBlocked: (key) => { + markAccountSemaphoreBlocked(key, Date.now() + 60_000); + }, + isolateProbeFailures: () => shouldIsolateProbeFailures(), + onCodexScopeRateLimited: async (params) => { + await markCodexScopeRateLimited({ + failedConnectionId: params.failedConnectionId, + model: params.model, + rateLimitedUntil: params.rateLimitedUntil, + credentials: (params.credentials || credentials) as { + connectionId?: string | null; + providerSpecificData?: unknown; + }, + }); + }, + onClearSessionAffinity: () => { + const key = + sessionAffinityKey || + extractSessionAffinityKey(body, clientRawRequest?.headers) || + null; + if (!key) return; + try { + deleteSessionAccountAffinity(key, "codex"); + } catch { + // best-effort + } + }, + onAuditAccountRotation: (params) => { + logAuditEvent({ + action: params.action, + actor: apiKeyInfo?.name || "system", + target: params.newConnectionId, + details: { + failed_connection_id: params.failedConnectionId, + new_connection_id: params.newConnectionId, + attempt: params.attempt, + retry_after_ms: params.retryAfterMs, + }, + }); + }, + }, + sendProviderAttempt: (modelToCall, allowDedup) => + executeProviderRequest(modelToCall, allowDedup), + }); + + pipelineRecovered = true; + currentModel = pipelineOutcome.model; + if (pipelineOutcome.kind === "error") { + providerResponse = pipelineOutcome.result.response; + providerUrl = ""; + providerHeaders = normalizeHeaders(pipelineOutcome.result.response.headers); + finalBody = translatedBody; + } else { + const result = { + response: pipelineOutcome.response, + url: pipelineOutcome.url, + headers: pipelineOutcome.headers, + transformedBody: pipelineOutcome.transformedBody, + }; + providerResponse = result.response; + providerUrl = result.url; + providerHeaders = result.headers; + finalBody = providerRequestCapture.body(result.transformedBody); + } + const responseConnectionId = getCurrentConnectionId(); + effectiveServiceTier = resolveEffectiveServiceTier(finalBody); + claudePromptCacheLogMeta = buildClaudePromptCacheLogMeta( + targetFormat, + finalBody, + providerHeaders, + clientRawRequest?.headers + ); + + // Log target request (final request to provider) + reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody); + updatePendingScope(pendingScope, { + providerRequest: finalBody, + providerUrl, + stage: "provider_response_started", + }); + // Update rate limiter from response headers (learn limits dynamically) + updateFromHeaders( + provider, + responseConnectionId, + providerResponse.headers, + providerResponse.status, + model + ); + + // Store rate-limit headers for quota saturation signals + try { + const { storeRateLimitHeaders } = await import("@/lib/quota/saturationSignals"); + storeRateLimitHeaders( + responseConnectionId, + provider, + providerResponse.headers as Record + ); + } catch { + // fail-open: saturation signal is best-effort + } + } catch (error) { + trackPendingRequest(model, provider, pendingConnId, false, undefined, pendingRequestId); + const errorMetadata = getSafeErrorMetadata(error); + const managedLeaseFenceCode = getManagedLeaseFenceErrorCode(errorMetadata.code); + if (managedLeaseFenceCode) + return { + response: managedLeaseFenceErrorResult(managedLeaseFenceCode), + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; + // isSemaphoreCapacityError already reads the code through getSafeErrorMetadata, + // so a hostile rejection cannot escape this classification. + if (isSemaphoreCapacityError(error)) { + const semaphoreCode = errorMetadata.code as string; + appendRequestLog({ + model, + provider, + connectionId, + status: `FAILED ${semaphoreCode}`, + }).catch(() => {}); + const failureMessage = sanitizeErrorMessage(errorMetadata.message) || "Semaphore timeout"; + persistAttemptLogs({ + status: HTTP_STATUS.RATE_LIMITED, + error: failureMessage, + providerRequest: finalBody || translatedBody, + clientResponse: buildErrorBody(HTTP_STATUS.RATE_LIMITED, failureMessage), + claudeCacheMeta: claudePromptCacheLogMeta, + cacheSource: "upstream", + }); + persistFailureUsage(HTTP_STATUS.RATE_LIMITED, semaphoreCode); + const result = stream + ? createStreamingErrorResult(HTTP_STATUS.RATE_LIMITED, failureMessage, semaphoreCode) + : createErrorResult(HTTP_STATUS.RATE_LIMITED, failureMessage); + return { + response: { + ...result, + errorType: "account_semaphore_capacity", + errorCode: semaphoreCode, + }, + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; + } + // abort(reason) can reject with a raw string lacking `name`/`status`; classify + // it through isLocalStreamLifecycleError so it maps to 499 rather than the + // 502 provider-failure default. + let isRequestAborted = errorMetadata.name === "AbortError"; + if (!isRequestAborted) { + try { + isRequestAborted = isLocalStreamLifecycleError(error); + } catch { + // A hostile Proxy must not escape the provider-error boundary during classification. + } + } + // #8376: proxyFetch tags unreachable transport failures so they remain + // distinguishable from ordinary provider 5xx responses. + const isProxyUnreachableFailure = + !isRequestAborted && errorMetadata.errorCode === "proxy_unreachable"; + const errorCode = errorMetadata.code; + const localRateLimitFailure = localLimiterErrors.getClientSafeLocalRateLimitError(error); + const failureStatus = isRequestAborted + ? 499 + : isProxyUnreachableFailure + ? HTTP_STATUS.BAD_GATEWAY + : localRateLimitFailure + ? localRateLimitFailure.status + : errorMetadata.name === "TimeoutError" || errorMetadata.name === "BodyTimeoutError" + ? HTTP_STATUS.GATEWAY_TIMEOUT + : errorMetadata.status + ? errorMetadata.status + : HTTP_STATUS.BAD_GATEWAY; + const failureMessage = isRequestAborted + ? "Request aborted" + : (() => { + try { + return formatProviderError( + localRateLimitFailure ?? error, + provider, + model, + failureStatus + ); + } catch { + // Formatting is diagnostic only; hostile rejection metadata falls back safely. + return errorMetadata.message || "Upstream provider error"; + } + })(); + const safeFailureMessage = sanitizeErrorMessage(failureMessage) || "Upstream provider error"; + const upstreamErrorCode = + localRateLimitFailure?.code ?? (isProxyUnreachableFailure ? "proxy_unreachable" : errorCode); + // Tag our own deadline timeouts (fetch-start TimeoutError / body BodyTimeoutError, + // both surfaced as a 504) as "upstream_timeout" so the cooldown layer can tell a + // slow-but-not-failed request apart from a real provider 5xx. (Antigravity already + // tags its pre-response timeout via the code below.) + const isOwnDeadlineTimeout = + failureStatus === HTTP_STATUS.GATEWAY_TIMEOUT && + (errorMetadata.name === "TimeoutError" || errorMetadata.name === "BodyTimeoutError"); + const upstreamErrorType = + upstreamErrorCode === ANTIGRAVITY_PRE_RESPONSE_TIMEOUT_CODE || isOwnDeadlineTimeout + ? "upstream_timeout" + : failureStatus === 401 + ? "authentication_error" + : undefined; + appendRequestLog({ + model, + provider, + connectionId, + status: `FAILED ${failureStatus}`, + }).catch(() => {}); + persistAttemptLogs({ + status: failureStatus, + error: safeFailureMessage, + providerRequest: finalBody || translatedBody, + // On a client-abort (AbortError), the client already disconnected before + // we ever got here — this body is what we WOULD have sent, not what was + // actually delivered. Logging it as `clientResponse` is misleading (the + // dashboard reads that field as "what the client received"), so omit it + // for this case; `error` above already records the failure reason. + clientResponse: + errorMetadata.name === "AbortError" + ? undefined + : buildErrorBody(failureStatus, failureMessage), + claudeCacheMeta: claudePromptCacheLogMeta, + cacheSource: "upstream", + }); + if (isRequestAborted) { + streamController.handleError(createSafeAbortError()); + return { + response: createErrorResult(499, "Request aborted"), + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; + } + const persistentErrorCode = projectFailureUsageErrorCode({ + statusCode: failureStatus, + message: failureMessage, + errorCode: projectPublicErrorIdentifier( + upstreamErrorCode || errorMetadata.name, + "upstream_error" + ), + errorType: upstreamErrorType, + }); + persistFailureUsage(failureStatus, persistentErrorCode); + console.log(`${COLORS.red}[ERROR] ${safeFailureMessage}${COLORS.reset}`); + if (stream && upstreamErrorCode) { + const result = createStreamingErrorResult( + failureStatus, + failureMessage, + upstreamErrorCode, + upstreamErrorType + ); + localLimiterErrors.markTrustedLocalRateLimitResponse(result.response, error); + return { + response: { + ...result, + errorType: upstreamErrorType, + errorCode: upstreamErrorCode, + }, + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; + } + const result = createErrorResult( + failureStatus, + failureMessage, + null, + upstreamErrorCode, + upstreamErrorType + ); + localLimiterErrors.markTrustedLocalRateLimitResponse(result.response, error); + return { + response: result, + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; + } + let upstreamErrorParsed = false; + let parsedStatusCode = providerResponse.status; + let parsedMessage = ""; + let parsedRetryAfterMs: number | null = null; + let upstreamErrorBody: unknown = null; + + // Track whether stream_options was present and stripped — if so, 401/403 after + // that may be from the modification rather than a genuine auth failure, so we + // skip the credential refresh attempt in that case. + const hadStreamOptions = + targetFormat === FORMATS.OPENAI_RESPONSES && "stream_options" in translatedBody; + if (hadStreamOptions) { + delete translatedBody.stream_options; + } + + // Handle 401/403 - try token refresh using executor + // T-PROBE: probe-origin failures never attempt the refresh — a probe must + // not consume a rotating refresh token nor persist an "expired" + // deactivation on refresh failure (#9817). The 401/403 then flows into + // the normal providerFailure classification (record-only in probe mode). + if ( + (providerResponse.status === HTTP_STATUS.UNAUTHORIZED || + providerResponse.status === HTTP_STATUS.FORBIDDEN) && + !hadStreamOptions && // Skip refresh if failure may be from stream_options removal, not auth + !(await shouldIsolateProbeFailures()) && + !(await shouldSkipCredentialRefresh(provider, providerResponse)) + ) { + // Fix A: wrap refreshCredentials in runWithOnPersist so the persist callback + // executes INSIDE the per-connection mutex held by getAccessToken. This makes + // [network refresh + DB write + outer-state mutation] one atomic step and + // prevents concurrent requests from reading a stale refreshToken before the + // DB has been updated (refresh_token_reused on Codex/OpenAI). + // + // Not every executor routes refresh through getAccessToken (e.g. github.ts + // calls refreshCopilotToken directly). When the persistFn doesn't fire from + // inside getAccessToken, we still need to do the credentials mutation + user + // callback after refreshCredentials returns. The `persistFnRan` flag tracks + // which path executed so we don't double-fire (race-prone) or skip (regression). + // Front 3: remember the refresh_token we are about to present so that, if the + // refresh fails as unrecoverable, we can tell a genuine death apart from a + // stale-token reuse that a concurrent/sibling refresh already rotated past. + const attemptedRefreshToken = + typeof credentials?.refreshToken === "string" ? credentials.refreshToken : null; + let persistFnRan = false; + const persistFn = onCredentialsRefreshed + ? async (refreshResult: Record) => { + persistFnRan = true; + // Mutate the shared credentials object so subsequent executor calls + // in this request see the new tokens. Runs INSIDE the mutex. + Object.assign(credentials, refreshResult); + await onCredentialsRefreshed(refreshResult); + } + : undefined; + + // #4038: build a compare-and-swap reread so getAccessToken can skip the persist if a + // concurrent writer (sibling request / HealthCheck / replica) already rotated this + // connection's refresh_token past the one we presented — overwriting would revert it + // and revoke the token family. No connectionId ⇒ no guard (behavior unchanged). + const casConnectionId = + typeof credentials?.connectionId === "string" ? credentials.connectionId.trim() : ""; + const casReread = casConnectionId + ? async () => { + const latest = await getProviderConnectionById(casConnectionId); + return typeof latest?.refreshToken === "string" ? latest.refreshToken : null; + } + : null; + + const newCredentials = (await refreshWithRetry( + () => + runWithCasGuard( + casReread ? { expectedRefreshToken: attemptedRefreshToken, reread: casReread } : null, + () => runWithOnPersist(persistFn, () => executor.refreshCredentials(credentials, log)) + ), + 3, + log, + provider // Explicitly pass the provider to avoid universally tripping the "unknown" circuit breaker + )) as null | { + accessToken?: string; + copilotToken?: string; + }; + + if (newCredentials?.accessToken || newCredentials?.copilotToken) { + log?.info?.("TOKEN", `${provider?.toUpperCase()} | refreshed`); + + // Fall back to post-mutex mutation only for executors that don't route + // through getAccessToken (and therefore never fire onPersist). For + // executors that DO route through it (Codex, Claude, Gemini, etc.) the + // mutation already happened atomically inside the mutex. + if (!persistFnRan) { + Object.assign(credentials, newCredentials); + if (onCredentialsRefreshed) { + await onCredentialsRefreshed(newCredentials); + } + } + + // Retry with new credentials — model + extra headers follow translatedBody.model so they + // stay aligned if this block ever runs after a path that mutates body.model (e.g. fallback). + try { + const retryModelId = String(translatedBody.model || effectiveModel); + const retryBody = await prepareUpstreamBody({ + translatedBody, + modelToCall: retryModelId, + ...trustedEffortContext, + provider, + targetFormat, + credentials: getExecutionCredentials(), + log, + bypassDefaultToolLimit: isOpencodeClient, + isOpencodeClient, + rawBody: body, + clientRawRequest, + }); + assertManagedLeaseFence(getExecutionConnectionId(getExecutionCredentials())); + const retryResult = normalizeExecutorResult( + await runWithCapture(providerRequestCapture, () => + executor.execute({ + model: retryModelId, + body: retryBody, + stream: upstreamStream, + credentials: getExecutionCredentials(), + signal: streamController.signal, + log, + extendedContext, + upstreamExtraHeaders: buildUpstreamHeadersForExecute(retryModelId), + clientHeaders: getExecutorClientHeaders(), + clientResponseFormat, + onCredentialsRefreshed, + skipUpstreamRetry: isCombo, + contextEditing: { enabled: contextEditingEnabled }, + correlationId, + }) + ) + ); + + if (retryResult.response.ok) { + providerResponse = retryResult.response; + providerUrl = retryResult.url; + providerHeaders = new Headers(retryResult.headers || {}); + finalBody = providerRequestCapture.body(retryResult.transformedBody); + reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody); + updatePendingScope(pendingScope, { + providerRequest: finalBody, + providerUrl, + stage: "provider_response_started", + }); + upstreamErrorParsed = false; // Reset since new response is OK + } else { + providerResponse = retryResult.response; + upstreamErrorParsed = false; // Let it be parsed downstream + } + } catch (retryErr) { + const retryLeaseFenceCode = getManagedLeaseFenceErrorCode( + getUpstreamErrorIdentifier(retryErr) + ); + if (retryLeaseFenceCode) + return { + response: managedLeaseFenceErrorResult(retryLeaseFenceCode), + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; + // Refresh succeeded but the retry leg failed (network blip, AbortError, + // executor throw). Don't swallow — the operator-visible signal "the user + // saw 401 even though auth was actually fixed" is much more confusing + // than the original 401 alone. Surface at error level with sanitization. + log?.error?.( + "TOKEN", + `${provider?.toUpperCase()} | retry after refresh failed: ${sanitizeErrorMessage(retryErr)}` + ); + } + } else { + log?.warn?.("TOKEN", `${provider?.toUpperCase()} | refresh failed`); + if (isUnrecoverableRefreshError(newCredentials) && onCredentialsRefreshed) { + // Front 3 (reuse-race tolerance): before deactivating, re-read the DB. + // If a sibling/concurrent refresh already rotated this connection's + // refresh_token (common for Codex/OpenAI under one shared Auth0 client), + // the failure we saw was a stale-token reuse — the account is healthy + // with the newer token, so keep it active instead of killing it. + let alreadyRotated = false; + if (typeof connectionId === "string" && connectionId && attemptedRefreshToken) { + try { + const latest = await getProviderConnectionById(connectionId); + if (wasRefreshTokenRotated(attemptedRefreshToken, latest?.refreshToken)) { + alreadyRotated = true; + log?.warn?.( + "TOKEN", + `${provider.toUpperCase()} | refresh_token already rotated by a concurrent refresh — keeping connection active` + ); + } + } catch { + // DB read failed — fall through to the safe default (deactivate). + } + } + if (!alreadyRotated) { + await onCredentialsRefreshed({ testStatus: "expired", isActive: false }); + } + } + } + } + + // Check provider response - return error info for fallback handling + providerFailure: if (!providerResponse.ok) { + trackPendingRequest(model, provider, pendingConnId, false, undefined, pendingRequestId); + + let statusCode = providerResponse.status; + let message = ""; + let retryAfterMs: number | null = null; + let upstreamErrorCode: string | undefined; + let upstreamErrorType: string | undefined; + + if (upstreamErrorParsed) { + statusCode = parsedStatusCode; + message = parsedMessage; + retryAfterMs = parsedRetryAfterMs; + } else { + const details = await parseUpstreamError(providerResponse, provider); + statusCode = details.statusCode; + message = details.message; + retryAfterMs = details.retryAfterMs; + upstreamErrorBody = details.responseBody; + upstreamErrorCode = typeof details.errorCode === "string" ? details.errorCode : undefined; + upstreamErrorType = typeof details.errorType === "string" ? details.errorType : undefined; + } + + // Gateways like agentrouter misstate temporary quota exhaustion as 403/400, + // which downstream classification treats as AUTH_ERROR and clients like + // Claude Code treat as permanent. Restate to 429 (+ synthetic Retry-After) + // BEFORE any classification so both the fallback engine and the surfaced + // client status see a retryable error. Registry-scoped per provider. + const restatement = applyStatusRestatement({ + provider, + status: statusCode, + message, + body: upstreamErrorBody, + retryAfterMs, + }); + if (restatement.ruleId) { + statusCode = restatement.status; + retryAfterMs = restatement.retryAfterMs; + log?.info?.( + "STATUS_RESTATE", + `${provider} ${restatement.fromStatus}→${statusCode} (${restatement.ruleId})` + ); + } + + const signatureRecovery = pipelineRecovered + ? { attempted: false, succeeded: false, execution: null, error: null, recoveryBody: null } + : await recoverAnthropicThinkingSignature({ + provider, + statusCode, + message, + body: translatedBody, + execute: async (recoveryBody) => { + translatedBody = recoveryBody as typeof translatedBody; + syncExecuteTranslatedBody(translatedBody); + return executeProviderRequest(currentModel, false); + }, + parseError: (response) => parseUpstreamError(response, provider), + }); + if (!pipelineRecovered && signatureRecovery.attempted && signatureRecovery.execution) { + providerResponse = signatureRecovery.execution.response; + if (signatureRecovery.succeeded) { + providerUrl = signatureRecovery.execution.url; + providerHeaders = signatureRecovery.execution.headers; + finalBody = providerRequestCapture.body(signatureRecovery.execution.transformedBody); + reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody); + updatePendingScope(pendingScope, { + providerRequest: finalBody, + providerUrl, + stage: "provider_response_started", + }); + log?.info?.( + "THINKING_SIGNATURE", + `Recovered ${provider}/${currentModel} after one historical-thinking retry` + ); + } else if (signatureRecovery.error) { + statusCode = signatureRecovery.error.statusCode; + message = signatureRecovery.error.message; + retryAfterMs = signatureRecovery.error.retryAfterMs; + upstreamErrorBody = signatureRecovery.error.responseBody; + upstreamErrorCode = + typeof signatureRecovery.error.errorCode === "string" + ? signatureRecovery.error.errorCode + : undefined; + upstreamErrorType = + typeof signatureRecovery.error.errorType === "string" + ? signatureRecovery.error.errorType + : undefined; + } + } + + if (signatureRecovery.succeeded) break providerFailure; + + // #10281 — tiny-budget reasoning probes (e.g. Claude Code's `/model` check + // sends `max_tokens: 1`): the model burns the whole budget on thinking, and + // some upstreams (e.g. api.cline.bot for deepseek-v4-flash) answer the empty + // outcome with a 5xx ("empty response content") instead of a truncated 200. + // Answer such probes with a valid truncated response rather than relaying the + // upstream failure — which would also mark the connection unavailable and + // poison fallback/cooldown bookkeeping for a request that is only a probe. + if ( + !stream && + isTinyBudgetReasoningProbe({ model: currentModel, body: finalBody || translatedBody }) && + isEmptyContentUpstreamFailure(statusCode, message) + ) { + providerResponse = buildReasoningProbeTruncatedResponse({ + model: currentModel, + maxTokens: toPositiveInteger( + (finalBody || translatedBody)?.max_tokens ?? + (finalBody || translatedBody)?.max_completion_tokens + ), + requestId: skillRequestId, + }); + log?.warn?.( + "PROBE", + `Reasoning probe (max_tokens < ${REASONING_BUFFER_MIN_TRIGGER}) answered with truncated 200 — upstream reported "${message}"` + ); + break providerFailure; + } + + const errorConnectionId = getCurrentConnectionId() || connectionId; + await applyProviderFailureClassification({ + statusCode, + message, + headers: providerResponse.headers, + upstreamErrorBody, + retryAfterMs, + targetModel: currentModel, + }); + + appendRequestLog({ + model, + provider, + connectionId: errorConnectionId, + status: `FAILED ${statusCode}`, + }).catch(() => {}); + + const errMsg = formatProviderError(new Error(message), provider, model, statusCode); + const safeErrMsg = sanitizeErrorMessage(errMsg) || "Upstream provider error"; + const safeUpstreamErrorBody = sanitizeUpstreamDetails(upstreamErrorBody); + console.log(`${COLORS.red}[ERROR] ${safeErrMsg}${COLORS.reset}`); + + // Log Antigravity retry time if available + if (retryAfterMs && provider === "antigravity") { + const retrySeconds = Math.ceil(retryAfterMs / 1000); + log?.debug?.("RETRY", `Antigravity quota reset in ${retrySeconds}s (${retryAfterMs}ms)`); + } + + // Log error with full request body for debugging + reqLogger.logError(new Error(message), finalBody || translatedBody); + reqLogger.logProviderResponse( + providerResponse.status, + providerResponse.statusText, + providerResponse.headers, + safeUpstreamErrorBody + ); + + // Rate limiter updated in applyProviderFailureClassification + + // ── T5: Intra-family model fallback ────────────────────────────────────── + // Before returning a model-unavailable error upstream, try sibling models + // from the same family. This keeps the request alive on the same account + // instead of failing the entire combo. + if (!pipelineRecovered && isModelUnavailableError(statusCode, message, provider)) { + const nextModel = getNextFamilyFallback(currentModel, triedModels, provider); + if (nextModel) { + triedModels.add(nextModel); + currentModel = nextModel; + translatedBody.model = nextModel; + log?.info?.("MODEL_FALLBACK", `${model} unavailable (${statusCode}) → trying ${nextModel}`); + // Re-execute with the fallback model + try { + const fallbackResult = await executeProviderRequest(nextModel, false); + if (fallbackResult.response.ok) { + providerResponse = fallbackResult.response; + providerUrl = fallbackResult.url; + providerHeaders = fallbackResult.headers; + finalBody = providerRequestCapture.body(fallbackResult.transformedBody); + reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody); + updatePendingScope(pendingScope, { + providerRequest: finalBody, + providerUrl, + stage: "provider_response_started", + }); + // Continue processing with the fallback response — skip error return + log?.info?.("MODEL_FALLBACK", `Serving ${nextModel} as fallback for ${model}`); + // Jump to streaming/non-streaming handling below + // We fall through by NOT returning here + } else { + // Fallback also failed — return original error + persistAttemptLogs({ + status: statusCode, + error: safeErrMsg, + providerRequest: finalBody || translatedBody, + providerResponse: safeUpstreamErrorBody, + clientResponse: buildErrorBody(statusCode, errMsg), + cacheSource: "upstream", + }); + persistFailureUsage(statusCode, "model_unavailable"); + return { + response: createErrorResult( + statusCode, + errMsg, + retryAfterMs, + upstreamErrorCode, + upstreamErrorType, + upstreamErrorBody, + { passthrough: sourceFormat === FORMATS.CLAUDE } + ), + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; + } + } catch { + persistAttemptLogs({ + status: statusCode, + error: safeErrMsg, + providerRequest: finalBody || translatedBody, + providerResponse: safeUpstreamErrorBody, + clientResponse: buildErrorBody(statusCode, errMsg), + cacheSource: "upstream", + }); + persistFailureUsage(statusCode, "model_unavailable"); + return { + response: createErrorResult( + statusCode, + errMsg, + retryAfterMs, + upstreamErrorCode, + upstreamErrorType, + upstreamErrorBody, + { passthrough: sourceFormat === FORMATS.CLAUDE } + ), + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; + } + } else { + persistAttemptLogs({ + status: statusCode, + error: safeErrMsg, + providerRequest: finalBody || translatedBody, + providerResponse: safeUpstreamErrorBody, + clientResponse: buildErrorBody(statusCode, errMsg), + cacheSource: "upstream", + }); + persistFailureUsage(statusCode, "model_unavailable"); + return { + response: createErrorResult( + statusCode, + errMsg, + retryAfterMs, + upstreamErrorCode, + upstreamErrorType, + upstreamErrorBody, + { passthrough: sourceFormat === FORMATS.CLAUDE } + ), + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; + } + } else if (isContextOverflowError(statusCode, message)) { + const familyCandidates = getModelFamily(currentModel, provider).filter( + (m) => m !== currentModel && !triedModels.has(m) + ); + const nextModel = + findLargerContextModel(currentModel, familyCandidates, provider) ?? + getNextFamilyFallback(currentModel, triedModels, provider); + if (nextModel) { + triedModels.add(nextModel); + currentModel = nextModel; + translatedBody.model = nextModel; + log?.info?.("CONTEXT_OVERFLOW_FALLBACK", `${model} context overflow → trying ${nextModel}`); + try { + const fallbackResult = await executeProviderRequest(nextModel, false); + if (fallbackResult.response.ok) { + providerResponse = fallbackResult.response; + providerUrl = fallbackResult.url; + providerHeaders = fallbackResult.headers; + finalBody = providerRequestCapture.body(fallbackResult.transformedBody); + reqLogger.logTargetRequest(providerUrl, providerHeaders, finalBody); + updatePendingScope(pendingScope, { + providerRequest: finalBody, + providerUrl, + stage: "provider_response_started", + }); + log?.info?.( + "CONTEXT_OVERFLOW_FALLBACK", + `Serving ${nextModel} as fallback for ${model}` + ); + } else { + persistAttemptLogs({ + status: statusCode, + error: safeErrMsg, + providerRequest: finalBody || translatedBody, + providerResponse: safeUpstreamErrorBody, + clientResponse: buildErrorBody(statusCode, errMsg), + cacheSource: "upstream", + }); + persistFailureUsage(statusCode, "context_overflow"); + return { + response: createErrorResult( + statusCode, + errMsg, + retryAfterMs, + upstreamErrorCode, + upstreamErrorType, + upstreamErrorBody, + { passthrough: sourceFormat === FORMATS.CLAUDE } + ), + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; + } + } catch { + persistAttemptLogs({ + status: statusCode, + error: safeErrMsg, + providerRequest: finalBody || translatedBody, + providerResponse: safeUpstreamErrorBody, + clientResponse: buildErrorBody(statusCode, errMsg), + cacheSource: "upstream", + }); + persistFailureUsage(statusCode, "context_overflow"); + return { + response: createErrorResult( + statusCode, + errMsg, + retryAfterMs, + upstreamErrorCode, + upstreamErrorType, + upstreamErrorBody, + { passthrough: sourceFormat === FORMATS.CLAUDE } + ), + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; + } + } else { + persistAttemptLogs({ + status: statusCode, + error: safeErrMsg, + providerRequest: finalBody || translatedBody, + providerResponse: safeUpstreamErrorBody, + clientResponse: buildErrorBody(statusCode, errMsg), + cacheSource: "upstream", + }); + persistFailureUsage(statusCode, "context_overflow"); + return { + response: createErrorResult( + statusCode, + errMsg, + retryAfterMs, + upstreamErrorCode, + upstreamErrorType, + upstreamErrorBody, + { passthrough: sourceFormat === FORMATS.CLAUDE } + ), + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; + } + } else { + persistAttemptLogs({ + status: statusCode, + error: safeErrMsg, + providerRequest: finalBody || translatedBody, + providerResponse: safeUpstreamErrorBody, + clientResponse: buildErrorBody(statusCode, errMsg), + cacheSource: "upstream", + }); + persistFailureUsage(statusCode, `upstream_${statusCode}`); + + // Emergency budget fallback is orchestrated exclusively by the routing layer + // (src/sse/handlers/chat.ts), which resolves credentials FOR the emergency + // provider through account selection. The executor-level hop that used to + // live here re-sent the FAILING provider's credentials to the emergency + // provider's endpoint (e.g. the OpenAI API key to integrate.api.nvidia.com) + // — a cross-provider credential leak that also never succeeded upstream. + return { + response: createErrorResult( + statusCode, + errMsg, + retryAfterMs, + upstreamErrorCode, + upstreamErrorType, + upstreamErrorBody, + { passthrough: sourceFormat === FORMATS.CLAUDE } + ), + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; + } + // ── End T5 ─────────────────────────────────────────────────────────────── + } + return { + carry: { + translatedBody, + currentModel, + effectiveServiceTier, + finalBody, + pipelineRecovered, + providerHeaders, + providerResponse, + providerUrl, + }, + }; +} diff --git a/open-sse/handlers/chatCore/streamingTail.ts b/open-sse/handlers/chatCore/streamingTail.ts new file mode 100644 index 000000000000..93d8431a4101 --- /dev/null +++ b/open-sse/handlers/chatCore/streamingTail.ts @@ -0,0 +1,749 @@ +import { getCodexClientSessionId } from "../../config/codexIdentity.ts"; +import { + noteCodexTurnStateProvenance, + readCodexTurnStateHeader, +} from "../../config/codexTurnState.ts"; +import { + STREAM_DISCONNECT_GRACE_PERIOD_MS, + STREAM_READINESS_MAX_TIMEOUT_MS, + STREAM_READINESS_TIMEOUT_MS, + STREAM_RECOVERY, +} from "../../config/constants.ts"; +import { incrementTokenUsage } from "../../services/geminiRateLimitTracker.ts"; +import { requestTtftMs, streamEmittedOutput } from "../../utils/streamTiming.ts"; +import { clearPostOutputFailureStreak } from "../../services/accountFallback/postOutputFailureStreak.ts"; +import { + createRoutingEvent, + emitRoutingEvent, + outcomeFromStatus, +} from "../../services/routing/index.ts"; +import { FORMATS } from "../../translator/formats.ts"; +import { needsTranslation } from "../../translator/index.ts"; +import { runEmptyTurnRetryLoop } from "./emptyTurnRetryLoop.ts"; +import { buildErrorBody } from "../../utils/error.ts"; +import { + createPassthroughStreamWithLogger, + createSSETransformStreamWithLogger, +} from "../../utils/stream.ts"; +import * as streamFailure from "../../utils/streamFailureFinalization.ts"; +import { ensureStreamReadiness } from "../../utils/streamReadiness.ts"; +import { resolveStreamReadinessTimeout } from "../../utils/streamReadinessPolicy.ts"; +import { maybeFallbackAfterReadiness } from "./streamReadinessFallback.ts"; +import { captureStreamReasoningForReplay } from "./streamReasoningCapture.ts"; +import { meteredBudgetCost } from "@/lib/usage/meteredBudgetPolicy"; +import { resolveSuppressThinkClose } from "../../utils/thinkCloseMarker.ts"; +import { hasActiveClaudeThinking } from "../../utils/thinkingBudget.ts"; +import { buildCacheUsageLogMeta } from "./cacheUsageMeta.ts"; +import { recordContextEditingTelemetryHook } from "./contextEditingTelemetry.ts"; +import { readCpaAuthIndex } from "./failureUsage.ts"; +import { emitRequestGamificationEvent } from "./gamificationEvent.ts"; +import { maybeConvertJsonBodyToSse } from "./jsonBodyToSse.ts"; +import { runMemoryExtractionGate } from "./memoryExtraction.ts"; +import { mergeResponseToolNameMap } from "./passthroughToolNames.ts"; +import { runPluginOnResponseHook, runPluginOnStreamCompleteHook } from "./pluginOnResponse.ts"; +import { routingFinishReason } from "./routingFinishReason.ts"; +import { wrapReadableStreamWithFinalize } from "./streamFinalize.ts"; +import { buildStreamLedgerDetails, recordStreamingCost } from "./streamingCost.ts"; +import { assembleStreamingPipeline } from "./streamingPipeline.ts"; +import { scheduleStreamingQuotaShareConsumption } from "./streamingQuotaShare.ts"; +import { assembleStreamingResponseHeaders } from "./streamingResponseHeaders.ts"; +import { storeStreamingSemanticCacheResponse } from "./streamingSemanticCacheStore.ts"; +import { recordStreamingUsageStats } from "./streamingUsageStats.ts"; +import { maybeSyncClaudeExtraUsageState } from "./telemetryHelpers.ts"; +import { getExecutorTimeoutMs, resolveConnectionTimeoutMs } from "./upstreamTimeouts.ts"; +import { recordCost } from "@/domain/costRules"; +import { extractFacts } from "@/lib/memory/extraction"; +import { calculateCost } from "@/lib/usage/costCalculator"; +import { appendRequestLog, trackPendingRequest } from "@/lib/usageDb"; +import { isFeatureFlagEnabled } from "@/shared/utils/featureFlags.ts"; +import { getProviderCredentials } from "@/sse/services/auth"; + +// eslint-disable-next-line @typescript-eslint/no-explicit-any -- deps bag mirrors the barrel closure +type Loose = any; + +export type StreamingTailDeps = Record; + +export async function runStreamingTail(deps: StreamingTailDeps) { + const { + agentGoalPolicy, + apiKeyInfo, + attachCompressionUsageReceiptAfterAnalytics, + body, + bodyForCacheWrite, + claudePromptCacheLogMeta, + clientRawRequest, + clientResponseFormat, + comboStrategy, + compressionResponseMeta, + connectionId, + contextEditingEnabled, + copilotCompatibleReasoning, + correlationId, + createPiiTransform, + credentials, + currentModel, + customToolNames, + echoModel, + endpointPath, + executeProviderRequest, + executor, + fallbackAttempts, + getCurrentConnectionId, + forcedConnectionId, + isCombo, + isDroidCLI, + isResponsesEndpoint, + log, + managedLease, + memoryOwnerId, + memorySettings, + model, + modelInfo, + onRequestSuccess, + onStreamFailure, + pendingConnId, + pendingRequestId, + persistAttemptLogs, + persistFailureUsage, + pipelineSessionId, + provider, + providerHeaders, + providerRequestCapture, + providerUrl, + reasoningCacheScope, + reasoningReplayHistory, + releaseTurnExecution, + reqLogger, + requestToolIdentityMap, + resolveReportedServiceTier, + semanticCacheEnabled, + skillRequestId, + startTime, + stream, + streamController, + streamUserAgent, + targetFormat, + thinkingMarkerHeader, + toolNameMap, + traceId, + translatedBody, + videoBridgeObserved, + } = deps; + let providerResponse, finalBody, effectiveServiceTier; + providerResponse = deps.providerResponse; + finalBody = deps.finalBody; + effectiveServiceTier = deps.effectiveServiceTier; + let onPipelineStreamError = null; + let onClientDisconnectFinalize = null; + let turnExecutionHandedOffToStream = false; + + // Streaming response + // #3089 — some "reasoning" openai-compatible upstreams ignore a stream:true + // request and return a complete application/json chat-completion body instead + // of an SSE stream. The readiness check below only recognizes SSE `data:` + // frames, so that body produced a spurious STREAM_EARLY_EOF / HTTP 502 even + // though it carried valid content/reasoning_content. Detect a JSON (non-SSE) + // upstream body and synthesize an equivalent OpenAI SSE stream so the + // streaming pipeline (and the client) get a valid stream. + providerResponse = await maybeConvertJsonBodyToSse(providerResponse, { log, provider, model }); + const streamReadinessPolicy = resolveStreamReadinessTimeout({ + baseTimeoutMs: STREAM_READINESS_TIMEOUT_MS, + provider, + model, + body: (finalBody || translatedBody) as Record | null | undefined, + sourceBody: body as Record | null | undefined, + maxTimeoutMs: agentGoalPolicy.detected + ? Math.max(STREAM_READINESS_MAX_TIMEOUT_MS, agentGoalPolicy.readinessMaxTimeoutMs) + : STREAM_READINESS_MAX_TIMEOUT_MS, + cascadeTimeoutMs: getExecutorTimeoutMs( + executor, + provider, + model, + resolveConnectionTimeoutMs(credentials?.providerSpecificData) + ), + }); + if (streamReadinessPolicy.timeoutMs !== streamReadinessPolicy.baseTimeoutMs) { + log?.debug?.( + "STREAM", + `adaptive readiness timeout=${streamReadinessPolicy.timeoutMs}ms base=${streamReadinessPolicy.baseTimeoutMs}ms reason=${streamReadinessPolicy.reasons.join(",")}` + ); + } + + let streamReadiness = await ensureStreamReadiness(providerResponse, { + timeoutMs: streamReadinessPolicy.timeoutMs, + maxTimeoutMs: streamReadinessPolicy.maxTimeoutMs, + provider, + model, + log, + }); + // A stall is an upstream issue, not an account fault — the executor loop + // already ended at headers, so this bounded retry is the only recovery left. + const fallback = await maybeFallbackAfterReadiness({ + streamReadiness, + clientAborted: streamController.signal.aborted, + failedConnectionId: getCurrentConnectionId(), + failedBody: providerResponse, + currentModel, + streamReadinessPolicy, + provider, + model, + log, + reqLogger, + providerUrl, + providerHeaders, + finalBody, + translatedBody, + executeProviderRequest, + providerRequestCapture, + }); + streamReadiness = fallback.readiness; + providerResponse = fallback.providerResponse; + finalBody = fallback.finalBody; + if (streamReadiness.ok === false) { + const { response: failureResponse, reason } = streamReadiness; + const { classificationReason, upstreamDiagnostic } = streamReadiness; + trackPendingRequest(model, provider, pendingConnId, false, undefined, pendingRequestId); + appendRequestLog({ + model, + provider, + connectionId, + status: `FAILED ${failureResponse.status}`, + }).catch(() => {}); + persistAttemptLogs({ + status: failureResponse.status, + error: reason, + providerRequest: finalBody || translatedBody, + clientResponse: buildErrorBody( + failureResponse.status, + classificationReason, + upstreamDiagnostic ? { error: { message: upstreamDiagnostic } } : undefined + ), + claudeCacheMeta: claudePromptCacheLogMeta, + cacheSource: "upstream", + }); + persistFailureUsage(failureResponse.status, streamReadiness.code); + // Do NOT call onStreamFailure — a stream stall is an upstream issue, + // not an account/quota failure. Marking the account unavailable here + // would lock out legitimate accounts when the upstream hangs. + return { + result: { + success: false, + status: failureResponse.status, + error: reason, + classificationError: classificationReason, + errorType: streamReadiness.type, + errorCode: streamReadiness.code, + response: failureResponse, + }, + carry: { onPipelineStreamError, onClientDisconnectFinalize, turnExecutionHandedOffToStream }, + }; + } + providerResponse = streamReadiness.response; + + // Flush-empty retry (opt-in `FLUSH_EMPTY_RETRY_ENABLED`, default off): when the + // upstream turn carries no usable content (reasoning-only 200, or a + // zero-valuable-chunk turn that the empty-stream guard would turn into a 502), + // issue bounded retries through the normal credential path BEFORE anything is + // exposed to the client — in particular before `onRequestSuccess` below. + // Empty turns are stochastic upstream misses, not account faults, so no + // cooldown and no forced exclusion: the round-robin picker may rotate + // fingerprint slots opportunistically, a single slot simply replays the same + // account. Budget: `STREAM_RECOVERY.EMPTY_TURN_RETRY_MAX` retries, then fall + // back to the current behavior. Translate-path streams only (mirror of the + // empty-stream guard); flag off = byte-for-byte unchanged. Bounded reader + // (abandon past the cap, never a full `text()` read); the original + // reconstructed response is piped, only the bounded copy is classified. + // Known TTFT cost when armed: a small valid turn under the cap is fully + // buffered before the first client byte (flag off by default, so the + // streaming path is untouched unless opted in). + if (stream && providerResponse.ok && providerResponse.body) { + let flushEmptyRetryArmed = false; + try { + flushEmptyRetryArmed = isFeatureFlagEnabled("FLUSH_EMPTY_RETRY_ENABLED"); + } catch { + flushEmptyRetryArmed = false; + } + const isTranslatePath = + targetFormat === FORMATS.OPENAI_RESPONSES || + needsTranslation(targetFormat, clientResponseFormat); + if (flushEmptyRetryArmed && isTranslatePath) { + const retried = await runEmptyTurnRetryLoop({ + providerResponse, + credentials, + provider, + currentModel, + model, + targetFormat, + clientResponseFormat, + isAborted: () => clientRawRequest?.signal?.aborted === true, + timeoutMs: streamReadinessPolicy.timeoutMs, + maxTimeoutMs: streamReadinessPolicy.maxTimeoutMs, + maxRetries: STREAM_RECOVERY.EMPTY_TURN_RETRY_MAX, + translatedBody, + finalBody, + providerUrl, + providerHeaders, + correlationId, + traceId, + log, + getProviderCredentials, + routing: { leased: Boolean(managedLease), forcedConnectionId, apiKey: apiKeyInfo }, + executeProviderRequest, + logTargetRequest: (url, headers, body) => reqLogger.logTargetRequest(url, headers, body), + captureBody: (body) => providerRequestCapture.body(body), + }); + providerResponse = retried.providerResponse; + if (retried.adopted) finalBody = retried.finalBody; + } + } + + // Notify success - caller can clear error status if needed + if (onRequestSuccess) { + await onRequestSuccess(); + } + + const responseHeaders = assembleStreamingResponseHeaders({ + providerHeaders: providerResponse.headers, + provider, + model, + pendingRequestId, + compressionResponseMeta, + comboStrategy, + fallbackAttempts, + isCombo, // #14116: foreign-account quota-header strip (only meaningful when true) + requestedConnectionId: forcedConnectionId || null, + selectedConnectionId: credentials?.connectionId ?? null, + }); + + // The streaming headers (turn-state included, when present) are committed to + // the client from here on — record which connection minted the blob so a + // later cross-account echo can be stripped (Codex failover guard). The + // in-place failover update means `credentials` is the winning account. + if (provider === "codex" && readCodexTurnStateHeader(providerResponse.headers)) { + noteCodexTurnStateProvenance( + getCodexClientSessionId(clientRawRequest?.headers), + credentials?.connectionId + ); + } + + // Create transform stream with logger for streaming response + let transformStream; + const responseToolNameMap = mergeResponseToolNameMap( + toolNameMap, + (finalBody as Record | null | undefined) ?? null + ); + + let streamCompletionRecorded = false; + let streamFailureCompletionRecorded = false; + + // Callback to save call log when stream completes (include responseBody when provided by stream) + let streamTimingOriginOffsetMs: number | null = null; // startTime → StreamTiming start + const onStreamComplete = ({ + status: streamStatus, + usage: streamUsage, + responseBody: streamResponseBody, + providerPayload, + clientPayload, + reasoningMeta: streamReasoningMeta, + error: streamError, + errorCode: streamErrorCode, + firstOutputMs, + itlMs: streamItlMs, + interrupted: _streamInterrupted, + }) => { + const ttft = requestTtftMs(streamTimingOriginOffsetMs, firstOutputMs); + const normalizedStreamStatus = streamStatus || 200; + if (streamCompletionRecorded) return; + streamCompletionRecorded = true; + if (normalizedStreamStatus !== 200) { + if (streamFailureCompletionRecorded) return; + streamFailureCompletionRecorded = true; + } + const cacheUsageLogMeta = buildCacheUsageLogMeta(streamUsage); + const streamConnectionId = getCurrentConnectionId(); + + if (normalizedStreamStatus === 200) { + clearPostOutputFailureStreak(provider, streamConnectionId, modelInfo.model); + void maybeSyncClaudeExtraUsageState({ + provider, + connectionId: streamConnectionId, + providerSpecificData: credentials?.providerSpecificData, + log, + }); + } + + if (normalizedStreamStatus === 200 && streamResponseBody) { + captureStreamReasoningForReplay({ + streamResponseBody, + clientResponseFormat, + responseToolNameMap, + providerRequestBody: finalBody || translatedBody || body, + translatedBody, + reasoningReplayHistory, + provider, + model, + reasoningCacheScope, + videoTranscriptSensitive: videoBridgeObserved, + }); + } + effectiveServiceTier = resolveReportedServiceTier(streamResponseBody) ?? effectiveServiceTier; + + // Context Editing telemetry (streaming): the reconstructed stream body now carries + // context_management.applied_edits from the final message_delta snapshot. Mirror the + // non-streaming hook so streaming context-clear savings also surface under engine + // "context-editing" in compression analytics. Best-effort, Claude-only. + if (normalizedStreamStatus === 200) { + recordContextEditingTelemetryHook({ + contextEditingEnabled, + provider, + responseBody: streamResponseBody, + skillRequestId, + log, + }); + } + + streamFailure.finalizeStreamRequestLog({ + pendingRequestId, + model, + provider, + connectionId: streamConnectionId, + providerResponse: providerPayload ?? streamResponseBody ?? undefined, + clientResponse: clientPayload ?? streamResponseBody ?? undefined, + status: normalizedStreamStatus, + error: streamError, + errorCode: streamErrorCode, + }); + + // Track cache token metrics for streaming responses + if (streamUsage && typeof streamUsage === "object") { + attachCompressionUsageReceiptAfterAnalytics(streamUsage as Record, "stream"); + // Track Gemini token consumption for TPM rate-limit pre-check + if (provider === "gemini") { + const promptTokens = + typeof (streamUsage as Record).prompt_tokens === "number" + ? ((streamUsage as Record).prompt_tokens as number) + : 0; + if (promptTokens > 0) incrementTokenUsage(model, promptTokens); + } + } + recordStreamingUsageStats(streamUsage, { + provider, + model, + streamStatus: normalizedStreamStatus, + startTime, + ttft, + streamErrorCode, + connectionId: streamConnectionId, + apiKeyInfo, + effectiveServiceTier, + isCombo, + comboStrategy, + endpoint: endpointPath, + cpaAuthIndex: readCpaAuthIndex(providerResponse), + }); + + // Routing event (feedback foundation) — fire-and-forget, cheap, never blocks + // the stream. Feeds the quality tracker + optional OTel exporter. + void emitRoutingEvent( + createRoutingEvent({ + requestId: traceId || pendingRequestId || "unknown", + provider: provider || "unknown", + model: model || "unknown", + strategy: isCombo ? (comboStrategy ?? "combo") : "direct", + latencyMs: Date.now() - startTime, + ttftMs: typeof ttft === "number" && Number.isFinite(ttft) && ttft >= 0 ? ttft : null, + itlMs: + typeof streamItlMs === "number" && Number.isFinite(streamItlMs) && streamItlMs >= 0 + ? streamItlMs + : null, + inputTokens: + streamUsage && typeof streamUsage === "object" + ? (() => { + const promptTokens = (streamUsage as Record).prompt_tokens; + return typeof promptTokens === "number" && Number.isFinite(promptTokens) + ? promptTokens + : null; + })() + : null, + outputTokens: + streamUsage && typeof streamUsage === "object" + ? (() => { + const completionTokens = (streamUsage as Record).completion_tokens; + return typeof completionTokens === "number" && Number.isFinite(completionTokens) + ? completionTokens + : null; + })() + : null, + cost: null, + retries: 0, + fallbackUsed: false, // combo-level fallback tracked by decisionTrace + outcome: + normalizedStreamStatus === 200 + ? "success" + : streamErrorCode === "stream_interrupted" || streamErrorCode === "aborted" + ? "stream_interrupted" + : outcomeFromStatus(normalizedStreamStatus), + status: normalizedStreamStatus, + finishReason: routingFinishReason(streamResponseBody), + connectionId: streamConnectionId ?? credentials?.connectionId ?? null, + }) + ); + + persistAttemptLogs({ + status: normalizedStreamStatus, + error: streamError || undefined, + tokens: streamUsage || {}, + responseBody: streamResponseBody ?? undefined, + providerRequest: finalBody || translatedBody, + providerResponse: providerPayload, + clientResponse: clientPayload ?? streamResponseBody ?? undefined, + claudeCacheMeta: claudePromptCacheLogMeta, + claudeCacheUsageMeta: cacheUsageLogMeta, + cacheSource: "upstream", + // #13130: persist TTFT so call_logs.ttft_ms lets the dashboard compute + // generation-time TPS instead of wall-clock TPS. + ttft, + reasoningMeta: streamReasoningMeta ?? null, + }); + + recordStreamingCost({ + apiKeyId: apiKeyInfo?.id, + provider, + model, + streamUsage, + serviceTier: effectiveServiceTier, + calculateCost, + // Only the budget-consumable share may draw down the allowance. + recordCost: (apiKeyId, cost, details) => { + const budgetCost = meteredBudgetCost(provider, cost); + if (budgetCost > 0) recordCost(apiKeyId, budgetCost, details); + }, + ledger: buildStreamLedgerDetails(effectiveServiceTier, normalizedStreamStatus < 400, traceId), + }); + + // === Quota Share POST-hook streaming (B/F7) — fire-and-forget, fail-open === + // Resolve the real per-request cost (calculateCost) so USD-unit pools accrue + // on streaming traffic too; this previously recorded usd:0 hardcoded, which + // meant DeepSeek-style `usd/monthly` shared pools never blocked on streams. + scheduleStreamingQuotaShareConsumption({ + apiKeyId: apiKeyInfo?.id, + connectionId: credentials?.connectionId, + provider, + model, + streamUsage, + streamStatus: normalizedStreamStatus, + serviceTier: effectiveServiceTier, + calculateCost, + log, + }); + // === /Quota Share POST-hook streaming === + + if (streamStatus === 200) { + // #12150 P1b surface 3 (fix round 1): see the matching non-streaming + // gate above — an observed request populates NO durable memory from + // either the request-derived text or this streamed response. + runMemoryExtractionGate({ + memoryOwnerId, + memorySettings, + videoBridgeObserved, + pipelineSessionId, + requestBody: body as Record, + responseBody: (streamResponseBody ?? null) as Record | null, + extractFacts, + log, + }); + } + + // Semantic cache: store assembled streaming response for future cache hits + storeStreamingSemanticCacheResponse({ + enabled: semanticCacheEnabled, + streamStatus, + streamResponseBody, + body: bodyForCacheWrite, + headers: clientRawRequest?.headers, + model, + provider, + apiKeyId: apiKeyInfo?.id ?? undefined, + streamUsage, + log, + videoTranscriptSensitive: videoBridgeObserved, + }); + + // Plugin onStreamComplete hook — fire-and-forget, fail-open (#9571) + // Pass traceId as requestId so plugins can correlate the stream-completion event + // with the originating request (the same id used for onRequest/onResponse). (#11825) + runPluginOnStreamCompleteHook({ + status: normalizedStreamStatus, + usage: streamUsage as Record | undefined, + ttft, + model, + provider, + errorCode: streamErrorCode, + startTime, + requestId: traceId, + }); + }; + + const streamFailureFinalizers = streamFailure.createStreamFailureFinalizers({ + isFailureCompletionRecorded: () => streamFailureCompletionRecorded, + isStreamCompletionRecorded: () => streamCompletionRecorded, + onStreamComplete, + persistFailureUsage, + onStreamFailure, + hasEmittedOutput: () => streamEmittedOutput(transformStream), + }); + const handleStreamFailure = streamFailureFinalizers.handleStreamFailure; + onPipelineStreamError = streamFailureFinalizers.onPipelineStreamError; + // #9653: gives a genuine, race-delayed completion a chance to land (see + // createClientDisconnectGraceHandler's doc comment) before persisting a false + // 499/0-tokens for a request that actually delivered its full response. + onClientDisconnectFinalize = streamFailure.createClientDisconnectGraceHandler({ + isStreamCompletionRecorded: () => streamCompletionRecorded, + gracePeriodMs: STREAM_DISCONNECT_GRACE_PERIOD_MS, + finalize: (event) => + handleStreamFailure({ + status: 499, + message: `Client disconnected: ${event.reason}`, + code: "client_disconnected", + type: "client_disconnected", + }), + }); + + // For providers using Responses API format, translate stream back to openai (Chat Completions) format + // UNLESS client is Droid CLI which expects openai-responses format back + const needsResponsesTranslation = + targetFormat === FORMATS.OPENAI_RESPONSES && + clientResponseFormat === FORMATS.OPENAI && + !isResponsesEndpoint && + !isDroidCLI; + const streamStateBody = finalBody || body; + + // Client's explicit thinking intent (Anthropic Messages shape). Claude Code + // sends `{type:"enabled"}` or `{type:"adaptive"}` to opt into relaying + // upstream reasoning_content as Claude thinking blocks; `{type:"disabled"}` + // or an omitted `thinking` field opts out. Kept false for every other + // client schema (OpenAI / Responses), which never express intent through + // `body.thinking`. Mirrors hasActiveClaudeThinking() so the request and + // response sides agree on what counts as "thinking requested" — a prior + // inline `=== "enabled"` check silently suppressed `adaptive` (the intent + // Claude Code actually sends), leaking the mismatch as a broken tool-call + // turn (call log 1787566395384-bab9ab: reasoning dropped → model emitted + // DSML tool-call markers as plain text → incomplete `stop` finish). + const requestedThinking = hasActiveClaudeThinking((body ?? {}) as Record); + + streamTimingOriginOffsetMs = Date.now() - startTime; + if (needsResponsesTranslation) { + // Provider returns openai-responses, translate to openai (Chat Completions) that clients expect + log?.debug?.("STREAM", `Responses translation mode: openai-responses → openai`); + transformStream = createSSETransformStreamWithLogger( + "openai-responses", + "openai", + provider, + reqLogger, + responseToolNameMap, + model, + connectionId, + streamStateBody, + onStreamComplete, + apiKeyInfo, + handleStreamFailure, + copilotCompatibleReasoning, + false, + requestedThinking, + customToolNames, + // openai-responses → openai translation still wants the namespace identity + // map for #7936-style round-trip closure when the client also speaks + // Responses (Codex CLI). + requestToolIdentityMap + ); + } else if (needsTranslation(targetFormat, clientResponseFormat)) { + // Standard translation for other providers + log?.debug?.("STREAM", `Translation mode: ${targetFormat} → ${clientResponseFormat}`); + transformStream = createSSETransformStreamWithLogger( + targetFormat, + clientResponseFormat, + provider, + reqLogger, + responseToolNameMap, + model, + connectionId, + streamStateBody, + onStreamComplete, + apiKeyInfo, + handleStreamFailure, + copilotCompatibleReasoning, + // Suppress the `` close marker for clients that render it verbatim + // (e.g. OpenCode by UA; any client via `x-omniroute-thinking-marker: off`); + // preserved for Claude Code / Cursor and unknown clients by default (#5245 / + // #5312). Responses API clients always suppress it (structured reasoning + // items make the marker meaningless); otherwise the header wins over the + // UA allowlist. + resolveSuppressThinkClose({ + userAgent: streamUserAgent, + thinkingMarkerHeader, + clientResponseFormat, + }), + requestedThinking, + customToolNames, + requestToolIdentityMap + ); + } else { + log?.debug?.("STREAM", `Standard passthrough mode`); + transformStream = createPassthroughStreamWithLogger( + provider, + reqLogger, + responseToolNameMap, + model, + connectionId, + streamStateBody, + onStreamComplete, + apiKeyInfo, + handleStreamFailure, + clientResponseFormat, + requestToolIdentityMap + ); + } + + const finalStream = assembleStreamingPipeline({ + providerResponse, + transformStream, + streamController, + createPiiTransform, + clientRawRequestHeaders: clientRawRequest?.headers, + clientResponseFormat, + echoModel, + responseHeaders, + // Same adaptive budget the pre-handoff readiness gate above just used — + // reasoning models that legitimately take a while to say anything keep + // that same patience for their first REAL content, not just their first + // lifecycle frame. See pipeWithDisconnect's own doc comment. + contentStallTimeoutMs: streamReadinessPolicy.timeoutMs, + }); + const clientFacingStream = wrapReadableStreamWithFinalize(finalStream, releaseTurnExecution); + + // ── Gamification event (fire-and-forget) ── + await emitRequestGamificationEvent({ apiKeyId: apiKeyInfo?.id, model, provider }); + + // ── Plugin onResponse hook (fire-and-forget) ── + await runPluginOnResponseHook({ + requestId: traceId, + body, + model, + provider, + apiKeyInfo, + headers: clientRawRequest?.headers, + response: { status: 200, streamed: true }, + }); + + const response = new Response(clientFacingStream, { + headers: responseHeaders, + }); + turnExecutionHandedOffToStream = true; + return { + result: { + success: true, + response, + }, + carry: { onPipelineStreamError, onClientDisconnectFinalize, turnExecutionHandedOffToStream }, + }; +} diff --git a/package.json b/package.json index af49b7a541d2..cc527986bb96 100644 --- a/package.json +++ b/package.json @@ -131,17 +131,17 @@ "electron:build:mac": "npm run build && cd electron && npm run build:mac", "electron:build:linux": "npm run build && cd electron && npm run build:linux", "electron:smoke:packaged": "node scripts/dev/smoke-electron-packaged.mjs", - "test": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", - "test:unit": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", - "test:unit:ci": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", - "test:unit:ci:shard": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 --test-shard=$TEST_SHARD tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 --test-shard=$TEST_SHARD \"tests/unit/dashboard/**/*.test.ts\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=1 --test-shard=$TEST_SHARD \"tests/unit/serial/**/*.test.ts\"", - "test:unit:fast": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-isolation=none tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-isolation=none \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", + "test": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", + "test:unit": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", + "test:unit:ci": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", + "test:unit:ci:shard": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 --test-shard=$TEST_SHARD tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=4 --test-shard=$TEST_SHARD \"tests/unit/dashboard/**/*.test.ts\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=1 --test-shard=$TEST_SHARD \"tests/unit/serial/**/*.test.ts\"", + "test:unit:fast": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-isolation=none tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-isolation=none \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", "test:scoped": "bash scripts/quality/test-scoped.sh", "test:scoped:staged": "bash scripts/quality/test-scoped.sh --staged", "test:scoped:full": "bash scripts/quality/test-scoped.sh --full", "test:unit:shard": "concurrently --kill-others-on-fail -n s1,s2 \"npm:test:unit:shard:1\" \"npm:test:unit:shard:2\"", - "test:unit:shard:1": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=1/2 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=1/2 \"tests/unit/dashboard/**/*.test.ts\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=1 --test-shard=1/2 \"tests/unit/serial/**/*.test.ts\"", - "test:unit:shard:2": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=2/2 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=2/2 \"tests/unit/dashboard/**/*.test.ts\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=1 --test-shard=2/2 \"tests/unit/serial/**/*.test.ts\"", + "test:unit:shard:1": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=1/2 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=1/2 \"tests/unit/dashboard/**/*.test.ts\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=1 --test-shard=1/2 \"tests/unit/serial/**/*.test.ts\"", + "test:unit:shard:2": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=2/2 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=10 --test-shard=2/2 \"tests/unit/dashboard/**/*.test.ts\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=1 --test-shard=2/2 \"tests/unit/serial/**/*.test.ts\"", "test:bun:db": "bun test tests/unit/db-adapters/bunSqliteAdapter.test.ts tests/unit/db-adapters/driverFactory.test.ts tests/unit/db-adapters/cliSqlite.test.mjs", "test:plan3": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test tests/unit/plan3-p0.test.ts", "test:fixes": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test tests/unit/fixes-p1.test.ts", @@ -294,7 +294,7 @@ "release:contributors": "node scripts/release/gen-contributors.mjs", "release:uncovered": "node scripts/release/list-uncovered-commits.mjs", "release:reconcile": "node scripts/release/reconcile-changelog.mjs", - "test:coverage:runner": "node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=8 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true NODE_OPTIONS=--max-old-space-size=8192 c8 --merge-async --output-dir coverage --exclude=tests/** --exclude=**/*.test.* --reporter=text-summary --reporter=html --reporter=json-summary --reporter=lcov --check-coverage --statements 60 --lines 60 --functions 60 --branches 60 node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=8 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", + "test:coverage:runner": "node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=8 tests/unit/*.test.ts \"tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts\" \"tests/unit/**/*.test.mjs\" && cross-env DISABLE_SQLITE_AUTO_BACKUP=true NODE_OPTIONS=--max-old-space-size=8192 c8 --merge-async --output-dir coverage --exclude=tests/** --exclude=**/*.test.* --reporter=text-summary --reporter=html --reporter=json-summary --reporter=lcov --check-coverage --statements 60 --lines 60 --functions 60 --branches 60 node --max-old-space-size=8192 --import tsx --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=8 \"tests/unit/dashboard/**/*.test.ts\" && npm run test:unit:serial", "test:unit:serial": "cross-env DISABLE_SQLITE_AUTO_BACKUP=true node --max-old-space-size=8192 --import tsx/esm --import ./open-sse/utils/setupPolyfill.ts --import ./tests/_setup/isolateDataDir.ts --test --test-force-exit --test-concurrency=1 \"tests/unit/serial/**/*.test.ts\"", "alibaba:sync-allowlist": "node --import tsx/esm scripts/ops/sync-alibaba-allowlist.mjs", "check:vitest-exclusions": "node scripts/check/check-vitest-exclusions.mjs", diff --git a/scripts/check/check-test-discovery.mjs b/scripts/check/check-test-discovery.mjs index 89301ebb4556..b3060b459320 100644 --- a/scripts/check/check-test-discovery.mjs +++ b/scripts/check/check-test-discovery.mjs @@ -58,7 +58,7 @@ export const COLLECTORS = [ // abaixo). Subdir novo: adicione aqui E nos scripts (o drift-check + o gate de // órfãos forçam a manutenção em sincronia). { - glob: "tests/unit/{api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts", + glob: "tests/unit/{api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage}/**/*.test.ts", sources: ["package.json"], }, // Node native runner — tests/unit/dashboard/** roda numa invocação separada com o hook diff --git a/scripts/dev/diff-chatcore-leaves.mjs b/scripts/dev/diff-chatcore-leaves.mjs new file mode 100644 index 000000000000..9fe065a82c53 --- /dev/null +++ b/scripts/dev/diff-chatcore-leaves.mjs @@ -0,0 +1,400 @@ +#!/usr/bin/env node +// scripts/dev/diff-chatcore-leaves.mjs +// +// Region diff that proves the four response-path leaves in open-sse/handlers/chatCore/ are +// mechanical lifts of the monolithic handleChatCore on the base ref. +// +// For every leaf it (1) slices the lifted region out of the BASE chatCore.ts (region.base: +// startPattern/endPattern), (2) slices the corresponding body out of the leaf (region.leaf), (3) +// tokenizes both with the TypeScript scanner (comments, whitespace and prettier re-wrap noise +// such as trailing commas vanish), (4) runs a Myers token diff and groups the edits into hunks. +// A hunk is "mechanical" only when it is one of the documented lift seams (see classifyHunk): +// the syncExecuteTranslatedBody(translatedBody) re-sync, the `return X` -> +// `return { response: X, carry: {...} }` wrapping with the exact per-leaf carry keys, and the +// region-boundary tokens that stay behind in handleChatCore. Every other hunk is printed as +// NON-MECHANICAL and the script exits 1 (exit 2 = a region could not be extracted). +// +// Usage: node scripts/dev/diff-chatcore-leaves.mjs [--base ] [--verbose] +// default base: origin/release/v3.8.52 (the pre-split chatCore.ts that the leaves lift from). + +import { execFileSync } from "node:child_process"; +import fs from "node:fs"; +import path from "node:path"; +import { createRequire } from "node:module"; +import { fileURLToPath } from "node:url"; + +const require = createRequire(import.meta.url); +const ts = require("typescript"); + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../.."); +const argv = process.argv.slice(2); +const flag = (name) => argv.includes(name); +const opt = (name, fallback) => (argv.includes(name) ? argv[argv.indexOf(name) + 1] : fallback); +const BASE_REF = opt("--base", "origin/release/v3.8.52"); +const VERBOSE = flag("--verbose"); +const BASE_FILE = "open-sse/handlers/chatCore.ts"; + +/** Normalize a source slice to comparable lines (drops blanks and comment-only lines). */ +export function normalize(text) { + const out = []; + let inBlock = false; + for (const raw of text.split("\n")) { + let line = raw.trim(); + if (inBlock) { + if (line.includes("*/")) inBlock = false; + continue; + } + if (line.startsWith("/*")) { + if (!line.includes("*/")) inBlock = true; + continue; + } + if (line === "" || line.startsWith("//")) continue; + line = line.replace(/\s+/g, " "); + // prettier re-wraps change trailing commas / semicolons-only differences after dedent + out.push(line); + } + return out; +} + +/** + * Slice a region out of NORMALIZED lines. The region starts at the first `startPattern` match + * (searched after the first `after` match when given) and ends at the first `endPattern` match + * after the start (the LAST match when `endMode: "last"`). The end line is excluded unless + * `inclusiveEnd`. + */ +export function sliceRegion( + lines, + { after, startPattern, endPattern, endMode = "first", inclusiveEnd } +) { + let from = 0; + if (after) { + from = lines.findIndex((l) => after.test(l)); + if (from < 0) throw new Error(`anchor ${after} not found`); + } + let s = -1; + for (let i = from; i < lines.length; i++) { + if (startPattern.test(lines[i])) { + s = i; + break; + } + } + if (s < 0) throw new Error(`start pattern ${startPattern} not found`); + let e = -1; + if (endMode === "last") { + for (let i = lines.length - 1; i > s; i--) { + if (endPattern.test(lines[i])) { + e = i; + break; + } + } + } else { + for (let i = s + 1; i < lines.length; i++) { + if (endPattern.test(lines[i])) { + e = i; + break; + } + } + } + if (e < 0) throw new Error(`end pattern ${endPattern} not found after the start`); + return lines.slice(s, inclusiveEnd ? e + 1 : e); +} + +/** Tokenize TypeScript source (comments/whitespace dropped, prettier-wrap noise removed). */ +export function tokenize(text) { + const scanner = ts.createScanner(ts.ScriptTarget.ES2022, true, ts.LanguageVariant.Standard, text); + const tokens = []; + const stack = []; // "tpl" | "brace" + let prevKind = ts.SyntaxKind.Unknown; + const exprEnd = new Set([ + ts.SyntaxKind.Identifier, + ts.SyntaxKind.CloseParenToken, + ts.SyntaxKind.CloseBracketToken, + ts.SyntaxKind.NumericLiteral, + ts.SyntaxKind.StringLiteral, + ts.SyntaxKind.NoSubstitutionTemplateLiteral, + ts.SyntaxKind.TemplateTail, + ts.SyntaxKind.RegularExpressionLiteral, + ts.SyntaxKind.ThisKeyword, + ts.SyntaxKind.TrueKeyword, + ts.SyntaxKind.FalseKeyword, + ts.SyntaxKind.NullKeyword, + ]); + for (let kind = scanner.scan(); kind !== ts.SyntaxKind.EndOfFileToken; kind = scanner.scan()) { + if (kind === ts.SyntaxKind.CloseBraceToken && stack[stack.length - 1] === "tpl") { + kind = scanner.reScanTemplateToken(false); + if (kind === ts.SyntaxKind.TemplateTail) stack.pop(); + } else if (kind === ts.SyntaxKind.CloseBraceToken) { + stack.pop(); + } else if (kind === ts.SyntaxKind.OpenBraceToken) { + stack.push("brace"); + } else if (kind === ts.SyntaxKind.TemplateHead) { + stack.push("tpl"); + } else if ( + (kind === ts.SyntaxKind.SlashToken || kind === ts.SyntaxKind.SlashEqualsToken) && + !exprEnd.has(prevKind) + ) { + kind = scanner.reScanSlashToken(); + } + tokens.push(scanner.getTokenText()); + prevKind = kind; + } + // prettier adds/removes trailing commas when it re-wraps after the dedent + return tokens.filter((t, i) => !(t === "," && /^[)\]}]$/.test(tokens[i + 1] ?? ""))); +} + +/** Myers O(ND) diff over token arrays. Returns edit hunks {aStart,aEnd,bStart,bEnd}. */ +export function myersHunks(a, b, maxD = 20000) { + const n = a.length; + const m = b.length; + const max = Math.min(n + m, maxD); + const trace = []; + let v = new Int32Array(2 * max + 2); + const off = max; + let found = -1; + for (let d = 0; d <= max && found < 0; d++) { + trace.push(v.slice()); + for (let k = -d; k <= d; k += 2) { + let x = + k === -d || (k !== d && v[off + k - 1] < v[off + k + 1]) + ? v[off + k + 1] + : v[off + k - 1] + 1; + let y = x - k; + while (x < n && y < m && a[x] === b[y]) { + x++; + y++; + } + v[off + k] = x; + if (x >= n && y >= m) { + found = d; + break; + } + } + } + if (found < 0) throw new Error(`token diff exceeds ${maxD} edits - regions are not comparable`); + // backtrack + const edits = []; // {type: "del"|"ins", ai, bi} + let x = n; + let y = m; + for (let d = found; d > 0; d--) { + const vv = trace[d]; + const k = x - y; + const prevK = k === -d || (k !== d && vv[off + k - 1] < vv[off + k + 1]) ? k + 1 : k - 1; + const prevX = vv[off + prevK]; + const prevY = prevX - prevK; + while (x > prevX && y > prevY) { + x--; + y--; + } + if (x === prevX) edits.push({ type: "ins", ai: x, bi: y - 1 }); + else edits.push({ type: "del", ai: x - 1, bi: y }); + x = prevX; + y = prevY; + } + edits.reverse(); + // group into hunks (merge edits separated by < 4 matching tokens) + const hunks = []; + for (const e of edits) { + const last = hunks[hunks.length - 1]; + const aPos = e.ai; + const bPos = e.bi; + if (last && aPos - last.aEnd < 4 && bPos - last.bEnd < 4 + (aPos - last.aEnd)) { + // extend + if (e.type === "del") last.aEnd = Math.max(last.aEnd, e.ai + 1); + else last.bEnd = Math.max(last.bEnd, e.bi + 1); + if (e.type === "del") last.bEnd = Math.max(last.bEnd, e.bi); + else last.aEnd = Math.max(last.aEnd, e.ai); + } else { + hunks.push({ + aStart: e.ai, + aEnd: e.type === "del" ? e.ai + 1 : e.ai, + bStart: e.bi, + bEnd: e.type === "ins" ? e.bi + 1 : e.bi, + }); + } + } + return hunks; +} + +// Regions. `base` slices the pre-split chatCore.ts, `leaf` slices the leaf file. Patterns run +// against NORMALIZED lines (trimmed, comments and blanks dropped), so indentation is irrelevant. +const LEAF_AFTER_DEPS = /^\} = deps;$/; +export const REGIONS = [ + { + name: "executeProviderRequest", + file: "open-sse/handlers/chatCore/executeProviderRequest.ts", + carryKeys: [], + allowRemoved: [], + base: { + startPattern: /^const execute = async \(\) => \{$/, + endPattern: /^return execute\(\);$/, + inclusiveEnd: true, + }, + leaf: { + after: LEAF_AFTER_DEPS, + startPattern: /^const execute = async \(\) => \{$/, + endPattern: /^return execute\(\);$/, + inclusiveEnd: true, + }, + }, + { + name: "streamingResponse", + file: "open-sse/handlers/chatCore/streamingResponse.ts", + carryKeys: [ + "translatedBody", + "currentModel", + "effectiveServiceTier", + "finalBody", + "pipelineRecovered", + "providerHeaders", + "providerResponse", + "providerUrl", + ], + allowRemoved: ["; } } }"], // closers of `if (stream) {` / handleChatCoreInner left in the barrel + base: { + after: /^let pipelineRecovered = false;$/, + startPattern: /^try \{$/, + endPattern: /^if \(!stream\) \{$/, + }, + leaf: { + after: LEAF_AFTER_DEPS, + startPattern: /^try \{$/, + endPattern: /^return \{$/, + endMode: "last", + }, + }, + { + name: "nonStreamingResponse", + file: "open-sse/handlers/chatCore/nonStreamingResponse.ts", + carryKeys: [ + "translatedBody", + "currentModel", + "finalBody", + "providerResponse", + "providerHeaders", + "effectiveServiceTier", + "claudePromptCacheLogMeta", + "reasoningReplayHistory", + "pipelineRecovered", + ], + allowRemoved: ["if ( ! stream ) {", "}"], // the `if (!stream)` opener / its closer stay in the barrel + base: { + startPattern: /^if \(!stream\) \{$/, + endPattern: /^providerResponse = await maybeConvertJsonBodyToSse\(/, + }, + leaf: { + after: LEAF_AFTER_DEPS, + startPattern: /^try \{$/, + endPattern: /^\}$/, + endMode: "last", + }, + }, + { + name: "streamingTail", + file: "open-sse/handlers/chatCore/streamingTail.ts", + carryKeys: [ + "onPipelineStreamError", + "onClientDisconnectFinalize", + "turnExecutionHandedOffToStream", + ], + allowRemoved: [ + // handleChatCoreInner's finally (turn-execution release) stays in the barrel + "; } finally { if ( ! turnExecutionHandedOffToStream ) { releaseTurnExecution ( ) ; } } }", + ], + base: { + startPattern: /^providerResponse = await maybeConvertJsonBodyToSse\(/, + endPattern: /^(export )?function isTokenExpiringSoon/, + }, + leaf: { + after: LEAF_AFTER_DEPS, + startPattern: /^providerResponse = await maybeConvertJsonBodyToSse\(/, + endPattern: /^\}$/, + endMode: "last", + }, + }, +]; + +const SYNC_CALL = "syncExecuteTranslatedBody ( translatedBody ) ;"; +const WRAP_OPEN = new Set(["{ response :", "response : {", "result : {"]); + +/** Returns a short reason when the hunk is a documented lift seam, else null. */ +export function classifyHunk(removed, added, region) { + const carry = `carry : { ${region.carryKeys.join(" , ")} }`; + if (removed === "" && added === SYNC_CALL) return "re-sync of the live translatedBody closure"; + if (removed === "" && WRAP_OPEN.has(added)) + return "return value wrapped as { response|result, carry }"; + const closeRe = new RegExp(`^, ${carry.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")} \\}( ;( \\})*)?$`); + if ((removed === "" || region.allowRemoved.includes(removed)) && closeRe.test(added)) { + return "carry object closing the wrapped return"; + } + const wrapValue = removed.match(/^[\w$]+$/); + if (wrapValue && added === `{ response : ${removed} , ${carry} }`) { + return "returned identifier wrapped with the carry object"; + } + if (region.allowRemoved.includes(removed) && added === "") + return "region boundary left in handleChatCore"; + return null; +} + +function readBase() { + try { + return execFileSync("git", ["show", `${BASE_REF}:${BASE_FILE}`], { + cwd: ROOT, + encoding: "utf8", + maxBuffer: 64 * 1024 * 1024, + }); + } catch (err) { + console.error( + `[diff-chatcore-leaves] cannot read ${BASE_FILE} from ${BASE_REF}: ${err.message}` + ); + process.exit(2); + } +} + +if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) { + main(); +} + +function main() { + console.log(`[diff-chatcore-leaves] base ref: ${BASE_REF}`); + const baseLines = normalize(readBase()); + let total = 0; + for (const region of REGIONS) { + const leafLines = normalize(fs.readFileSync(path.join(ROOT, region.file), "utf8")); + let bt; + let lt; + try { + bt = tokenize(sliceRegion(baseLines, region.base).join("\n")); + lt = tokenize(sliceRegion(leafLines, region.leaf).join("\n")); + } catch (err) { + console.error(`[diff-chatcore-leaves] ${region.name}: cannot extract region: ${err.message}`); + process.exit(2); + } + const hunks = myersHunks(bt, lt); + let mechanical = 0; + const bad = []; + for (const h of hunks) { + const removed = bt.slice(h.aStart, h.aEnd).join(" "); + const added = lt.slice(h.bStart, h.bEnd).join(" "); + const reason = classifyHunk(removed, added, region); + if (reason) { + mechanical++; + if (VERBOSE) console.log(` ok ${reason}: -[${removed}] +[${added.slice(0, 120)}]`); + } else { + bad.push({ h, removed, added }); + } + } + total += bad.length; + console.log( + `${bad.length === 0 ? "OK " : "FAIL"} ${region.name}: base ${bt.length} tokens, leaf ${lt.length} tokens, ` + + `${mechanical} mechanical hunk(s), ${bad.length} NON-MECHANICAL` + ); + for (const { h, removed, added } of bad) { + console.log(` @@ after: ${bt.slice(Math.max(0, h.aStart - 8), h.aStart).join(" ")}`); + if (removed) console.log(` - ${removed}`); + if (added) console.log(` + ${added}`); + } + } + console.log(`\n[diff-chatcore-leaves] non-mechanical hunks: ${total}`); + if (total === 0) console.log("[diff-chatcore-leaves] all four leaves are mechanical lifts"); + process.exit(total === 0 ? 0 : 1); +} diff --git a/scripts/release/merge-train.sh b/scripts/release/merge-train.sh index f29a25dc710c..be9c84c5a7f0 100755 --- a/scripts/release/merge-train.sh +++ b/scripts/release/merge-train.sh @@ -319,7 +319,7 @@ if [ "$FAST" = "1" ]; then | grep -E '\.test\.(ts|mjs)$' || true) # Subdirs test:unit actually runs (package.json allowlist) — anything else under # tests/unit// belongs to another runner (e.g. autoCombo -> vitest). - UNIT_SUBDIRS=",api,auth,authz,build,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage," + UNIT_SUBDIRS=",api,auth,authz,build,chatcore,cli,cli-helper,combo,compression,correctness,cors,db,db-adapters,docs,gamification,guardrails,lib,mcp,memory,runtime,security,services,settings,shared,translator,ui,usage," MAIN=() DASH=() SERIAL=() diff --git a/tests/integration/integration-wiring.test.ts b/tests/integration/integration-wiring.test.ts index bb3d2abf3cce..1d31bc49cd07 100644 --- a/tests/integration/integration-wiring.test.ts +++ b/tests/integration/integration-wiring.test.ts @@ -82,6 +82,8 @@ describe("Pipeline Wiring — instrumentation-node.ts", () => { describe("Pipeline Wiring — sse chat handler", () => { const src = readProjectFile("src/sse/handlers/chat.ts"); const coreSrc = readProjectFile("open-sse/handlers/chatCore.ts"); + // recordCost moved with the streaming tail when handleChatCore was split into leaves. + const streamingTailSrc = readProjectFile("open-sse/handlers/chatCore/streamingTail.ts"); it("should import and use guardrail pre-call validation", () => { assert.ok(src, "src/sse/handlers/chat.ts should exist"); @@ -107,8 +109,9 @@ describe("Pipeline Wiring — sse chat handler", () => { it("should keep cost tracking integration in the chat pipeline", () => { assert.ok(coreSrc, "open-sse/handlers/chatCore.ts should exist"); + assert.ok(streamingTailSrc, "open-sse/handlers/chatCore/streamingTail.ts should exist"); assert.match(coreSrc, /calculateCost/); - assert.match(coreSrc, /recordCost/); + assert.match(streamingTailSrc, /recordCost/); }); it("should not track backup artifacts in the active src/sse shim", () => { diff --git a/tests/unit/8395-plugin-hooks-fire.test.ts b/tests/unit/8395-plugin-hooks-fire.test.ts index a41cf501a7d6..b3c0224ddb50 100644 --- a/tests/unit/8395-plugin-hooks-fire.test.ts +++ b/tests/unit/8395-plugin-hooks-fire.test.ts @@ -123,29 +123,34 @@ test("loadPlugin no longer spawns the plugin host with stdout/stderr fully ignor // contract test for runPluginOnResponseHook itself // (tests/unit/chatcore-plugin-onresponse.test.ts). test("chatCore.ts calls runPluginOnResponseHook from both the non-streaming and streaming success paths", async () => { - const source = await readFile( - join(import.meta.dirname, "../../open-sse/handlers/chatCore.ts"), + // After the chatCore decomposition the two success-path call sites live in the + // leg leaves: non-streaming in nonStreamingResponse.ts, streaming in streamingTail.ts. + const nonStreamingSource = await readFile( + join(import.meta.dirname, "../../open-sse/handlers/chatCore/nonStreamingResponse.ts"), + "utf-8" + ); + const streamingSource = await readFile( + join(import.meta.dirname, "../../open-sse/handlers/chatCore/streamingTail.ts"), "utf-8" ); - const nonStreamingReturnIndex = source.indexOf("maybeWrapForcedNonStreamingResponsesJson({"); const hookCallNeedle = "await runPluginOnResponseHook({"; - const hookCallIndex = source.indexOf(hookCallNeedle); - const secondHookCallIndex = source.indexOf(hookCallNeedle, hookCallIndex + 1); + const nonStreamingReturnIndex = nonStreamingSource.indexOf("maybeWrapForcedNonStreamingResponsesJson({"); + const hookCallIndex = nonStreamingSource.indexOf(hookCallNeedle); + const streamingHookCallIndex = streamingSource.indexOf(hookCallNeedle); - assert.notEqual(hookCallIndex, -1, "expected at least one runPluginOnResponseHook call site"); + assert.notEqual(hookCallIndex, -1, "expected a runPluginOnResponseHook call site in the non-streaming leg"); assert.notEqual( - secondHookCallIndex, + streamingHookCallIndex, -1, - "expected TWO runPluginOnResponseHook call sites — one per success branch " + - "(non-streaming JSON return and streaming SSE return)" + "expected a runPluginOnResponseHook call site in the streaming leg — one per success branch" ); assert.ok( - source.indexOf("response: { status: 200, data: translatedResponse }") !== -1, + nonStreamingSource.indexOf("response: { status: 200, data: translatedResponse }") !== -1, "non-streaming branch must pass translatedResponse as plugin response data (#8711)" ); assert.ok( - source.indexOf("response: { status: 200, streamed: true }") !== -1, + streamingSource.indexOf("response: { status: 200, streamed: true }") !== -1, "streaming branch must pass streamed:true without materializing SSE body (#8711)" ); assert.ok( diff --git a/tests/unit/antigravity-429-quota-cooldown.test.ts b/tests/unit/antigravity-429-quota-cooldown.test.ts index eccdc8942d1f..6bdbcfba9b0c 100644 --- a/tests/unit/antigravity-429-quota-cooldown.test.ts +++ b/tests/unit/antigravity-429-quota-cooldown.test.ts @@ -164,13 +164,17 @@ test("direct Antigravity has one downstream model-lock owner and clamps body pro path.resolve(import.meta.dirname, "../../open-sse/handlers/chatCore.ts"), "utf8" ); + const eprSource = fs.readFileSync( + path.resolve(import.meta.dirname, "../../open-sse/handlers/chatCore/executeProviderRequest.ts"), + "utf8" + ); assert.match( chatCoreSource, /accountSemaphoreKey && !deferAntigravityQuotaStateToCaller/, "chatCore must not apply a prose-derived Antigravity semaphore TTL" ); assert.match( - chatCoreSource, + eprSource, /Dropped generic quota cache after 429/, "non-Codex 429 must leave a QUOTA debug breadcrumb" ); diff --git a/tests/unit/body-timeout-integration.test.ts b/tests/unit/body-timeout-integration.test.ts index 935cea918a4c..71ac6d8056d3 100644 --- a/tests/unit/body-timeout-integration.test.ts +++ b/tests/unit/body-timeout-integration.test.ts @@ -25,8 +25,10 @@ test("FETCH_BODY_TIMEOUT_MS defaults to FETCH_TIMEOUT_MS when no env override", // ── 2. BodyTimeoutError classification in chatCore ────────────────────── test("chatCore error classification maps BodyTimeoutError to 504 GATEWAY_TIMEOUT", () => { - // Read the source to verify the error classification logic includes BodyTimeoutError - const content = fs.readFileSync("open-sse/handlers/chatCore.ts", "utf8"); + // Read the leaf to verify the error classification logic includes BodyTimeoutError + // (the streaming leg's catch block lives in chatCore/streamingResponse.ts since + // the chatCore decomposition; the barrel no longer holds the classification). + const content = fs.readFileSync("open-sse/handlers/chatCore/streamingResponse.ts", "utf8"); // The error classification block should include BodyTimeoutError alongside TimeoutError. // Match whatever identifier carries the error (#13910 renamed it to `errorMetadata`); @@ -40,14 +42,19 @@ test("chatCore error classification maps BodyTimeoutError to 504 GATEWAY_TIMEOUT }); test("chatCore catch block decrements pending requests for all error types", () => { - const content = fs.readFileSync("open-sse/handlers/chatCore.ts", "utf8"); - - // The catch block should call trackPendingRequest with false before error classification - const catchBlockPattern = /catch\s*\(error\)\s*\{[^}]*trackPendingRequest\([^)]*,\s*false\b/; - assert.ok( - catchBlockPattern.test(content), - "chatCore catch block should decrement pending requests" - ); + // After the chatCore decomposition the per-leg catches live in the leaves; + // assert both legs decrement pending requests at the top of their catch. + for (const leaf of [ + "open-sse/handlers/chatCore/nonStreamingResponse.ts", + "open-sse/handlers/chatCore/streamingResponse.ts", + ]) { + const content = fs.readFileSync(leaf, "utf8"); + const catchBlockPattern = /catch\s*\(error\)\s*\{[^}]*trackPendingRequest\([^)]*,\s*false\b/; + assert.ok( + catchBlockPattern.test(content), + `${leaf} catch block should decrement pending requests` + ); + } }); test("withBodyTimeout error name is BodyTimeoutError", async () => { diff --git a/tests/unit/chatcore-hierarchical-admission.test.ts b/tests/unit/chatcore-hierarchical-admission.test.ts index 72d53290ec8a..d99eca98b2d6 100644 --- a/tests/unit/chatcore-hierarchical-admission.test.ts +++ b/tests/unit/chatcore-hierarchical-admission.test.ts @@ -2,8 +2,10 @@ import assert from "node:assert/strict"; import { readFileSync } from "node:fs"; import { test } from "node:test"; +// After the chatCore decomposition the wire-send path lives in +// chatCore/executeProviderRequest.ts; the barrel only wires it up. const source = readFileSync( - new URL("../../open-sse/handlers/chatCore.ts", import.meta.url), + new URL("../../open-sse/handlers/chatCore/executeProviderRequest.ts", import.meta.url), "utf8" ); @@ -12,6 +14,17 @@ const pipeline = readFileSync( "utf8" ); +const legs = [ + readFileSync( + new URL("../../open-sse/handlers/chatCore/nonStreamingResponse.ts", import.meta.url), + "utf8" + ), + readFileSync( + new URL("../../open-sse/handlers/chatCore/streamingResponse.ts", import.meta.url), + "utf8" + ), +]; + test("chatCore acquires cumulative gates immediately before withRateLimit", () => { const acquire = source.indexOf("await acquireConcurrencyGates("); const rateLimit = source.indexOf("await withRateLimit(", acquire); @@ -35,7 +48,7 @@ test("chatCore acquires cumulative gates immediately before withRateLimit", () = // single index check covered it; the loop is now split across two files, so the // guard checks both halves of the same invariant. test("each rotated account attempt acquires and releases a fresh composite slot", () => { - const sendFn = source.indexOf("const executeProviderRequest = async ("); + const sendFn = source.indexOf("export async function executeProviderRequest("); const attemptLoop = source.indexOf("while (attempts < maxAttempts)", sendFn); const acquire = source.indexOf("await acquireConcurrencyGates(", attemptLoop); const release = source.indexOf("releaseAccountSemaphore();", acquire); @@ -52,10 +65,12 @@ test("each rotated account attempt acquires and releases a fresh composite slot" "a throwing attempt must release the composite slot" ); - const sendWirings = - source.match( - /sendProviderAttempt: \(modelToCall, allowDedup\) =>\s*executeProviderRequest\(modelToCall, allowDedup\)/g - ) ?? []; + const sendWirings = legs.flatMap( + (leg) => + leg.match( + /sendProviderAttempt: \(modelToCall, allowDedup\) =>\s*executeProviderRequest\(modelToCall, allowDedup\)/g + ) ?? [] + ); assert.equal(sendWirings.length, 2, "both legs send every pipeline attempt through the gate"); const rotationLoop = pipeline.search(/while \(\s*attempts < maxAttempts\b/); diff --git a/tests/unit/chatcore-stream-error-result.test.ts b/tests/unit/chatcore-stream-error-result.test.ts index 2fb68dbf62e8..38f9e94ee1c5 100644 --- a/tests/unit/chatcore-stream-error-result.test.ts +++ b/tests/unit/chatcore-stream-error-result.test.ts @@ -156,13 +156,15 @@ test("getUpstreamErrorIdentifier returns a non-empty string code or undefined", test("non-streaming runNonStreamingProviderLeg is inside a try that maps semaphore errors", async () => { const fs = await import("node:fs"); - const src = fs.readFileSync("open-sse/handlers/chatCore.ts", "utf8"); + // After the decomposition the non-streaming branch lives in the leaf; its + // function-level try/catch is what maps semaphore errors. + const src = fs.readFileSync("open-sse/handlers/chatCore/nonStreamingResponse.ts", "utf8"); // `let`, not `const`, since 6077b9dd (#12867) made the finalization step reassign // legResult. The guard is about the try/catch that wraps the call, not the keyword. const idx = src.search(/(?:const|let) legResult = await runNonStreamingProviderLeg/); assert.ok(idx >= 0, "non-streaming branch must exist"); - const start = src.lastIndexOf("if (!stream)", idx); - const end = src.indexOf("// Streaming response", idx); + const start = src.lastIndexOf("export async function runNonStreamingResponse(", idx); + const end = src.indexOf("} catch (error)", idx); assert.ok(start >= 0 && end > start, "non-stream block bounds"); const block = src.slice(start, end); assert.match( @@ -170,8 +172,9 @@ test("non-streaming runNonStreamingProviderLeg is inside a try that maps semapho /try\s*\{[\s\S]*runNonStreamingProviderLeg/, "non-stream leg must sit in a try so SEMAPHORE_TIMEOUT cannot escape handleChatCore" ); + const catchBlock = src.slice(end); assert.match( - block, + catchBlock, /isSemaphoreCapacityError/, "same catch that maps stream semaphore errors must cover the non-stream leg" ); diff --git a/tests/unit/chatcore/diff-chatcore-leaves.test.ts b/tests/unit/chatcore/diff-chatcore-leaves.test.ts new file mode 100644 index 000000000000..0def1c655a5c --- /dev/null +++ b/tests/unit/chatcore/diff-chatcore-leaves.test.ts @@ -0,0 +1,71 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { + classifyHunk, + myersHunks, + normalize, + sliceRegion, + tokenize, + // @ts-expect-error - plain .mjs dev script without type declarations +} from "../../../scripts/dev/diff-chatcore-leaves.mjs"; + +const region = { + carryKeys: ["a", "b"], + allowRemoved: [], +}; + +test("tokenize ignores comments, whitespace and prettier re-wrap commas", () => { + const wrapped = tokenize("foo(\n a, // note\n b,\n);"); + const flat = tokenize("foo(a, b);"); + assert.deepEqual(wrapped, flat); +}); + +test("normalize drops blank, line-comment and block-comment lines", () => { + assert.deepEqual(normalize(" a = 1;\n\n// c\n/* x\n y */\n b;"), ["a = 1;", "b;"]); +}); + +test("sliceRegion honours the anchor, start and end patterns", () => { + const lines = ["x", "start", "y", "end", "start", "z", "end"]; + assert.deepEqual(sliceRegion(lines, { startPattern: /^start$/, endPattern: /^end$/ }), [ + "start", + "y", + ]); + assert.deepEqual( + sliceRegion(lines, { + after: /^end$/, + startPattern: /^start$/, + endPattern: /^end$/, + inclusiveEnd: true, + }), + ["start", "z", "end"] + ); +}); + +test("identical token streams produce no hunks; a changed literal produces one", () => { + const a = tokenize('return fail(499, "Request aborted");'); + assert.equal(myersHunks(a, [...a]).length, 0); + const b = tokenize('return fail(499, "Request aborted!");'); + const hunks = myersHunks(a, b); + assert.equal(hunks.length, 1); + assert.equal( + classifyHunk( + a.slice(hunks[0].aStart, hunks[0].aEnd).join(" "), + b.slice(hunks[0].bStart, hunks[0].bEnd).join(" "), + region + ), + null, + "a rewritten literal is NOT a mechanical seam" + ); +}); + +test("classifyHunk accepts only the documented lift seams", () => { + assert.ok(classifyHunk("", "syncExecuteTranslatedBody ( translatedBody ) ;", region)); + assert.ok(classifyHunk("", "{ response :", region)); + assert.ok(classifyHunk("", ", carry : { a , b } }", region)); + assert.ok(classifyHunk("result", "{ response : result , carry : { a , b } }", region)); + // wrong carry keys / unknown additions are rejected + assert.equal(classifyHunk("", ", carry : { a , c } }", region), null); + assert.equal(classifyHunk("result", "{ response : other , carry : { a , b } }", region), null); + assert.equal(classifyHunk("", "doSomethingNew ( ) ;", region), null); +}); diff --git a/tests/unit/chatcore/execute-provider-request.test.ts b/tests/unit/chatcore/execute-provider-request.test.ts new file mode 100644 index 000000000000..1e575ffb14b0 --- /dev/null +++ b/tests/unit/chatcore/execute-provider-request.test.ts @@ -0,0 +1,101 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../../.."); +const BARREL = path.join(ROOT, "open-sse/handlers/chatCore.ts"); +const LEAF = path.join(ROOT, "open-sse/handlers/chatCore/executeProviderRequest.ts"); + +function read(file: string): string { + return fs.readFileSync(file, "utf8"); +} + +test("executeProviderRequest lives in its own leaf", () => { + assert.equal(fs.existsSync(LEAF), true); + const leaf = read(LEAF); + assert.match(leaf, /export async function executeProviderRequest\(/); + assert.equal( + /const executeProviderRequest = async/.test(read(BARREL)), + false + ); +}); + +test("the leaf keeps the tip send shape", () => { + const leaf = read(LEAF); + assert.match(leaf, /prepareUpstreamBody\(/); + assert.match(leaf, /injectSystemPromptPostTranslation\(/); + assert.equal( + leaf.includes("isClaudePassthrough"), + false, + "tip executor.execute does not take isClaudePassthrough" + ); +}); + +test("handleChatCore calls the leaf and still records the cost ledger", () => { + const barrel = read(BARREL); + assert.match(barrel, /executeProviderRequestLeaf\(executeProviderRequestDeps/); + const nonStreaming = read(path.join(ROOT, "open-sse/handlers/chatCore/nonStreamingResponse.ts")); + assert.match(nonStreaming, /recordChatCallCost\(/); + const tail = read(path.join(ROOT, "open-sse/handlers/chatCore/streamingTail.ts")); + assert.match(tail, /buildStreamLedgerDetails\(/); + assert.match(nonStreaming, /buildCostCtx\(/); +}); + +test("syncExecuteTranslatedBody seam keeps executeProviderRequestDeps synchronized across rebindings", () => { + const barrel = read(BARREL); + assert.match( + barrel, + /const syncExecuteTranslatedBody = \(next: unknown\) => \{[\s\S]*?executeProviderRequestDeps\.translatedBody = next as Record;/, + "chatCore must define syncExecuteTranslatedBody and update executeProviderRequestDeps.translatedBody" + ); + + // Functional simulation of the closure / deps reference seam: + const executeProviderRequestDeps: Record = { + translatedBody: { model: "base-model", messages: [] }, + }; + const syncExecuteTranslatedBody = (next: unknown) => { + executeProviderRequestDeps.translatedBody = next as Record; + }; + const simulateDispatch = (deps: typeof executeProviderRequestDeps) => { + return deps.translatedBody; + }; + + assert.deepEqual(simulateDispatch(executeProviderRequestDeps), { + model: "base-model", + messages: [], + }); + + // Rebind via syncExecuteTranslatedBody (e.g. tool follow-up or pipeline wire update): + const reboundBody = { model: "base-model", messages: [{ role: "assistant", content: "followup" }] }; + syncExecuteTranslatedBody(reboundBody); + + assert.strictEqual( + simulateDispatch(executeProviderRequestDeps), + reboundBody, + "subsequent dispatches must observe the rebound translatedBody reference" + ); + + // Verify that the seam is wired into both streaming and non-streaming response leaves: + assert.match( + barrel, + /runNonStreamingResponse\(\{[\s\S]*?syncExecuteTranslatedBody,/, + "runNonStreamingResponse must receive syncExecuteTranslatedBody" + ); + assert.match( + barrel, + /runStreamingResponse\(\{[\s\S]*?syncExecuteTranslatedBody,/, + "runStreamingResponse must receive syncExecuteTranslatedBody" + ); + + // Invariant: every reassignment of translatedBody in the leaves must immediately sync back + const nonStreaming = read(path.join(ROOT, "open-sse/handlers/chatCore/nonStreamingResponse.ts")); + const streaming = read(path.join(ROOT, "open-sse/handlers/chatCore/streamingResponse.ts")); + + const nonStreamingSyncCount = (nonStreaming.match(/syncExecuteTranslatedBody\(translatedBody\)/g) ?? []).length; + assert.ok(nonStreamingSyncCount >= 5, "nonStreamingResponse must sync on every translatedBody rebinding"); + + const streamingSyncCount = (streaming.match(/syncExecuteTranslatedBody\(translatedBody\)/g) ?? []).length; + assert.ok(streamingSyncCount >= 2, "streamingResponse must sync on every translatedBody rebinding"); +}); diff --git a/tests/unit/chatcore/leaf-deps-contract.test.ts b/tests/unit/chatcore/leaf-deps-contract.test.ts new file mode 100644 index 000000000000..bee1cb486775 --- /dev/null +++ b/tests/unit/chatcore/leaf-deps-contract.test.ts @@ -0,0 +1,86 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import ts from "typescript"; + +// The streaming/non-streaming/tail leaves take a loosely typed deps bag +// (Record), so the compiler cannot notice a key the leaf reads +// but chatCore.ts never passes — the leaf then dies at runtime with +// "x is not a function" (the streaming direct-refresh retry lost +// getExecutorClientHeaders this way). Assert the contract structurally. + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../../.."); +const BARREL = path.join(ROOT, "open-sse/handlers/chatCore.ts"); +const LEAVES: Record = { + runStreamingResponse: "open-sse/handlers/chatCore/streamingResponse.ts", + runNonStreamingResponse: "open-sse/handlers/chatCore/nonStreamingResponse.ts", + runStreamingTail: "open-sse/handlers/chatCore/streamingTail.ts", +}; + +function parse(file: string): ts.SourceFile { + return ts.createSourceFile(file, fs.readFileSync(file, "utf8"), ts.ScriptTarget.Latest, true); +} + +function passedKeys(sf: ts.SourceFile, fn: string): Set { + const keys = new Set(); + const walk = (node: ts.Node) => { + if ( + ts.isCallExpression(node) && + ts.isIdentifier(node.expression) && + node.expression.text === fn + ) { + const arg = node.arguments[0]; + if (arg && ts.isObjectLiteralExpression(arg)) { + for (const prop of arg.properties) { + const name = prop.name; + if (name && (ts.isIdentifier(name) || ts.isStringLiteral(name))) keys.add(name.text); + } + } + } + ts.forEachChild(node, walk); + }; + walk(sf); + return keys; +} + +function readKeys(sf: ts.SourceFile): Set { + const keys = new Set(); + const walk = (node: ts.Node) => { + if ( + ts.isVariableDeclaration(node) && + ts.isObjectBindingPattern(node.name) && + node.initializer && + ts.isIdentifier(node.initializer) && + node.initializer.text === "deps" + ) { + for (const el of node.name.elements) { + keys.add(((el.propertyName ?? el.name) as ts.Identifier).text); + } + } + if ( + ts.isPropertyAccessExpression(node) && + ts.isIdentifier(node.expression) && + node.expression.text === "deps" + ) { + keys.add(node.name.text); + } + ts.forEachChild(node, walk); + }; + walk(sf); + return keys; +} + +const barrel = parse(BARREL); + +for (const [fn, leafPath] of Object.entries(LEAVES)) { + test(`${fn}: chatCore.ts passes every deps key the leaf reads`, () => { + const read = readKeys(parse(path.join(ROOT, leafPath))); + const given = passedKeys(barrel, fn); + assert.ok(read.size > 0, `no deps keys found in ${leafPath}`); + assert.ok(given.size > 0, `no ${fn}({...}) call found in chatCore.ts`); + const missing = [...read].filter((key) => !given.has(key)); + assert.deepEqual(missing, [], `${fn} reads deps keys chatCore.ts never passes`); + }); +} diff --git a/tests/unit/chatcore/non-streaming-response.test.ts b/tests/unit/chatcore/non-streaming-response.test.ts new file mode 100644 index 000000000000..ef88f7970daa --- /dev/null +++ b/tests/unit/chatcore/non-streaming-response.test.ts @@ -0,0 +1,40 @@ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import path from "node:path"; +import test from "node:test"; + +const ROOT = path.resolve(import.meta.dirname, "../../.."); +const read = (p) => readFileSync(path.join(ROOT, p), "utf8"); +const LEAF = "open-sse/handlers/chatCore/nonStreamingResponse.ts"; +const BARREL = "open-sse/handlers/chatCore.ts"; + +test("the non-streaming path lives in its own leaf", () => { + const leaf = read(LEAF); + assert.match(leaf, /export async function runNonStreamingResponse\(/); + assert.doesNotMatch(read(BARREL), /const runNonStreamingPipeline = async/); +}); + +test("the barrel calls the leaf and writes the carried state back", () => { + const barrel = read(BARREL); + assert.match(barrel, /runNonStreamingResponse\(/); + for (const name of ["translatedBody", "currentModel", "finalBody", "providerResponse", "pipelineRecovered"]) { + assert.match(barrel, new RegExp("carry\\." + name + "\\b")); + } +}); + +test("the cost ledger stays recorded on the non-streaming path", () => { + const leaf = read(LEAF); + assert.match(leaf, /recordChatCallCost\(/); + assert.match(leaf, /buildCostCtx\(/); +}); + +test("the barrel imports the leaf (no-undef is disabled for TS, so assert it)", () => { + const barrel = read(BARREL); + assert.match(barrel, /import \{\s*runNonStreamingResponse,?\s*\} from "\.\/chatCore\/nonStreamingResponse\.ts";/); +}); + +test("every leaf exit carries state (the leg-error exit must not return bare)", () => { + const leaf = read(LEAF); + const bareReturns = leaf.match(/^\s*return err;$/gm); + assert.equal(bareReturns, null, "a bare `return err;` leaves the barrel reading .carry of undefined"); +}); diff --git a/tests/unit/chatcore/streaming-response.test.ts b/tests/unit/chatcore/streaming-response.test.ts new file mode 100644 index 000000000000..158f90a4e64e --- /dev/null +++ b/tests/unit/chatcore/streaming-response.test.ts @@ -0,0 +1,42 @@ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import path from "node:path"; +import test from "node:test"; + +const ROOT = path.resolve(import.meta.dirname, "../../.."); +const read = (p) => readFileSync(path.join(ROOT, p), "utf8"); +const LEAF = "open-sse/handlers/chatCore/streamingResponse.ts"; +const BARREL = "open-sse/handlers/chatCore.ts"; + +test("the streaming path lives in its own leaf", () => { + const leaf = read(LEAF); + assert.match(leaf, /export async function runStreamingResponse\(/); + assert.doesNotMatch(read(BARREL), /const pipelineOutcome = await runProviderExecutionPipeline/); +}); + +test("the barrel calls the leaf and writes the carried state back", () => { + const barrel = read(BARREL); + assert.match(barrel, /runStreamingResponse\(/); + for (const name of ["translatedBody", "currentModel", "providerUrl", "pipelineRecovered"]) { + assert.match(barrel, new RegExp("carry\\." + name + "\\b")); + } +}); + +test("the barrel imports the leaf (no-undef is disabled for TS, so assert it)", () => { + const barrel = read(BARREL); + assert.match(barrel, /import \{\s*runStreamingResponse,?\s*\} from "\.\/chatCore\/streamingResponse\.ts";/); +}); + +test("message stays block-local (declared inside providerFailure, never carried)", () => { + const leaf = read(LEAF); + const barrel = read(BARREL); + assert.doesNotMatch(leaf, /finalBody, message, pipelineRecovered/); + assert.doesNotMatch(barrel, /message = streamingOutcome\.carry\.message/); +}); + +test("streaming cost accounting stays out of the response leaf", () => { + // It lives in the streaming tail (see streaming-tail.test.ts); the response + // leaf only executes the provider call and must not touch the ledger. + const leaf = read(LEAF); + assert.doesNotMatch(leaf, /recordStreamingCost\(/); +}); diff --git a/tests/unit/chatcore/streaming-tail.test.ts b/tests/unit/chatcore/streaming-tail.test.ts new file mode 100644 index 000000000000..cccdb33762f2 --- /dev/null +++ b/tests/unit/chatcore/streaming-tail.test.ts @@ -0,0 +1,43 @@ +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import path from "node:path"; +import test from "node:test"; + +const ROOT = path.resolve(import.meta.dirname, "../../.."); +const read = (p) => readFileSync(path.join(ROOT, p), "utf8"); +const LEAF = "open-sse/handlers/chatCore/streamingTail.ts"; +const BARREL = "open-sse/handlers/chatCore.ts"; + +test("the streaming tail lives in its own leaf", () => { + const leaf = read(LEAF); + assert.match(leaf, /export async function runStreamingTail\(/); + assert.doesNotMatch(read(BARREL), /const onStreamComplete = \(\{/); +}); + +test("the barrel calls the leaf and writes the carried state back", () => { + const barrel = read(BARREL); + assert.match(barrel, /runStreamingTail\(/); + for (const name of [ + "onPipelineStreamError", + "onClientDisconnectFinalize", + "turnExecutionHandedOffToStream", + ]) { + assert.match(barrel, new RegExp("carry\\." + name + "\\b")); + } +}); + +test("the barrel imports the leaf (no-undef is disabled for TS, so assert it)", () => { + const barrel = read(BARREL); + assert.match( + barrel, + /import \{\s*runStreamingTail,?\s*\} from "\.\/chatCore\/streamingTail\.ts";/ + ); +}); + +test("streaming cost accounting lives in the tail leaf, not the barrel", () => { + const leaf = read(LEAF); + assert.match(leaf, /recordStreamingCost\(/); + assert.match(leaf, /buildStreamLedgerDetails\(/); + const barrel = read(BARREL); + assert.doesNotMatch(barrel, /recordStreamingCost\(/); +}); diff --git a/tests/unit/fixtures/error-public-boundaries-hardening.fixture.ts b/tests/unit/fixtures/error-public-boundaries-hardening.fixture.ts index a47937b4c903..b775a585fa23 100644 --- a/tests/unit/fixtures/error-public-boundaries-hardening.fixture.ts +++ b/tests/unit/fixtures/error-public-boundaries-hardening.fixture.ts @@ -398,10 +398,15 @@ test("stream request finalization never warns with a raw error object", () => { test("chatCore provider-failure writes use the projected persistent message", () => { const source = fs.readFileSync(path.join(REPO_ROOT, "open-sse/handlers/chatCore.ts"), "utf8"); - const failureStart = source.indexOf("providerFailure: if (!providerResponse.ok)"); - const failureEnd = source.indexOf("// Non-streaming response", failureStart); - assert.ok(failureStart >= 0 && failureEnd > failureStart, "providerFailure block must exist"); - const failureBlock = source.slice(failureStart, failureEnd); + // The providerFailure block moved into the streaming leaf with the handleChatCore split; the + // classification helper (applyProviderFailureClassification) stayed in the barrel. + const streamingLeaf = fs.readFileSync( + path.join(REPO_ROOT, "open-sse/handlers/chatCore/streamingResponse.ts"), + "utf8" + ); + const failureStart = streamingLeaf.indexOf("providerFailure: if (!providerResponse.ok)"); + assert.ok(failureStart >= 0, "providerFailure block must exist in streamingResponse.ts"); + const failureBlock = streamingLeaf.slice(failureStart); const classifierStart = source.indexOf("const applyProviderFailureClassification = async"); const classifierEnd = source.indexOf("\n };\n", classifierStart); assert.ok( diff --git a/tests/unit/hard-session-lease-bypass-inventory.test.ts b/tests/unit/hard-session-lease-bypass-inventory.test.ts index 6a6f194587dc..09e8d65c7cd4 100644 --- a/tests/unit/hard-session-lease-bypass-inventory.test.ts +++ b/tests/unit/hard-session-lease-bypass-inventory.test.ts @@ -24,9 +24,11 @@ const EXPECTED: Record> = { // executeProviderRequest(), whose assertManagedLeaseFence(attemptConnectionId) rejects a // connection other than the leased one — so it is fenced centrally (class A). // #14914 moved that loop (and its credential rollback) into - // chatCore/emptyTurnRetryLoop.ts; chatCore.ts now passes `getProviderCredentials` in - // as a dependency (a reference, not a call), so the site is inventoried at its new - // home — still dispatched through executeProviderRequest(), still class A. + // chatCore/emptyTurnRetryLoop.ts; the response-path split (chatCore.ts split into + // response-path leaves) then moved the call site into streamingTail.ts, which now + // passes `getProviderCredentials` in as a dependency (a reference, not a call) to + // emptyTurnRetryLoop.ts — still dispatched through executeProviderRequest(), still + // fenced centrally (class A). "open-sse/handlers/chatCore/emptyTurnRetryLoop.ts": 1, "open-sse/handlers/chatCore/providerExecutionPipeline.ts": 2, "open-sse/services/imageCombo.ts": 1, @@ -80,7 +82,11 @@ const EXPECTED: Record> = { "src/sse/services/imageCredentialRetry.ts": 1, }, executor: { - "open-sse/handlers/chatCore.ts": 3, + // The three executor.execute() sites that used to sit in chatCore.ts moved + // with the decomposition: two into the wire-send leaf and one into the + // streaming leg (same sites, new homes). + "open-sse/handlers/chatCore/executeProviderRequest.ts": 2, + "open-sse/handlers/chatCore/streamingResponse.ts": 1, "open-sse/handlers/chatCore/cliproxyModelMapping.ts": 1, "open-sse/handlers/chatCore/cliproxyapiCredentials.ts": 1, // v3.8.51 #11754: the legacy common ChatGPT Web's synthetic @@ -97,7 +103,10 @@ const EXPECTED: Record> = { }, connection: { "open-sse/handlers/autoComboCandidates.ts": 1, - "open-sse/handlers/chatCore.ts": 3, + // Two of the three connection re-resolution sites moved into the streaming + // leg with the decomposition (same sites, new home). + "open-sse/handlers/chatCore.ts": 1, + "open-sse/handlers/chatCore/streamingResponse.ts": 2, "open-sse/handlers/cursorCliProxy.ts": 1, "open-sse/services/alibabaFreeTier.ts": 1, "open-sse/services/alibabaFreeTierQuotaFetcher.ts": 1, @@ -257,7 +266,8 @@ const CLASSIFICATION: Record> = { ]) ), executor: { - "open-sse/handlers/chatCore.ts": "A", + "open-sse/handlers/chatCore/executeProviderRequest.ts": "A", + "open-sse/handlers/chatCore/streamingResponse.ts": "A", "open-sse/handlers/chatCore/cliproxyModelMapping.ts": "A", "open-sse/handlers/chatCore/cliproxyapiCredentials.ts": "A", "open-sse/handlers/videoGeneration.ts": "B", @@ -271,6 +281,7 @@ const CLASSIFICATION: Record> = { [ "open-sse/handlers/autoComboCandidates.ts", "open-sse/handlers/chatCore.ts", + "open-sse/handlers/chatCore/streamingResponse.ts", "open-sse/services/alibabaFreeTier.ts", "open-sse/services/alibabaFreeTierQuotaFetcher.ts", "open-sse/services/combo/executeTargetGates.ts", @@ -380,7 +391,6 @@ test("hard-lease credential, executor, and connection-query inventory has no unc test("managed request surfaces are fenced centrally or rejected before independent dispatch", () => { const chat = fs.readFileSync(path.join(REPO_ROOT, "src/sse/handlers/chat.ts"), "utf8"); - const core = fs.readFileSync(path.join(REPO_ROOT, "open-sse/handlers/chatCore.ts"), "utf8"); const ws = fs.readFileSync( path.join(REPO_ROOT, "src/app/api/internal/codex-responses-ws/route.ts"), "utf8" @@ -407,9 +417,22 @@ test("managed request surfaces are fenced centrally or rejected before independe assert.match(chat, /parseManagedLeaseRequestContext\(request\.headers\)/); assert.match(chat, /isManagedComboUnsupported/); - assert.match(core, /assertManagedLeaseFence\(attemptConnectionId\)/); + // The fence call sites moved into the leg leaves with the decomposition. + const eprSource = fs.readFileSync( + path.join(REPO_ROOT, "open-sse/handlers/chatCore/executeProviderRequest.ts"), + "utf8" + ); + const streamingLeg = fs.readFileSync( + path.join(REPO_ROOT, "open-sse/handlers/chatCore/streamingResponse.ts"), + "utf8" + ); + const nonStreamingLeg = fs.readFileSync( + path.join(REPO_ROOT, "open-sse/handlers/chatCore/nonStreamingResponse.ts"), + "utf8" + ); + assert.match(eprSource, /assertManagedLeaseFence\(attemptConnectionId\)/); assert.match( - core, + streamingLeg, /assertManagedLeaseFence\(getExecutionConnectionId\(getExecutionCredentials\(\)\)\)/ ); // #12867 (d6f315018) extracted codex 429 / antigravity 422 account rotation out @@ -423,9 +446,14 @@ test("managed request surfaces are fenced centrally or rejected before independe path.join(REPO_ROOT, "open-sse/handlers/chatCore/providerExecutionPipeline.ts"), "utf8" ); - const rotationPolicySites = core.match( - /allowAccountRotation: !managedLease && comboStrategy !== "context-relay"/g - ); + const rotationPolicySites = [ + ...nonStreamingLeg.matchAll( + /allowAccountRotation: !managedLease && comboStrategy !== "context-relay"/g + ), + ...streamingLeg.matchAll( + /allowAccountRotation: !managedLease && comboStrategy !== "context-relay"/g + ), + ]; assert.equal( rotationPolicySites?.length, 2, diff --git a/tests/unit/responses-json-to-sse-13033.test.ts b/tests/unit/responses-json-to-sse-13033.test.ts index e46e7d8029f3..11a08707ac63 100644 --- a/tests/unit/responses-json-to-sse-13033.test.ts +++ b/tests/unit/responses-json-to-sse-13033.test.ts @@ -69,12 +69,20 @@ test("chatCore stamps clientRequestedResponsesStream before forcing stream:false ); const stamp = source.indexOf("clientRequestedResponsesStream = true"); const force = source.indexOf("(body as Record).stream = false"); - const wrap = source.indexOf("maybeWrapForcedNonStreamingResponsesJson({"); + const leafSource = await readFile( + join(import.meta.dirname, "../../open-sse/handlers/chatCore/nonStreamingResponse.ts"), + "utf-8" + ); + const wrap = leafSource.indexOf("maybeWrapForcedNonStreamingResponsesJson({"); + const leafCall = source.indexOf("await runNonStreamingResponse("); assert.ok(stamp !== -1, "must stamp the client-requested stream flag"); assert.ok(force !== -1, "must still force stream:false for the web_search fallback"); assert.ok(wrap !== -1, "must wrap the non-streaming JSON return"); assert.ok(stamp < force, "stamp must happen before stream:false"); - assert.ok(wrap > force, "wrap must happen on the non-streaming return after the force"); + assert.ok( + leafCall > force, + "the non-streaming leaf runs after the force, so its wrap lands after it" + ); }); // When the client speaks the Responses API, the forced non-streaming leg is diff --git a/tests/unit/upstream-status-restatement.test.ts b/tests/unit/upstream-status-restatement.test.ts index ceccfda6b8ec..4c4a47dfb5cc 100644 --- a/tests/unit/upstream-status-restatement.test.ts +++ b/tests/unit/upstream-status-restatement.test.ts @@ -112,25 +112,61 @@ test("R10: chatCore wires applyStatusRestatement into the providerFailure block" // (side-effectful DB/env wiring), so the wiring contract is asserted at the // source level: the hook must exist, run against the parsed error, and // reassign both statusCode and retryAfterMs BEFORE classification. + // The classification helper stays in the barrel; the providerFailure block + // moved into the streaming leg with the decomposition. const { readFile } = await import("node:fs/promises"); const src = await readFile( new URL("../../open-sse/handlers/chatCore.ts", import.meta.url), "utf8" ); - assert.match(src, /applyStatusRestatement\(/, "chatCore must call applyStatusRestatement"); + const legSrc = await readFile( + new URL("../../open-sse/handlers/chatCore/streamingResponse.ts", import.meta.url), + "utf8" + ); + assert.match( + legSrc, + /applyStatusRestatement\(/, + "streaming leg must call applyStatusRestatement" + ); const helperIndex = src.indexOf("const applyProviderFailureClassification = async ("); - const blockIndex = src.indexOf("providerFailure: if (!providerResponse.ok)"); - assert.ok(helperIndex > -1 && blockIndex > helperIndex, "classification helper and block exist"); + const helperEnd = src.indexOf("const streamingOutcome = await runStreamingResponse("); + assert.ok( + helperIndex > -1 && helperEnd > helperIndex, + "classification helper exists before streaming dispatch" + ); const classifyCalls = src.match(/classifyProviderError\(/g) ?? []; assert.equal(classifyCalls.length, 1, "chatCore classifies provider errors in exactly one place"); const classifyIndex = src.indexOf("classifyProviderError(statusCode"); assert.ok( - classifyIndex > helperIndex && classifyIndex < blockIndex, + classifyIndex > helperIndex && classifyIndex < helperEnd, "classifyProviderError must live inside applyProviderFailureClassification" ); - const block = src.slice(blockIndex); + const blockIndex = legSrc.indexOf("providerFailure: if (!providerResponse.ok)"); + assert.ok(blockIndex > -1, "providerFailure block exists in streamingResponse.ts"); + // Cross-file form of the original `classifyIndex < blockIndex` ordering: the classification + // helper (holding the single classifyProviderError call) must be defined before the dispatch + // that hands it to the leaf containing the providerFailure block, and the dispatch must + // actually pass it in. + const dispatchIndex = src.indexOf("await runStreamingResponse({"); + assert.ok(dispatchIndex > -1, "streaming dispatch exists in chatCore.ts"); + assert.ok( + classifyIndex < dispatchIndex, + "classifyProviderError must be classified-before the block: helper precedes the streaming dispatch" + ); + assert.match( + src.slice(dispatchIndex, src.indexOf("});", dispatchIndex)), + /\bapplyProviderFailureClassification,/, + "streaming dispatch must pass applyProviderFailureClassification into the leaf" + ); + assert.equal( + legSrc.indexOf("classifyProviderError("), + -1, + "streaming leg does not classify directly; it delegates to applyProviderFailureClassification" + ); + + const block = legSrc.slice(blockIndex); const hookIndex = block.indexOf("applyStatusRestatement("); const statusReassign = block.indexOf("statusCode = restatement.status;"); const retryReassign = block.indexOf("retryAfterMs = restatement.retryAfterMs;");