diff --git a/docs-site/src/content/docs/guides/codex-integration.md b/docs-site/src/content/docs/guides/codex-integration.md index 3c4a7ba0d84..864fd3b4e07 100644 --- a/docs-site/src/content/docs/guides/codex-integration.md +++ b/docs-site/src/content/docs/guides/codex-integration.md @@ -98,6 +98,13 @@ is a `POST` to the canonical Responses URL or a configured WebSocket route, and falls back to it when the request cannot be prepared, the `response.create` frame exceeds its size limit, or the proxy route cannot carry the socket. +To keep the built-in ChatGPT provider on HTTP/SSE, set `providers.openai.upstreamWebsocket` +to `false` in `~/.opencodex/config.json` and restart the proxy. Merge this field into the +existing `openai` provider; preserve its account mode and other settings. Omit the field +to restore the default upstream WebSocket selection. This setting does not change the +client-facing `websockets` switch or the ChatGPT account used for the request. Native +mid-turn steering and injection need upstream WebSocket and are unavailable while it is off. + Local provider pacing can also hold a request before it is dispatched at all. So a slow first output has several possible contributors, and upstream queueing is only one of them. `ocx doctor` classifies configuration and measures none of these: compare actual transport, pacing, network, diff --git a/src/server/auth-cors.ts b/src/server/auth-cors.ts index bb5c05e10d2..e2ad66f1ca0 100644 --- a/src/server/auth-cors.ts +++ b/src/server/auth-cors.ts @@ -744,6 +744,12 @@ export function providerManagementConfigError( // validation and then rejected by the seed comparison, so canonical OpenAI could never // set OR clear it — the value was admitted and then refused in the same request. delete canonicalCandidate.annotateEmptyToolOutputs; + // Canonical ChatGPT keeps WebSocket as the default, but an operator may + // select the existing HTTP/SSE path without changing its auth or endpoint. + if (raw.upstreamWebsocket !== undefined) { + if (raw.upstreamWebsocket !== false) return "provider openai upstreamWebsocket must be false or omitted"; + delete canonicalCandidate.upstreamWebsocket; + } const canonical = seed && (options?.allowOperatorOverlays ? matchesCanonicalProviderSeed(canonicalCandidate, seed) : sameCanonicalProviderSeed(canonicalCandidate, seed)); diff --git a/src/server/responses/fetch-helpers.ts b/src/server/responses/fetch-helpers.ts index 61981aac401..1c27951019a 100644 --- a/src/server/responses/fetch-helpers.ts +++ b/src/server/responses/fetch-helpers.ts @@ -268,7 +268,7 @@ export function providerFetch( // transport (measured ~3s faster TTFT than the SSE POST queue); everything // else keeps the provider's HTTP fetch. See ws-upstream.ts for the details. const unpaced = async (input: Parameters[0], init?: RequestInit) => { - const upstreamWebsocket = provider.upstreamWebsocket === true; + const upstreamWebsocket = provider.upstreamWebsocket; if (!options.httpOnly && typeof input === "string" && init && shouldUseCodexWsUpstream(input, init, runtime, upstreamWebsocket)) { const egress = egressFor(input); diff --git a/src/server/responses/native-response-control.ts b/src/server/responses/native-response-control.ts index 42b926df4d3..13c4b4b48f9 100644 --- a/src/server/responses/native-response-control.ts +++ b/src/server/responses/native-response-control.ts @@ -31,9 +31,9 @@ export function markNativeControlResponse(response: Response): Response { native /** Recognize a marked native response by identity, not by caller-controlled content. */ export function isNativeControlResponse(response: Response): boolean { return nativeControlResponses.has(response); } -/** Preserve canonical ChatGPT eligibility; only injection may use the separately billed public API. */ +/** Canonical ChatGPT needs its upstream WS enabled; only injection may use the separately billed public API. */ export function nativeResponseControlEligible(provider: OcxProviderConfig, control?: NativeResponseControl): boolean { - if (isCanonicalOpenAiForwardProvider(provider)) return true; + if (isCanonicalOpenAiForwardProvider(provider)) return provider.upstreamWebsocket !== false; return control?.kind === "injection" && provider.adapter === "openai-responses" && provider.upstreamWebsocket === true && provider.authMode !== "forward" && provider.baseUrl?.replace(/\/+$/, "") === "https://api.openai.com/v1"; diff --git a/src/server/responses/ws-upstream.ts b/src/server/responses/ws-upstream.ts index 4bba3ad9a15..7e7af0791e4 100644 --- a/src/server/responses/ws-upstream.ts +++ b/src/server/responses/ws-upstream.ts @@ -85,10 +85,11 @@ export function shouldUseCodexWsUpstream( url: string, init?: RequestInit, runtime: BunRuntimeGateInput = currentBunRuntimeIdentity(), - upstreamWebsocketConfigured = false, + upstreamWebsocketConfigured?: boolean, ): boolean { if (!bunSupportsBoundedCodexWsRelay(runtime)) return false; if (socks5ProxyFromEnv()) return false; + if (url === CODEX_RESPONSES_HTTP_URL && upstreamWebsocketConfigured === false) return false; // Bun's client WebSocket API delivers only fully assembled messages and has // no enforceable inbound payload limit. Keep arbitrary provider endpoints on // bounded HTTP/SSE until the client can reject fragmented text and binary diff --git a/structure/config.md b/structure/config.md index 9330ca6dfff..57337722c7a 100644 --- a/structure/config.md +++ b/structure/config.md @@ -101,6 +101,7 @@ merge cannot turn them into a valid config while discarding the original bytes. | Retained state | `appOwnedMemoryBudgetMb` | Process-wide eviction target for app-owned logs, caches, blobs, and continuation payloads. Default 256 MiB, valid 64..4096; pinned state may temporarily exceed the target, but every pin-capable store has a finite local cap and their documented aggregate stays below `APP_OWNED_WORST_CASE_PINNED_BYTES` (512 MiB). Neither value caps RSS or native runtime memory. | | Spend | `spend.root`, `spend.identity`, `spend.pool`, `spend.retentionDays` | Durable token ceilings for the spend-reservation ledger. Absent is the default and means observe-only accounting: spend is still journaled and nothing is refused, so observe-only and enforced servers take the same state-directory writer lease. One live process may write one directory; explicit sibling instances need separate `OPENCODEX_HOME` directories. There is no default figure for any scope — the ledger is on by default, so a shipped ceiling would refuse real traffic on upgrade against a number nobody chose. Strictly validated and positive-integer only, because 0 would read as a budget and refuse everything; a malformed section degrades to no ceiling, which is why the write path rejects it and load diagnostics report it. Resolution and application live in `src/lib/spend-reservation-ledger.ts`; see [`transports/responses.md`](transports/responses.md). | | Transport | stream mode, timeouts, proxy settings, `websockets`, `emptyCompletionRetry` | `streamMode` persists in config.json; Windows services need a persisted input, and macOS uses it for explicit eager-relay opt-in. Empty-completion replay is an explicit top-level opt-in because its second upstream request may be billable. | +| Canonical ChatGPT upstream transport | `providers.openai.upstreamWebsocket` | Omitted uses upstream WebSocket when eligible; explicit `false` selects HTTP/SSE without changing the canonical provider identity. `true` is rejected on the canonical row. This is independent of the client-facing `websockets` setting. | | Provider egress | `providers..proxy`, `providers..noProxy` | An absent `proxy` inherits global egress; `"direct"` or `null` forces direct egress; HTTP(S) and SOCKS5(H) URLs select a provider-owned proxy. `noProxy` uses NO_PROXY syntax and sends a matching destination direct across either a provider-owned or inherited global proxy. `src/lib/provider-egress.ts` owns parsing and request-local resolution. | | Credentials | `apiKeys` | Data-plane only; never admitted to `/api/*`. | | Lifecycle | `codexAutoStart`, shim/start behavior, resume-history sync, storage cleanup | Startup safety reads these; see [`gui-and-management-api.md`](gui-and-management-api.md). | diff --git a/structure/transports/streaming-health.md b/structure/transports/streaming-health.md index ce8954be79f..f1906696397 100644 --- a/structure/transports/streaming-health.md +++ b/structure/transports/streaming-health.md @@ -211,7 +211,12 @@ the upgrade with 426 so Codex falls back to HTTP cleanly. That setting controls the client-facing upgrade only. The transparent upstream ChatGPT WS optimization described above is selected independently and still -returns the same downstream SSE contract. Its WSS route checks NO_PROXY first, then selects the +returns the same downstream SSE contract. The canonical `openai` provider uses +upstream WebSocket by default; `providers.openai.upstreamWebsocket: false` sends +its streaming turns over HTTP/SSE instead. This explicit choice also makes +native mid-turn steering and injection unavailable on that provider. It does +not change the endpoint, credential, or downstream event format. +Its WSS route checks NO_PROXY first, then selects the first non-empty HTTPS_PROXY, https_proxy, ALL_PROXY, or all_proxy value. HTTP_PROXY alone does not route WSS. Unsupported or malformed selected proxy values skip the WebSocket attempt and use the existing SSE path immediately; they never fall through to a lower-priority proxy or direct WebSocket diff --git a/tests/helpers/ws-upstream-fixtures.ts b/tests/helpers/ws-upstream-fixtures.ts index 8fd6c719558..248b4608e28 100644 --- a/tests/helpers/ws-upstream-fixtures.ts +++ b/tests/helpers/ws-upstream-fixtures.ts @@ -13,7 +13,7 @@ import { */ export const BOUNDED_WS_RUNTIME = "1.4.0"; -export function shouldUseCodexWsUpstream(url: string, init?: RequestInit, upstreamWebsocket = false): boolean { +export function shouldUseCodexWsUpstream(url: string, init?: RequestInit, upstreamWebsocket?: boolean): boolean { return rawShouldUseCodexWsUpstream(url, init, BOUNDED_WS_RUNTIME, upstreamWebsocket); } diff --git a/tests/responses/ws-native-injection.test.ts b/tests/responses/ws-native-injection.test.ts index 9c67aa83410..9001c1a7a1a 100644 --- a/tests/responses/ws-native-injection.test.ts +++ b/tests/responses/ws-native-injection.test.ts @@ -416,6 +416,9 @@ test("public injection excludes custom gateways, forwarded auth and an unopted A expect(nativeResponseControlEligible({ ...provider, upstreamWebsocket: false }, channel)).toBe(false); expect(nativeResponseControlEligible({ ...provider, authMode: "forward" }, channel)).toBe(false); expect(nativeResponseControlEligible(provider)).toBe(false); + const canonical = injectionConfig().providers.openai; + expect(nativeResponseControlEligible(canonical, channel)).toBe(true); + expect(nativeResponseControlEligible({ ...canonical, upstreamWebsocket: false }, channel)).toBe(false); }); test("injection mode refuses simultaneous steering instead of fabricating protocol equivalence", () => { diff --git a/tests/responses/ws-upstream.test.ts b/tests/responses/ws-upstream.test.ts index f306942bc47..43d6b3aeef1 100644 --- a/tests/responses/ws-upstream.test.ts +++ b/tests/responses/ws-upstream.test.ts @@ -133,8 +133,7 @@ describe("shouldUseCodexWsUpstream", () => { }); test("keeps configured provider endpoints on bounded HTTP SSE", () => { - // The canonical backend ignores the flag. - expect(shouldUseCodexWsUpstream(CODEX_URL, streamingInit(), false)).toBe(true); + expect(shouldUseCodexWsUpstream(CODEX_URL, streamingInit(), false)).toBe(false); // Bun cannot reject oversized messages before assembling them, so even an // opted-in provider cannot join the WebSocket lane. expect(shouldUseCodexWsUpstream("https://sub2api.example.com/v1/responses", streamingInit(), true)).toBe(false); @@ -274,20 +273,21 @@ describe("providerFetch routing", () => { } as unknown as OcxProviderConfig; const wrapped = providerFetch(provider, BOUNDED_WS_RUNTIME); - // Eligible: WS adapter serves it, base fetch untouched. const wsResponse = await wrapped(CODEX_URL, streamingInit()); expect(wsResponse.headers.get("content-type")).toContain("text/event-stream"); expect(baseCalls).toHaveLength(0); expect(FakeWebSocket.instances).toHaveLength(1); - // Non-streaming body: base fetch. await wrapped(CODEX_URL, { method: "POST", body: JSON.stringify({ model: "m" }) }); - // Different host: base fetch. await wrapped("https://api.openai.com/v1/responses", streamingInit()); - // Request-object input: base fetch (WS path only handles string URLs). await wrapped(new Request(CODEX_URL, streamingInit() as RequestInit)); expect(baseCalls).toHaveLength(3); expect(FakeWebSocket.instances).toHaveLength(1); + provider.upstreamWebsocket = false; + const httpOnly = providerFetch(provider, BOUNDED_WS_RUNTIME); + expect(await (await httpOnly(CODEX_URL, streamingInit())).text()).toBe("base"); + expect(baseCalls).toHaveLength(4); + expect(FakeWebSocket.instances).toHaveLength(1); }); test("routes an opt-in provider's Responses streams over bounded HTTP SSE", async () => { diff --git a/tests/server/management-provider-validation.test.ts b/tests/server/management-provider-validation.test.ts index a663550a511..46da86d4234 100644 --- a/tests/server/management-provider-validation.test.ts +++ b/tests/server/management-provider-validation.test.ts @@ -941,6 +941,8 @@ describe("provider management validation", () => { }); test("provider management permits snapshot repair only on canonical OpenAI forward seeds", () => { + expect(providerManagementConfigError("openai", { ...canonicalDirect, upstreamWebsocket: false })).toBeNull(); + expect(providerManagementConfigError("openai", { ...canonicalDirect, upstreamWebsocket: true })).toContain("must be false or omitted"); for (const mode of ["pool", "direct"] as const) { expect(providerManagementConfigError("openai", { ...canonicalDirect, @@ -3367,14 +3369,14 @@ describe("provider management validation", () => { createManagementConvergeCodex: catalogConvergenceFactory(), }); }; - const canonical = await post({ name: "openai", provider: canonicalDirect }); + const canonical = await post({ name: "openai", provider: { ...canonicalDirect, upstreamWebsocket: false } }); expect(canonical?.status).toBe(200); + expect(loadConfig().providers.openai?.upstreamWebsocket).toBe(false); expect(resolvedError).toHaveBeenCalledWith( "openai", expect.objectContaining({ baseUrl: canonicalDirect.baseUrl }), { allowBenchmarkAddresses: true }, ); - resolvedError.mockResolvedValueOnce( "baseUrl hostname custom.example.test resolves to a benchmark address (198.18.0.30); set allowPrivateNetwork:true only for intentionally local/self-hosted providers", );