Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions docs-site/src/content/docs/guides/codex-integration.md
Original file line number Diff line number Diff line change
Expand Up @@ -98,6 +98,13 @@ is a `POST` to the canonical Responses URL or a configured WebSocket route, and
falls back to it when the request cannot be prepared, the `response.create` frame exceeds its size
limit, or the proxy route cannot carry the socket.

To keep the built-in ChatGPT provider on HTTP/SSE, set `providers.openai.upstreamWebsocket`
to `false` in `~/.opencodex/config.json` and restart the proxy. Merge this field into the
existing `openai` provider; preserve its account mode and other settings. Omit the field
to restore the default upstream WebSocket selection. This setting does not change the
client-facing `websockets` switch or the ChatGPT account used for the request. Native
mid-turn steering and injection need upstream WebSocket and are unavailable while it is off.

Local provider pacing can also hold a request before it is dispatched at all. So a slow first
output has several possible contributors, and upstream queueing is only one of them. `ocx doctor`
classifies configuration and measures none of these: compare actual transport, pacing, network,
Expand Down
6 changes: 6 additions & 0 deletions src/server/auth-cors.ts
Original file line number Diff line number Diff line change
Expand Up @@ -744,6 +744,12 @@ export function providerManagementConfigError(
// validation and then rejected by the seed comparison, so canonical OpenAI could never
// set OR clear it — the value was admitted and then refused in the same request.
delete canonicalCandidate.annotateEmptyToolOutputs;
// Canonical ChatGPT keeps WebSocket as the default, but an operator may
// select the existing HTTP/SSE path without changing its auth or endpoint.
if (raw.upstreamWebsocket !== undefined) {
if (raw.upstreamWebsocket !== false) return "provider openai upstreamWebsocket must be false or omitted";
delete canonicalCandidate.upstreamWebsocket;
}
const canonical = seed && (options?.allowOperatorOverlays
? matchesCanonicalProviderSeed(canonicalCandidate, seed)
: sameCanonicalProviderSeed(canonicalCandidate, seed));
Expand Down
2 changes: 1 addition & 1 deletion src/server/responses/fetch-helpers.ts
Original file line number Diff line number Diff line change
Expand Up @@ -268,7 +268,7 @@ export function providerFetch(
// transport (measured ~3s faster TTFT than the SSE POST queue); everything
// else keeps the provider's HTTP fetch. See ws-upstream.ts for the details.
const unpaced = async (input: Parameters<typeof globalThis.fetch>[0], init?: RequestInit) => {
const upstreamWebsocket = provider.upstreamWebsocket === true;
const upstreamWebsocket = provider.upstreamWebsocket;
if (!options.httpOnly && typeof input === "string" && init
&& shouldUseCodexWsUpstream(input, init, runtime, upstreamWebsocket)) {
const egress = egressFor(input);
Expand Down
4 changes: 2 additions & 2 deletions src/server/responses/native-response-control.ts
Original file line number Diff line number Diff line change
Expand Up @@ -31,9 +31,9 @@ export function markNativeControlResponse(response: Response): Response { native
/** Recognize a marked native response by identity, not by caller-controlled content. */
export function isNativeControlResponse(response: Response): boolean { return nativeControlResponses.has(response); }

/** Preserve canonical ChatGPT eligibility; only injection may use the separately billed public API. */
/** Canonical ChatGPT needs its upstream WS enabled; only injection may use the separately billed public API. */
export function nativeResponseControlEligible(provider: OcxProviderConfig, control?: NativeResponseControl): boolean {
if (isCanonicalOpenAiForwardProvider(provider)) return true;
if (isCanonicalOpenAiForwardProvider(provider)) return provider.upstreamWebsocket !== false;
return control?.kind === "injection" && provider.adapter === "openai-responses"
&& provider.upstreamWebsocket === true && provider.authMode !== "forward"
&& provider.baseUrl?.replace(/\/+$/, "") === "https://api.openai.com/v1";
Expand Down
3 changes: 2 additions & 1 deletion src/server/responses/ws-upstream.ts
Original file line number Diff line number Diff line change
Expand Up @@ -85,10 +85,11 @@ export function shouldUseCodexWsUpstream(
url: string,
init?: RequestInit,
runtime: BunRuntimeGateInput = currentBunRuntimeIdentity(),
upstreamWebsocketConfigured = false,
upstreamWebsocketConfigured?: boolean,
): boolean {
if (!bunSupportsBoundedCodexWsRelay(runtime)) return false;
if (socks5ProxyFromEnv()) return false;
if (url === CODEX_RESPONSES_HTTP_URL && upstreamWebsocketConfigured === false) return false;
// Bun's client WebSocket API delivers only fully assembled messages and has
// no enforceable inbound payload limit. Keep arbitrary provider endpoints on
// bounded HTTP/SSE until the client can reject fragmented text and binary
Expand Down
1 change: 1 addition & 0 deletions structure/config.md
Original file line number Diff line number Diff line change
Expand Up @@ -101,6 +101,7 @@ merge cannot turn them into a valid config while discarding the original bytes.
| Retained state | `appOwnedMemoryBudgetMb` | Process-wide eviction target for app-owned logs, caches, blobs, and continuation payloads. Default 256 MiB, valid 64..4096; pinned state may temporarily exceed the target, but every pin-capable store has a finite local cap and their documented aggregate stays below `APP_OWNED_WORST_CASE_PINNED_BYTES` (512 MiB). Neither value caps RSS or native runtime memory. |
| Spend | `spend.root`, `spend.identity`, `spend.pool`, `spend.retentionDays` | Durable token ceilings for the spend-reservation ledger. Absent is the default and means observe-only accounting: spend is still journaled and nothing is refused, so observe-only and enforced servers take the same state-directory writer lease. One live process may write one directory; explicit sibling instances need separate `OPENCODEX_HOME` directories. There is no default figure for any scope — the ledger is on by default, so a shipped ceiling would refuse real traffic on upgrade against a number nobody chose. Strictly validated and positive-integer only, because 0 would read as a budget and refuse everything; a malformed section degrades to no ceiling, which is why the write path rejects it and load diagnostics report it. Resolution and application live in `src/lib/spend-reservation-ledger.ts`; see [`transports/responses.md`](transports/responses.md). |
| Transport | stream mode, timeouts, proxy settings, `websockets`, `emptyCompletionRetry` | `streamMode` persists in config.json; Windows services need a persisted input, and macOS uses it for explicit eager-relay opt-in. Empty-completion replay is an explicit top-level opt-in because its second upstream request may be billable. |
| Canonical ChatGPT upstream transport | `providers.openai.upstreamWebsocket` | Omitted uses upstream WebSocket when eligible; explicit `false` selects HTTP/SSE without changing the canonical provider identity. `true` is rejected on the canonical row. This is independent of the client-facing `websockets` setting. |
| Provider egress | `providers.<name>.proxy`, `providers.<name>.noProxy` | An absent `proxy` inherits global egress; `"direct"` or `null` forces direct egress; HTTP(S) and SOCKS5(H) URLs select a provider-owned proxy. `noProxy` uses NO_PROXY syntax and sends a matching destination direct across either a provider-owned or inherited global proxy. `src/lib/provider-egress.ts` owns parsing and request-local resolution. |
| Credentials | `apiKeys` | Data-plane only; never admitted to `/api/*`. |
| Lifecycle | `codexAutoStart`, shim/start behavior, resume-history sync, storage cleanup | Startup safety reads these; see [`gui-and-management-api.md`](gui-and-management-api.md). |
Expand Down
7 changes: 6 additions & 1 deletion structure/transports/streaming-health.md
Original file line number Diff line number Diff line change
Expand Up @@ -211,7 +211,12 @@ the upgrade with 426 so Codex falls back to HTTP cleanly.

That setting controls the client-facing upgrade only. The transparent upstream
ChatGPT WS optimization described above is selected independently and still
returns the same downstream SSE contract. Its WSS route checks NO_PROXY first, then selects the
returns the same downstream SSE contract. The canonical `openai` provider uses
upstream WebSocket by default; `providers.openai.upstreamWebsocket: false` sends
its streaming turns over HTTP/SSE instead. This explicit choice also makes
native mid-turn steering and injection unavailable on that provider. It does
not change the endpoint, credential, or downstream event format.
Its WSS route checks NO_PROXY first, then selects the
first non-empty HTTPS_PROXY, https_proxy, ALL_PROXY, or all_proxy value. HTTP_PROXY alone does not
route WSS. Unsupported or malformed selected proxy values skip the WebSocket attempt and use the
existing SSE path immediately; they never fall through to a lower-priority proxy or direct WebSocket
Expand Down
2 changes: 1 addition & 1 deletion tests/helpers/ws-upstream-fixtures.ts
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,7 @@ import {
*/
export const BOUNDED_WS_RUNTIME = "1.4.0";

export function shouldUseCodexWsUpstream(url: string, init?: RequestInit, upstreamWebsocket = false): boolean {
export function shouldUseCodexWsUpstream(url: string, init?: RequestInit, upstreamWebsocket?: boolean): boolean {
return rawShouldUseCodexWsUpstream(url, init, BOUNDED_WS_RUNTIME, upstreamWebsocket);
}

Expand Down
3 changes: 3 additions & 0 deletions tests/responses/ws-native-injection.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -416,6 +416,9 @@ test("public injection excludes custom gateways, forwarded auth and an unopted A
expect(nativeResponseControlEligible({ ...provider, upstreamWebsocket: false }, channel)).toBe(false);
expect(nativeResponseControlEligible({ ...provider, authMode: "forward" }, channel)).toBe(false);
expect(nativeResponseControlEligible(provider)).toBe(false);
const canonical = injectionConfig().providers.openai;
expect(nativeResponseControlEligible(canonical, channel)).toBe(true);
expect(nativeResponseControlEligible({ ...canonical, upstreamWebsocket: false }, channel)).toBe(false);
});

test("injection mode refuses simultaneous steering instead of fabricating protocol equivalence", () => {
Expand Down
12 changes: 6 additions & 6 deletions tests/responses/ws-upstream.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -133,8 +133,7 @@ describe("shouldUseCodexWsUpstream", () => {
});

test("keeps configured provider endpoints on bounded HTTP SSE", () => {
// The canonical backend ignores the flag.
expect(shouldUseCodexWsUpstream(CODEX_URL, streamingInit(), false)).toBe(true);
expect(shouldUseCodexWsUpstream(CODEX_URL, streamingInit(), false)).toBe(false);
// Bun cannot reject oversized messages before assembling them, so even an
// opted-in provider cannot join the WebSocket lane.
expect(shouldUseCodexWsUpstream("https://sub2api.example.com/v1/responses", streamingInit(), true)).toBe(false);
Expand Down Expand Up @@ -274,20 +273,21 @@ describe("providerFetch routing", () => {
} as unknown as OcxProviderConfig;
const wrapped = providerFetch(provider, BOUNDED_WS_RUNTIME);

// Eligible: WS adapter serves it, base fetch untouched.
const wsResponse = await wrapped(CODEX_URL, streamingInit());
expect(wsResponse.headers.get("content-type")).toContain("text/event-stream");
expect(baseCalls).toHaveLength(0);
expect(FakeWebSocket.instances).toHaveLength(1);

// Non-streaming body: base fetch.
await wrapped(CODEX_URL, { method: "POST", body: JSON.stringify({ model: "m" }) });
// Different host: base fetch.
await wrapped("https://api.openai.com/v1/responses", streamingInit());
// Request-object input: base fetch (WS path only handles string URLs).
await wrapped(new Request(CODEX_URL, streamingInit() as RequestInit));
expect(baseCalls).toHaveLength(3);
expect(FakeWebSocket.instances).toHaveLength(1);
provider.upstreamWebsocket = false;
const httpOnly = providerFetch(provider, BOUNDED_WS_RUNTIME);
expect(await (await httpOnly(CODEX_URL, streamingInit())).text()).toBe("base");
expect(baseCalls).toHaveLength(4);
expect(FakeWebSocket.instances).toHaveLength(1);
});

test("routes an opt-in provider's Responses streams over bounded HTTP SSE", async () => {
Expand Down
6 changes: 4 additions & 2 deletions tests/server/management-provider-validation.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -941,6 +941,8 @@ describe("provider management validation", () => {
});

test("provider management permits snapshot repair only on canonical OpenAI forward seeds", () => {
expect(providerManagementConfigError("openai", { ...canonicalDirect, upstreamWebsocket: false })).toBeNull();
expect(providerManagementConfigError("openai", { ...canonicalDirect, upstreamWebsocket: true })).toContain("must be false or omitted");
for (const mode of ["pool", "direct"] as const) {
expect(providerManagementConfigError("openai", {
...canonicalDirect,
Expand Down Expand Up @@ -3367,14 +3369,14 @@ describe("provider management validation", () => {
createManagementConvergeCodex: catalogConvergenceFactory(),
});
};
const canonical = await post({ name: "openai", provider: canonicalDirect });
const canonical = await post({ name: "openai", provider: { ...canonicalDirect, upstreamWebsocket: false } });
expect(canonical?.status).toBe(200);
expect(loadConfig().providers.openai?.upstreamWebsocket).toBe(false);
expect(resolvedError).toHaveBeenCalledWith(
"openai",
expect.objectContaining({ baseUrl: canonicalDirect.baseUrl }),
{ allowBenchmarkAddresses: true },
);

resolvedError.mockResolvedValueOnce(
"baseUrl hostname custom.example.test resolves to a benchmark address (198.18.0.30); set allowPrivateNetwork:true only for intentionally local/self-hosted providers",
);
Expand Down
Loading