From d25837d20b61f72f7bbc6d963273fde14f9b9827 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Fri, 21 Aug 2026 12:52:02 +0900 Subject: [PATCH] fix(xai): default Grok OAuth to chat --- src/providers/fastwire.ts | 42 ++-- src/providers/registry.ts | 11 +- structure/04_transports-and-sidecars.md | 16 +- tests/adapter-resolve.test.ts | 14 +- tests/fastwire-policy.test.ts | 108 ++++++++-- ...erver-xai-chat-reasoning-streaming.test.ts | 195 ++++++++++++++++++ tests/server-xai-oauth-401-replay.test.ts | 3 +- tests/server-xai-responses-streaming.test.ts | 4 +- 8 files changed, 334 insertions(+), 59 deletions(-) create mode 100644 tests/server-xai-chat-reasoning-streaming.test.ts diff --git a/src/providers/fastwire.ts b/src/providers/fastwire.ts index 1972c4339f2..63cd6425320 100644 --- a/src/providers/fastwire.ts +++ b/src/providers/fastwire.ts @@ -143,34 +143,40 @@ function resolvePolicyAdapter( inbound: InboundWire, ): { adapter: string; hardPinned: boolean; forwardCallerServiceTier?: boolean } { // Hard pins and configured overrides deliberately use the same exact-key semantics as - // resolveWireProtocolOverride(). Registry defaults alone normalize ids at their boundary. + // resolveWireProtocolOverride(). Registry route policy normalizes ids at its boundary. const hardPin = Object.hasOwn(authority.hardPins, modelId) ? authority.hardPins[modelId] : undefined; if (typeof hardPin === "string") return { adapter: hardPin, hardPinned: true }; if (authority.modelWireOverrideAllowed) { + const registryDefault = MODEL_ADAPTER_OVERRIDE_ALLOWED.has(authority.providerAdapter) + ? registryDefaultForModel( + authority.registryWireDefaults, + modelId, + inbound, + authority.providerAuthMode, + ) + : undefined; const configured = Object.hasOwn(authority.modelAdapters, modelId) ? authority.modelAdapters[modelId] : undefined; if (typeof configured === "string" && MODEL_ADAPTER_OVERRIDE_ALLOWED.has(configured)) { - return { adapter: configured, hardPinned: false }; + return { + adapter: configured, + hardPinned: false, + ...(registryDefault?.forwardCallerServiceTier === false + ? { forwardCallerServiceTier: false } + : {}), + }; } - if (MODEL_ADAPTER_OVERRIDE_ALLOWED.has(authority.providerAdapter)) { - const registryDefault = registryDefaultForModel( - authority.registryWireDefaults, - modelId, - inbound, - authority.providerAuthMode, - ); - if (registryDefault !== undefined) { - return { - adapter: registryDefault.adapter, - hardPinned: false, - ...(registryDefault.forwardCallerServiceTier !== undefined - ? { forwardCallerServiceTier: registryDefault.forwardCallerServiceTier } - : {}), - }; - } + if (registryDefault !== undefined) { + return { + adapter: registryDefault.adapter, + hardPinned: false, + ...(registryDefault.forwardCallerServiceTier !== undefined + ? { forwardCallerServiceTier: registryDefault.forwardCallerServiceTier } + : {}), + }; } } return { adapter: authority.providerAdapter, hardPinned: false }; diff --git a/src/providers/registry.ts b/src/providers/registry.ts index f4441557858..7ac39720d50 100644 --- a/src/providers/registry.ts +++ b/src/providers/registry.ts @@ -1026,19 +1026,18 @@ export const PROVIDER_REGISTRY: readonly ProviderRegistryEntry[] = [ // grok-4.5; the reasoning ladder does not — 4.6 adds the documented xhigh rung. models: ["grok-4.6", "grok-4.5", "grok-4.3", "grok-4.20-0309-reasoning", "grok-4.20-0309-non-reasoning", "grok-build-0.1", "grok-composer-2.5-fast"], defaultModel: "grok-4.5", - // The current Grok CLI catalog declares both subscription models as native Responses - // backends. Keep API-key and translated Chat/Anthropic callers on their existing wire; - // Codex Responses traffic can relay xAI's SSE as it arrives instead of waiting for the - // Chat Completions compatibility stream to flush at the end of a reasoning turn. + // Keep Codex Responses callers on the compatibility Chat wire until xAI can replay + // opaque reasoning continuation and compaction state across later turns. The scoped + // declaration also keeps caller-owned service tiers off the OAuth subscription route. modelWireDefaults: { "grok-4.6": { - wire: "openai-responses", + wire: "openai-chat", inbound: ["responses"], authModes: ["oauth"], forwardCallerServiceTier: false, }, "grok-4.5": { - wire: "openai-responses", + wire: "openai-chat", inbound: ["responses"], authModes: ["oauth"], forwardCallerServiceTier: false, diff --git a/structure/04_transports-and-sidecars.md b/structure/04_transports-and-sidecars.md index 64cf8e83834..d63e594b052 100644 --- a/structure/04_transports-and-sidecars.md +++ b/structure/04_transports-and-sidecars.md @@ -80,14 +80,14 @@ different custom destination does not inherit its upstream assumptions. Object-f also narrow the decision by inbound protocol and authentication mode; an auth-scoped default must not leak from a subscription transport into an API-key or forwarded-credential route. -xAI keeps `openai-chat` as its provider-wide compatibility wire. The official Grok CLI catalog -declares the Grok 4.5 and 4.6 subscription models as Responses backends, so only OAuth-backed native -Responses traffic for those exact models selects `openai-responses`. API-key requests, translated -Chat/Anthropic callers, other Grok models, and explicit model adapter overrides retain their -existing wire. This lets Codex receive native xAI SSE deltas as they arrive without widening the -credential or compatibility boundary. These OAuth subscription defaults drop caller-owned -`service_tier`; they neither advertise nor inject Fast. The API-key transport remains governed by -its separate capability declaration. +xAI keeps `openai-chat` as both its provider-wide compatibility wire and the default for Grok 4.5 +and 4.6 subscription traffic. The official Grok CLI catalog declares those models as Responses +backends, but the current gateway rejects opaque reasoning continuation and compaction state on +later turns. Operators may still select `openai-responses` with an explicit model adapter override +while that compatibility work continues. The OAuth route drops caller-owned `service_tier` even +when an override selects Responses, and native Responses OAuth 401 replay remains available to +explicit opt-ins. API-key requests, translated Chat/Anthropic callers, and other Grok models retain +their existing wire and tier policy. OpenCode Go documents `gpt-5.6-luna` on `/zen/go/v1/responses` while sibling models use its Chat or Anthropic endpoints. The built-in preset therefore selects `openai-responses` only for Luna and diff --git a/tests/adapter-resolve.test.ts b/tests/adapter-resolve.test.ts index ceffd921491..927864eb9ff 100644 --- a/tests/adapter-resolve.test.ts +++ b/tests/adapter-resolve.test.ts @@ -100,10 +100,10 @@ describe("registry per-model wire defaults", () => { }); } - test("routes current xAI subscription models through Responses for native Codex traffic", () => { + test("keeps current xAI subscription models on Chat by default", () => { for (const model of ["grok-4.6", "grok-4.5"]) { expect(resolveWireProtocolOverride("xai", model, xai("oauth"), "responses").adapter) - .toBe("openai-responses"); + .toBe("openai-chat"); } }); @@ -118,10 +118,12 @@ describe("registry per-model wire defaults", () => { .toBe("openai-chat"); }); - test("an explicit xAI Chat override opts out of the subscription Responses default", () => { - const provider = xai("oauth", { modelAdapters: { "grok-4.6": "openai-chat" } }); - expect(resolveWireProtocolOverride("xai", "grok-4.6", provider, "responses").adapter) - .toBe("openai-chat"); + test("an explicit xAI Responses override opts into the native wire", () => { + for (const model of ["grok-4.6", "grok-4.5"]) { + const provider = xai("oauth", { modelAdapters: { [model]: "openai-responses" } }); + expect(resolveWireProtocolOverride("xai", model, provider, "responses").adapter) + .toBe("openai-responses"); + } }); function deepseek(overrides: Partial = {}): OcxProviderConfig { diff --git a/tests/fastwire-policy.test.ts b/tests/fastwire-policy.test.ts index 6cd7529a949..412d0524a47 100644 --- a/tests/fastwire-policy.test.ts +++ b/tests/fastwire-policy.test.ts @@ -256,25 +256,95 @@ describe("resolveFastPolicy matrix", () => { expect(fastPolicyForModel(provider, MODEL, "fixture").capability).toBe(false); }); - test("captured xAI registry defaults keep OAuth and key transports separate", () => { - const oauthProvider = Object.freeze({ - adapter: "openai-chat", - baseUrl: "https://api.x.ai/v1", - authMode: "oauth" as const, - }); - const keyProvider = Object.freeze({ - adapter: "openai-chat", - baseUrl: "https://api.x.ai/v1", - authMode: "key" as const, - }); - - expect(fastPolicyForModel(oauthProvider, "grok-4.6", "xai")).toMatchObject({ - adapter: "openai-responses", - eligibility: "unclassified", - forwardCallerTier: false, - }); - expect(fastPolicyForModel(keyProvider, "grok-4.6", "xai").adapter) - .toBe("openai-chat"); + test("locks the five-row xAI and DeepSeek wire/tier regression matrix", () => { + const rows = [ + { + name: "xAI OAuth default", + providerName: "xai", + modelIds: ["grok-4.6", "grok-4.5"], + provider: { + adapter: "openai-chat", + baseUrl: "https://api.x.ai/v1", + authMode: "oauth" as const, + }, + adapter: "openai-chat", + forwardCallerTier: false, + callerTier: undefined, + settledCallerTier: undefined, + }, + { + name: "xAI OAuth Responses override", + providerName: "xai", + modelIds: ["grok-4.6", "grok-4.5"], + provider: { + adapter: "openai-chat", + baseUrl: "https://api.x.ai/v1", + authMode: "oauth" as const, + modelAdapters: { "grok-4.6": "openai-responses", "grok-4.5": "openai-responses" }, + }, + adapter: "openai-responses", + forwardCallerTier: false, + callerTier: "flex", + settledCallerTier: undefined, + }, + { + name: "xAI API-key default", + providerName: "xai", + modelIds: ["grok-4.6", "grok-4.5"], + provider: { + adapter: "openai-chat", + baseUrl: "https://api.x.ai/v1", + authMode: "key" as const, + }, + adapter: "openai-chat", + forwardCallerTier: false, + callerTier: undefined, + settledCallerTier: undefined, + }, + { + name: "xAI API-key Responses override", + providerName: "xai", + modelIds: ["grok-4.6", "grok-4.5"], + provider: { + adapter: "openai-chat", + baseUrl: "https://api.x.ai/v1", + authMode: "key" as const, + modelAdapters: { "grok-4.6": "openai-responses", "grok-4.5": "openai-responses" }, + }, + adapter: "openai-responses", + forwardCallerTier: true, + callerTier: "flex", + settledCallerTier: "flex", + }, + { + name: "DeepSeek V4 defaults", + providerName: "deepseek", + modelIds: ["deepseek-v4-flash", "deepseek-v4-pro"], + provider: { + adapter: "openai-chat", + baseUrl: "https://api.deepseek.com", + authMode: "key" as const, + }, + adapter: "openai-responses", + forwardCallerTier: false, + callerTier: undefined, + settledCallerTier: undefined, + }, + ] as const; + + for (const row of rows) { + for (const modelId of row.modelIds) { + const policy = fastPolicyForModel(row.provider, modelId, row.providerName); + expect(policy.adapter).toBe(row.adapter); + expect(policy.forwardCallerTier).toBe(row.forwardCallerTier); + expect(tierValueAfterDecision(decideTier(policy, undefined, undefined), undefined)) + .toBeUndefined(); + if (row.callerTier !== undefined) { + expect(tierValueAfterDecision(decideTier(policy, undefined, row.callerTier), row.callerTier)) + .toBe(row.settledCallerTier); + } + } + } }); test("prototype-named providers and models use only own wire-policy rows", () => { diff --git a/tests/server-xai-chat-reasoning-streaming.test.ts b/tests/server-xai-chat-reasoning-streaming.test.ts new file mode 100644 index 00000000000..cbe71782510 --- /dev/null +++ b/tests/server-xai-chat-reasoning-streaming.test.ts @@ -0,0 +1,195 @@ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import { mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { saveConfig } from "../src/config"; +import { saveCredential } from "../src/oauth/store"; +import { + XAI_GROK_CLI_BASE_URL, + XAI_GROK_CLIENT_VERSION, +} from "../src/providers/xai-transport"; +import { startServer } from "../src/server"; +import type { OcxConfig } from "../src/types"; +import { installIsolatedCodexHome, type IsolatedCodexHome } from "./helpers/isolated-codex-home"; + +const CHAT_ENDPOINT = `${XAI_GROK_CLI_BASE_URL}/chat/completions`; +const encoder = new TextEncoder(); + +let testDir = ""; +let previousHome: string | undefined; +let isolatedCodexHome: IsolatedCodexHome | null = null; +let originalFetch: typeof fetch; + +beforeEach(async () => { + originalFetch = globalThis.fetch; + previousHome = process.env.OPENCODEX_HOME; + isolatedCodexHome = installIsolatedCodexHome("ocx-xai-chat-reasoning-codex-"); + testDir = mkdtempSync(join(tmpdir(), "ocx-xai-chat-reasoning-")); + process.env.OPENCODEX_HOME = testDir; + await saveCredential("xai", { + access: "stream-access", + refresh: "stream-refresh", + expires: Date.now() + 3_600_000, + accountId: "xai-stream-account", + source: "oauth", + }); +}); + +afterEach(() => { + globalThis.fetch = originalFetch; + if (previousHome === undefined) delete process.env.OPENCODEX_HOME; + else process.env.OPENCODEX_HOME = previousHome; + isolatedCodexHome?.restore(); + isolatedCodexHome = null; + if (testDir) rmSync(testDir, { recursive: true, force: true }); +}); + +function config(): OcxConfig { + return { + port: 0, + hostname: "127.0.0.1", + defaultProvider: "xai", + providers: { + xai: { + adapter: "openai-chat", + baseUrl: "https://api.x.ai/v1", + authMode: "oauth", + models: ["grok-4.6"], + }, + }, + } as OcxConfig; +} + +function chatSse(payload: unknown): Uint8Array { + return encoder.encode(`data: ${JSON.stringify(payload)}\n\n`); +} + +describe("xAI OAuth Chat reasoning streaming", () => { + test("bridges reasoning_content before content and response.completed", async () => { + let releaseCompletion!: () => void; + const completionGate = new Promise(resolve => { releaseCompletion = resolve; }); + let completionReleased = false; + let outboundBody: Record | undefined; + let outboundHeaders: Headers | undefined; + let upstreamCalls = 0; + + globalThis.fetch = (async (input, init) => { + const url = input instanceof Request ? input.url : String(input); + if (url !== CHAT_ENDPOINT) return originalFetch(input, init); + upstreamCalls += 1; + outboundHeaders = new Headers(init?.headers); + outboundBody = JSON.parse(String(init?.body)) as Record; + + const body = new ReadableStream({ + start(controller) { + controller.enqueue(chatSse({ + id: "chatcmpl_xai_reasoning", + object: "chat.completion.chunk", + model: "grok-4.6", + choices: [{ index: 0, delta: { role: "assistant" }, finish_reason: null }], + })); + controller.enqueue(chatSse({ + id: "chatcmpl_xai_reasoning", + object: "chat.completion.chunk", + model: "grok-4.6", + choices: [{ index: 0, delta: { reasoning_content: "first thought" }, finish_reason: null }], + })); + controller.enqueue(chatSse({ + id: "chatcmpl_xai_reasoning", + object: "chat.completion.chunk", + model: "grok-4.6", + choices: [{ index: 0, delta: { reasoning_content: " then second" }, finish_reason: null }], + })); + void completionGate.then(() => { + completionReleased = true; + controller.enqueue(chatSse({ + id: "chatcmpl_xai_reasoning", + object: "chat.completion.chunk", + model: "grok-4.6", + choices: [{ index: 0, delta: { content: "answer" }, finish_reason: null }], + })); + controller.enqueue(chatSse({ + id: "chatcmpl_xai_reasoning", + object: "chat.completion.chunk", + model: "grok-4.6", + choices: [{ index: 0, delta: {}, finish_reason: "stop" }], + usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 }, + })); + controller.enqueue(encoder.encode("data: [DONE]\n\n")); + controller.close(); + }); + }, + }); + return new Response(body, { headers: { "content-type": "text/event-stream" } }); + }) as typeof fetch; + + saveConfig(config()); + const server = startServer(0); + let reader: ReadableStreamDefaultReader | undefined; + try { + const response = await originalFetch(new URL("/v1/responses", server.url), { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ + model: "xai/grok-4.6", + input: "hello", + stream: true, + store: false, + service_tier: "priority", + reasoning: { effort: "xhigh", summary: "auto" }, + }), + }); + expect(response.status).toBe(200); + reader = response.body!.getReader(); + const decoder = new TextDecoder(); + let received = ""; + await Promise.race([ + (async () => { + while (!received.includes("response.reasoning_summary_text.delta")) { + const chunk = await reader!.read(); + if (chunk.done) throw new Error("stream ended before the first xAI reasoning delta"); + received += decoder.decode(chunk.value, { stream: true }); + } + })(), + new Promise((_, reject) => setTimeout( + () => reject(new Error("xAI reasoning was not relayed before completion")), + 1_500, + )), + ]); + + expect(received).toContain("first thought"); + expect(received).not.toContain("response.completed"); + expect(completionReleased).toBe(false); + expect(upstreamCalls).toBe(1); + expect(outboundBody?.model).toBe("grok-4.6"); + expect(outboundBody?.messages).toBeArray(); + expect(outboundBody?.stream).toBe(true); + expect(outboundBody?.service_tier).toBeUndefined(); + expect(outboundBody?.reasoning_effort).toBe("xhigh"); + expect(outboundBody?.input).toBeUndefined(); + expect(outboundBody?.reasoning).toBeUndefined(); + expect(outboundHeaders?.get("authorization")).toBe("Bearer stream-access"); + expect(outboundHeaders?.get("x-grok-client-identifier")).toBe("opencodex"); + expect(outboundHeaders?.get("x-grok-client-version")).toBe(XAI_GROK_CLIENT_VERSION); + + releaseCompletion(); + while (true) { + const chunk = await reader.read(); + if (chunk.done) break; + received += decoder.decode(chunk.value, { stream: true }); + } + const reasoningIndex = received.indexOf("response.reasoning_summary_text.delta"); + const contentIndex = received.indexOf("response.output_text.delta"); + const completedIndex = received.indexOf("response.completed"); + expect(reasoningIndex).toBeGreaterThanOrEqual(0); + expect(contentIndex).toBeGreaterThan(reasoningIndex); + expect(completedIndex).toBeGreaterThan(contentIndex); + expect(received).toContain("then second"); + expect(received).toContain("answer"); + } finally { + releaseCompletion(); + await reader?.cancel().catch(() => {}); + await server.stop(true); + } + }, 10_000); +}); diff --git a/tests/server-xai-oauth-401-replay.test.ts b/tests/server-xai-oauth-401-replay.test.ts index abe2e62673c..fd7f6dff0d1 100644 --- a/tests/server-xai-oauth-401-replay.test.ts +++ b/tests/server-xai-oauth-401-replay.test.ts @@ -60,6 +60,7 @@ function xaiConfig(authMode: "oauth" | "key" = "oauth"): OcxConfig { baseUrl: "https://api.x.ai/v1", authMode, ...(authMode === "key" ? { apiKey: "xai-api-key" } : {}), + ...(authMode === "oauth" ? { modelAdapters: { "grok-4.5": "openai-responses" } } : {}), models: ["grok-4.5"], }, }, @@ -142,7 +143,7 @@ function installOAuthFetch( return { chatAuth, counts }; } -describe("xAI OAuth upstream 401 replay", () => { +describe("xAI OAuth Responses opt-in upstream 401 replay", () => { test("initial OAuth refresh projects raw provider failures before responding", async () => { await seedOAuth(0); saveConfig(xaiConfig()); diff --git a/tests/server-xai-responses-streaming.test.ts b/tests/server-xai-responses-streaming.test.ts index 7195436a60a..5051814c908 100644 --- a/tests/server-xai-responses-streaming.test.ts +++ b/tests/server-xai-responses-streaming.test.ts @@ -56,6 +56,8 @@ function config(): OcxConfig { baseUrl: "https://api.x.ai/v1", authMode: "oauth", models: ["grok-4.6"], + modelAdapters: { "grok-4.6": "openai-responses" }, + modelSupportsServiceTier: { "grok-4.6": false }, }, }, } as OcxConfig; @@ -65,7 +67,7 @@ function sse(payload: unknown): Uint8Array { return encoder.encode(`data: ${JSON.stringify(payload)}\n\n`); } -describe("xAI OAuth Responses streaming", () => { +describe("xAI OAuth Responses streaming opt-in", () => { test("uses the native Responses wire and relays the first delta before completion", async () => { let releaseCompletion!: () => void; const completionGate = new Promise(resolve => { releaseCompletion = resolve; });