Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
- **fix(command-code):** route Responses-shaped bodies to `/provider/v1/responses`, where `reasoning: {"effort":"none"}` is honored. The chat endpoint's effort enum is `low|medium|high|xhigh|max` — no disable value — and silently drops a nested `reasoning.effort`, so reasoning-off was impossible for GPT-5.6 and DeepSeek models. Verified live 2026-09-24: `/responses` + effort none → `reasoning_tokens: 0`; `/chat` + `reasoning:{effort:"none"}` → 41 reasoning tokens. ([#14692](https://github.com/diegosouzapw/OmniRoute/pull/14692)) — thanks @adivekar-utexas
185 changes: 134 additions & 51 deletions open-sse/executors/commandCode.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,10 @@ import { randomUUID } from "node:crypto";
import { isVisionModelId } from "@/shared/constants/visionModels";
import { MUSE_SPARK_PATTERN } from "./base/reasoningEffort.ts";
import { REGISTRY } from "../config/providerRegistry.ts";
import {
isResponsesShapedBody,
projectResponsesForCli,
} from "./commandCode/responsesProjection.ts";
import {
BaseExecutor,
mergeUpstreamExtraHeaders,
Expand All @@ -19,6 +23,14 @@ export const COMMAND_CODE_VERSION = process.env.COMMAND_CODE_VERSION?.trim() ||
// "Too big: expected number to be <=200000 at params.max_tokens". We only clamp
// a client-supplied value down; we never fabricate this number for requests
// that omit the field.
//
// The quoted error names `params.max_tokens`, which is the /alpha/generate shape.
// The flat Chat surface uses top-level `max_tokens` and the Responses surface
// uses `max_output_tokens`; Command Code documents the 200_000 limit only for
// the first. The same gateway fronts all three, so this constant is applied to
// every output-cap field on the assumption the ceiling is endpoint-wide. If a
// live request ever 400s on max_output_tokens at a different bound, split the
// constant per surface rather than loosening this one.
const MAX_COMMAND_CODE_TOKENS = 200_000;
const encoder = new TextEncoder();

Expand Down Expand Up @@ -75,12 +87,16 @@ function clampMaxTokens(value: unknown): number | undefined {
*/
const MUSE_SPARK_MIN_OUTPUT_TOKENS = 512;

function applyMuseSparkMinOutputTokens(model: string, body: JsonRecord): void {
function applyMuseSparkMinOutputTokens(
model: string,
body: JsonRecord,
field: "max_tokens" | "max_output_tokens" = "max_tokens"
): void {
if (!MUSE_SPARK_PATTERN.test(model)) return;
const current = body.max_tokens;
const current = body[field];
if (typeof current !== "number" || !Number.isFinite(current)) return;
if (current >= MUSE_SPARK_MIN_OUTPUT_TOKENS) return;
body.max_tokens = MUSE_SPARK_MIN_OUTPUT_TOKENS;
body[field] = MUSE_SPARK_MIN_OUTPUT_TOKENS;
}

const COMMAND_CODE_PASSTHROUGH_FIELDS = [
Expand Down Expand Up @@ -114,6 +130,13 @@ function normalizeCommandCodeWireModel(model: string): string {
return COMMAND_CODE_BARE_MODEL_VENDOR_PREFIX[bare] ?? bare;
}

// Responses-shape detection and Responses -> Chat projection live in
// ./commandCode/responsesProjection.ts (kept out of this file for the 1200-line
// file-size gate).




// ── OpenAi Flat Body Builder (/provider/v1/chat/completions) ─────────────────

function buildOpenAiBody(model: string, body: unknown, stream: boolean): { body: JsonRecord } {
Expand All @@ -129,6 +152,24 @@ function buildOpenAiBody(model: string, body: unknown, stream: boolean): { body:
stream: stream === true,
};

// Forward max_tokens only when the client actually supplied a positive value
// (clamped to the endpoint ceiling). Omitting it lets the provider's upstream
// apply the model's own native default; a non-positive value such as -1
// ("let the server choose") must be omitted, NOT coerced to 1 (#5166).
if (isResponsesShapedBody(out)) {
// Responses shape: the cap lives on max_output_tokens; leave it clamped and
// never fabricate a Chat-shaped max_tokens alongside it.
const maxOutput = clampMaxTokens(out.max_output_tokens);
delete out.max_tokens;
delete out.max_completion_tokens;
delete out.max_output_tokens;
if (maxOutput !== undefined) {
out.max_output_tokens = maxOutput;
}
applyMuseSparkMinOutputTokens(resolvedModel, out, "max_output_tokens");
return { body: out };
}

const maxTokens = clampMaxTokens(input.max_tokens ?? input.max_completion_tokens);
delete out.max_tokens;
delete out.max_completion_tokens;
Expand Down Expand Up @@ -915,18 +956,99 @@ export class CommandCodeExecutor extends BaseExecutor {
return `${baseUrl}${this.config.chatPath || "/provider/v1/chat/completions"}`;
}

/**
* OpenAI Responses endpoint for models whose targetFormat is
* `openai-responses`. Same base + auth as chat; Command Code serves OpenAI and
* open models on both surfaces, but only /responses honors
* `reasoning: {"effort": "none"}` — the chat validator's effort enum has no
* disable value and silently drops a nested `reasoning.effort` (verified live
* 2026-09-24: /responses + effort none → reasoning_tokens 0; /chat +
* `reasoning:{effort:"none"}` → reasoning_tokens 41).
*/
buildResponsesUrl() {
const baseUrl = (this.config.baseUrl || "https://api.commandcode.ai").replace(/\/$/, "");
return `${baseUrl}/provider/v1/responses`;
}

buildCliUrl() {
const baseUrl = (this.config.baseUrl || "https://api.commandcode.ai").replace(/\/$/, "");
return `${baseUrl}/alpha/generate`;
}

/**
* Fallback path when the flat provider endpoint rejects with 403/404 (e.g. a Go
* plan without Provider API access, or a model only served on the CLI endpoint).
* Rebuilds the body in Command Code's CLI shape and posts to /alpha/generate.
*/
private async executeCliFallback(input: {
model: string;
sanitizedBody: unknown;
stream: boolean;
apiKey: string;
signal?: AbortSignal;
upstreamExtraHeaders?: Record<string, string>;
}) {
const { model, sanitizedBody, stream, apiKey, signal, upstreamExtraHeaders } = input;
const cliUrl = this.buildCliUrl();
const cliHeaders: Record<string, string> = {
"Content-Type": "application/json",
Authorization: `Bearer ${apiKey}`,
"x-command-code-version": COMMAND_CODE_VERSION,
"x-cli-environment": "external",
"x-project-slug": "pi-cc",
"x-taste-learning": "false",
"x-co-flag": "false",
"x-session-id": randomUUID(),
};
mergeUpstreamExtraHeaders(cliHeaders, upstreamExtraHeaders);

const { body: cliTransformedBody, toolNameMap } = buildCommandCodeCliBody(
model,
sanitizedBody,
stream
);

const cliUpstream = await fetch(cliUrl, {
method: "POST",
headers: cliHeaders,
body: JSON.stringify(cliTransformedBody),
signal: signal || undefined,
});

if (!cliUpstream.ok) {
const errorText = await cliUpstream.text().catch(() => {
console.warn("[commandCode] cli upstream text failed");
return "";
});
return {
response: new Response(errorText || `Command Code API error ${cliUpstream.status}`, {
status: cliUpstream.status,
statusText: cliUpstream.statusText,
headers: cliUpstream.headers,
}),
url: cliUrl,
headers: cliHeaders,
transformedBody: cliTransformedBody,
};
}

const response = stream
? createStreamResponse(cliUpstream, model, signal, toolNameMap)
: await createJsonResponse(cliUpstream, model, signal, toolNameMap);

return { response, url: cliUrl, headers: cliHeaders, transformedBody: cliTransformedBody };
}

async execute({ model, body, stream, credentials, signal, upstreamExtraHeaders }: ExecuteInput) {
const apiKey = credentials?.apiKey || credentials?.accessToken;
if (!apiKey) throw new Error("Command Code API key required");

const sanitizedBody = sanitizeReasoningEffortForProvider(body, this.provider, model);
const { body: transformedBody } = buildOpenAiBody(model, sanitizedBody, stream);
const url = this.buildUrl();
// Route by body shape: a Responses-shaped body (targetFormat openai-responses)
// must hit /provider/v1/responses, where `reasoning: {"effort":"none"}` is
// honored; the chat endpoint silently drops it.
const url = isResponsesShapedBody(transformedBody) ? this.buildResponsesUrl() : this.buildUrl();

const headers: Record<string, string> = {
"Content-Type": "application/json",
Expand All @@ -949,54 +1071,15 @@ export class CommandCodeExecutor extends BaseExecutor {
// Fallback: If /provider/v1/chat/completions returns 403 (e.g. Go plan without Provider
// API access) or 404, fallback to /alpha/generate (CLI endpoint).
if (upstream.status === 403 || upstream.status === 404) {
const cliUrl = this.buildCliUrl();
const cliHeaders: Record<string, string> = {
"Content-Type": "application/json",
Authorization: `Bearer ${apiKey}`,
"x-command-code-version": COMMAND_CODE_VERSION,
"x-cli-environment": "external",
"x-project-slug": "pi-cc",
"x-taste-learning": "false",
"x-co-flag": "false",
"x-session-id": randomUUID(),
};
mergeUpstreamExtraHeaders(cliHeaders, upstreamExtraHeaders);

const { body: cliTransformedBody, toolNameMap } = buildCommandCodeCliBody(
model,
sanitizedBody,
stream
);

const cliUpstream = await fetch(cliUrl, {
method: "POST",
headers: cliHeaders,
body: JSON.stringify(cliTransformedBody),
signal: signal || undefined,
});

if (!cliUpstream.ok) {
const errorText = await cliUpstream.text().catch(() => {
console.warn("[commandCode] cli upstream text failed");
return "";
});
return {
response: new Response(errorText || `Command Code API error ${cliUpstream.status}`, {
status: cliUpstream.status,
statusText: cliUpstream.statusText,
headers: cliUpstream.headers,
}),
url: cliUrl,
headers: cliHeaders,
transformedBody: cliTransformedBody,
};
// /alpha/generate is Chat-shaped, so a Responses request must be projected
// onto `messages` first. When it has no faithful CLI form, skip the fallback
// and surface the upstream error rather than replaying a mangled body.
const cliShaped = isResponsesShapedBody(sanitizedBody)
? projectResponsesForCli(sanitizedBody as JsonRecord)
: sanitizedBody;
if (cliShaped !== null) {
return this.executeCliFallback({ model, sanitizedBody: cliShaped, stream, apiKey, signal, upstreamExtraHeaders });
}

const response = stream
? createStreamResponse(cliUpstream, model, signal, toolNameMap)
: await createJsonResponse(cliUpstream, model, signal, toolNameMap);

return { response, url: cliUrl, headers: cliHeaders, transformedBody: cliTransformedBody };
}

const errorText = await upstream.text().catch(() => {
Expand Down
146 changes: 146 additions & 0 deletions open-sse/executors/commandCode/responsesProjection.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,146 @@
// Responses -> Chat projection for the /alpha/generate CLI fallback.
//
// Split out of commandCode.ts to keep that file under the 1200-line file-size
// gate; the logic is self-contained and only two symbols are needed by callers.

type JsonRecord = Record<string, unknown>;

function isRecord(value: unknown): value is JsonRecord {
return typeof value === "object" && value !== null && !Array.isArray(value);
}

function stringValue(value: unknown): string | undefined {
return typeof value === "string" ? value : undefined;
}

/**
* True for an OpenAI Responses-shaped body (`input`, not `messages`). Command Code
* serves OpenAI models on BOTH /chat/completions and /responses, but only the
* Responses endpoint honors `reasoning: {"effort": "none"}` (the chat validator
* accepts low|medium|high|xhigh|max only, and silently ignores a nested
* `reasoning.effort`). Route Responses-shaped bodies there so a no-thinking
* request can actually disable reasoning.
*
* `messages` is the Chat discriminator and WINS when both are present: `input` is
* only consulted when `messages` is absent. A body carrying both is malformed,
* and defaulting to the flat Chat surface keeps routing predictable if a payload
* override ever injects `messages` into a Responses request.
*/
export function isResponsesShapedBody(body: unknown): boolean {
if (!isRecord(body)) return false;
return body.input !== undefined && body.messages === undefined;
}

/**
* Per-type projections from a Responses input item onto Chat messages. The map
* keys are exactly the item types that have a faithful Chat form. Anything else
* — `reasoning` items, built-in tool calls (`web_search_call`,
* `local_shell_call`, …), and any future opaque item — has no CLI
* representation. Bailing on those is the only honest option: silently dropping
* the item would corrupt the replay.
*/
const RESPONSES_ITEM_PROJECTORS: Record<
string,
(item: JsonRecord, messages: JsonRecord[]) => boolean
> = {
message: projectResponsesMessageItem,
function_call: projectResponsesFunctionCallItem,
function_call_output: projectResponsesToolOutputItem,
};

function projectResponsesMessageItem(item: JsonRecord, messages: JsonRecord[]): boolean {
const role = stringValue(item.role);
if (role !== "user" && role !== "system" && role !== "assistant") return false;
messages.push({ role, content: item.content });
return true;
}

function projectResponsesToolOutputItem(item: JsonRecord, messages: JsonRecord[]): boolean {
const callId = stringValue(item.call_id) ?? "";
if (!callId) return false;
messages.push({ role: "tool", tool_call_id: callId, content: item.output });
return true;
}

/**
* Project a Responses `function_call` item onto a Chat assistant `tool_calls`
* entry. Consecutive calls — and a preceding assistant `message` — collapse into
* one assistant turn, matching how Chat groups `tool_calls`.
*/
function projectResponsesFunctionCallItem(item: JsonRecord, messages: JsonRecord[]): boolean {
const callId = stringValue(item.call_id) ?? "";
const name = stringValue(item.name) ?? "";
if (!callId || !name) return false;
const call: JsonRecord = {
id: callId,
type: "function",
function: { name, arguments: stringValue(item.arguments) ?? "{}" },
};
const last = messages[messages.length - 1];
if (isRecord(last) && last.role === "assistant") {
if (!Array.isArray(last.tool_calls)) last.tool_calls = [];
(last.tool_calls as JsonRecord[]).push(call);
} else {
messages.push({ role: "assistant", content: null, tool_calls: [call] });
}
return true;
}

/**
* Project Responses input items onto Chat messages in place. Returns false when
* an item has no faithful Chat form.
*/
function projectResponsesItems(items: unknown[], messages: JsonRecord[]): boolean {
for (const item of items) {
if (!isRecord(item)) return false;
const type = typeof item.type === "string" ? item.type : "message";
const project = RESPONSES_ITEM_PROJECTORS[type];
if (!project || !project(item, messages)) return false;
}
return true;
}

/**
* Project a Responses-shaped body onto the Chat shape `/alpha/generate` speaks.
*
* The CLI fallback is `messages`-based (`buildCommandCodeCliBody` reads
* `input.messages` and `input.max_tokens`), but a Responses request carries
* `input` and `max_output_tokens` instead. Replaying it unchanged sends an empty
* message list and drops the output cap, so the two have to be projected.
*
* Returns `null` when the request has no faithful CLI form (see
* `projectResponsesItems`); the caller then surfaces the upstream error rather
* than replaying a mangled body. `instructions` maps onto `system`, since the
* CLI body has no `instructions` field and would otherwise drop it.
*/
export function projectResponsesForCli(body: JsonRecord): JsonRecord | null {
const raw = body.input;
const messages: JsonRecord[] = [];

if (typeof raw === "string") {
messages.push({ role: "user", content: raw });
} else if (Array.isArray(raw)) {
if (!projectResponsesItems(raw, messages)) return null;
if (messages.length === 0) return null;
} else {
return null;
}

const out: JsonRecord = { ...body, messages };
delete out.input;
if (typeof body.max_output_tokens === "number") {
out.max_tokens = body.max_output_tokens;
}
delete out.max_output_tokens;

// Responses carries the system prompt as `instructions`; the CLI body has no
// such field and would silently drop it. Map it onto `system` when the request
// did not already set one.
const instructions = stringValue(body.instructions);
if (instructions && !stringValue(out.system)) {
out.system = instructions;
}
delete out.instructions;

return out;
}
Loading
Loading