Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
43 changes: 35 additions & 8 deletions cli/scripts/build-cli.js
Original file line number Diff line number Diff line change
Expand Up @@ -131,23 +131,50 @@ if (fs.existsSync(cliAppDir)) {
console.log("✅ Cleaned\n");

// Step 3: Copy Next.js standalone build to app/cli/app.
// Newer Next.js standalone output writes server.js/package.json plus .next/, src/, and
// node_modules/ directly under .next/standalone. Older builds may still use a nested app/.
// Layout history (newest → oldest):
// 1. Next 16 + workspace tracing root: writes server.js/package.json/.next/src/node_modules
// under .next/standalone/<workspaceFolderName>/ (e.g. .next/standalone/9router/).
// 2. Next 13–15: directly under .next/standalone/ (server.js at the root).
// 3. Pre-13 nested layout: under .next/standalone/app/.
// We probe each in turn so the same script keeps working across upgrades.
console.log("3️⃣ Copying Next.js standalone build to app/cli/app...");
const standaloneRoot = path.join(appDir, ".next", "standalone");
const standaloneRootResolved = path.join(buildDistDir, "standalone");
const standaloneRootToUse = fs.existsSync(standaloneRootResolved) ? standaloneRootResolved : standaloneRoot;
const standaloneApp = fs.existsSync(path.join(standaloneRootToUse, "server.js"))
? standaloneRootToUse
: path.join(standaloneRootToUse, "app");
if (!fs.existsSync(standaloneApp)) {

function findStandaloneApp(root) {
if (!fs.existsSync(root)) return null;
// 1. Flat layout: .next/standalone/server.js
if (fs.existsSync(path.join(root, "server.js"))) return root;
// 2. Nested-by-workspace-folder layout: .next/standalone/<folder>/server.js
// Pick the first subdir that has a server.js; usually only one (the
// workspace folder name = `9router`).
for (const entry of fs.readdirSync(root, { withFileTypes: true })) {
if (!entry.isDirectory()) continue;
if (entry.name === "node_modules") continue; // never the app
if (fs.existsSync(path.join(root, entry.name, "server.js"))) {
return path.join(root, entry.name);
}
}
// 3. Legacy nested-app layout: .next/standalone/app/
if (fs.existsSync(path.join(root, "app"))) return path.join(root, "app");
return null;
}

const standaloneApp = findStandaloneApp(standaloneRootToUse);
if (!standaloneApp) {
console.error("❌ Next.js standalone build not found under .next/standalone");
console.error("Expected either .next/standalone/server.js or .next/standalone/app/");
console.error("Expected one of:");
console.error(" - .next/standalone/server.js");
console.error(" - .next/standalone/<workspaceFolder>/server.js");
console.error(" - .next/standalone/app/");
process.exit(1);
}
console.log(` using standalone root: ${path.relative(appDir, standaloneApp) || "."}`);
copyRecursive(standaloneApp, cliAppDir);

// Older nested-app layout stores traced node_modules at standalone root.
// When the standalone app sits one level deep, the traced node_modules is at
// the standalone root and must be merged in alongside the app files.
const standaloneNodeModules = path.join(standaloneRootToUse, "node_modules");
if (standaloneApp !== standaloneRootToUse && fs.existsSync(standaloneNodeModules)) {
copyRecursive(standaloneNodeModules, path.join(cliAppDir, "node_modules"));
Expand Down
26 changes: 26 additions & 0 deletions cli/src/cli/menus/providers.js
Original file line number Diff line number Diff line change
Expand Up @@ -76,8 +76,34 @@ const PROVIDER_MODELS = {
{ id: "grok-code-fast-1" },
],
kr: [
// Base
{ id: "claude-sonnet-4.5" },
{ id: "claude-haiku-4.5" },
{ id: "deepseek-3.2" },
{ id: "qwen3-coder-next" },
{ id: "glm-5" },
{ id: "MiniMax-M2.5" },
// Thinking
{ id: "claude-sonnet-4.5-thinking" },
{ id: "claude-haiku-4.5-thinking" },
{ id: "deepseek-3.2-thinking" },
{ id: "qwen3-coder-next-thinking" },
{ id: "glm-5-thinking" },
{ id: "MiniMax-M2.5-thinking" },
// Agentic
{ id: "claude-sonnet-4.5-agentic" },
{ id: "claude-haiku-4.5-agentic" },
{ id: "deepseek-3.2-agentic" },
{ id: "qwen3-coder-next-agentic" },
{ id: "glm-5-agentic" },
{ id: "MiniMax-M2.5-agentic" },
// Thinking + Agentic
{ id: "claude-sonnet-4.5-thinking-agentic" },
{ id: "claude-haiku-4.5-thinking-agentic" },
{ id: "deepseek-3.2-thinking-agentic" },
{ id: "qwen3-coder-next-thinking-agentic" },
{ id: "glm-5-thinking-agentic" },
{ id: "MiniMax-M2.5-thinking-agentic" },
],
openai: [
{ id: "gpt-4o" },
Expand Down
27 changes: 25 additions & 2 deletions open-sse/config/providerModels.js
Original file line number Diff line number Diff line change
Expand Up @@ -132,24 +132,47 @@ export const PROVIDER_MODELS = {
{ id: "text-embedding-3-large", name: "Text Embedding 3 Large (GitHub)", type: "embedding" },
],
kr: [ // Kiro AI
// --- Base Claude variants ---
// --- Base models ---
// Same upstream model id Kiro accepts. The `-thinking`, `-agentic`, and
// `-thinking-agentic` rows below are 9router synthetic variants:
// * `-thinking` → injects `<thinking_mode>enabled</thinking_mode>` so
// Kiro emits reasoningContentEvent frames.
// * `-agentic` → injects the chunked-write system prompt to dodge
// Kiro's 2-3 min server timeout on big writes.
// * `-thinking-agentic` → both.
// The translator strips these suffixes before the request leaves this
// process. See `open-sse/config/kiroConstants.js` and
// `open-sse/services/kiroModels.js`.
// { id: "claude-opus-4.5", name: "Claude Opus 4.5" },
{ id: "claude-sonnet-4.5", name: "Claude Sonnet 4.5" },
{ id: "claude-haiku-4.5", name: "Claude Haiku 4.5" },
{ id: "deepseek-3.2", name: "DeepSeek 3.2", strip: ["image", "audio"] },
{ id: "qwen3-coder-next", name: "Qwen3 Coder Next", strip: ["image", "audio"] },
{ id: "glm-5", name: "GLM 5" },
{ id: "MiniMax-M2.5", name: "MiniMax M2.5" },
// --- Thinking variants (alias to base; thinking is enabled at request time
// --- Thinking variants (alias to base; thinking turns on at request time
// via <thinking_mode>enabled</thinking_mode> system-prompt injection) ---
{ id: "claude-sonnet-4.5-thinking", name: "Claude Sonnet 4.5 (Thinking)" },
{ id: "claude-haiku-4.5-thinking", name: "Claude Haiku 4.5 (Thinking)" },
{ id: "deepseek-3.2-thinking", name: "DeepSeek 3.2 (Thinking)", strip: ["image", "audio"] },
{ id: "qwen3-coder-next-thinking", name: "Qwen3 Coder Next (Thinking)", strip: ["image", "audio"] },
{ id: "glm-5-thinking", name: "GLM 5 (Thinking)" },
{ id: "MiniMax-M2.5-thinking", name: "MiniMax M2.5 (Thinking)" },
// --- Agentic variants (synthetic; same upstream model + chunked-write
// system prompt to dodge Kiro's 2-3 min server timeout on big writes) ---
{ id: "claude-sonnet-4.5-agentic", name: "Claude Sonnet 4.5 (Agentic)" },
{ id: "claude-haiku-4.5-agentic", name: "Claude Haiku 4.5 (Agentic)" },
{ id: "deepseek-3.2-agentic", name: "DeepSeek 3.2 (Agentic)", strip: ["image", "audio"] },
{ id: "qwen3-coder-next-agentic", name: "Qwen3 Coder Next (Agentic)", strip: ["image", "audio"] },
{ id: "glm-5-agentic", name: "GLM 5 (Agentic)" },
{ id: "MiniMax-M2.5-agentic", name: "MiniMax M2.5 (Agentic)" },
// --- Thinking + Agentic combined ---
{ id: "claude-sonnet-4.5-thinking-agentic", name: "Claude Sonnet 4.5 (Thinking + Agentic)" },
{ id: "claude-haiku-4.5-thinking-agentic", name: "Claude Haiku 4.5 (Thinking + Agentic)" },
{ id: "deepseek-3.2-thinking-agentic", name: "DeepSeek 3.2 (Thinking + Agentic)", strip: ["image", "audio"] },
{ id: "qwen3-coder-next-thinking-agentic", name: "Qwen3 Coder Next (Thinking + Agentic)", strip: ["image", "audio"] },
{ id: "glm-5-thinking-agentic", name: "GLM 5 (Thinking + Agentic)" },
{ id: "MiniMax-M2.5-thinking-agentic", name: "MiniMax M2.5 (Thinking + Agentic)" },
],
cu: [ // Cursor IDE
{ id: "default", name: "Auto (Server Picks)" },
Expand Down
148 changes: 128 additions & 20 deletions open-sse/executors/kiro.js
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@ import { v4 as uuidv4 } from "uuid";
import { refreshKiroToken } from "../services/tokenRefresh.js";
import { proxyAwareFetch } from "../utils/proxyFetch.js";
import { HTTP_STATUS, RETRY_CONFIG, DEFAULT_RETRY_CONFIG, resolveRetryEntry } from "../config/runtimeConfig.js";
import { splitInlineThinking, flushPendingThinking } from "./kiroThinking.js";

/**
* KiroExecutor - Executor for Kiro AI (AWS CodeWhisperer)
Expand Down Expand Up @@ -68,17 +69,38 @@ export class KiroExecutor extends BaseExecutor {

// Success - transform and return
// For Kiro, we need to transform the binary EventStream to SSE
// Create a TransformStream to convert binary to SSE text
const transformedResponse = this.transformEventStreamToSSE(response, model);
// Create a TransformStream to convert binary to SSE text.
//
// We pass a `thinkingExpected` hint based on the request body. When the
// user enabled thinking, Claude on Kiro streams its reasoning **inline**
// as `<thinking>…</thinking>` blocks inside `assistantResponseEvent`
// rather than as separate `reasoningContentEvent` frames. The transform
// stream uses this flag to split that inline reasoning back into the
// OpenAI `delta.reasoning_content` channel.
const upstreamUserContent =
transformedBody?.conversationState?.currentMessage?.userInputMessage?.content || "";
const thinkingExpected = upstreamUserContent.includes("<thinking_mode>enabled</thinking_mode>");
const transformedResponse = this.transformEventStreamToSSE(response, model, { thinkingExpected });
return { response: transformedResponse, url, headers, transformedBody };
}
}

/**
* Transform AWS EventStream binary response to SSE text stream
* Using TransformStream instead of ReadableStream.pull() to avoid Workers timeout
*
* @param {Response} response Upstream raw fetch response (binary EventStream).
* @param {string} model Logical model id (kept in OpenAI chunks for clients).
* @param {object} [opts]
* @param {boolean} [opts.thinkingExpected=false]
* When true, scan inbound `assistantResponseEvent.content` for inline
* `<thinking>…</thinking>` blocks and split them out into the OpenAI
* `delta.reasoning_content` channel. Required for Claude on Kiro because
* it streams reasoning inline (not as `reasoningContentEvent`) when
* `<thinking_mode>enabled</thinking_mode>` is in the system prompt.
*/
transformEventStreamToSSE(response, model) {
transformEventStreamToSSE(response, model, opts = {}) {
const thinkingExpected = !!opts.thinkingExpected;
let buffer = new Uint8Array(0);
let chunkIndex = 0;
const responseId = `chatcmpl-${Date.now()}`;
Expand All @@ -90,7 +112,91 @@ export class KiroExecutor extends BaseExecutor {
hasReasoningContent: false,
reasoningChunkCount: 0,
toolCallIndex: 0,
seenToolIds: new Map()
seenToolIds: new Map(),
// Inline-thinking splitter state. `thinkingMode === true` means we are
// currently inside a `<thinking>…</thinking>` block, so subsequent text
// should be routed to `delta.reasoning_content`. `pendingTag` carries
// an unfinished tag fragment (e.g. `<thi`) across frames.
thinkingMode: false,
pendingTag: ""
};

// ---- Helpers ----------------------------------------------------------
// We declare these as locals (not arrow methods on `state`) so the
// TransformStream's `transform` function can call them directly via
// closure. They mutate `state.thinkingMode`, `state.pendingTag`, and
// emit OpenAI-shaped chunks via the supplied `controller`.
/**
* Emit a content delta. Sends `role: "assistant"` on the very first chunk
* of any kind (matching OpenAI's wire format).
*/
const emitContent = (controller, content) => {
if (!content) return;
const chunkOut = {
id: responseId,
object: "chat.completion.chunk",
created,
model,
choices: [{
index: 0,
delta: chunkIndex === 0
? { role: "assistant", content }
: { content },
finish_reason: null
}]
};
chunkIndex++;
controller.enqueue(new TextEncoder().encode(`data: ${JSON.stringify(chunkOut)}\n\n`));
};

/**
* Emit a reasoning delta. Behaves like emitContent but writes the text to
* `delta.reasoning_content` so downstream translators (Anthropic /
* thinking_blocks, Claude SSE, etc.) can re-wrap it correctly.
*/
const emitReasoning = (controller, reasoning) => {
if (!reasoning) return;
state.hasReasoningContent = true;
const reasoningDelta =
state.reasoningChunkCount === 0 && chunkIndex === 0
? { role: "assistant", reasoning_content: reasoning }
: { reasoning_content: reasoning };
const chunkOut = {
id: responseId,
object: "chat.completion.chunk",
created,
model,
choices: [{
index: 0,
delta: reasoningDelta,
finish_reason: null
}]
};
chunkIndex++;
state.reasoningChunkCount++;
controller.enqueue(new TextEncoder().encode(`data: ${JSON.stringify(chunkOut)}\n\n`));
};

/**
* Stream-safe `<thinking>` / `</thinking>` splitter.
*
* Walks one `assistantResponseEvent.content` slice at a time and emits
* either content or reasoning chunks based on the current mode. It carries
* a `pendingTag` buffer across slices so a tag split between frames
* (e.g. `…</think` then `ing>foo`) is still recognised.
*
* The splitter is only engaged when `thinkingExpected === true`. For
* non-thinking requests we keep the original passthrough path so any
* literal `<thinking>` text the user asked for in their answer is left
* alone.
*/
const runSplitter = (controller, raw) => {
splitInlineThinking(
state,
raw,
(s) => emitContent(controller, s),
(s) => emitReasoning(controller, s)
);
};

const transformStream = new TransformStream({
Expand Down Expand Up @@ -127,22 +233,16 @@ export class KiroExecutor extends BaseExecutor {
if (eventType === "assistantResponseEvent" && event.payload?.content) {
const content = event.payload.content;
state.totalContentLength += content.length;

const chunk = {
id: responseId,
object: "chat.completion.chunk",
created,
model,
choices: [{
index: 0,
delta: chunkIndex === 0
? { role: "assistant", content }
: { content },
finish_reason: null
}]
};
chunkIndex++;
controller.enqueue(new TextEncoder().encode(`data: ${JSON.stringify(chunk)}\n\n`));

if (thinkingExpected) {
// Claude on Kiro emits reasoning inline as `<thinking>…</thinking>`
// when the system prompt enables thinking. Split it into the
// OpenAI reasoning_content channel so downstream consumers see
// the same shape they would get from a native reasoning model.
runSplitter(controller, content);
} else {
emitContent(controller, content);
}
}

// Handle reasoningContentEvent (Kiro thinking / reasoning)
Expand Down Expand Up @@ -376,6 +476,14 @@ export class KiroExecutor extends BaseExecutor {
},

flush(controller) {
// Drain any pending inline-thinking tag fragment so we don't drop
// trailing characters when the stream ends mid-tag (e.g. `<thi`).
flushPendingThinking(
state,
(s) => emitContent(controller, s),
(s) => emitReasoning(controller, s)
);

// Emit finish chunk if not already sent
if (!state.finishEmitted) {
state.finishEmitted = true;
Expand Down
Loading