Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
56 changes: 56 additions & 0 deletions open-sse/executors/kiro.ts
Original file line number Diff line number Diff line change
Expand Up @@ -214,6 +214,12 @@ export class KiroExecutor extends BaseExecutor {
if (b.conversationState !== undefined) kiroPayload.conversationState = b.conversationState;
if (b.profileArn !== undefined) kiroPayload.profileArn = b.profileArn;
if (b.inferenceConfig !== undefined) kiroPayload.inferenceConfig = b.inferenceConfig;
// Thinking control: `additionalModelRequestFields` ({output_config.effort,
// thinking:{type:"adaptive"}, max_tokens}) is a recognized top-level field on
// GenerateAssistantResponse — it steers adaptive reasoning. Built by the
// openai-to-kiro translator only when the request asked for thinking.
if (b.additionalModelRequestFields !== undefined)
kiroPayload.additionalModelRequestFields = b.additionalModelRequestFields;

// Fallback: if somehow conversationState isn't there, return the rest without model
// (for backward compatibility if something else bypasses the translator)
Expand Down Expand Up @@ -382,6 +388,56 @@ export class KiroExecutor extends BaseExecutor {
if (!state.totalContentLength) state.totalContentLength = 0;
if (!state.contextUsagePercentage) state.contextUsagePercentage = 0;

// Native reasoning frames. Verified against the live CodeWhisperer
// stream (2026-07): with adaptive thinking enabled (via
// additionalModelRequestFields), Kiro streams reasoning as a dedicated
// `reasoningContentEvent` frame carrying `{ text, signature }` — NOT
// inline `<thinking>` tags and NOT `assistantResponseEvent`. Some
// models/variants instead use a `reasoningText` object or a flat
// `{ text }` (cf. javargasm/pi-kiro `src/event-parser.ts`). OmniRoute
// had no handler for this event, so the reasoning was silently dropped;
// route it to the OpenAI `reasoning_content` channel.
{
const rp = event.payload as Record<string, unknown> | undefined;
const rt = rp?.reasoningText;
if (eventType === "reasoningContentEvent" || rt !== undefined) {
let nativeReasoning = "";
if (rt && typeof rt === "object") {
const rto = rt as { text?: unknown; Text?: unknown };
nativeReasoning =
typeof rto.text === "string"
? rto.text
: typeof rto.Text === "string"
? rto.Text
: "";
} else if (typeof rt === "string") {
nativeReasoning = rt;
} else if (typeof rp?.text === "string") {
nativeReasoning = rp.text as string;
}
if (nativeReasoning) {
state.hasReasoningContent = true;
const reasoningDelta: JsonRecord =
(state.reasoningChunkCount ?? 0) === 0 && chunkIndex === 0
? { role: "assistant", reasoning_content: nativeReasoning }
: { reasoning_content: nativeReasoning };
const chunk: JsonRecord = {
id: responseId,
object: "chat.completion.chunk",
created,
model,
choices: [{ index: 0, delta: reasoningDelta, finish_reason: null }],
};
chunkIndex++;
state.reasoningChunkCount = (state.reasoningChunkCount ?? 0) + 1;
controller.enqueue(TEXT_ENCODER.encode(`data: ${JSON.stringify(chunk)}\n\n`));
}
// Consume the reasoning frame (incl. signature-only) so it never
// falls through to the content handlers below.
continue;
}
}

// Handle assistantResponseEvent
if (eventType === "assistantResponseEvent") {
const content =
Expand Down
129 changes: 129 additions & 0 deletions open-sse/translator/request/openai-to-kiro.ts
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@
import { register } from "../registry.ts";
import { FORMATS } from "../formats.ts";
import { v4 as uuidv4, v5 as uuidv5 } from "uuid";
import { capMaxOutputTokens, capThinkingBudget, supportsReasoning } from "@/lib/modelCapabilities";
import {
parseToolInput,
normalizeKiroToolSchema,
Expand Down Expand Up @@ -575,6 +576,80 @@ function convertMessages(messages, tools, model) {
return { history: alternatingHistory, currentMessage, toolsAttached };
}

/** Kiro's accepted reasoning-effort levels (`output_config.effort`). */
const KIRO_EFFORT_LEVELS = ["low", "medium", "high", "xhigh", "max"];

/**
* Resolve the Kiro effort level for a request, or "" when no reasoning was asked
* for. Effort sources, in priority order:
* 1. OpenAI-style `reasoning_effort`
* 2. Anthropic adaptive-thinking `output_config.effort` (the canonical field)
* 3. Anthropic `thinking` block — `{type:"enabled", budget_tokens}` mapped to a
* level via {@link effortFromBudget}; `{type:"adaptive"}` (no explicit
* effort) defaults to `high`, matching Anthropic's documented default
* (omitting `effort` ≡ `high`).
* OpenAI's `minimal` collapses to `low` (Kiro has no `minimal`).
*/
function resolveKiroEffort(body: Record<string, unknown>): string {
let effort = typeof body.reasoning_effort === "string" ? body.reasoning_effort.toLowerCase() : "";

if (!effort) {
const outputConfig = body.output_config as Record<string, unknown> | undefined;
if (
outputConfig &&
typeof outputConfig === "object" &&
typeof outputConfig.effort === "string"
) {
effort = outputConfig.effort.toLowerCase();
}
}

if (!effort) {
const thinking = body.thinking as Record<string, unknown> | undefined;
if (thinking && typeof thinking === "object") {
if (thinking.type === "enabled") {
effort = effortFromBudget(Number(thinking.budget_tokens) || 0);
} else if (thinking.type === "adaptive") {
effort = "high";
}
}
}

if (effort === "minimal") effort = "low";
return KIRO_EFFORT_LEVELS.includes(effort) ? effort : "";
}
Comment on lines +593 to +620

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

high

The current implementation of resolveKiroEffort only extracts the effort level from body.reasoning_effort or body.thinking when type is "enabled". However, standard Anthropic adaptive thinking requests (especially for Claude 4.6/4.7/4.8 models) use thinking: { type: "adaptive" } and specify the effort level in output_config.effort.

If a client sends these standard adaptive thinking parameters, resolveKiroEffort will return "", which silently disables thinking on Kiro. We should update resolveKiroEffort to also check body.output_config.effort and handle thinking.type === "adaptive" (defaulting to "high" effort if not specified).

function resolveKiroEffort(body: Record<string, unknown>): string {
  let effort = typeof body.reasoning_effort === "string" ? body.reasoning_effort.toLowerCase() : "";

  if (!effort) {
    const outputConfig = body.output_config as Record<string, unknown> | undefined;
    if (outputConfig && typeof outputConfig === "object" && typeof outputConfig.effort === "string") {
      effort = outputConfig.effort.toLowerCase();
    }
  }

  if (!effort) {
    const thinking = body.thinking as Record<string, unknown> | undefined;
    if (thinking && typeof thinking === "object") {
      if (thinking.type === "enabled") {
        effort = effortFromBudget(Number(thinking.budget_tokens) || 0);
      } else if (thinking.type === "adaptive") {
        effort = "high";
      }
    }
  }

  if (effort === "minimal") effort = "low";
  return KIRO_EFFORT_LEVELS.includes(effort) ? effort : "";
}


/** Map an Anthropic `thinking.budget_tokens` to a coarse Kiro effort level. */
function effortFromBudget(budget: number): string {
if (budget >= 32000) return "high";
if (budget >= 16000) return "medium";
if (budget > 0) return "low";
return "";
}

/**
* Soft `<max_thinking_length>` budget for the Kiro prompt directive, per effort
* level. Anthropic publishes no effort→token mapping (effort is "a behavioral
* signal, not a strict token budget"), so this is a heuristic tuned against the
* live CodeWhisperer stream, where a larger budget measurably deepens reasoning
* up to the model cap. It is a hint the model may honor, not a hard cap (the hard
* enable signal is `<thinking_mode>`); the caller clamps it to the model's cap.
*/
function thinkingLengthForEffort(effort: string): number {
switch (effort) {
case "max":
return 120000;
case "xhigh":
return 64000;
case "high":
return 32000;
case "medium":
return 16000;
default:
return 8000; // low
}
}

/**
* Build Kiro payload from OpenAI format
*/
Expand Down Expand Up @@ -683,6 +758,11 @@ export function buildKiroPayload(model, body, stream, credentials) {
temperature?: number;
topP?: number;
};
additionalModelRequestFields?: {
thinking?: { type: string; display?: string };
output_config?: { effort: string };
max_tokens?: number;
};
} = {
conversationState: {
chatTriggerType: "MANUAL",
Expand Down Expand Up @@ -754,6 +834,55 @@ export function buildKiroPayload(model, body, stream, credentials) {
if (topP !== undefined) payload.inferenceConfig.topP = topP;
}

// Thinking mode for Claude models on Kiro (ported from javargasm/pi-kiro).
// Two coordinated signals steer reasoning on the CodeWhisperer surface:
// 1. a `<thinking_mode>enabled</thinking_mode><max_thinking_length>N</...>`
// directive prepended to the current user message — makes Claude emit its
// reasoning INLINE as `<thinking>…</thinking>`, which the Kiro executor
// splits back into the OpenAI `reasoning_content` channel (kiroThinking.ts);
// 2. top-level `additionalModelRequestFields` (output_config.effort +
// thinking:{type:"adaptive"} + a clamped max_tokens), forwarded to AWS by
// the Kiro executor's transformRequest allowlist — this is the graded
// effort lever. Gated on models that advertise thinking support.
const kiroEffort = supportsReasoning(normalizedModel) ? resolveKiroEffort(body) : "";
if (kiroEffort) {
// `<thinking_mode>` / `<max_thinking_length>` are Kiro/CodeWhisperer prompt
// conventions (NOT Anthropic API params); the length is a soft hint (the hard
// enable signal is `<thinking_mode>`), clamped to the model's thinking cap.
const thinkingLength = capThinkingBudget(normalizedModel, thinkingLengthForEffort(kiroEffort));
const directive =
`<thinking_mode>enabled</thinking_mode>` +
`<max_thinking_length>${thinkingLength}</max_thinking_length>`;
payload.conversationState.currentMessage.userInputMessage.content = `${directive}\n\n${payload.conversationState.currentMessage.userInputMessage.content}`;

const fields: {
output_config: { effort: string };
thinking: { type: string; display: string };
max_tokens?: number;
} = {
output_config: { effort: kiroEffort },
thinking: { type: "adaptive", display: "summarized" },
};
// Forward max_tokens only when the client set one, clamped to the model's
// output window (floor 1024) — matches pi-kiro and avoids an over-budget reject.
if (maxTokens > 0) {
const capped = capMaxOutputTokens(normalizedModel, maxTokens) ?? maxTokens;
fields.max_tokens = Math.max(Math.floor(capped), 1024);
}
payload.additionalModelRequestFields = fields;

// Adaptive-only Claude models (Opus 4.7/4.8, Sonnet 5, Fable 5) reject a
// non-default temperature / top_p with a 400 while thinking is active, so
// strip both. Drop inferenceConfig entirely if nothing else remains.
if (payload.inferenceConfig) {
delete payload.inferenceConfig.temperature;
delete payload.inferenceConfig.topP;
if (Object.keys(payload.inferenceConfig).length === 0) {
delete payload.inferenceConfig;
}
}
}

return payload;
}

Expand Down
58 changes: 58 additions & 0 deletions tests/unit/executor-kiro.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -168,6 +168,31 @@ test("KiroExecutor.transformRequest removes the top-level model field", () => {
);
});

test("KiroExecutor.transformRequest forwards additionalModelRequestFields (thinking) to AWS", () => {
const executor = new KiroExecutor();
const body = {
model: "kiro-model",
conversationState: {
currentMessage: { userInputMessage: { modelId: "kiro-model" } },
},
additionalModelRequestFields: {
output_config: { effort: "high" },
thinking: { type: "adaptive", display: "summarized" },
max_tokens: 32000,
},
};

const result = executor.transformRequest("kiro-model", body, true, {}) as any;
// The thinking control must survive the strict allowlist — otherwise graded
// reasoning never reaches CodeWhisperer (the field the openai-to-kiro
// translator builds would be silently dropped).
assert.deepEqual(result.additionalModelRequestFields, {
output_config: { effort: "high" },
thinking: { type: "adaptive", display: "summarized" },
max_tokens: 32000,
});
});

test("KiroExecutor.transformEventStreamToSSE converts text, tool calls, usage and DONE", async () => {
const executor = new KiroExecutor();
const invalidPreludeFrame = buildEventFrame("assistantResponseEvent", { content: "skip me" });
Expand Down Expand Up @@ -202,6 +227,39 @@ test("KiroExecutor.transformEventStreamToSSE converts text, tool calls, usage an
assert.match(text, /\[DONE\]/);
});

test("KiroExecutor.transformEventStreamToSSE surfaces native reasoning frames as reasoning_content", async () => {
const executor = new KiroExecutor();
// Verified live wire format: Kiro streams adaptive-thinking reasoning as a
// dedicated `reasoningContentEvent` frame carrying `{ text, signature }`. Also
// cover the `reasoningText` object variant and a signature-only frame.
const response = buildEventStreamResponse([
buildEventFrame("reasoningContentEvent", { text: "Let me think... " }),
buildEventFrame("reasoningContentEvent", { text: "step two. " }),
buildEventFrame("reasoningContentEvent", { signature: "sig-only-frame" }),
buildEventFrame("assistantResponseEvent", { reasoningText: { text: "variant." } }),
buildEventFrame("assistantResponseEvent", { content: "The answer is 42." }),
buildEventFrame("metricsEvent", { inputTokens: 3, outputTokens: 5 }),
]);

const transformed = executor.transformEventStreamToSSE(response, "kiro-model");
const chunks = parseSSEJsonChunks(await transformed.text());
const reasoning = chunks
.map((c) => c.choices?.[0]?.delta?.reasoning_content)
.filter(Boolean)
.join("");
const content = chunks
.map((c) => c.choices?.[0]?.delta?.content)
.filter(Boolean)
.join("");

assert.equal(
reasoning,
"Let me think... step two. variant.",
"reasoningContentEvent frames + reasoningText variant must all surface"
);
assert.match(content, /The answer is 42\./, "normal content must still flow");
});

test("KiroExecutor.transformEventStreamToSSE parses fragmented frames and waits for post-stop usage", async () => {
const executor = new KiroExecutor();
const bytes = concatArrays(
Expand Down
Loading