Skip to content
This repository was archived by the owner on Aug 25, 2026. It is now read-only.
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions .changeset/preserve-openai-compaction-state.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,10 @@
---
"@moonshot-ai/kimi-code": patch
"@moonshot-ai/agent-core": patch
"@moonshot-ai/agent-core-v2": patch
"@moonshot-ai/kosong": patch
---

Preserve opaque OpenAI Responses compaction state across turns and automatically
use `/responses/compact` when the active provider exposes that capability,
falling back to Kimi's existing local summarizer when it does not.
3 changes: 3 additions & 0 deletions apps/kimi-code/src/tui/utils/message-replay.ts
Original file line number Diff line number Diff line change
Expand Up @@ -171,6 +171,7 @@ export function collectReplayMessageContent(
break;
case 'audio_url':
case 'image_url':
case 'openai_compaction':
case 'video_url':
break;
}
Expand Down Expand Up @@ -285,6 +286,8 @@ function contentPartToText(part: ContentPart): string {
return mediaUrlPartToText('video', part.videoUrl.url);
case 'audio_url':
return mediaUrlPartToText('audio', part.audioUrl.url);
case 'openai_compaction':
return '';
}
}

Expand Down
2 changes: 2 additions & 0 deletions apps/vscode/src/runtime/replay-adapter.ts
Original file line number Diff line number Diff line change
Expand Up @@ -511,6 +511,8 @@ function toLegacyContent(content: readonly ContentPart[]): LegacyContentPart[] {
case "video_url":
result.push({ type: "video_url", video_url: { ...part.videoUrl } });
break;
case "openai_compaction":
break;
}
}
return result;
Expand Down
4 changes: 3 additions & 1 deletion apps/vscode/src/utils/session-context.ts
Original file line number Diff line number Diff line change
Expand Up @@ -196,6 +196,7 @@ function formatPartMarkdown(part: ContentPart): string {
case "image_url": return "[image]";
case "audio_url": return "[audio]";
case "video_url": return "[video]";
case "openai_compaction": return "";
}
}

Expand All @@ -205,7 +206,8 @@ function stringifyParts(parts: readonly ContentPart[]): string {
if (part.type === "think") return part.think.trim() ? `<thinking>\n${part.think}\n</thinking>` : "";
if (part.type === "image_url") return "[image]";
if (part.type === "audio_url") return "[audio]";
return "[video]";
if (part.type === "video_url") return "[video]";
return "";
}).filter(Boolean).join("\n");
}

Expand Down
6 changes: 6 additions & 0 deletions docs/en/guides/sessions.md
Original file line number Diff line number Diff line change
Expand Up @@ -68,6 +68,12 @@ You can manage sessions without leaving the terminal. The following slash comman

As a conversation grows, Kimi Code CLI automatically compresses the message history when the context approaches the window limit, freeing up token space. You can also trigger compression manually at any time:

For OpenAI Responses-compatible providers, Kimi Code automatically tries the
provider's native `/responses/compact` endpoint and preserves its opaque
replacement state. If that capability is unavailable, or when `/compact`
includes a custom instruction, it uses Kimi Code's existing summary compaction
instead. No separate setting is required.

```
/compact
```
Expand Down
4 changes: 4 additions & 0 deletions docs/zh/guides/sessions.md
Original file line number Diff line number Diff line change
Expand Up @@ -68,6 +68,10 @@ kimi --session

对话变长时,Kimi Code CLI 会在上下文接近窗口上限时自动压缩历史消息,释放 token 空间。也可以随时手动触发:

对于兼容 OpenAI Responses 的提供商,Kimi Code 会自动尝试其原生
`/responses/compact` 接口,并保留接口返回的不透明替换状态。如果该能力不可用,
或 `/compact` 带有自定义指引,则回退到 Kimi Code 原有的摘要压缩。无需额外配置。

```
/compact
```
Expand Down
2 changes: 1 addition & 1 deletion packages/agent-core-v2/docs/wire-manifest.d.ts
Original file line number Diff line number Diff line change
Expand Up @@ -114,7 +114,7 @@ interface ContextAppendMessagePayload {
message: {
role: 'system' | 'user' | 'assistant' | 'tool';
name?: string;
content: ('text' | 'think' | 'image_url' | 'audio_url' | 'video_url')[];
content: ('text' | 'think' | 'image_url' | 'audio_url' | 'video_url' | 'openai_compaction')[];
toolCalls: {
type: 'function';
id: string;
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@ export interface CompactionUserSelection<T> {

export interface ContextCompactionShapeInput {
readonly summary: string;
readonly replacementMessages?: readonly ContextMessage[];
readonly legacySummaryMessage?: ContextMessage;
readonly contextSummary?: string;
readonly compactedCount: number;
Expand Down Expand Up @@ -51,6 +52,22 @@ export function buildContextCompactionShape(
history: readonly ContextMessage[],
input: ContextCompactionShapeInput,
): ContextCompactionShape {
if (input.replacementMessages !== undefined) {
const messages = [...input.replacementMessages];
const contextSummary = input.contextSummary ?? input.summary;
return {
summary: input.summary,
contextSummary,
compactedCount: input.compactedCount,
tokensBefore: input.tokensBefore,
tokensAfter: input.tokensAfter ?? estimateTokensForMessages(messages),
keptUserMessageCount:
input.keptUserMessageCount ?? messages.filter((message) => message.role === 'user').length,
keptHeadUserMessageCount: input.keptHeadUserMessageCount,
droppedCount: input.droppedCount,
messages,
};
}
if (usesLegacyTailShape(input)) {
const contextSummary = input.contextSummary ?? input.summary;
const messages = [
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,8 @@ import type { ContextMessage } from './types';

export interface ContextCompactionInput {
readonly summary: string;
/** Provider-owned canonical replacement window, when native compaction is used. */
readonly replacementMessages?: readonly ContextMessage[];
readonly contextSummary?: string;
readonly compactedCount: number;
readonly tokensBefore: number;
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -113,6 +113,8 @@ export class AgentContextMemoryService extends Disposable implements IAgentConte
this.wire.dispatch(
contextApplyCompaction({
summary: result.summary,
replacementMessages:
input.replacementMessages === undefined ? undefined : [...input.replacementMessages],
contextSummary: result.contextSummary,
compactedCount: result.compactedCount,
tokensBefore: result.tokensBefore,
Expand Down
7 changes: 7 additions & 0 deletions packages/agent-core-v2/src/agent/contextMemory/contextOps.ts
Original file line number Diff line number Diff line change
Expand Up @@ -156,6 +156,7 @@ const contextCompactionBaseShape = {
keptHeadUserMessageCount: z.number().optional(),
droppedCount: z.number().optional(),
legacyTail: z.boolean().optional(),
replacementMessages: z.array(contextMessageSchema).optional(),
};

const contextApplyCompactionSchema = z.union([
Expand Down Expand Up @@ -210,6 +211,7 @@ export function readContextCompactionShapeInput(
const keptUserMessageCount = readOptionalNumber(fields, 'keptUserMessageCount');
return {
summary: readContextCompactionRawSummary(fields),
replacementMessages: readReplacementMessages(fields),
legacySummaryMessage: readLegacySummaryMessage(fields),
contextSummary: readOptionalString(fields, 'contextSummary'),
compactedCount: readContextCompactedCount(fields),
Expand All @@ -222,6 +224,11 @@ export function readContextCompactionShapeInput(
};
}

function readReplacementMessages(record: UnknownRecord): readonly ContextMessage[] | undefined {
const value = record['replacementMessages'];
return Array.isArray(value) ? (value as ContextMessage[]) : undefined;
}

export function readContextCompactedCount(record: ContextCompactionRecord): number {
const fields = record as UnknownRecord;
const compactedCount = fields['compactedCount'];
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -36,7 +36,7 @@ function toProtocolRole(role: ContextMessage['role']): MessageRole {
return role as MessageRole;
}

function mapContentPart(part: ContextMessage['content'][number]): MessageContent {
function mapContentPart(part: ContextMessage['content'][number]): MessageContent | undefined {
switch (part.type) {
case 'text':
return { type: 'text', text: part.text };
Expand All @@ -59,13 +59,22 @@ function mapContentPart(part: ContextMessage['content'][number]): MessageContent
? { type: 'video', source: { kind: 'file', file_id: ref.fileId } }
: { type: 'video', source: { kind: 'url', url: part.videoUrl.url, id: part.videoUrl.id } };
}
case 'openai_compaction':
return undefined;
}
}

function mapContentParts(parts: ContextMessage['content']): MessageContent[] {
return parts.flatMap((part) => {
const mapped = mapContentPart(part);
return mapped === undefined ? [] : [mapped];
});
}

function buildProtocolContent(msg: ContextMessage): MessageContent[] {
if (msg.role === 'tool') {
if (msg.toolCallId === undefined) {
return msg.content.map((p) => mapContentPart(p));
return mapContentParts(msg.content);
}
const hasMediaPart = msg.content.some(
(p) => p.type === 'image_url' || p.type === 'video_url' || p.type === 'audio_url',
Expand All @@ -89,7 +98,7 @@ function buildProtocolContent(msg: ContextMessage): MessageContent[] {
return [part];
}

const base = msg.content.map((p) => mapContentPart(p));
const base = mapContentParts(msg.content);

if (msg.role === 'assistant' && msg.toolCalls.length > 0) {
for (const call of msg.toolCalls) {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -74,6 +74,7 @@ const OVERFLOW_CONTEXT_SAFETY_RATIO = 0.85;
const OVERFLOW_STATUS_RECOVERY_RATIO = 0.5;
const MAX_COMPACTION_OVERFLOW_SHRINK_ATTEMPTS = 3;
const COMPACTION_OVERFLOW_SHRINK_RATIOS = [0.7, 0.5, 0.35] as const;
const REMOTE_COMPACTION_SUMMARY = '[OpenAI server compaction checkpoint]';
const EMPTY_TOOL_PARAMETERS: Record<string, unknown> = {
type: 'object',
properties: {},
Expand Down Expand Up @@ -115,6 +116,7 @@ export class AgentFullCompactionService extends Disposable implements IAgentFull
private compactionCountInTurn = 0;
private _compacting: ActiveCompaction | null = null;
private readonly observedMaxContextTokensByModel = new Map<string, number>();
private readonly remoteCompactionUnavailableModels = new Set<string>();
private lastCompactedTokenCount: number | null = null;
private consecutiveOverflowCompactions = 0;
private activeTurnId: number | undefined;
Expand Down Expand Up @@ -537,6 +539,71 @@ export class AgentFullCompactionService extends Disposable implements IAgentFull

const resolvedModel = this.profile.resolveModelContext();
thinkingEffort = resolvedModel.thinkingLevel;

const modelAlias = resolvedModel.modelAlias;
const hasCustomInstruction = (data.instruction?.trim().length ?? 0) > 0;
// Provider-owned compaction has no portable custom-instruction field.
if (!hasCustomInstruction && !this.remoteCompactionUnavailableModels.has(modelAlias)) {
try {
const remote = await this.llmRequester.compact?.(
{
messages: stripDynamicToolContext(originalHistory),
source: {
type: 'operation',
turnId: active.originTurnId,
requestKind: 'remote_compaction',
},
},
signal,
);
if (remote !== undefined) {
if (!historySafeToCompact(this.context.get(), originalHistory)) {
this.cancelActive(active);
throw compactionCancelledReason(active);
}
const result = this.context.applyCompaction({
summary: REMOTE_COMPACTION_SUMMARY,
contextSummary: REMOTE_COMPACTION_SUMMARY,
replacementMessages: remote.messages as readonly ContextMessage[],
compactedCount: originalHistory.length,
tokensBefore,
});
const properties: CompactionFinishedEvent = {
turn_id: active.originTurnId,
source: data.source,
tokens_before: result.tokensBefore,
tokens_after: result.tokensAfter,
duration_ms: Date.now() - startedAt,
compacted_count: result.compactedCount,
retry_count: 0,
round: 1,
thinking_effort: thinkingEffort,
...usageTelemetry(remote.usage ?? null),
};
this.telemetry.track2('compaction_finished', properties);
return result;
}
this.remoteCompactionUnavailableModels.add(modelAlias);
} catch (error) {
if (isAbortError(error)) throw error;
if (
isError2(error) &&
(error.code === ErrorCodes.AUTH_LOGIN_REQUIRED ||
error.code === ErrorCodes.PROVIDER_AUTH_ERROR)
) {
throw error;
}
const status = findAPIStatusError(error)?.statusCode;
if (status === 400 || status === 404 || status === 405 || status === 501) {
this.remoteCompactionUnavailableModels.add(modelAlias);
}
this.log.warn('remote compaction unavailable; falling back to local summary', {
model: modelAlias,
status,
});
}
}

const maxContextTokens = resolvedModel.modelCapabilities.max_context_tokens;
const defaultCompactionCap =
maxContextTokens > 0
Expand Down
11 changes: 10 additions & 1 deletion packages/agent-core-v2/src/agent/llmRequester/llmRequester.ts
Original file line number Diff line number Diff line change
@@ -1,5 +1,9 @@
import { createDecorator } from '#/_base/di/instantiation';
import type { FinishReason, ThinkingEffort } from '#/kosong/contract/provider';
import type {
FinishReason,
ProviderCompactionResult,
ThinkingEffort,
} from '#/kosong/contract/provider';
import type { Message, StreamedMessagePart } from '#/kosong/contract/message';
import type { Tool } from '#/kosong/contract/tool';
import type { TokenUsage } from '#/kosong/contract/usage';
Expand Down Expand Up @@ -70,6 +74,11 @@ export interface IAgentLLMRequesterService {
onPart?: AgentLLMRequestPartHandler,
signal?: AbortSignal,
): AgentLLMRequestTask;

compact?(
overrides?: AgentLLMRequestOverrides,
signal?: AbortSignal,
): Promise<ProviderCompactionResult | undefined>;
}

export const IAgentLLMRequesterService = createDecorator<IAgentLLMRequesterService>(
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -55,7 +55,10 @@ import {
isRetryableGenerateError,
} from '#/kosong/contract/errors';
import { type Message } from '#/kosong/contract/message';
import { type ThinkingEffort } from '#/kosong/contract/provider';
import type {
ProviderCompactionResult,
ThinkingEffort,
} from '#/kosong/contract/provider';
import { type Tool } from '#/kosong/contract/tool';
import { emptyUsage, inputTotal, type TokenUsage } from '#/kosong/contract/usage';
import { ILogService, type LogContext } from '#/_base/log/log';
Expand Down Expand Up @@ -194,6 +197,23 @@ export class AgentLLMRequesterService implements IAgentLLMRequesterService {
};
}

async compact(
overrides: AgentLLMRequestOverrides = {},
signal?: AbortSignal,
): Promise<ProviderCompactionResult | undefined> {
const request = this.resolveRequest(overrides);
const input = {
systemPrompt: request.systemPrompt,
tools: request.tools,
messages: request.messages,
};
const result = await request.requester.compact?.(input, signal, request.params);
if (result?.usage !== undefined) {
this.usage.record(request.modelAlias, result.usage, request.source);
}
return result;
}

private async requestWithTrace(
trace: MutableLLMRequestTrace,
overrides: AgentLLMRequestOverrides,
Expand Down
1 change: 1 addition & 0 deletions packages/agent-core-v2/src/agent/loop/loopService.ts
Original file line number Diff line number Diff line change
Expand Up @@ -948,6 +948,7 @@ export class AgentLoopService extends Disposable implements IAgentLoopService {
case 'image_url':
case 'audio_url':
case 'video_url':
case 'openai_compaction':
return;
case 'function': {
onResponseEvent();
Expand Down
6 changes: 5 additions & 1 deletion packages/agent-core-v2/src/agent/mcp/output.ts
Original file line number Diff line number Diff line change
Expand Up @@ -246,7 +246,11 @@ function applyBinaryPartCap(parts: readonly ContentPart[]): {
const out: ContentPart[] = [];

for (const part of parts) {
if (part.type === 'text' || part.type === 'think') {
if (
part.type === 'text' ||
part.type === 'think' ||
part.type === 'openai_compaction'
) {
out.push(part);
continue;
}
Expand Down
Loading