Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 11 additions & 1 deletion src/adapters/openai-chat/summary-budget.ts
Original file line number Diff line number Diff line change
Expand Up @@ -37,6 +37,16 @@ function isTinyCap(value: unknown): value is number {
return typeof value === "number" && Number.isInteger(value) && value >= 1 && value <= 1024;
}

function hasConversationEnvelope(transcript: string): boolean {
// Case-insensitive tag search in place: request bodies can reach hundreds of MiB, so
// lowercasing a copy would roughly double peak memory for an already large request.
const opening = /<conversation>/i.exec(transcript);
if (opening === null) return false;
const closing = /<\/conversation>/gi;
closing.lastIndex = opening.index + opening[0].length;
return closing.test(transcript);
}

/**
* Aside's emergency checkpoint is a standalone summary, not an ordinary short answer (#5465).
* Runs at the physical Chat destination, after all combo effort overrides, so `effort` is the
Expand Down Expand Up @@ -64,7 +74,7 @@ export function protectGlmSummaryBudget(
const transcript = textContent(user.content);
if (instruction === undefined || transcript === undefined) return false;
if (!/\b(?:summari[sz](?:e|ation|er|ing)|summary|checkpoint)\b/i.test(instruction)) return false;
if (!/<conversation>[\s\S]*<\/conversation>/i.test(transcript)) return false;
if (!hasConversationEnvelope(transcript)) return false;
Comment thread
coderabbitai[bot] marked this conversation as resolved.
if (transcript.length < MIN_CHECKPOINT_TRANSCRIPT_CHARS) return false;
if (isTinyCap(body.max_tokens)) body.max_tokens = PROTECTED_SUMMARY_CAP;
if (isTinyCap(body.max_completion_tokens)) body.max_completion_tokens = PROTECTED_SUMMARY_CAP;
Expand Down
48 changes: 47 additions & 1 deletion tests/adapters/openai/openai-chat-glm-summary.test.ts
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import { describe, expect, test } from "bun:test";
import { describe, expect, spyOn, test } from "bun:test";
import { buildOpenAIChatPassthroughRequest, createOpenAIChatAdapter } from "../../../src/adapters/openai-chat";
import { protectGlmSummaryBudget } from "../../../src/adapters/openai-chat/summary-budget";
import { chatCompletionsToResponsesBody } from "../../../src/chat/inbound";
import { concreteComboRequestBody } from "../../../src/combos/request";
import { parseRequest } from "../../../src/responses/parser";
Expand Down Expand Up @@ -84,6 +85,51 @@ describe("GLM summary mitigation stays inside its boundary (#5953 review)", () =
const short = [messages[0], { role: "user", content: "<conversation>User: hi.</conversation>" }];
for (const body of bodies({ messages: short })) untouched(body);
});
/** The only patterns detection may scan the transcript with: the two fixed tags. */
const FIXED_TAG_SCANS = ["<conversation>", "<\\/conversation>"];
/** Runs detection while counting scans/copies targeted at the transcript — deterministic, unlike a wall-clock bound. */
const detectWithProbe = (userContent: string) => {
const scans: string[] = [];
const copies: string[] = [];
const origExec = RegExp.prototype.exec;
const execProbe = spyOn(RegExp.prototype, "exec").mockImplementation(function (this: RegExp, str: string) {
if (str === userContent) scans.push(this.source);
return origExec.call(this, str);
});
const copyProbes = (["toLowerCase", "toUpperCase", "toLocaleLowerCase", "toLocaleUpperCase"] as const).map(name => {
const orig = String.prototype[name];
return spyOn(String.prototype, name).mockImplementation(function (this: unknown) {
if (String(this) === userContent) copies.push(name);
return orig.call(this as string);
});
});
try {
const result = protectGlmSummaryBudget({
model, max_tokens: 512,
messages: [messages[0], { role: "user", content: userContent }],
}, provider.baseUrl, "max");
// A call count cannot bound work inside one scan — pinning every transcript scan to a
// fixed tag is what fails the original greedy `[\s\S]*` pass, not the count alone.
for (const source of scans) expect(FIXED_TAG_SCANS).toContain(source);
return { result, scans, copies };
} finally {
execProbe.mockRestore();
for (const probe of copyProbes) probe.mockRestore();
}
};
test("rejects many unmatched conversation openings with a bounded number of transcript scans", () => {
const { result, scans, copies } = detectWithProbe("<conversation>".repeat(32_000));
expect(result).toBeFalse();
expect(scans.length).toBeLessThanOrEqual(2);
expect(copies).toEqual([]);
});
test("detects an uppercase checkpoint in place, without a transcript-sized normalized copy", () => {
const transcript = `<CONVERSATION>${"LOREM IPSUM DOLOR SIT AMET.\n".repeat(200)}</CONVERSATION>`;
const { result, scans, copies } = detectWithProbe(transcript);
expect(result).toBeTrue();
expect(scans.length).toBeLessThanOrEqual(2);
expect(copies).toEqual([]);
});
test.each(["low", "medium"])("an effective %s effort is not overridden", effort => {
for (const body of bodies({ reasoning_effort: effort })) untouched(body, effort);
});
Expand Down
Loading