Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions changelog.d/fixes/15591-output-styles-hu-ja-zh-text.md
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
- **fix(compression):** legacy `cavemanOutputMode` injection now matches the old caveman text in Hungarian (it no longer falls back to English) and drops the extra space before the shared boundary sentence in Japanese and Chinese, including `terse-cjk` and multi-style selections. Prompt-cache prefixes for existing ja/zh selections change once on upgrade.
29 changes: 20 additions & 9 deletions open-sse/services/compression/outputStyles/apply.ts
Original file line number Diff line number Diff line change
Expand Up @@ -103,17 +103,27 @@ function resolveStyles(
return resolved;
}

/** Build the combined instruction body (no marker, no trailing boundary). Pure / deterministic. */
/**
* Build the combined instruction body (no marker, no boundary block), ending with one
* space when the last style's own text puts whitespace (or nothing) before
* SHARED_BOUNDARIES, and with none when a non-whitespace character directly precedes
* it. Pure / deterministic.
*/
function buildStyleInstructions(resolved: OutputStyleSelectionEntry[], language: string): string {
const parts: string[] = [];
let separator = " ";
for (const { id, level } of resolved) {
const meta = outputStyleMeta(id);
const localized = meta.i18n?.[language];
const levels = localized ?? meta.levels;
const text = levels[level];
// Strip the per-style boundary so the combined boundary block is appended once below.
parts.push(levels[level].replace(SHARED_BOUNDARIES, "").trim());
parts.push(text.replace(SHARED_BOUNDARIES, "").trim());
// The separator mirrors whatever spacing the style's own text puts before SHARED_BOUNDARIES.
const at = text.indexOf(SHARED_BOUNDARIES);
separator = at > 0 && !/\s/.test(text[at - 1]) ? "" : " ";
}
return parts.join("\n");
return `${parts.join("\n")}${separator}`;
}

/**
Expand Down Expand Up @@ -159,12 +169,13 @@ export function applyOutputStyles(
return { body, applied: false, skippedReason: "no_styles" };
}

// Single space before the boundary block so a legacy single-style
// (terse-prose) injection stays byte-identical to the old caveman output mode
// (D-A5 back-compat): terse-prose declares no `boundaries`, so its block is
// exactly SHARED_BOUNDARIES as before. A style that declares one gets the
// shared clause AND its own, in catalog order.
const combined = `${buildStyleInstructions(resolved, language)} ${buildStyleBoundaries(resolved, language)}`;
// The boundary block follows the instruction body with the spacing the last style's own
// text puts before SHARED_BOUNDARIES, so a legacy single-style (terse-prose) injection
// matches the old caveman output mode below the marker line in every legacy language
// (D-A5 back-compat): terse-prose declares no `boundaries`, so its block is exactly
// SHARED_BOUNDARIES as before.
// A style that declares one gets the shared clause AND its own, in catalog order.
const combined = `${buildStyleInstructions(resolved, language)}${buildStyleBoundaries(resolved, language)}`;
const instruction = `${OUTPUT_STYLE_MARKER}\n${combined}`;

const messages = Array.isArray(body.messages) ? body.messages : null;
Expand Down
3 changes: 2 additions & 1 deletion open-sse/services/compression/outputStyles/backCompat.ts
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,8 @@ interface ConfigSlice {
* Resolve the effective output-style selection (D-A5 back-compat).
* Precedence: an explicit non-empty `outputStyles` wins; otherwise a stored
* `cavemanOutputMode` (when enabled) maps to `[{ terse-prose, <intensity> }]`,
* keeping existing installs byte-identical until they opt into other styles.
* so existing installs keep the legacy instruction text below the marker line
* until they opt into other styles.
* Pure; never throws.
*/
export function resolveOutputStyleSelection(config: ConfigSlice): OutputStyleSelectionEntry[] {
Expand Down
28 changes: 12 additions & 16 deletions open-sse/services/compression/outputStyles/catalog.ts
Original file line number Diff line number Diff line change
Expand Up @@ -40,27 +40,23 @@ export const SAFETY_BOUNDARIES =
* settings panel both enumerate this object, so no other file needs to change (D-A1).
* Declaration order is the deterministic concatenation order used by the injector.
*/

// The terse-prose text mirrors the legacy caveman table: en is the default `levels`,
// every other language an `i18n` override. Derived rather than hand-copied, so a
// language added to the legacy table cannot be missed here;
// output-styles-legacy-parity.test.ts guards the mirror from the other side.
const { en: TERSE_PROSE_LEVELS, ...TERSE_PROSE_I18N } = CAVEMAN_INSTRUCTION_BY_LANGUAGE;

export const OUTPUT_STYLE_CATALOG: Record<string, OutputStyle> = {
"terse-prose": {
id: "terse-prose",
label: "Terse prose",
description: "Drop filler/articles/hedging; keep technical substance exact.",
// Migrated verbatim from the caveman output mode (outputMode.ts) — referenced (not
// re-typed) so the back-compat injection stays byte-identical across ALL languages,
// not just English (the legacy mode localized to en/pt-BR/ja/id).
levels: CAVEMAN_INSTRUCTION_BY_LANGUAGE.en,
i18n: {
"pt-BR": CAVEMAN_INSTRUCTION_BY_LANGUAGE["pt-BR"],
es: CAVEMAN_INSTRUCTION_BY_LANGUAGE.es,
de: CAVEMAN_INSTRUCTION_BY_LANGUAGE.de,
fr: CAVEMAN_INSTRUCTION_BY_LANGUAGE.fr,
it: CAVEMAN_INSTRUCTION_BY_LANGUAGE.it,
ru: CAVEMAN_INSTRUCTION_BY_LANGUAGE.ru,
zh: CAVEMAN_INSTRUCTION_BY_LANGUAGE.zh,
ja: CAVEMAN_INSTRUCTION_BY_LANGUAGE.ja,
id: CAVEMAN_INSTRUCTION_BY_LANGUAGE.id,
vi: CAVEMAN_INSTRUCTION_BY_LANGUAGE.vi,
},
// Referenced from the caveman output mode (outputMode.ts) so the back-compat injection
// matches the legacy text below the marker line in every language
// CAVEMAN_INSTRUCTION_BY_LANGUAGE covers.
levels: TERSE_PROSE_LEVELS,
i18n: TERSE_PROSE_I18N,
},
"less-code": {
id: "less-code",
Expand Down
5 changes: 3 additions & 2 deletions tests/unit/compression/output-styles-apply.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -151,8 +151,9 @@ test("Responses input (no messages) uses instructions field", () => {
});

test("terse-prose localizes per language (back-compat with the legacy caveman packs)", () => {
// Regression guard: the legacy caveman output mode localized to en/pt-BR/ja/id; the
// migrated terse-prose style must inject the SAME localized text, not fall back to English.
// Regression guard: the legacy caveman output mode carries a localized text per language;
// the migrated terse-prose style must inject the SAME localized text, not fall back to
// English. (Every language × level pair is pinned in output-styles-legacy-parity.test.ts.)
const ptBR = applyOutputStyles(
{ messages: [{ role: "user", content: "Resuma os logs." }] },
sel(["terse-prose", "lite"]),
Expand Down
23 changes: 23 additions & 0 deletions tests/unit/compression/output-styles-boundaries.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -183,6 +183,29 @@ test("terse-prose alone stays byte-identical to the legacy caveman injection (D-
assert.equal(text, `${OUTPUT_STYLE_MARKER}\n${legacy}`);
});

test("the boundary block keeps the spacing the last style's own text puts before it", () => {
// ja and zh texts run straight into SHARED_BOUNDARIES. less-code has no hu text, so its
// English text, with one space, decides the hu case.
const cases: Array<[string, OutputStyleSelectionEntry[], string]> = [
["ja", sel(["terse-prose", "full"], ["less-code", "full"]), "。"],
["zh", sel(["terse-prose", "full"], ["ponytail", "full"]), "。"],
["zh", sel(["terse-cjk", "full"]), "。"],
["hu", sel(["terse-prose", "full"], ["less-code", "full"]), " "],
["ja", sel(["terse-prose", "ultra"], ["less-code", "ultra"]), "。"],
];
for (const [language, selection, before] of cases) {
const text = injected(
{ messages: [{ role: "user", content: "Refactor this module." }] },
selection,
language
);
const at = text.indexOf(SHARED_BOUNDARIES);
assert.ok(at > 1, `${language}: SHARED_BOUNDARIES is present`);
assert.equal(text[at - 1], before, `${language}: character before SHARED_BOUNDARIES`);
assert.notEqual(text[at - 2], " ", `${language}: at most one space before SHARED_BOUNDARIES`);
}
});

test("boundary emission is deterministic (same selection + language → byte-identical)", () => {
const make = () =>
injected(
Expand Down
4 changes: 2 additions & 2 deletions tests/unit/compression/output-styles-i18n-matrix.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -38,8 +38,8 @@ const KNOWN_ENGLISH_ONLY: Record<string, string> = {};
*/
const BASELINE_LANGUAGES: Record<string, string[]> = {
// terse-prose reuses CAVEMAN_INSTRUCTION_BY_LANGUAGE (outputMode.ts), which
// localizes to pt-BR/es/de/fr/it/ru/zh/ja/id/vi — keep the two in sync.
"terse-prose": ["pt-BR", "es", "de", "fr", "it", "ru", "zh", "ja", "id", "vi"],
// localizes to pt-BR/es/de/fr/it/ru/zh/ja/id/vi/hu — keep the two in sync.
"terse-prose": ["pt-BR", "es", "de", "fr", "it", "ru", "zh", "ja", "id", "vi", "hu"],
"less-code": ["pt-BR", "vi", "ja", "id", "es", "de", "fr", "it", "ru", "zh"],
ponytail: ["pt-BR", "vi", "ja", "id", "es", "de", "fr", "it", "ru", "zh"],
"i-have-adhd": ["pt-BR", "vi", "ja", "id", "es", "de", "fr", "it", "ru", "zh"],
Expand Down
67 changes: 67 additions & 0 deletions tests/unit/compression/output-styles-legacy-parity.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,67 @@
import { test } from "node:test";
import assert from "node:assert/strict";
import {
applyCavemanOutputMode,
CAVEMAN_INSTRUCTION_BY_LANGUAGE,
} from "../../../open-sse/services/compression/outputMode.ts";
import { resolveOutputStyleSelection } from "../../../open-sse/services/compression/outputStyles/backCompat.ts";
import {
applyOutputStyles,
OUTPUT_STYLE_MARKER,
} from "../../../open-sse/services/compression/outputStyles/apply.ts";

const LEGACY_MARKER = "[OmniRoute Caveman Output Mode]";
const LEVELS = ["lite", "full", "ultra"] as const;

// The instruction text below the marker line. The instruction lands in a top-level
// `system` field or a trailing system message (#13383), so every text surface is
// gathered and read from the marker on — a placement change cannot fake a failure here.
function instructionText(
result: {
applied: boolean;
body: { messages?: Array<{ role?: string; content?: unknown }>; system?: unknown };
},
marker: string
): string {
assert.equal(result.applied, true, "the injector applied the instruction");
const parts: string[] = [];
if (typeof result.body.system === "string") parts.push(result.body.system);
else if (Array.isArray(result.body.system)) {
for (const block of result.body.system) {
const text = (block as { text?: unknown } | null)?.text;
if (typeof text === "string") parts.push(text);
}
}
for (const message of result.body.messages ?? []) {
if (typeof message.content === "string") parts.push(message.content);
}
const joined = parts.join("\n");
const markerAt = joined.indexOf(marker);
assert.ok(
markerAt >= 0 && joined[markerAt + marker.length] === "\n",
`the instruction starts with ${marker}`
);
return joined.slice(markerAt + marker.length + 1);
}

// The languages come from the legacy table at runtime, so a language pack added there without a
// matching terse-prose translation fails here.
for (const language of Object.keys(CAVEMAN_INSTRUCTION_BY_LANGUAGE)) {
for (const intensity of LEVELS) {
test(`legacy ${language} ${intensity}: the unified injector writes the legacy text below its marker`, () => {
const body = { messages: [{ role: "user", content: "Summarize this API response." }] };
const config = { enabled: true, intensity, autoClarity: true };
const legacy = applyCavemanOutputMode(structuredClone(body), config, language);
const next = applyOutputStyles(
structuredClone(body),
resolveOutputStyleSelection({ cavemanOutputMode: config }),
language,
{ autoClarity: true }
);
assert.equal(
instructionText(next, OUTPUT_STYLE_MARKER),
instructionText(legacy, LEGACY_MARKER)
);
});
}
}
Loading