Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion config/quality/file-size-baseline.json
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@
"_comment": "Catraca de tamanho (check-file-size.mjs). frozen so pode encolher; arquivos novos <= cap. --update ratcheta.",
"_rebaseline_v3.8.25": "Drift consciente do ciclo v3.8.24->v3.8.25 (features #3799-#3806: free-provider-rankings, plugins menu, proxy IP-family selector). 3 arquivos cresceram por feature legitima, nao por regressao de qualidade: ProxyRegistryManager.tsx 1072->1089, sidebarVisibility.ts 990->1006, schemas.ts 2519->2522. Encolher fica como debt para um refactor dedicado.",
"_rebaseline_2026_06_15_3860_compression_ui": "PR #3860 own growth: sidebarVisibility.ts 1006->1100 (+94 = Compression Hub menu entries: Hub + per-engine Lite/Aggressive/Ultra pages + combos editor) and chatCore.ts 5812->5815 (+3 = compression UI config wiring). Cohesive feature growth, not a quality regression.",
"_rebaseline_2026_06_15_3885_glm_5_2": "PR #3885 own growth: pricing.ts 1508->1529 (+21 = GLM-5.2 pricing rows for glm-5.2 + effort aliases glm-5.2-high/-max, same $1.2/$5 schedule as glm-5.1; pure data). Also adds glm-5.2 specs to glmProvider.ts/modelSpecs.ts (modelSpecs.ts stays under cap). Cohesive model registration; not extractable.",
"_rebaseline_2026_06_15_3870_alias_lookup": "PR #3870 own growth: providerRegistry.ts 4703->4708 (+5 = generateModels() also stores each provider's models under its raw id, not only its alias, so getProviderModels(rawId) works when alias != id e.g. github->gh; preserves the existing first-wins guard). Cohesive registry fix; not extractable.",
"_rebaseline_2026_06_15_3871_empty_pool": "PR #3871 own growth: combo.ts 5203->5204 (+1 = guard expandAutoComboCandidatePool against an empty candidatePool array — Array.isArray(pool) && pool.length > 0 so [] falls through to active-connection expansion instead of early-returning). One-line correctness fix; not extractable.",
"cap": 800,
Expand Down Expand Up @@ -101,7 +102,7 @@
"src/shared/components/RequestLoggerV2.tsx": 1282,
"src/shared/components/analytics/charts.tsx": 1558,
"src/shared/constants/cliTools.ts": 875,
"src/shared/constants/pricing.ts": 1508,
"src/shared/constants/pricing.ts": 1529,
"src/shared/constants/providers.ts": 3147,
"src/shared/constants/sidebarVisibility.ts": 1100,
"src/shared/services/cliRuntime.ts": 1090,
Expand Down
24 changes: 24 additions & 0 deletions open-sse/config/glmProvider.ts
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,30 @@ export const GLM_ANTHROPIC_DEFAULT_BASE_URLS = Object.freeze({
});

export const GLM_SHARED_MODELS = Object.freeze([
{
id: "glm-5.2",
name: "GLM 5.2",
contextLength: 1000000,
maxOutputTokens: 131072,
toolCalling: true,
supportsReasoning: true,
},
{
id: "glm-5.2-high",
name: "GLM 5.2 High",
contextLength: 1000000,
maxOutputTokens: 131072,
toolCalling: true,
supportsReasoning: true,
},
{
id: "glm-5.2-max",
name: "GLM 5.2 Max",
contextLength: 1000000,
maxOutputTokens: 131072,
toolCalling: true,
supportsReasoning: true,
},
{
id: "glm-5.1",
name: "GLM 5.1",
Expand Down
66 changes: 61 additions & 5 deletions open-sse/executors/glm.ts
Original file line number Diff line number Diff line change
Expand Up @@ -50,6 +50,19 @@ function getEffectiveKey(credentials: ProviderCredentials): string {
return credentials.apiKey || credentials.accessToken || "";
}

/**
* GLM-5.2 effort tiers route exclusively through the Anthropic transport,
* where Zhipu maps Claude Code effort selectors (high/max) to reasoning
* intensity. The base model ID sent upstream is always "glm-5.2".
*
* https://docs.z.ai/devpack/latest-model
*/
function parseGlm52Effort(model: string): { baseModel: string; effort: "high" | "max" } | null {
if (model === "glm-5.2-high") return { baseModel: "glm-5.2", effort: "high" };
if (model === "glm-5.2-max") return { baseModel: "glm-5.2", effort: "max" };
return null;
}

function applyGlmRequestDefaults(body: unknown, defaults?: JsonRecord | null): unknown {
const record = asRecord(body);
if (!record || !defaults) return body;
Expand Down Expand Up @@ -228,27 +241,61 @@ export class GlmExecutor extends DefaultExecutor {
credentials: ProviderCredentials,
transport: GlmTransport
) {
const transformed = this.transformRequest(model, body, stream, credentials);
const effortTier = parseGlm52Effort(model);
const effectiveModel = effortTier ? effortTier.baseModel : model;

const transformed = this.transformRequest(effectiveModel, body, stream, credentials);
const record = asRecord(transformed);

// Ensure upstream receives the base model ID, not the effort-suffixed alias
if (record && effortTier) {
record.model = effectiveModel;
}

if (transport === "openai") {
const record = asRecord(transformed);
if (record && stream && hasTools(record) && record.tool_stream === undefined) {
return { ...record, tool_stream: true };
}
return transformed;
}

return translateRequest(
const translated = translateRequest(
FORMATS.OPENAI,
FORMATS.CLAUDE,
model,
{ ...(transformed as JsonRecord), _disableToolPrefix: true },
effectiveModel,
{ ...(record ?? {}), _disableToolPrefix: true },
stream,
credentials,
this.provider,
null,
{ preserveCacheControl: false }
);

// Inject effort and thinking for the Anthropic transport.
// Zhipu's Anthropic endpoint requires thinking.type=enabled to emit
// thinking_delta blocks in the SSE response. Without it, reasoning is
// not surfaced and clients see no thinking content.
// The effort-2025-11-24 beta header (in GLM_ANTHROPIC_BETA) carries
// the high/max intensity selector.
if (effortTier) {
const translatedRecord = asRecord(translated);
if (translatedRecord) {
translatedRecord.effort = effortTier.effort;
// Zhipu's Anthropic endpoint only supports thinking.type
// "enabled"/"disabled" — not "adaptive". Clients like Claude Code
// default to "adaptive" for reasoning models, so force "enabled"
// here while preserving any other fields (e.g. budget_tokens).
const existingThinking = asRecord(translatedRecord.thinking);
if (!existingThinking || existingThinking.type !== "enabled") {
translatedRecord.thinking = {
...existingThinking,
type: "enabled",
};
}
}
}

return translated;
}

private async executeTransport(
Expand Down Expand Up @@ -343,6 +390,15 @@ export class GlmExecutor extends DefaultExecutor {
}

async execute(input: ExecuteInput): Promise<GlmExecuteResult> {
const effortTier = parseGlm52Effort(input.model);

// GLM-5.2 effort tiers route directly through Anthropic transport (no fallback).
// Zhipu only graduates effort on the Anthropic endpoint via the
// effort-2025-11-24 beta header included in GLM_ANTHROPIC_BETA.
if (effortTier) {
return this.executeTransport(input, "anthropic");
}

const primaryTransport = getGlmTransport(
input.credentials.providerSpecificData,
this.config.baseUrl
Expand Down
44 changes: 44 additions & 0 deletions open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -88,6 +88,9 @@ describe("GLM Coding provider registry surfaces", () => {
expect(PROVIDER_ID_TO_ALIAS.glm).toBe("glm");
expect(byProviderId).toEqual(byAlias);
expect(byProviderId.map((model) => model.id)).toEqual([
"glm-5.2",
"glm-5.2-high",
"glm-5.2-max",
"glm-5.1",
"glm-5",
"glm-5-turbo",
Expand All @@ -101,6 +104,30 @@ describe("GLM Coding provider registry surfaces", () => {
]);
});

it("registers GLM-5.2 with correct specs and effort tier aliases", () => {
const models = getModelsByProviderId("glm");
const get = (id: string) => models.find((m) => m.id === id);

// Base model
const base = get("glm-5.2");
expect(base).toBeDefined();
expect(base?.contextLength).toBe(1000000);
expect(base?.maxOutputTokens).toBe(131072);
expect(base?.supportsReasoning).toBe(true);
expect(base?.toolCalling).toBe(true);

// Effort tier aliases share the same specs
const high = get("glm-5.2-high");
expect(high).toBeDefined();
expect(high?.contextLength).toBe(1000000);
expect(high?.maxOutputTokens).toBe(131072);

const max = get("glm-5.2-max");
expect(max).toBeDefined();
expect(max?.contextLength).toBe(1000000);
expect(max?.maxOutputTokens).toBe(131072);
});

it("applies doc-backed context window overrides for GLM models", () => {
const models = getModelsByProviderId("glm");
const get = (id: string) => models.find((m) => m.id === id);
Expand All @@ -126,6 +153,9 @@ describe("GLM Coding provider registry surfaces", () => {
expect(supportsToolCalling("glm/glm-5")).toBe(true);
expect(supportsToolCalling("glm/glm-4.7-flash")).toBe(true);
expect(supportsToolCalling("glm/glm-4.5-air")).toBe(true);
expect(supportsToolCalling("glm/glm-5.2")).toBe(true);
expect(supportsToolCalling("glm/glm-5.2-high")).toBe(true);
expect(supportsToolCalling("glm/glm-5.2-max")).toBe(true);

expect(getPricingForModel("glm", "glm-5")).toEqual({
input: 1.0,
Expand All @@ -148,6 +178,20 @@ describe("GLM Coding provider registry surfaces", () => {
reasoning: 1.1,
cache_creation: 0.2,
});
expect(getPricingForModel("glm", "glm-5.2")).toEqual({
input: 1.2,
output: 5,
cached: 0.3,
reasoning: 5,
cache_creation: 1.2,
});
expect(getPricingForModel("glm", "glm-5.2-max")).toEqual({
input: 1.2,
output: 5,
cached: 0.3,
reasoning: 5,
cache_creation: 1.2,
});
});

it("keeps the repo-derived GLM inventory internally aligned across registry and pricing surfaces", () => {
Expand Down
20 changes: 20 additions & 0 deletions src/shared/constants/modelSpecs.ts
Original file line number Diff line number Diff line change
Expand Up @@ -354,6 +354,26 @@ export const MODEL_SPECS: Record<string, ModelSpec> = {
supportsTools: true,
},

// ── Z.AI GLM-5.2 (1M context, 128K max output, effort tiers) ────
"glm-5.2": {
maxOutputTokens: 131072,
contextWindow: 1000000,
supportsThinking: true,
supportsTools: true,
},
"glm-5.2-high": {
maxOutputTokens: 131072,
contextWindow: 1000000,
supportsThinking: true,
supportsTools: true,
},
"glm-5.2-max": {
maxOutputTokens: 131072,
contextWindow: 1000000,
supportsThinking: true,
supportsTools: true,
},

// ── Z.AI GLM-5.x (200K context, 128K max output) ─────────────────
"glm-5.1": {
maxOutputTokens: 128000,
Expand Down
21 changes: 21 additions & 0 deletions src/shared/constants/pricing.ts
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,27 @@ const CLAUDE_SONNET_46_PRICING = {
};

const GLM_PRICING = {
"glm-5.2": {
input: 1.2,
output: 5,
cached: 0.3,
reasoning: 5,
cache_creation: 1.2,
},
"glm-5.2-high": {
input: 1.2,
output: 5,
cached: 0.3,
reasoning: 5,
cache_creation: 1.2,
},
"glm-5.2-max": {
input: 1.2,
output: 5,
cached: 0.3,
reasoning: 5,
cache_creation: 1.2,
},
"glm-5.1": {
input: 1.2,
output: 5,
Expand Down