From 8cba7bffe9f26d16284e9a02fe8cd66524045ab3 Mon Sep 17 00:00:00 2001 From: dhaern Date: Mon, 15 Jun 2026 10:19:26 +0000 Subject: [PATCH 1/4] feat(glm): add GLM-5.2 with effort-tier routing (high/max) via Anthropic transport MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit GLM-5.2 (released 2026-06-13) brings a 1M context window, 131072 max output tokens and always-on thinking. While Zhipu's OpenAI-compatible endpoint only supports binary thinking (enabled/disabled), the Anthropic-compatible endpoint graduates reasoning intensity through Claude Code's effort selector (high/max), carried by the effort-2025-11-24 beta header already present in GLM_ANTHROPIC_BETA. This change introduces three catalog entries: - glm-5.2 → OpenAI transport (coding/paas/v4), thinking always-on - glm-5.2-high → Anthropic transport, effort: high, no fallback - glm-5.2-max → Anthropic transport, effort: max, no fallback The effort tiers route directly through the Anthropic transport (no fallback) because effort graduation is only supported there. The model suffix is stripped before the upstream call so Zhipu always receives the base id 'glm-5.2'. thinking.type=enabled is injected into the Anthropic body so the upstream emits thinking_delta blocks; these are translated back to reasoning_content by the existing claude-to-openai translator, surfacing thinking content to clients (OpenCode, Claude Code, Cursor, etc.). GLM_REQUEST_DEFAULTS (maxTokens 16384) is left untouched to avoid breaking existing 4.x/5.x models with smaller output caps. Pricing entries mirror glm-5.1 (same Coding Plan quota). Model specs (modelSpecs.ts) declare the 1M context / 131K output for the discovery surface. Refs: https://docs.z.ai/devpack/latest-model --- open-sse/config/glmProvider.ts | 24 ++++++++ open-sse/executors/glm.ts | 58 +++++++++++++++++-- .../__tests__/glmCodingProviderConfig.test.ts | 44 ++++++++++++++ src/shared/constants/modelSpecs.ts | 20 +++++++ src/shared/constants/pricing.ts | 21 +++++++ 5 files changed, 162 insertions(+), 5 deletions(-) diff --git a/open-sse/config/glmProvider.ts b/open-sse/config/glmProvider.ts index 26b0e833a8a..e18b5faa47c 100644 --- a/open-sse/config/glmProvider.ts +++ b/open-sse/config/glmProvider.ts @@ -16,6 +16,30 @@ export const GLM_ANTHROPIC_DEFAULT_BASE_URLS = Object.freeze({ }); export const GLM_SHARED_MODELS = Object.freeze([ + { + id: "glm-5.2", + name: "GLM 5.2", + contextLength: 1000000, + maxOutputTokens: 131072, + toolCalling: true, + supportsReasoning: true, + }, + { + id: "glm-5.2-high", + name: "GLM 5.2 High", + contextLength: 1000000, + maxOutputTokens: 131072, + toolCalling: true, + supportsReasoning: true, + }, + { + id: "glm-5.2-max", + name: "GLM 5.2 Max", + contextLength: 1000000, + maxOutputTokens: 131072, + toolCalling: true, + supportsReasoning: true, + }, { id: "glm-5.1", name: "GLM 5.1", diff --git a/open-sse/executors/glm.ts b/open-sse/executors/glm.ts index 6e78ead8db8..c01fe6f48cb 100644 --- a/open-sse/executors/glm.ts +++ b/open-sse/executors/glm.ts @@ -50,6 +50,19 @@ function getEffectiveKey(credentials: ProviderCredentials): string { return credentials.apiKey || credentials.accessToken || ""; } +/** + * GLM-5.2 effort tiers route exclusively through the Anthropic transport, + * where Zhipu maps Claude Code effort selectors (high/max) to reasoning + * intensity. The base model ID sent upstream is always "glm-5.2". + * + * https://docs.z.ai/devpack/latest-model + */ +function parseGlm52Effort(model: string): { baseModel: string; effort: "high" | "max" } | null { + if (model === "glm-5.2-high") return { baseModel: "glm-5.2", effort: "high" }; + if (model === "glm-5.2-max") return { baseModel: "glm-5.2", effort: "max" }; + return null; +} + function applyGlmRequestDefaults(body: unknown, defaults?: JsonRecord | null): unknown { const record = asRecord(body); if (!record || !defaults) return body; @@ -228,27 +241,53 @@ export class GlmExecutor extends DefaultExecutor { credentials: ProviderCredentials, transport: GlmTransport ) { - const transformed = this.transformRequest(model, body, stream, credentials); + const effortTier = parseGlm52Effort(model); + const effectiveModel = effortTier ? effortTier.baseModel : model; + + const transformed = this.transformRequest(effectiveModel, body, stream, credentials); + const record = asRecord(transformed); + + // Ensure upstream receives the base model ID, not the effort-suffixed alias + if (record && effortTier) { + record.model = effectiveModel; + } if (transport === "openai") { - const record = asRecord(transformed); if (record && stream && hasTools(record) && record.tool_stream === undefined) { return { ...record, tool_stream: true }; } return transformed; } - return translateRequest( + const translated = translateRequest( FORMATS.OPENAI, FORMATS.CLAUDE, - model, - { ...(transformed as JsonRecord), _disableToolPrefix: true }, + effectiveModel, + { ...(record ?? {}), _disableToolPrefix: true }, stream, credentials, this.provider, null, { preserveCacheControl: false } ); + + // Inject effort and thinking for the Anthropic transport. + // Zhipu's Anthropic endpoint requires thinking.type=enabled to emit + // thinking_delta blocks in the SSE response. Without it, reasoning is + // not surfaced and clients see no thinking content. + // The effort-2025-11-24 beta header (in GLM_ANTHROPIC_BETA) carries + // the high/max intensity selector. + if (effortTier) { + const translatedRecord = asRecord(translated); + if (translatedRecord) { + translatedRecord.effort = effortTier.effort; + if (!translatedRecord.thinking) { + translatedRecord.thinking = { type: "enabled" }; + } + } + } + + return translated; } private async executeTransport( @@ -343,6 +382,15 @@ export class GlmExecutor extends DefaultExecutor { } async execute(input: ExecuteInput): Promise { + const effortTier = parseGlm52Effort(input.model); + + // GLM-5.2 effort tiers route directly through Anthropic transport (no fallback). + // Zhipu only graduates effort on the Anthropic endpoint via the + // effort-2025-11-24 beta header included in GLM_ANTHROPIC_BETA. + if (effortTier) { + return this.executeTransport(input, "anthropic"); + } + const primaryTransport = getGlmTransport( input.credentials.providerSpecificData, this.config.baseUrl diff --git a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts index f937df26c24..7a2db82b6da 100644 --- a/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts +++ b/open-sse/mcp-server/__tests__/glmCodingProviderConfig.test.ts @@ -88,6 +88,9 @@ describe("GLM Coding provider registry surfaces", () => { expect(PROVIDER_ID_TO_ALIAS.glm).toBe("glm"); expect(byProviderId).toEqual(byAlias); expect(byProviderId.map((model) => model.id)).toEqual([ + "glm-5.2", + "glm-5.2-high", + "glm-5.2-max", "glm-5.1", "glm-5", "glm-5-turbo", @@ -101,6 +104,30 @@ describe("GLM Coding provider registry surfaces", () => { ]); }); + it("registers GLM-5.2 with correct specs and effort tier aliases", () => { + const models = getModelsByProviderId("glm"); + const get = (id: string) => models.find((m) => m.id === id); + + // Base model + const base = get("glm-5.2"); + expect(base).toBeDefined(); + expect(base?.contextLength).toBe(1000000); + expect(base?.maxOutputTokens).toBe(131072); + expect(base?.supportsReasoning).toBe(true); + expect(base?.toolCalling).toBe(true); + + // Effort tier aliases share the same specs + const high = get("glm-5.2-high"); + expect(high).toBeDefined(); + expect(high?.contextLength).toBe(1000000); + expect(high?.maxOutputTokens).toBe(131072); + + const max = get("glm-5.2-max"); + expect(max).toBeDefined(); + expect(max?.contextLength).toBe(1000000); + expect(max?.maxOutputTokens).toBe(131072); + }); + it("applies doc-backed context window overrides for GLM models", () => { const models = getModelsByProviderId("glm"); const get = (id: string) => models.find((m) => m.id === id); @@ -126,6 +153,9 @@ describe("GLM Coding provider registry surfaces", () => { expect(supportsToolCalling("glm/glm-5")).toBe(true); expect(supportsToolCalling("glm/glm-4.7-flash")).toBe(true); expect(supportsToolCalling("glm/glm-4.5-air")).toBe(true); + expect(supportsToolCalling("glm/glm-5.2")).toBe(true); + expect(supportsToolCalling("glm/glm-5.2-high")).toBe(true); + expect(supportsToolCalling("glm/glm-5.2-max")).toBe(true); expect(getPricingForModel("glm", "glm-5")).toEqual({ input: 1.0, @@ -148,6 +178,20 @@ describe("GLM Coding provider registry surfaces", () => { reasoning: 1.1, cache_creation: 0.2, }); + expect(getPricingForModel("glm", "glm-5.2")).toEqual({ + input: 1.2, + output: 5, + cached: 0.3, + reasoning: 5, + cache_creation: 1.2, + }); + expect(getPricingForModel("glm", "glm-5.2-max")).toEqual({ + input: 1.2, + output: 5, + cached: 0.3, + reasoning: 5, + cache_creation: 1.2, + }); }); it("keeps the repo-derived GLM inventory internally aligned across registry and pricing surfaces", () => { diff --git a/src/shared/constants/modelSpecs.ts b/src/shared/constants/modelSpecs.ts index 7dbd2764c7e..0d1f3caa225 100644 --- a/src/shared/constants/modelSpecs.ts +++ b/src/shared/constants/modelSpecs.ts @@ -354,6 +354,26 @@ export const MODEL_SPECS: Record = { supportsTools: true, }, + // ── Z.AI GLM-5.2 (1M context, 128K max output, effort tiers) ──── + "glm-5.2": { + maxOutputTokens: 131072, + contextWindow: 1000000, + supportsThinking: true, + supportsTools: true, + }, + "glm-5.2-high": { + maxOutputTokens: 131072, + contextWindow: 1000000, + supportsThinking: true, + supportsTools: true, + }, + "glm-5.2-max": { + maxOutputTokens: 131072, + contextWindow: 1000000, + supportsThinking: true, + supportsTools: true, + }, + // ── Z.AI GLM-5.x (200K context, 128K max output) ───────────────── "glm-5.1": { maxOutputTokens: 128000, diff --git a/src/shared/constants/pricing.ts b/src/shared/constants/pricing.ts index d245c9aab97..392afafd50c 100644 --- a/src/shared/constants/pricing.ts +++ b/src/shared/constants/pricing.ts @@ -60,6 +60,27 @@ const CLAUDE_SONNET_46_PRICING = { }; const GLM_PRICING = { + "glm-5.2": { + input: 1.2, + output: 5, + cached: 0.3, + reasoning: 5, + cache_creation: 1.2, + }, + "glm-5.2-high": { + input: 1.2, + output: 5, + cached: 0.3, + reasoning: 5, + cache_creation: 1.2, + }, + "glm-5.2-max": { + input: 1.2, + output: 5, + cached: 0.3, + reasoning: 5, + cache_creation: 1.2, + }, "glm-5.1": { input: 1.2, output: 5, From af8414d95a1d8835a8f8e6ce655ad6928bf32409 Mon Sep 17 00:00:00 2001 From: dhaern Date: Mon, 15 Jun 2026 10:38:33 +0000 Subject: [PATCH 2/4] fix(glm): force thinking.type=enabled for effort tiers, ignore adaptive Clients like Claude Code default to thinking.type=adaptive for reasoning models, but Zhipu only supports enabled/disabled. The previous check (!translatedRecord.thinking) passed adaptive through, causing 400s. Force enabled while preserving other thinking fields (e.g. budget_tokens). Fixes review feedback from @gemini-code-assist --- open-sse/executors/glm.ts | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/open-sse/executors/glm.ts b/open-sse/executors/glm.ts index c01fe6f48cb..418a8a7c0ab 100644 --- a/open-sse/executors/glm.ts +++ b/open-sse/executors/glm.ts @@ -281,8 +281,15 @@ export class GlmExecutor extends DefaultExecutor { const translatedRecord = asRecord(translated); if (translatedRecord) { translatedRecord.effort = effortTier.effort; - if (!translatedRecord.thinking) { - translatedRecord.thinking = { type: "enabled" }; + // Zhipu's Anthropic endpoint only supports thinking.type + // "enabled"/"disabled" — not "adaptive". Clients like Claude Code + // default to "adaptive" for reasoning models, so force "enabled" + // here while preserving any other fields (e.g. budget_tokens). + if (!translatedRecord.thinking || asRecord(translatedRecord.thinking)?.type !== "enabled") { + translatedRecord.thinking = { + ...(asRecord(translatedRecord.thinking) ?? {}), + type: "enabled", + }; } } } From b8066cea693e28b8f8377f862a8d4f6fb17430e9 Mon Sep 17 00:00:00 2001 From: dhaern Date: Mon, 15 Jun 2026 10:44:13 +0000 Subject: [PATCH 3/4] refactor(glm): extract existingThinking to avoid redundant asRecord calls Addresses readability feedback from @gemini-code-assist --- open-sse/executors/glm.ts | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/open-sse/executors/glm.ts b/open-sse/executors/glm.ts index 418a8a7c0ab..f83c6ae260c 100644 --- a/open-sse/executors/glm.ts +++ b/open-sse/executors/glm.ts @@ -285,9 +285,10 @@ export class GlmExecutor extends DefaultExecutor { // "enabled"/"disabled" — not "adaptive". Clients like Claude Code // default to "adaptive" for reasoning models, so force "enabled" // here while preserving any other fields (e.g. budget_tokens). - if (!translatedRecord.thinking || asRecord(translatedRecord.thinking)?.type !== "enabled") { + const existingThinking = asRecord(translatedRecord.thinking); + if (!existingThinking || existingThinking.type !== "enabled") { translatedRecord.thinking = { - ...(asRecord(translatedRecord.thinking) ?? {}), + ...existingThinking, type: "enabled", }; } From 6f85a470b70ae0d0dee00e98f2b0838cc4eaa427 Mon Sep 17 00:00:00 2001 From: dhaern Date: Mon, 15 Jun 2026 12:25:46 -0300 Subject: [PATCH 4/4] chore(quality): re-baseline pricing.ts +21 for GLM-5.2 pricing rows PR #3885 own growth: pricing.ts 1508->1529 (+21 = glm-5.2 + effort aliases). Updates the frozen file-size baseline so Fast Quality Gates pass on release/v3.8.26. Co-authored-by: diegosouzapw --- config/quality/file-size-baseline.json | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index e83ca193aa4..14c5e8bfbe1 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -2,6 +2,7 @@ "_comment": "Catraca de tamanho (check-file-size.mjs). frozen so pode encolher; arquivos novos <= cap. --update ratcheta.", "_rebaseline_v3.8.25": "Drift consciente do ciclo v3.8.24->v3.8.25 (features #3799-#3806: free-provider-rankings, plugins menu, proxy IP-family selector). 3 arquivos cresceram por feature legitima, nao por regressao de qualidade: ProxyRegistryManager.tsx 1072->1089, sidebarVisibility.ts 990->1006, schemas.ts 2519->2522. Encolher fica como debt para um refactor dedicado.", "_rebaseline_2026_06_15_3860_compression_ui": "PR #3860 own growth: sidebarVisibility.ts 1006->1100 (+94 = Compression Hub menu entries: Hub + per-engine Lite/Aggressive/Ultra pages + combos editor) and chatCore.ts 5812->5815 (+3 = compression UI config wiring). Cohesive feature growth, not a quality regression.", + "_rebaseline_2026_06_15_3885_glm_5_2": "PR #3885 own growth: pricing.ts 1508->1529 (+21 = GLM-5.2 pricing rows for glm-5.2 + effort aliases glm-5.2-high/-max, same $1.2/$5 schedule as glm-5.1; pure data). Also adds glm-5.2 specs to glmProvider.ts/modelSpecs.ts (modelSpecs.ts stays under cap). Cohesive model registration; not extractable.", "_rebaseline_2026_06_15_3870_alias_lookup": "PR #3870 own growth: providerRegistry.ts 4703->4708 (+5 = generateModels() also stores each provider's models under its raw id, not only its alias, so getProviderModels(rawId) works when alias != id e.g. github->gh; preserves the existing first-wins guard). Cohesive registry fix; not extractable.", "_rebaseline_2026_06_15_3871_empty_pool": "PR #3871 own growth: combo.ts 5203->5204 (+1 = guard expandAutoComboCandidatePool against an empty candidatePool array — Array.isArray(pool) && pool.length > 0 so [] falls through to active-connection expansion instead of early-returning). One-line correctness fix; not extractable.", "cap": 800, @@ -101,7 +102,7 @@ "src/shared/components/RequestLoggerV2.tsx": 1282, "src/shared/components/analytics/charts.tsx": 1558, "src/shared/constants/cliTools.ts": 875, - "src/shared/constants/pricing.ts": 1508, + "src/shared/constants/pricing.ts": 1529, "src/shared/constants/providers.ts": 3147, "src/shared/constants/sidebarVisibility.ts": 1100, "src/shared/services/cliRuntime.ts": 1090,