From 98d86b2b5e1a4f1ab7de447070d55de5f42273b5 Mon Sep 17 00:00:00 2001 From: Koosha Pari Date: Sun, 13 Sep 2026 02:19:56 -0700 Subject: [PATCH 1/2] fix(zed-hosted): strip provider prefix from model to prevent max_tokens inflation (#13364) When a client sends a model like 'zed-hosted/claude-haiku-4-5-20251001', the provider prefix was passed through to openaiToClaudeRequest unchanged. The capability lookup in fitThinkingToMaxTokens uses static MODEL_SPECS keyed by bare model id, so a prefixed name silently misses the output cap (64000 for Haiku 4.5) and causes max_tokens inflation: 32000 + 131072 = 163072, which Anthropic rejects with 'max_tokens: 163072 > 64000'. Strip the provider prefix before passing the model to translators, the wire payload, and the stream wrapper so capability lookups resolve correctly. Fixes #13364 --- open-sse/executors/zed-hosted.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/open-sse/executors/zed-hosted.ts b/open-sse/executors/zed-hosted.ts index ba66706f10c..807d42db429 100644 --- a/open-sse/executors/zed-hosted.ts +++ b/open-sse/executors/zed-hosted.ts @@ -467,7 +467,7 @@ export class ZedHostedExecutor extends BaseExecutor { thread_id: bodyRecord.thread_id || (credentials as Record)?._clientSessionId, prompt_id: bodyRecord.prompt_id, provider, - model, + model: bareModel, provider_request: providerRequest, }; @@ -496,7 +496,7 @@ export class ZedHostedExecutor extends BaseExecutor { const suppressThinkClose = resolveZedSuppressThinkClose(clientHeaders, clientResponseFormat); const wrapped = response.ok - ? wrapZedCompletionStream(response, provider, model, { suppressThinkClose }) + ? wrapZedCompletionStream(response, provider, bareModel, { suppressThinkClose }) : response; return { response: wrapped, From 35f9a0abc8ba1db28f69cd5a036a44301d7407b9 Mon Sep 17 00:00:00 2001 From: Koosha Pari Date: Sun, 13 Sep 2026 02:22:22 -0700 Subject: [PATCH 2/2] fix: add CJK quota-exhaustion patterns to 429 classifier (fixes #13194) CJK providers (z.ai/GLM, Kimi, Qwen, MiniMax) return Chinese error bodies that none of the English patterns match, causing the gateway to classify long-window quota exhaustion as transient rate_limit. Added Chinese patterns for usage limit, quota exhausted, call limit, and insufficient balance to both QUOTA_PATTERNS and TERMINAL_QUOTA_PATTERNS. --- src/shared/utils/classify429.ts | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/src/shared/utils/classify429.ts b/src/shared/utils/classify429.ts index d4823347782..b70c26abc70 100644 --- a/src/shared/utils/classify429.ts +++ b/src/shared/utils/classify429.ts @@ -99,6 +99,17 @@ const QUOTA_PATTERNS: ReadonlyArray = [ /organization TPD rate limit/i, /\bTPD rate limit\b/i, /insufficient balance/i, + + // #13194 — CJK providers (z.ai/GLM, Kimi, Qwen, MiniMax) return Chinese + // quota-exhaustion messages that none of the English patterns above match. + // Without these, the gateway misclassifies them as transient rate_limit + // and retries a dead model every 60s instead of failing over. + /使用上限/, + /额度已用尽/, + /额度已用完/, + /调用上限/, + /已达上限/, + /余额不足/, ]; /** @@ -166,6 +177,13 @@ const TERMINAL_QUOTA_PATTERNS: ReadonlyArray = [ /organization TPD rate limit/i, /\bTPD rate limit\b/i, /insufficient balance/i, + // #13194 — CJK equivalents for terminal quota signals. + /使用上限/, + /额度已用尽/, + /额度已用完/, + /调用上限/, + /已达上限/, + /余额不足/, ]; /**