From 51e0fa3ff20caa44f2b0c9a57c5c74693a7eb869 Mon Sep 17 00:00:00 2001 From: Luca Steeb Date: Mon, 17 Aug 2026 17:44:13 +0200 Subject: [PATCH 1/2] feat(models): sync Novita mappings and pricing Add the missing Novita mappings for qwen3.8-max, kimi-k2.5, kimi-k2.7-code and glm-5.2, and reconcile the existing ones against Novita's rate card and live deployments. Prices, context and max-output come from Novita's /openai/v1/models endpoint; capability flags were probed against the live deployments. Co-Authored-By: Claude Opus 5 (1M context) --- packages/models/src/models/alibaba.ts | 37 ++++++++++++++++ packages/models/src/models/moonshot.ts | 61 +++++++++++++++++++++++++- packages/models/src/models/tencent.ts | 2 + packages/models/src/models/zai.ts | 60 +++++++++++++++++-------- 4 files changed, 140 insertions(+), 20 deletions(-) diff --git a/packages/models/src/models/alibaba.ts b/packages/models/src/models/alibaba.ts index e50d18610a..df69908bd8 100644 --- a/packages/models/src/models/alibaba.ts +++ b/packages/models/src/models/alibaba.ts @@ -1782,6 +1782,43 @@ export const alibabaModels = [ "reasoning_effort", ], }, + { + providerId: "novita", + externalId: "qwen/qwen3.8-max", + inputPrice: "2e-6", + cachedInputPrice: "0.25e-6", + outputPrice: "6e-6", + requestPrice: "0", + contextSize: 1000000, + maxOutput: 131072, + // novita accepts every reasoning_effort tier but none of them changes + // the deployment's behaviour (thinking stays on even for "none"), so + // no tier is declared and reasoning_effort is left out of + // supportedParameters below + reasoning: true, + reasoningOutput: "omit", + streaming: true, + vision: true, + tools: true, + // Qwen thinking models reject tool_choice "required" or object + supportedToolChoices: ["auto", "none"], + jsonOutput: true, + jsonOutputSchema: true, + // novita's deployment 400s on the developer role + supportsDeveloperRole: false, + supportedParameters: [ + "temperature", + "max_tokens", + "top_p", + "frequency_penalty", + "presence_penalty", + "stop", + "stream", + "tools", + "tool_choice", + "response_format", + ], + }, { providerId: "scx-ai-gp", externalId: "qwen3.8-max", diff --git a/packages/models/src/models/moonshot.ts b/packages/models/src/models/moonshot.ts index 6ca6896725..fcde1e59a9 100644 --- a/packages/models/src/models/moonshot.ts +++ b/packages/models/src/models/moonshot.ts @@ -259,6 +259,40 @@ export const moonshotModels = [ "reasoning_effort", ], }, + { + providerId: "novita", + externalId: "moonshotai/kimi-k2.5", + // About 40% of requests to novita's kimi-k2.5 deployment never + // respond — the connection just stays open until it times out. Plain + // chat, tool calls and tool results all hang at the same rate, while + // kimi-k2.6 and kimi-k2.7-code on the same key are unaffected + // (measured 2026-08-17), so keep it out of routing and e2e. + stability: "unstable", + test: "skip", + inputPrice: "0.6e-6", + cachedInputPrice: "0.1e-6", + outputPrice: "3.0e-6", + requestPrice: "0", + contextSize: 262144, + maxOutput: 262144, + reasoning: true, + // Moonshot thinking is a binary toggle (`thinking.type`), not a + // graduated effort: none/minimal disable it, low..max enable it. + reasoningEfforts: [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max", + ], + streaming: true, + vision: true, + tools: true, + jsonOutput: true, + jsonOutputSchema: true, + }, { providerId: "together-ai", externalId: "moonshotai/Kimi-K2.5", @@ -455,9 +489,9 @@ export const moonshotModels = [ { providerId: "novita", externalId: "moonshotai/kimi-k2.6", - inputPrice: "0.95e-6", + inputPrice: "0.8e-6", cachedInputPrice: "0.16e-6", - outputPrice: "4.0e-6", + outputPrice: "3.4e-6", requestPrice: "0", contextSize: 262144, maxOutput: 262144, @@ -601,6 +635,29 @@ export const moonshotModels = [ "reasoning_effort", ], }, + { + providerId: "novita", + externalId: "moonshotai/kimi-k2.7-code", + inputPrice: "0.95e-6", + cachedInputPrice: "0.19e-6", + outputPrice: "4.0e-6", + requestPrice: "0", + contextSize: 262144, + maxOutput: 262144, + reasoning: true, + // Thinking is always on for kimi-k2.7-code: novita accepts `none` and + // `minimal` but keeps reasoning, so only low..max are offered. + reasoningEfforts: ["low", "medium", "high", "xhigh", "max"], + streaming: true, + vision: true, + tools: true, + // novita 400s on forced tool_choice for this deployment + supportedToolChoices: ["auto", "none"], + jsonOutput: true, + jsonOutputSchema: true, + // novita's deployment 400s on the developer role + supportsDeveloperRole: false, + }, { providerId: "nebius", externalId: "moonshotai/Kimi-K2.7-Code", diff --git a/packages/models/src/models/tencent.ts b/packages/models/src/models/tencent.ts index f24e53b921..09f5070c03 100644 --- a/packages/models/src/models/tencent.ts +++ b/packages/models/src/models/tencent.ts @@ -41,6 +41,8 @@ export const tencentModels = [ reasoningEfforts: ["none", "low", "high"], vision: false, tools: true, + // novita 400s on a named function choice for this deployment + supportedToolChoices: ["auto", "none", "required"], jsonOutputSchema: true, }, ], diff --git a/packages/models/src/models/zai.ts b/packages/models/src/models/zai.ts index 6354994410..8f856b15b6 100644 --- a/packages/models/src/models/zai.ts +++ b/packages/models/src/models/zai.ts @@ -27,6 +27,33 @@ export const zaiModels = [ webSearchPrice: "0.01", jsonOutput: true, }, + { + providerId: "novita", + externalId: "zai-org/glm-5.2", + inputPrice: "1.4e-6", + cachedInputPrice: "0.26e-6", + outputPrice: "4.4e-6", + requestPrice: "0", + contextSize: 1048576, + maxOutput: 131072, + streaming: true, + reasoning: true, + reasoningEfforts: [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max", + ], + vision: false, + tools: true, + // JSON mode is unreliable on this deployment: roughly half of the + // responses put the object in reasoning_content with empty content, and + // the rest markdown-fence it. json_schema behaves the same way. + jsonOutput: false, + }, { providerId: "canopywave", test: "skip", // over-reasons heavily and streams slowly (~1 tok/s), so the 60s streaming timeout is flaky @@ -235,36 +262,33 @@ export const zaiModels = [ { providerId: "novita", externalId: "zai-org/glm-5.1", - inputPrice: "1.4e-6", + inputPrice: "1.38e-6", cachedInputPrice: "0.26e-6", outputPrice: "4.4e-6", requestPrice: "0", contextSize: 204800, - maxOutput: 131100, + maxOutput: 131072, quantization: "fp8", streaming: true, reasoning: true, + reasoningEfforts: [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max", + ], // novita's glm-5.1 reasons adaptively and omits reasoning_content // for simple prompts; no parameter forces it on reasoningOutput: "omit", vision: false, tools: true, jsonOutput: true, - // novita disables thinking when reasoning_effort is forwarded - // (empty reasoning_content); omitting it reasons by default, so - // exclude reasoning_effort here (verified live 2026-07-14) - supportedParameters: [ - "temperature", - "max_tokens", - "top_p", - "frequency_penalty", - "presence_penalty", - "stop", - "stream", - "response_format", - "tools", - "tool_choice", - ], + // JSON mode consistently markdown-fences the object on this + // deployment, so normalize it defensively in both modes. + healStreamingJsonOutput: true, }, { providerId: "together-ai", @@ -398,7 +422,7 @@ export const zaiModels = [ outputPrice: "3.2e-6", requestPrice: "0", contextSize: 202800, - maxOutput: 131100, + maxOutput: 131072, quantization: "fp8", streaming: true, reasoning: true, From bd838fc05e8ac669f43307ac4d64cf40d64ed2d8 Mon Sep 17 00:00:00 2001 From: Luca Steeb Date: Mon, 17 Aug 2026 18:20:48 +0200 Subject: [PATCH 2/2] fix(models): drop unstable Novita kimi-k2.5 mapping About 40% of requests to novita's kimi-k2.5 deployment never respond, so ship nothing rather than a mapping fenced off by stability/test flags. Co-Authored-By: Claude Opus 5 (1M context) --- packages/models/src/models/moonshot.ts | 34 -------------------------- 1 file changed, 34 deletions(-) diff --git a/packages/models/src/models/moonshot.ts b/packages/models/src/models/moonshot.ts index fcde1e59a9..b755a80969 100644 --- a/packages/models/src/models/moonshot.ts +++ b/packages/models/src/models/moonshot.ts @@ -259,40 +259,6 @@ export const moonshotModels = [ "reasoning_effort", ], }, - { - providerId: "novita", - externalId: "moonshotai/kimi-k2.5", - // About 40% of requests to novita's kimi-k2.5 deployment never - // respond — the connection just stays open until it times out. Plain - // chat, tool calls and tool results all hang at the same rate, while - // kimi-k2.6 and kimi-k2.7-code on the same key are unaffected - // (measured 2026-08-17), so keep it out of routing and e2e. - stability: "unstable", - test: "skip", - inputPrice: "0.6e-6", - cachedInputPrice: "0.1e-6", - outputPrice: "3.0e-6", - requestPrice: "0", - contextSize: 262144, - maxOutput: 262144, - reasoning: true, - // Moonshot thinking is a binary toggle (`thinking.type`), not a - // graduated effort: none/minimal disable it, low..max enable it. - reasoningEfforts: [ - "none", - "minimal", - "low", - "medium", - "high", - "xhigh", - "max", - ], - streaming: true, - vision: true, - tools: true, - jsonOutput: true, - jsonOutputSchema: true, - }, { providerId: "together-ai", externalId: "moonshotai/Kimi-K2.5",