Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
37 changes: 37 additions & 0 deletions packages/models/src/models/alibaba.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1782,6 +1782,43 @@ export const alibabaModels = [
"reasoning_effort",
],
},
{
providerId: "novita",
externalId: "qwen/qwen3.8-max",
inputPrice: "2e-6",
cachedInputPrice: "0.25e-6",
outputPrice: "6e-6",
requestPrice: "0",
contextSize: 1000000,
maxOutput: 131072,
// novita accepts every reasoning_effort tier but none of them changes
// the deployment's behaviour (thinking stays on even for "none"), so
// no tier is declared and reasoning_effort is left out of
// supportedParameters below
reasoning: true,
reasoningOutput: "omit",
streaming: true,
vision: true,
tools: true,
// Qwen thinking models reject tool_choice "required" or object
supportedToolChoices: ["auto", "none"],
jsonOutput: true,
jsonOutputSchema: true,
// novita's deployment 400s on the developer role
supportsDeveloperRole: false,
supportedParameters: [
"temperature",
"max_tokens",
"top_p",
"frequency_penalty",
"presence_penalty",
"stop",
"stream",
"tools",
"tool_choice",
"response_format",
],
},
{
providerId: "scx-ai-gp",
externalId: "qwen3.8-max",
Expand Down
27 changes: 25 additions & 2 deletions packages/models/src/models/moonshot.ts
Original file line number Diff line number Diff line change
Expand Up @@ -455,9 +455,9 @@ export const moonshotModels = [
{
providerId: "novita",
externalId: "moonshotai/kimi-k2.6",
inputPrice: "0.95e-6",
inputPrice: "0.8e-6",
cachedInputPrice: "0.16e-6",
outputPrice: "4.0e-6",
outputPrice: "3.4e-6",
requestPrice: "0",
contextSize: 262144,
maxOutput: 262144,
Expand Down Expand Up @@ -601,6 +601,29 @@ export const moonshotModels = [
"reasoning_effort",
],
},
{
providerId: "novita",
externalId: "moonshotai/kimi-k2.7-code",
inputPrice: "0.95e-6",
cachedInputPrice: "0.19e-6",
outputPrice: "4.0e-6",
requestPrice: "0",
contextSize: 262144,
maxOutput: 262144,
reasoning: true,
// Thinking is always on for kimi-k2.7-code: novita accepts `none` and
// `minimal` but keeps reasoning, so only low..max are offered.
reasoningEfforts: ["low", "medium", "high", "xhigh", "max"],
streaming: true,
vision: true,
tools: true,
// novita 400s on forced tool_choice for this deployment
supportedToolChoices: ["auto", "none"],
jsonOutput: true,
jsonOutputSchema: true,
// novita's deployment 400s on the developer role
supportsDeveloperRole: false,
},
{
providerId: "nebius",
externalId: "moonshotai/Kimi-K2.7-Code",
Expand Down
2 changes: 2 additions & 0 deletions packages/models/src/models/tencent.ts
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,8 @@ export const tencentModels = [
reasoningEfforts: ["none", "low", "high"],
vision: false,
tools: true,
// novita 400s on a named function choice for this deployment
supportedToolChoices: ["auto", "none", "required"],
jsonOutputSchema: true,
},
],
Expand Down
60 changes: 42 additions & 18 deletions packages/models/src/models/zai.ts
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,33 @@ export const zaiModels = [
webSearchPrice: "0.01",
jsonOutput: true,
},
{
providerId: "novita",
externalId: "zai-org/glm-5.2",
inputPrice: "1.4e-6",
cachedInputPrice: "0.26e-6",
outputPrice: "4.4e-6",
requestPrice: "0",
contextSize: 1048576,
maxOutput: 131072,
streaming: true,
reasoning: true,
reasoningEfforts: [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh",
"max",
],
vision: false,
tools: true,
// JSON mode is unreliable on this deployment: roughly half of the
// responses put the object in reasoning_content with empty content, and
// the rest markdown-fence it. json_schema behaves the same way.
jsonOutput: false,
},
{
providerId: "canopywave",
test: "skip", // over-reasons heavily and streams slowly (~1 tok/s), so the 60s streaming timeout is flaky
Expand Down Expand Up @@ -235,36 +262,33 @@ export const zaiModels = [
{
providerId: "novita",
externalId: "zai-org/glm-5.1",
inputPrice: "1.4e-6",
inputPrice: "1.38e-6",
cachedInputPrice: "0.26e-6",
outputPrice: "4.4e-6",
requestPrice: "0",
contextSize: 204800,
maxOutput: 131100,
maxOutput: 131072,
quantization: "fp8",
streaming: true,
reasoning: true,
reasoningEfforts: [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh",
"max",
],
// novita's glm-5.1 reasons adaptively and omits reasoning_content
// for simple prompts; no parameter forces it on
reasoningOutput: "omit",
vision: false,
tools: true,
jsonOutput: true,
// novita disables thinking when reasoning_effort is forwarded
// (empty reasoning_content); omitting it reasons by default, so
// exclude reasoning_effort here (verified live 2026-07-14)
supportedParameters: [
"temperature",
"max_tokens",
"top_p",
"frequency_penalty",
"presence_penalty",
"stop",
"stream",
"response_format",
"tools",
"tool_choice",
],
// JSON mode consistently markdown-fences the object on this
// deployment, so normalize it defensively in both modes.
healStreamingJsonOutput: true,
},
{
providerId: "together-ai",
Expand Down Expand Up @@ -398,7 +422,7 @@ export const zaiModels = [
outputPrice: "3.2e-6",
requestPrice: "0",
contextSize: 202800,
maxOutput: 131100,
maxOutput: 131072,
quantization: "fp8",
streaming: true,
reasoning: true,
Expand Down
Loading