diff --git a/package.json b/package.json index 31ac53895..fa2f538b2 100644 --- a/package.json +++ b/package.json @@ -100,6 +100,7 @@ "test:openai-compat-catalog": "npx tsx test/continuous-test-suite-openai-compat-catalog.ts", "test:provider-descriptors": "npx tsx test/continuous-test-suite-provider-descriptors.ts", "test:provider-structure": "npx tsx test/continuous-test-suite-provider-structure.ts", + "test:model-manifests": "npx tsx test/continuous-test-suite-model-manifests.ts", "test:provider-fallback": "npx tsx test/continuous-test-suite-provider-fallback.ts", "test:provider-fallback-latency": "npx tsx test/continuous-test-suite-provider-fallback-latency.ts", "test:provider-wiring": "npx tsx test/continuous-test-suite-provider-wiring.ts", @@ -196,7 +197,7 @@ "quality:metrics": "tsx scripts/quality-metrics.ts", "quality:report": "pnpm run quality:metrics && echo 'Quality metrics saved to quality-metrics.json'", "pre-commit": "lint-staged", - "pre-push": "pnpm run build && pnpm run test:providers-mocked && pnpm run test:provider-structure", + "pre-push": "pnpm run build && pnpm run test:providers-mocked && pnpm run test:provider-structure && pnpm run test:model-manifests", "check:all": "pnpm run lint && pnpm run format --check && pnpm run validate && pnpm run validate:commit", "test:file-formats": "npx tsx test/continuous-test-suite-file-formats.ts", "test:multimodal": "pnpm run test:file-formats && pnpm run test:multimodal:sdk", diff --git a/scripts/generate-remaining-manifests.ts b/scripts/generate-remaining-manifests.ts new file mode 100644 index 000000000..4c2e442ed --- /dev/null +++ b/scripts/generate-remaining-manifests.ts @@ -0,0 +1,172 @@ +#!/usr/bin/env npx tsx +/** + * One-time generator for the 28 model manifests not hand-authored in Tasks + * 2-3. Reads only exported symbols from the pre-migration model-metadata + * stores (MODEL_REGISTRY, getContextWindowSize, ProviderImageAdapter, + * PROVIDER_MAX_TOKENS) — see Task 5's design note for why the private + * PRICING/VISION_CAPABILITIES tables aren't imported directly. Run once; + * the output files are committed and hand-editable afterward like Tasks 2-3. + * + * calculateCost is imported from utils/pricing.js, NOT models/modelRegistry.js + * — both modules export a same-named function, but only pricing.ts's version + * is the actual cost calculator that reads the PRICING table hasPricing() + * gates on. Reading modelRegistry.js's calculateCost here would read a + * different store than the one hasPricing() just checked, producing + * internally-inconsistent data in exactly the cases the gate exists to catch. + */ +import { writeFileSync } from "node:fs"; +import { AIProviderName } from "../src/lib/constants/enums.js"; +import { getModelsByProvider } from "../src/lib/models/modelRegistry.js"; +import { calculateCost, hasPricing } from "../src/lib/utils/pricing.js"; +import { getContextWindowSize } from "../src/lib/constants/contextWindows.js"; +import { ProviderImageAdapter } from "../src/lib/adapters/providerImageAdapter.js"; +import { PROVIDER_MAX_TOKENS } from "../src/lib/core/constants.js"; + +const FULL_PROVIDERS = [ + AIProviderName.AZURE, + AIProviderName.BEDROCK, + AIProviderName.OLLAMA, + AIProviderName.MISTRAL, + AIProviderName.GOOGLE_AI, +] as const; + +const MINIMAL_PROVIDERS = [ + AIProviderName.OPENAI_COMPATIBLE, + AIProviderName.OPENROUTER, + AIProviderName.VERTEX, + AIProviderName.HUGGINGFACE, + AIProviderName.LITELLM, + AIProviderName.SAGEMAKER, + AIProviderName.DEEPSEEK, + AIProviderName.NVIDIA_NIM, + AIProviderName.LM_STUDIO, + AIProviderName.LLAMACPP, + AIProviderName.XAI, + AIProviderName.GROQ, + AIProviderName.COHERE, + AIProviderName.TOGETHER_AI, + AIProviderName.FIREWORKS, + AIProviderName.PERPLEXITY, + AIProviderName.CLOUDFLARE, + AIProviderName.REPLICATE, + AIProviderName.VOYAGE, + AIProviderName.JINA, + AIProviderName.STABILITY, + AIProviderName.IDEOGRAM, + AIProviderName.RECRAFT, +] as const; + +function toCamel(provider: string): string { + return provider.replace(/-([a-z])/g, (_, c: string) => c.toUpperCase()); +} + +function quoteKey(key: string): string { + return /^[a-zA-Z_$][a-zA-Z0-9_$]*$/.test(key) ? key : JSON.stringify(key); +} + +function generateFullManifest(provider: AIProviderName): string { + const models = getModelsByProvider(provider); + const entries = models + .map((m) => { + const priced = hasPricing(provider, m.id); + // usage.total is required by TokenUsage's type but unused by + // calculateCost's body (src/lib/utils/pricing.ts:750-779) — it only + // reads .input/.output/.cacheReadTokens/.cacheCreationTokens. Probing + // with 1_000_000 units on one side and 0 on the other isolates each + // per-token rate scaled back up to a per-million-token price. + const inputRate = priced + ? calculateCost(provider, m.id, { + input: 1_000_000, + output: 0, + total: 1_000_000, + }) + : undefined; + const outputRate = priced + ? calculateCost(provider, m.id, { + input: 0, + output: 1_000_000, + total: 1_000_000, + }) + : undefined; + const pricing = + priced && inputRate !== undefined && outputRate !== undefined + ? `{ input: ${inputRate}, output: ${outputRate} }` + : undefined; + const vision = ProviderImageAdapter.supportsVision(provider, m.id); + return ` ${quoteKey(m.id)}: { + aliases: ${JSON.stringify(m.aliases)}, + displayName: ${JSON.stringify(m.name)}, + contextWindow: ${m.limits.maxContextTokens}, + maxOutputTokens: ${m.limits.maxOutputTokens}, + ${pricing ? `pricingPerMTok: ${pricing},` : "// pricingPerMTok omitted: hasPricing() reports no verified rate"} + vision: ${vision}, + functionCalling: ${m.capabilities.functionCalling}, + reasoning: ${m.capabilities.reasoning}, + jsonMode: ${m.capabilities.jsonMode}, + },`; + }) + .join("\n"); + const defaultWindow = getContextWindowSize(provider); + return `import type { ProviderModelManifest } from "../../types/index.js"; + +export const ${toCamel(provider)}Manifest: ProviderModelManifest = { + defaultContextWindow: ${defaultWindow}, + models: { +${entries} + }, +}; +`; +} + +function generateMinimalManifest(provider: AIProviderName): string { + const defaultWindow = getContextWindowSize(provider); + const providerLimits = PROVIDER_MAX_TOKENS[ + provider as keyof typeof PROVIDER_MAX_TOKENS + ] as { default?: number } | number | undefined; + const rawMaxOutput = + typeof providerLimits === "number" + ? providerLimits + : (providerLimits?.default ?? PROVIDER_MAX_TOKENS.default); + // An output ceiling can never legitimately exceed total context — most + // MINIMAL_PROVIDERS aren't in PROVIDER_MAX_TOKENS's explicit 9-provider + // list, so they fall to its flat 64000 default regardless of how small + // their real context window is. Clamp rather than let the two diverge. + const maxOutput = Math.min(rawMaxOutput, defaultWindow); + return `import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: ${provider} has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const ${toCamel(provider)}Manifest: ProviderModelManifest = { + defaultContextWindow: ${defaultWindow}, + models: { + _default: { + aliases: [], + contextWindow: ${defaultWindow}, + maxOutputTokens: ${maxOutput}, + vision: false, + functionCalling: false, + }, + }, +}; +`; +} + +for (const provider of FULL_PROVIDERS) { + writeFileSync( + `src/lib/models/manifests/${provider}.ts`, + generateFullManifest(provider), + ); + console.log(`wrote src/lib/models/manifests/${provider}.ts`); +} + +for (const provider of MINIMAL_PROVIDERS) { + writeFileSync( + `src/lib/models/manifests/${provider}.ts`, + generateMinimalManifest(provider), + ); + console.log(`wrote src/lib/models/manifests/${provider}.ts`); +} diff --git a/src/lib/constants/enums.ts b/src/lib/constants/enums.ts index aff0779d1..aefffcec1 100644 --- a/src/lib/constants/enums.ts +++ b/src/lib/constants/enums.ts @@ -522,7 +522,12 @@ export enum GoogleAIModels { * Supported Models for Anthropic (Direct API) */ export enum AnthropicModels { - // Claude 4.6 Series (Latest - February 2026) + // Claude 5 Series (mid 2026) — see MODEL_CONTEXT_WINDOWS.anthropic + // (src/lib/constants/contextWindows.ts) for the 1M context window this id + // already carries there. + CLAUDE_SONNET_5 = "claude-sonnet-5", + + // Claude 4.6 Series (February 2026) CLAUDE_OPUS_4_6 = "claude-opus-4-6", CLAUDE_SONNET_4_6 = "claude-sonnet-4-6", diff --git a/src/lib/models/manifestRegistry.ts b/src/lib/models/manifestRegistry.ts new file mode 100644 index 000000000..7ce6ab538 --- /dev/null +++ b/src/lib/models/manifestRegistry.ts @@ -0,0 +1,208 @@ +import type { + ProviderModelManifest, + ProviderModelManifestEntry, +} from "../types/index.js"; +import { PROVIDER_MAX_TOKENS } from "../core/constants.js"; +import { anthropicManifest } from "./manifests/anthropic.js"; +import { openaiManifest } from "./manifests/openai.js"; +import { azureManifest } from "./manifests/azure.js"; +import { bedrockManifest } from "./manifests/bedrock.js"; +import { ollamaManifest } from "./manifests/ollama.js"; +import { mistralManifest } from "./manifests/mistral.js"; +import { googleAiManifest } from "./manifests/google-ai.js"; +import { openaiCompatibleManifest } from "./manifests/openai-compatible.js"; +import { openrouterManifest } from "./manifests/openrouter.js"; +import { vertexManifest } from "./manifests/vertex.js"; +import { huggingfaceManifest } from "./manifests/huggingface.js"; +import { litellmManifest } from "./manifests/litellm.js"; +import { sagemakerManifest } from "./manifests/sagemaker.js"; +import { deepseekManifest } from "./manifests/deepseek.js"; +import { nvidiaNimManifest } from "./manifests/nvidia-nim.js"; +import { lmStudioManifest } from "./manifests/lm-studio.js"; +import { llamacppManifest } from "./manifests/llamacpp.js"; +import { xaiManifest } from "./manifests/xai.js"; +import { groqManifest } from "./manifests/groq.js"; +import { cohereManifest } from "./manifests/cohere.js"; +import { togetherAiManifest } from "./manifests/together-ai.js"; +import { fireworksManifest } from "./manifests/fireworks.js"; +import { perplexityManifest } from "./manifests/perplexity.js"; +import { cloudflareManifest } from "./manifests/cloudflare.js"; +import { replicateManifest } from "./manifests/replicate.js"; +import { voyageManifest } from "./manifests/voyage.js"; +import { jinaManifest } from "./manifests/jina.js"; +import { stabilityManifest } from "./manifests/stability.js"; +import { ideogramManifest } from "./manifests/ideogram.js"; +import { recraftManifest } from "./manifests/recraft.js"; + +/** + * Every provider's model manifest, keyed by the exact AIProviderName enum + * value (kebab-case) — e.g. "google-ai", "nvidia-nim". Manifests are pure + * data with zero heavy dependencies, so they are imported statically here + * (Critical Rule 1's dynamic-import mandate targets providerRegistry.ts's + * *provider* factories, which pull in real SDK clients — not this). + */ +export const MANIFEST_REGISTRY: Record = { + anthropic: anthropicManifest, + openai: openaiManifest, + azure: azureManifest, + bedrock: bedrockManifest, + ollama: ollamaManifest, + mistral: mistralManifest, + "google-ai": googleAiManifest, + "openai-compatible": openaiCompatibleManifest, + openrouter: openrouterManifest, + vertex: vertexManifest, + huggingface: huggingfaceManifest, + litellm: litellmManifest, + sagemaker: sagemakerManifest, + deepseek: deepseekManifest, + "nvidia-nim": nvidiaNimManifest, + "lm-studio": lmStudioManifest, + llamacpp: llamacppManifest, + xai: xaiManifest, + groq: groqManifest, + cohere: cohereManifest, + "together-ai": togetherAiManifest, + fireworks: fireworksManifest, + perplexity: perplexityManifest, + cloudflare: cloudflareManifest, + replicate: replicateManifest, + voyage: voyageManifest, + jina: jinaManifest, + stability: stabilityManifest, + ideogram: ideogramManifest, + recraft: recraftManifest, +}; + +export function getManifestForProvider( + provider: string, +): ProviderModelManifest | undefined { + return MANIFEST_REGISTRY[provider]; +} + +export function getAllManifestProviders(): string[] { + return Object.keys(MANIFEST_REGISTRY); +} + +/** + * Apply every matching family rule's patch, in declaration order, on top of + * a base entry. Later rules win on overlapping fields (last patch applied + * wins), matching the "later registrations overwrite earlier ones" idiom + * used elsewhere in this subsystem (see registerRuntimeContextWindow's + * docblock, src/lib/constants/contextWindows.ts). + */ +function applyFamilyRules( + manifest: ProviderModelManifest, + model: string, + base: ProviderModelManifestEntry, +): ProviderModelManifestEntry { + if (!manifest.familyRules) { + return base; + } + let result = base; + for (const rule of manifest.familyRules) { + if (rule.pattern.test(model)) { + result = { ...result, ...rule.patch }; + } + } + return result; +} + +/** + * Resolve a model against a provider's manifest WITHOUT ever falling back to + * the provider's `_default` entry. Used by callers that need to insert their + * own special-case fallback (cross-provider pricing, proxy pass-through) + * between "no real match" and "give up" — see resolveManifestEntry's + * docblock for why the split exists. + * + * Resolution order: exact canonical id, then a declared alias, then + * longest-prefix match (for tagged/gateway-shaped ids like Ollama's + * "llama3.2:latest" or OpenRouter's "openai/gpt-4o"), then undefined. + * Family rules are applied on top of whichever entry matched. + */ +export function resolveManifestEntryExact( + provider: string, + model: string, +): ProviderModelManifestEntry | undefined { + const manifest = MANIFEST_REGISTRY[provider]; + if (!manifest) { + return undefined; + } + const exact = manifest.models[model]; + if (exact) { + return applyFamilyRules(manifest, model, exact); + } + const aliasMatch = Object.entries(manifest.models).find( + ([canonicalId, entry]) => + canonicalId !== "_default" && entry.aliases.includes(model), + ); + if (aliasMatch) { + return applyFamilyRules(manifest, model, aliasMatch[1]); + } + const sortedKeys = Object.keys(manifest.models) + .filter((k) => k !== "_default") + .sort((a, b) => b.length - a.length); + const prefixKey = sortedKeys.find((k) => model.startsWith(k)); + if (prefixKey) { + return applyFamilyRules(manifest, model, manifest.models[prefixKey]); + } + return undefined; +} + +/** + * Provider-specific output-token ceiling used to synthesize a `_default` + * entry when a manifest declares no explicit one (currently every + * MODEL_REGISTRY-generated manifest — azure, bedrock, ollama, mistral, + * google-ai — since `generateFullManifest` only emits real model ids). + * Mirrors the same table + fallback chain the minimal-manifest generator + * already uses (scripts/generate-remaining-manifests.ts), so a provider + * with no PROVIDER_MAX_TOKENS entry still gets an honest ceiling instead of + * `undefined`. Clamped against the manifest's own `defaultContextWindow` — + * an output ceiling can never legitimately exceed total context. + */ +function defaultMaxOutputTokens( + provider: string, + defaultContextWindow: number, +): number { + const providerLimits = PROVIDER_MAX_TOKENS[ + provider as keyof typeof PROVIDER_MAX_TOKENS + ] as { default?: number } | number | undefined; + const rawMax = + typeof providerLimits === "number" + ? providerLimits + : (providerLimits?.default ?? PROVIDER_MAX_TOKENS.default); + return Math.min(rawMax, defaultContextWindow); +} + +/** + * Resolve a model against a provider's manifest, falling back to the + * provider's `_default` entry (synthesized from `PROVIDER_MAX_TOKENS`, + * clamped to `defaultContextWindow`, when no explicit `_default` model + * entry exists) when no real model matches. Family rules are tested + * against the ORIGINAL model string even on the `_default` path, so an + * unmatched gateway-shaped id still gets patched. + */ +export function resolveManifestEntry( + provider: string, + model: string, +): ProviderModelManifestEntry | undefined { + const exact = resolveManifestEntryExact(provider, model); + if (exact) { + return exact; + } + const manifest = MANIFEST_REGISTRY[provider]; + if (!manifest) { + return undefined; + } + const defaultEntry: ProviderModelManifestEntry = manifest.models._default ?? { + aliases: [], + contextWindow: manifest.defaultContextWindow, + maxOutputTokens: defaultMaxOutputTokens( + provider, + manifest.defaultContextWindow, + ), + vision: false, + functionCalling: false, + }; + return applyFamilyRules(manifest, model, defaultEntry); +} diff --git a/src/lib/models/manifests/anthropic.ts b/src/lib/models/manifests/anthropic.ts new file mode 100644 index 000000000..baf238b3c --- /dev/null +++ b/src/lib/models/manifests/anthropic.ts @@ -0,0 +1,347 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Anthropic model manifest. Canonical ids match AnthropicModels + * (src/lib/constants/enums.ts:524+) and MODEL_CONTEXT_WINDOWS.anthropic + * (src/lib/constants/contextWindows.ts:160-182). maxOutputTokens values + * come from getClaudeMaxOutputTokens (src/lib/utils/tokenLimits.ts) — the + * regex ladder already authoritative for native Claude request paths. + */ +export const anthropicManifest: ProviderModelManifest = { + defaultContextWindow: 200_000, + familyRules: [ + { + // claude-{opus,sonnet,haiku}-N (N>=4) and claude-{fable,mythos}-N — + // mirrors CLAUDE_MODERN_VISION_FAMILIES + // (src/lib/adapters/providerImageAdapter.ts:70-73). Applied to + // gateway-shaped ids (e.g. "vertex_ai/claude-sonnet-5@20260203") that + // don't exact- or prefix-match any entry below. + pattern: /claude-(?:opus|sonnet|haiku)-(?:[4-9]|\d{2,})/i, + patch: { vision: true }, + }, + { + pattern: /claude-(?:fable|mythos)-\d/i, + patch: { vision: true }, + }, + ], + models: { + "claude-sonnet-5": { + aliases: ["sonnet-5", "claude-sonnet"], + displayName: "Claude Sonnet 5", + contextWindow: 1_000_000, + maxOutputTokens: 64_000, + // No pricingPerMTok: genuinely absent from PRICING.anthropic today. + // Do not invent a rate — hasPricing()/findRates() (pricing.ts) must + // keep reporting this model as unpriced. + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + samplingParams: false, // matches SAMPLING_PARAM_REJECTING_FAMILIES /sonnet[-_.]?5(?![0-9])/i + }, + "claude-opus-4-6": { + aliases: ["opus-4.6", "claude-opus-latest"], + displayName: "Claude Opus 4.6", + contextWindow: 1_000_000, + maxOutputTokens: 32_000, + pricingPerMTok: { + input: 5.0, + output: 25.0, + cacheRead: 0.5, + cacheWrite: 6.25, + }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "claude-sonnet-4-6": { + aliases: ["sonnet-4.6"], + displayName: "Claude Sonnet 4.6", + contextWindow: 1_000_000, + maxOutputTokens: 64_000, + pricingPerMTok: { + input: 3.0, + output: 15.0, + cacheRead: 0.3, + cacheWrite: 3.75, + }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "claude-opus-4-5-20251101": { + aliases: ["claude-opus-4-5", "opus-4.5"], + displayName: "Claude Opus 4.5", + contextWindow: 200_000, + maxOutputTokens: 32_000, + pricingPerMTok: { + input: 5.0, + output: 25.0, + cacheRead: 0.5, + cacheWrite: 6.25, + }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[AnthropicModels.CLAUDE_OPUS_4_5] + // (src/lib/models/modelRegistry.ts:1084-1133) so Task 9's equality test + // holds for performance/useCases/category, not just pricing/limits. + curated: { + performance: { speed: "medium", quality: "high", accuracy: "high" }, + useCases: { + coding: 10, + creative: 10, + analysis: 10, + conversation: 9, + reasoning: 10, + translation: 9, + summarization: 9, + }, + category: "reasoning", + }, + }, + "claude-sonnet-4-5-20250929": { + aliases: ["claude-sonnet-4-5", "sonnet-4.5"], + displayName: "Claude Sonnet 4.5", + contextWindow: 200_000, + maxOutputTokens: 64_000, + pricingPerMTok: { + input: 3.0, + output: 15.0, + cacheRead: 0.3, + cacheWrite: 3.75, + }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[AnthropicModels.CLAUDE_SONNET_4_5] + // (src/lib/models/modelRegistry.ts:1135-1179). + curated: { + performance: { speed: "medium", quality: "high", accuracy: "high" }, + useCases: { + coding: 10, + creative: 9, + analysis: 9, + conversation: 9, + reasoning: 10, + translation: 8, + summarization: 8, + }, + category: "coding", + }, + }, + "claude-haiku-4-5-20251001": { + aliases: ["claude-haiku-4-5", "haiku-4.5", "claude-4-5-haiku"], + displayName: "Claude 4.5 Haiku", + contextWindow: 200_000, + maxOutputTokens: 64_000, + pricingPerMTok: { + input: 1.0, + output: 5.0, + cacheRead: 0.1, + cacheWrite: 1.25, + }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[AnthropicModels.CLAUDE_4_5_HAIKU] + // (src/lib/models/modelRegistry.ts:1181-1224). + curated: { + performance: { speed: "fast", quality: "high", accuracy: "high" }, + useCases: { + coding: 8, + creative: 8, + analysis: 8, + conversation: 9, + reasoning: 8, + translation: 8, + summarization: 9, + }, + category: "general", + }, + }, + "claude-opus-4-1-20250805": { + aliases: ["claude-opus-4-1", "opus-4.1"], + displayName: "Claude Opus 4.1", + contextWindow: 200_000, + maxOutputTokens: 32_000, + pricingPerMTok: { + input: 15.0, + output: 75.0, + cacheRead: 1.5, + cacheWrite: 18.75, + }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "claude-opus-4-20250514": { + aliases: ["claude-opus-4"], + displayName: "Claude Opus 4", + contextWindow: 200_000, + maxOutputTokens: 32_000, + pricingPerMTok: { + input: 15.0, + output: 75.0, + cacheRead: 1.5, + cacheWrite: 18.75, + }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "claude-sonnet-4-20250514": { + aliases: ["claude-sonnet-4"], + displayName: "Claude Sonnet 4", + contextWindow: 200_000, + maxOutputTokens: 64_000, + pricingPerMTok: { + input: 3.0, + output: 15.0, + cacheRead: 0.3, + cacheWrite: 3.75, + }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "claude-3-7-sonnet-20250219": { + aliases: ["claude-3-7-sonnet"], + displayName: "Claude 3.7 Sonnet", + contextWindow: 200_000, + maxOutputTokens: 64_000, + pricingPerMTok: { + input: 3.0, + output: 15.0, + cacheRead: 0.3, + cacheWrite: 3.75, + }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "claude-3-5-sonnet-20241022": { + aliases: ["claude-3-5-sonnet"], + displayName: "Claude 3.5 Sonnet", + contextWindow: 200_000, + maxOutputTokens: 8_192, + pricingPerMTok: { + input: 3.0, + output: 15.0, + cacheRead: 0.3, + cacheWrite: 3.75, + }, + vision: true, + functionCalling: true, + reasoning: false, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[AnthropicModels.CLAUDE_3_5_SONNET] + // (src/lib/models/modelRegistry.ts:1226-1275). + curated: { + performance: { speed: "medium", quality: "high", accuracy: "high" }, + useCases: { + coding: 10, + creative: 9, + analysis: 9, + conversation: 9, + reasoning: 10, + translation: 8, + summarization: 8, + }, + category: "coding", + }, + }, + "claude-3-5-haiku-20241022": { + aliases: ["claude-3-5-haiku"], + displayName: "Claude 3.5 Haiku", + contextWindow: 200_000, + maxOutputTokens: 8_192, + pricingPerMTok: { + input: 0.8, + output: 4.0, + cacheRead: 0.08, + cacheWrite: 1.0, + }, + // Deliberately false: the last non-vision Claude. Does not appear in + // VISION_CAPABILITIES.anthropic and does not match the modern-family + // regex (family word precedes the version digit for 3.x ids), see + // providerImageAdapter.ts:66-68's comment. + vision: false, + functionCalling: true, + reasoning: false, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[AnthropicModels.CLAUDE_3_5_HAIKU] + // (src/lib/models/modelRegistry.ts:1277-1320). + curated: { + performance: { speed: "fast", quality: "high", accuracy: "high" }, + useCases: { + coding: 8, + creative: 7, + analysis: 8, + conversation: 8, + reasoning: 8, + translation: 8, + summarization: 9, + }, + category: "general", + }, + }, + "claude-3-opus-20240229": { + aliases: ["claude-3-opus"], + displayName: "Claude 3 Opus", + contextWindow: 200_000, + maxOutputTokens: 4_096, + pricingPerMTok: { + input: 15.0, + output: 75.0, + cacheRead: 1.5, + cacheWrite: 18.75, + }, + vision: true, + functionCalling: true, + reasoning: false, + jsonMode: true, + }, + "claude-3-sonnet-20240229": { + aliases: ["claude-3-sonnet"], + displayName: "Claude 3 Sonnet", + contextWindow: 200_000, + maxOutputTokens: 4_096, + pricingPerMTok: { + input: 3.0, + output: 15.0, + cacheRead: 0.3, + cacheWrite: 3.75, + }, + vision: true, + functionCalling: true, + reasoning: false, + jsonMode: true, + }, + "claude-3-haiku-20240307": { + aliases: ["claude-3-haiku"], + displayName: "Claude 3 Haiku", + contextWindow: 200_000, + maxOutputTokens: 4_096, + pricingPerMTok: { + input: 0.25, + output: 1.25, + cacheRead: 0.025, + cacheWrite: 0.3125, + }, + vision: true, + functionCalling: true, + reasoning: false, + jsonMode: true, + }, + }, +}; diff --git a/src/lib/models/manifests/azure.ts b/src/lib/models/manifests/azure.ts new file mode 100644 index 000000000..1382f90d7 --- /dev/null +++ b/src/lib/models/manifests/azure.ts @@ -0,0 +1,88 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +export const azureManifest: ProviderModelManifest = { + defaultContextWindow: 128000, + models: { + "gpt-5.1": { + aliases: ["azure-gpt-5.1", "gpt51-azure", "azure-flagship"], + displayName: "GPT-5.1 (Azure)", + contextWindow: 300000, + maxOutputTokens: 64000, + pricingPerMTok: { input: 0.625, output: 5 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "gpt-5.1-chat": { + aliases: ["azure-gpt-5.1-chat", "gpt51-chat-azure"], + displayName: "GPT-5.1 Chat (Azure)", + contextWindow: 300000, + maxOutputTokens: 32000, + pricingPerMTok: { input: 0.625, output: 5 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "gpt-5.1-codex": { + aliases: ["azure-gpt-5.1-codex", "gpt51-codex-azure", "azure-code"], + displayName: "GPT-5.1 Codex (Azure)", + contextWindow: 300000, + maxOutputTokens: 64000, + pricingPerMTok: { input: 1.25, output: 10 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "gpt-5.1-codex-mini": { + aliases: ["azure-gpt-5.1-codex-mini", "gpt51-codex-mini-azure"], + displayName: "GPT-5.1 Codex Mini (Azure)", + contextWindow: 200000, + maxOutputTokens: 32000, + pricingPerMTok: { input: 1.25, output: 10 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "gpt-5.1-codex-max": { + aliases: [ + "azure-gpt-5.1-codex-max", + "gpt51-codex-max-azure", + "azure-enterprise", + ], + displayName: "GPT-5.1 Codex Max (Azure)", + contextWindow: 500000, + maxOutputTokens: 128000, + pricingPerMTok: { input: 1.25, output: 10 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "gpt-5-pro": { + aliases: ["azure-gpt-5-pro", "gpt5-pro-azure"], + displayName: "GPT-5 Pro (Azure)", + contextWindow: 256000, + maxOutputTokens: 64000, + pricingPerMTok: { input: 1.25, output: 10 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "gpt-5-turbo": { + aliases: ["azure-gpt-5-turbo", "gpt5-turbo-azure"], + displayName: "GPT-5 Turbo (Azure)", + contextWindow: 200000, + maxOutputTokens: 32768, + pricingPerMTok: { input: 1.25, output: 10 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + }, +}; diff --git a/src/lib/models/manifests/bedrock.ts b/src/lib/models/manifests/bedrock.ts new file mode 100644 index 000000000..eb83e0315 --- /dev/null +++ b/src/lib/models/manifests/bedrock.ts @@ -0,0 +1,76 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +// The capability and limit fields below come from that model's AWS Bedrock +// model card (docs.aws.amazon.com/bedrock/latest/userguide/model-card-*.html): +// the Model Details block for contextWindow and maxOutputTokens, and the +// "Features supported using bedrock-runtime endpoint" table for jsonMode (AWS +// calls it "Structured outputs") and functionCalling ("Client-side tool +// calling"). `reasoning` is true only where the card states "Reasoning: +// Supported"; the card omits the line entirely for models that do not. +// +// `pricingPerMTok` is NOT from the model cards — they link out to +// aws.amazon.com/bedrock/pricing rather than carrying rates. Entries here +// either omit it or carry a rate verified separately, so do not assume the +// card backs a pricing figure. +export const bedrockManifest: ProviderModelManifest = { + defaultContextWindow: 200000, + models: { + "amazon.nova-premier-v1:0": { + aliases: ["nova-premier", "aws-flagship"], + displayName: "Amazon Nova Premier", + contextWindow: 1000000, + maxOutputTokens: 25000, + // pricingPerMTok omitted: hasPricing() reports no verified rate + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: false, + }, + "amazon.nova-pro-v1:0": { + aliases: ["nova-pro", "aws-balanced"], + displayName: "Amazon Nova Pro", + contextWindow: 300000, + maxOutputTokens: 5000, + // pricingPerMTok omitted: hasPricing() reports no verified rate + vision: true, + functionCalling: true, + reasoning: false, + jsonMode: false, + }, + "amazon.nova-lite-v1:0": { + aliases: ["nova-lite", "aws-lite", "aws-cheap"], + displayName: "Amazon Nova Lite", + contextWindow: 300000, + maxOutputTokens: 5000, + // pricingPerMTok omitted: hasPricing() reports no verified rate + vision: true, + functionCalling: true, + reasoning: false, + jsonMode: false, + }, + // 20251101, not the 20251124 launch date — AWS issues the ID from the + // model snapshot date. See the model card's Programmatic Access table. + "anthropic.claude-opus-4-5-20251101-v1:0": { + aliases: ["bedrock-claude-4.5-opus", "bedrock-claude-flagship"], + displayName: "Claude 4.5 Opus (Bedrock)", + contextWindow: 200000, + maxOutputTokens: 64000, + pricingPerMTok: { input: 5, output: 25 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "meta.llama4-maverick-17b-instruct-v1:0": { + aliases: ["bedrock-llama4", "bedrock-llama-maverick"], + displayName: "Llama 4 Maverick (Bedrock)", + contextWindow: 1000000, + maxOutputTokens: 8192, + // pricingPerMTok omitted: hasPricing() reports no verified rate + vision: true, + functionCalling: true, + reasoning: false, + jsonMode: false, + }, + }, +}; diff --git a/src/lib/models/manifests/cloudflare.ts b/src/lib/models/manifests/cloudflare.ts new file mode 100644 index 000000000..687024c70 --- /dev/null +++ b/src/lib/models/manifests/cloudflare.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: cloudflare has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const cloudflareManifest: ProviderModelManifest = { + defaultContextWindow: 8192, + models: { + _default: { + aliases: [], + contextWindow: 8192, + maxOutputTokens: 8192, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/cohere.ts b/src/lib/models/manifests/cohere.ts new file mode 100644 index 000000000..aed03e6bd --- /dev/null +++ b/src/lib/models/manifests/cohere.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: cohere has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const cohereManifest: ProviderModelManifest = { + defaultContextWindow: 128000, + models: { + _default: { + aliases: [], + contextWindow: 128000, + maxOutputTokens: 64000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/deepseek.ts b/src/lib/models/manifests/deepseek.ts new file mode 100644 index 000000000..a8d02bf78 --- /dev/null +++ b/src/lib/models/manifests/deepseek.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: deepseek has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const deepseekManifest: ProviderModelManifest = { + defaultContextWindow: 64000, + models: { + _default: { + aliases: [], + contextWindow: 64000, + maxOutputTokens: 64000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/fireworks.ts b/src/lib/models/manifests/fireworks.ts new file mode 100644 index 000000000..3600d95a2 --- /dev/null +++ b/src/lib/models/manifests/fireworks.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: fireworks has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const fireworksManifest: ProviderModelManifest = { + defaultContextWindow: 128000, + models: { + _default: { + aliases: [], + contextWindow: 128000, + maxOutputTokens: 64000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/google-ai.ts b/src/lib/models/manifests/google-ai.ts new file mode 100644 index 000000000..729d4f803 --- /dev/null +++ b/src/lib/models/manifests/google-ai.ts @@ -0,0 +1,29 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +export const googleAiManifest: ProviderModelManifest = { + defaultContextWindow: 1048576, + models: { + "gemini-2.5-pro": { + aliases: ["gemini-pro", "google-flagship", "best-analysis"], + displayName: "Gemini 2.5 Pro", + contextWindow: 2097152, + maxOutputTokens: 8192, + pricingPerMTok: { input: 1.25, output: 10 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "gemini-2.5-flash": { + aliases: ["gemini-flash", "google-fast", "best-value"], + displayName: "Gemini 2.5 Flash", + contextWindow: 1048576, + maxOutputTokens: 8192, + pricingPerMTok: { input: 0.3, output: 2.5 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + }, +}; diff --git a/src/lib/models/manifests/groq.ts b/src/lib/models/manifests/groq.ts new file mode 100644 index 000000000..6921573b1 --- /dev/null +++ b/src/lib/models/manifests/groq.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: groq has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const groqManifest: ProviderModelManifest = { + defaultContextWindow: 128000, + models: { + _default: { + aliases: [], + contextWindow: 128000, + maxOutputTokens: 64000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/huggingface.ts b/src/lib/models/manifests/huggingface.ts new file mode 100644 index 000000000..fd5001dd1 --- /dev/null +++ b/src/lib/models/manifests/huggingface.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: huggingface has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const huggingfaceManifest: ProviderModelManifest = { + defaultContextWindow: 32000, + models: { + _default: { + aliases: [], + contextWindow: 32000, + maxOutputTokens: 32000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/ideogram.ts b/src/lib/models/manifests/ideogram.ts new file mode 100644 index 000000000..dca10179f --- /dev/null +++ b/src/lib/models/manifests/ideogram.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: ideogram has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const ideogramManifest: ProviderModelManifest = { + defaultContextWindow: 2000, + models: { + _default: { + aliases: [], + contextWindow: 2000, + maxOutputTokens: 2000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/jina.ts b/src/lib/models/manifests/jina.ts new file mode 100644 index 000000000..40f4edd12 --- /dev/null +++ b/src/lib/models/manifests/jina.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: jina has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const jinaManifest: ProviderModelManifest = { + defaultContextWindow: 8192, + models: { + _default: { + aliases: [], + contextWindow: 8192, + maxOutputTokens: 8192, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/litellm.ts b/src/lib/models/manifests/litellm.ts new file mode 100644 index 000000000..510b5fc1a --- /dev/null +++ b/src/lib/models/manifests/litellm.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: litellm has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const litellmManifest: ProviderModelManifest = { + defaultContextWindow: 128000, + models: { + _default: { + aliases: [], + contextWindow: 128000, + maxOutputTokens: 128000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/llamacpp.ts b/src/lib/models/manifests/llamacpp.ts new file mode 100644 index 000000000..156b6abc6 --- /dev/null +++ b/src/lib/models/manifests/llamacpp.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: llamacpp has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const llamacppManifest: ProviderModelManifest = { + defaultContextWindow: 8192, + models: { + _default: { + aliases: [], + contextWindow: 8192, + maxOutputTokens: 8192, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/lm-studio.ts b/src/lib/models/manifests/lm-studio.ts new file mode 100644 index 000000000..1ead16473 --- /dev/null +++ b/src/lib/models/manifests/lm-studio.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: lm-studio has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const lmStudioManifest: ProviderModelManifest = { + defaultContextWindow: 8192, + models: { + _default: { + aliases: [], + contextWindow: 8192, + maxOutputTokens: 8192, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/mistral.ts b/src/lib/models/manifests/mistral.ts new file mode 100644 index 000000000..d40a13d71 --- /dev/null +++ b/src/lib/models/manifests/mistral.ts @@ -0,0 +1,51 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +export const mistralManifest: ProviderModelManifest = { + defaultContextWindow: 128000, + models: { + "mistral-large-latest": { + aliases: ["mistral-large", "mistral-flagship"], + displayName: "Mistral Large", + contextWindow: 131072, + maxOutputTokens: 8192, + pricingPerMTok: { input: 2, output: 6 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "mistral-small-latest": { + aliases: ["mistral-small", "mistral-cheap"], + displayName: "Mistral Small", + contextWindow: 32768, + maxOutputTokens: 8192, + pricingPerMTok: { input: 0.2, output: 0.6 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "codestral-latest": { + aliases: ["codestral", "mistral-code"], + displayName: "Codestral", + contextWindow: 32768, + maxOutputTokens: 8192, + pricingPerMTok: { input: 0.3, output: 0.9 }, + vision: false, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "pixtral-large": { + aliases: ["pixtral", "mistral-vision"], + displayName: "Pixtral Large", + contextWindow: 131072, + maxOutputTokens: 8192, + // pricingPerMTok omitted: hasPricing() reports no verified rate + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + }, +}; diff --git a/src/lib/models/manifests/nvidia-nim.ts b/src/lib/models/manifests/nvidia-nim.ts new file mode 100644 index 000000000..4cd1e0fbd --- /dev/null +++ b/src/lib/models/manifests/nvidia-nim.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: nvidia-nim has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const nvidiaNimManifest: ProviderModelManifest = { + defaultContextWindow: 128000, + models: { + _default: { + aliases: [], + contextWindow: 128000, + maxOutputTokens: 64000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/ollama.ts b/src/lib/models/manifests/ollama.ts new file mode 100644 index 000000000..5bf77e34c --- /dev/null +++ b/src/lib/models/manifests/ollama.ts @@ -0,0 +1,73 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +export const ollamaManifest: ProviderModelManifest = { + defaultContextWindow: 128000, + models: { + "llama4:latest": { + aliases: ["llama4", "llama4-local"], + displayName: "Llama 4", + contextWindow: 131072, + maxOutputTokens: 8192, + // pricingPerMTok omitted: hasPricing() reports no verified rate + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "llama3.3:latest": { + aliases: ["llama3.3", "llama3.3-local"], + displayName: "Llama 3.3", + contextWindow: 131072, + maxOutputTokens: 8192, + // pricingPerMTok omitted: hasPricing() reports no verified rate + vision: false, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "llama3.2:latest": { + aliases: ["llama3.2", "llama", "local", "offline"], + displayName: "Llama 3.2 Latest", + contextWindow: 131072, + maxOutputTokens: 8192, + // pricingPerMTok omitted: hasPricing() reports no verified rate + vision: false, + functionCalling: false, + reasoning: true, + jsonMode: false, + }, + "deepseek-r1:70b": { + aliases: ["deepseek-r1", "deepseek-reasoning", "local-reasoning"], + displayName: "DeepSeek-R1 70B", + contextWindow: 65536, + maxOutputTokens: 8192, + // pricingPerMTok omitted: hasPricing() reports no verified rate + vision: false, + functionCalling: false, + reasoning: true, + jsonMode: false, + }, + "qwen3:72b": { + aliases: ["qwen3", "qwen3-72b-local"], + displayName: "Qwen 3 72B", + contextWindow: 131072, + maxOutputTokens: 8192, + // pricingPerMTok omitted: hasPricing() reports no verified rate + vision: false, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + "mistral-large:latest": { + aliases: ["mistral-large-local"], + displayName: "Mistral Large (Local)", + contextWindow: 131072, + maxOutputTokens: 8192, + // pricingPerMTok omitted: hasPricing() reports no verified rate + vision: false, + functionCalling: true, + reasoning: true, + jsonMode: true, + }, + }, +}; diff --git a/src/lib/models/manifests/openai-compatible.ts b/src/lib/models/manifests/openai-compatible.ts new file mode 100644 index 000000000..2cad7c47e --- /dev/null +++ b/src/lib/models/manifests/openai-compatible.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: openai-compatible has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const openaiCompatibleManifest: ProviderModelManifest = { + defaultContextWindow: 128000, + models: { + _default: { + aliases: [], + contextWindow: 128000, + maxOutputTokens: 64000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/openai.ts b/src/lib/models/manifests/openai.ts new file mode 100644 index 000000000..b04075eea --- /dev/null +++ b/src/lib/models/manifests/openai.ts @@ -0,0 +1,561 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * OpenAI model manifest. contextWindow values come from + * MODEL_CONTEXT_WINDOWS.openai (src/lib/constants/contextWindows.ts), which + * disagrees with MODEL_REGISTRY.limits.maxContextTokens for several ids + * (e.g. gpt-5: 400_000 here vs 256_000 there) — MODEL_CONTEXT_WINDOWS wins + * as the more actively-maintained store. maxOutputTokens/aliases/capability + * flags come from MODEL_REGISTRY (src/lib/models/modelRegistry.ts:30-984). + * pricingPerMTok comes from PRICING.openai (src/lib/utils/pricing.ts). + */ +export const openaiManifest: ProviderModelManifest = { + defaultContextWindow: 128_000, + models: { + "gpt-4o": { + aliases: ["gpt4o", "gpt-4-omni", "openai-flagship"], + displayName: "GPT-4 Omni", + contextWindow: 128_000, + maxOutputTokens: 4_096, + pricingPerMTok: { input: 2.5, output: 10.0, cacheRead: 0.625 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_4O] + // (src/lib/models/modelRegistry.ts:30-73). + curated: { + performance: { speed: "medium", quality: "high", accuracy: "high" }, + useCases: { + coding: 9, + creative: 8, + analysis: 9, + conversation: 9, + reasoning: 9, + translation: 8, + summarization: 8, + }, + category: "general", + }, + }, + "gpt-4o-mini": { + aliases: ["gpt4o-mini", "gpt-4-mini", "fastest", "cheap"], + displayName: "GPT-4 Omni Mini", + contextWindow: 128_000, + maxOutputTokens: 16_384, + pricingPerMTok: { input: 0.15, output: 0.6, cacheRead: 0.0375 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_4O_MINI] + // (src/lib/models/modelRegistry.ts:75-118). + curated: { + performance: { speed: "fast", quality: "high", accuracy: "high" }, + useCases: { + coding: 8, + creative: 7, + analysis: 8, + conversation: 8, + reasoning: 8, + translation: 8, + summarization: 9, + }, + category: "general", + }, + }, + "gpt-5": { + aliases: ["gpt5", "gpt-5-flagship", "openai-latest"], + displayName: "GPT-5", + contextWindow: 400_000, + maxOutputTokens: 32_768, + pricingPerMTok: { input: 1.25, output: 10.0, cacheRead: 0.3125 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_5] + // (src/lib/models/modelRegistry.ts:121-165). + curated: { + performance: { speed: "medium", quality: "high", accuracy: "high" }, + useCases: { + coding: 10, + creative: 10, + analysis: 10, + conversation: 10, + reasoning: 10, + translation: 9, + summarization: 9, + }, + category: "reasoning", + }, + }, + "gpt-5-mini": { + aliases: ["gpt5-mini", "gpt-5-fast"], + displayName: "GPT-5 Mini", + contextWindow: 400_000, + maxOutputTokens: 16_384, + pricingPerMTok: { input: 0.25, output: 2.0, cacheRead: 0.0625 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_5_MINI] + // (src/lib/models/modelRegistry.ts:167-210). + curated: { + performance: { speed: "fast", quality: "high", accuracy: "high" }, + useCases: { + coding: 8, + creative: 8, + analysis: 8, + conversation: 9, + reasoning: 8, + translation: 8, + summarization: 9, + }, + category: "general", + }, + }, + o3: { + aliases: ["o3-reasoning", "o3-thinking"], + displayName: "O3", + contextWindow: 200_000, + maxOutputTokens: 100_000, + pricingPerMTok: { input: 2.0, output: 8.0, cacheRead: 0.5 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.O3] + // (src/lib/models/modelRegistry.ts:213-257). + curated: { + performance: { speed: "slow", quality: "high", accuracy: "high" }, + useCases: { + coding: 10, + creative: 8, + analysis: 10, + conversation: 7, + reasoning: 10, + translation: 7, + summarization: 8, + }, + category: "reasoning", + }, + }, + "o3-mini": { + aliases: ["o3-mini-reasoning"], + displayName: "O3 Mini", + contextWindow: 200_000, + maxOutputTokens: 65_536, + pricingPerMTok: { input: 1.1, output: 4.4, cacheRead: 0.275 }, + vision: false, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.O3_MINI] + // (src/lib/models/modelRegistry.ts:259-303). + curated: { + performance: { speed: "medium", quality: "high", accuracy: "high" }, + useCases: { + coding: 9, + creative: 6, + analysis: 9, + conversation: 7, + reasoning: 9, + translation: 6, + summarization: 7, + }, + category: "reasoning", + }, + }, + "gpt-5-nano": { + aliases: ["gpt5-nano", "gpt-5-cheapest"], + displayName: "GPT-5 Nano", + contextWindow: 400_000, + maxOutputTokens: 128_000, + pricingPerMTok: { input: 0.05, output: 0.4, cacheRead: 0.0125 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_5_NANO] + // (src/lib/models/modelRegistry.ts:305-349). + curated: { + performance: { speed: "fast", quality: "medium", accuracy: "medium" }, + useCases: { + coding: 6, + creative: 6, + analysis: 6, + conversation: 8, + reasoning: 6, + translation: 7, + summarization: 8, + }, + category: "general", + }, + }, + "gpt-5.2": { + aliases: ["gpt52", "gpt-5.2-thinking", "openai-latest-reasoning"], + displayName: "GPT-5.2 Thinking", + contextWindow: 400_000, + maxOutputTokens: 64_000, + pricingPerMTok: { input: 1.75, output: 14.0, cacheRead: 0.4375 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_5_2] + // (src/lib/models/modelRegistry.ts:352-396). + curated: { + performance: { speed: "medium", quality: "high", accuracy: "high" }, + useCases: { + coding: 10, + creative: 10, + analysis: 10, + conversation: 9, + reasoning: 10, + translation: 9, + summarization: 9, + }, + category: "reasoning", + }, + }, + "gpt-5.2-chat-latest": { + aliases: ["gpt52-chat", "gpt-5.2-instant", "gpt52-fast"], + displayName: "GPT-5.2 Instant", + contextWindow: 128_000, + maxOutputTokens: 32_000, + // Inherits gpt-5.2's rate — matches findRates()'s existing prefix-match + // resolution for this id today (no distinct PRICING.openai key exists). + pricingPerMTok: { input: 1.75, output: 14.0, cacheRead: 0.4375 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_5_2_CHAT_LATEST] + // (src/lib/models/modelRegistry.ts:398-442). + curated: { + performance: { speed: "fast", quality: "high", accuracy: "high" }, + useCases: { + coding: 9, + creative: 9, + analysis: 9, + conversation: 10, + reasoning: 9, + translation: 9, + summarization: 9, + }, + category: "general", + }, + }, + "gpt-5.2-pro": { + aliases: ["gpt52-pro", "gpt-5.2-professional", "openai-science"], + displayName: "GPT-5.2 Pro", + contextWindow: 400_000, + maxOutputTokens: 128_000, + // Inherits gpt-5.2's rate — see gpt-5.2-chat-latest's comment above. + pricingPerMTok: { input: 1.75, output: 14.0, cacheRead: 0.4375 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_5_2_PRO] + // (src/lib/models/modelRegistry.ts:444-488). + curated: { + performance: { speed: "slow", quality: "high", accuracy: "high" }, + useCases: { + coding: 10, + creative: 9, + analysis: 10, + conversation: 8, + reasoning: 10, + translation: 9, + summarization: 9, + }, + category: "reasoning", + }, + }, + "gpt-4.1": { + aliases: ["gpt-4.1", "gpt41", "million-context"], + displayName: "GPT-4.1", + contextWindow: 1_047_576, + maxOutputTokens: 128_000, + pricingPerMTok: { input: 2.0, output: 8.0, cacheRead: 0.5 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_4_1] + // (src/lib/models/modelRegistry.ts:491-534). + curated: { + performance: { speed: "medium", quality: "high", accuracy: "high" }, + useCases: { + coding: 10, + creative: 8, + analysis: 9, + conversation: 8, + reasoning: 9, + translation: 8, + summarization: 9, + }, + category: "coding", + }, + }, + "gpt-4.1-mini": { + aliases: ["gpt-4.1-mini", "gpt41-mini"], + displayName: "GPT-4.1 Mini", + contextWindow: 1_047_576, + maxOutputTokens: 128_000, + pricingPerMTok: { input: 0.4, output: 1.6, cacheRead: 0.1 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_4_1_MINI] + // (src/lib/models/modelRegistry.ts:536-579). + curated: { + performance: { speed: "fast", quality: "high", accuracy: "high" }, + useCases: { + coding: 9, + creative: 7, + analysis: 8, + conversation: 8, + reasoning: 8, + translation: 8, + summarization: 9, + }, + category: "coding", + }, + }, + "gpt-4.1-nano": { + aliases: ["gpt-4.1-nano", "gpt41-nano"], + displayName: "GPT-4.1 Nano", + contextWindow: 1_047_576, + maxOutputTokens: 128_000, + pricingPerMTok: { input: 0.1, output: 0.4, cacheRead: 0.025 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_4_1_NANO] + // (src/lib/models/modelRegistry.ts:581-624). + curated: { + performance: { speed: "fast", quality: "medium", accuracy: "medium" }, + useCases: { + coding: 7, + creative: 6, + analysis: 7, + conversation: 7, + reasoning: 7, + translation: 7, + summarization: 8, + }, + category: "coding", + }, + }, + "o3-pro": { + aliases: ["o3-pro", "o3-professional"], + displayName: "O3 Pro", + contextWindow: 200_000, + maxOutputTokens: 100_000, + // Inherits o3's rate — matches findRates()'s existing prefix-match + // resolution for this id today (no distinct PRICING.openai key exists). + pricingPerMTok: { input: 2.0, output: 8.0, cacheRead: 0.5 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.O3_PRO] + // (src/lib/models/modelRegistry.ts:627-671). + curated: { + performance: { speed: "slow", quality: "high", accuracy: "high" }, + useCases: { + coding: 10, + creative: 7, + analysis: 10, + conversation: 6, + reasoning: 10, + translation: 6, + summarization: 7, + }, + category: "reasoning", + }, + }, + "o4-mini": { + aliases: ["o4-mini", "o4-fast"], + displayName: "O4 Mini", + contextWindow: 200_000, + maxOutputTokens: 100_000, + pricingPerMTok: { input: 1.1, output: 4.4, cacheRead: 0.275 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.O4_MINI] + // (src/lib/models/modelRegistry.ts:673-717). + curated: { + performance: { speed: "medium", quality: "high", accuracy: "high" }, + useCases: { + coding: 9, + creative: 6, + analysis: 9, + conversation: 7, + reasoning: 10, + translation: 6, + summarization: 7, + }, + category: "reasoning", + }, + }, + o1: { + aliases: ["o1-full", "o1-premium"], + displayName: "O1", + contextWindow: 200_000, + maxOutputTokens: 32_768, + pricingPerMTok: { input: 15.0, output: 60.0, cacheRead: 3.75 }, + // DELIBERATE DEVIATION FROM task-3-brief.md (which specifies `true`): + // set to `false` to mirror live behavior exactly. Today, + // ProviderImageAdapter.supportsVision("openai", "o1") returns false — + // VISION_CAPABILITIES.openai (providerImageAdapter.ts) has no "o1" + // entry (o1-preview/o1-mini are also absent, but so is o1 itself) and + // there is no "openai" key in VISION_FAMILY_RULES for a regex fallback + // to catch it. This manifest's job in this PR is to be a faithful, + // behavior-preserving replacement for the tables it will supersede + // (Tasks 7-11 migrate the consumers; Task 14 proves behavior was + // preserved) — that proof only holds if the manifest encodes what the + // system actually does today, not what it should do. + // + // This is very likely a live bug, not a fact about o1: OpenAI's real + // o1 (distinct from o1-preview/o1-mini) is believed to accept image + // input, and MODEL_REGISTRY[OpenAIModels.O1].capabilities.vision is + // itself `true` (modelRegistry.ts:726) — i.e. the existing hand-curated + // registry already disagrees with VISION_CAPABILITIES.openai on this + // exact point. Correcting VISION_CAPABILITIES.openai to add "o1" is a + // separate, deliberately deferred change with its own commit and its + // own test, not something to smuggle into a purely-additive data port. + vision: false, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.O1] + // (src/lib/models/modelRegistry.ts:719-763). + curated: { + performance: { speed: "slow", quality: "high", accuracy: "high" }, + useCases: { + coding: 10, + creative: 7, + analysis: 10, + conversation: 6, + reasoning: 10, + translation: 6, + summarization: 7, + }, + category: "reasoning", + }, + }, + "o1-mini": { + aliases: ["o1-mini", "o1-budget"], + displayName: "O1 Mini", + contextWindow: 128_000, + maxOutputTokens: 65_536, + pricingPerMTok: { input: 0.55, output: 2.2, cacheRead: 0.1375 }, + vision: false, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.O1_MINI] + // (src/lib/models/modelRegistry.ts:810-853). + curated: { + performance: { speed: "medium", quality: "high", accuracy: "high" }, + useCases: { + coding: 8, + creative: 5, + analysis: 8, + conversation: 6, + reasoning: 8, + translation: 5, + summarization: 6, + }, + category: "reasoning", + }, + }, + "gpt-4": { + aliases: ["gpt4", "gpt-4-base"], + displayName: "GPT-4", + contextWindow: 8_192, + maxOutputTokens: 4_096, + pricingPerMTok: { input: 30.0, output: 60.0 }, + vision: false, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_4] + // (src/lib/models/modelRegistry.ts:856-899). + curated: { + performance: { speed: "slow", quality: "high", accuracy: "high" }, + useCases: { + coding: 8, + creative: 8, + analysis: 8, + conversation: 8, + reasoning: 8, + translation: 8, + summarization: 8, + }, + category: "general", + }, + }, + "gpt-4-turbo": { + aliases: ["gpt4-turbo", "gpt-4-turbo-preview"], + displayName: "GPT-4 Turbo", + contextWindow: 128_000, + maxOutputTokens: 4_096, + pricingPerMTok: { input: 10.0, output: 30.0 }, + vision: true, + functionCalling: true, + reasoning: true, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_4_TURBO] + // (src/lib/models/modelRegistry.ts:901-944). + curated: { + performance: { speed: "medium", quality: "high", accuracy: "high" }, + useCases: { + coding: 8, + creative: 8, + analysis: 9, + conversation: 8, + reasoning: 8, + translation: 8, + summarization: 8, + }, + category: "general", + }, + }, + "gpt-3.5-turbo": { + aliases: ["gpt35", "gpt-3.5", "chatgpt"], + displayName: "GPT-3.5 Turbo", + contextWindow: 16_385, + maxOutputTokens: 4_096, + pricingPerMTok: { input: 0.5, output: 1.0 }, + vision: false, + functionCalling: true, + reasoning: false, + jsonMode: true, + // Carried forward verbatim from MODEL_REGISTRY[OpenAIModels.GPT_3_5_TURBO] + // (src/lib/models/modelRegistry.ts:946-989). + curated: { + performance: { speed: "fast", quality: "medium", accuracy: "medium" }, + useCases: { + coding: 6, + creative: 6, + analysis: 6, + conversation: 7, + reasoning: 5, + translation: 7, + summarization: 7, + }, + category: "general", + }, + }, + }, +}; diff --git a/src/lib/models/manifests/openrouter.ts b/src/lib/models/manifests/openrouter.ts new file mode 100644 index 000000000..ac2c5fb14 --- /dev/null +++ b/src/lib/models/manifests/openrouter.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: openrouter has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const openrouterManifest: ProviderModelManifest = { + defaultContextWindow: 128000, + models: { + _default: { + aliases: [], + contextWindow: 128000, + maxOutputTokens: 64000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/perplexity.ts b/src/lib/models/manifests/perplexity.ts new file mode 100644 index 000000000..b2abed066 --- /dev/null +++ b/src/lib/models/manifests/perplexity.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: perplexity has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const perplexityManifest: ProviderModelManifest = { + defaultContextWindow: 127000, + models: { + _default: { + aliases: [], + contextWindow: 127000, + maxOutputTokens: 64000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/recraft.ts b/src/lib/models/manifests/recraft.ts new file mode 100644 index 000000000..b3f0d2849 --- /dev/null +++ b/src/lib/models/manifests/recraft.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: recraft has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const recraftManifest: ProviderModelManifest = { + defaultContextWindow: 2000, + models: { + _default: { + aliases: [], + contextWindow: 2000, + maxOutputTokens: 2000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/replicate.ts b/src/lib/models/manifests/replicate.ts new file mode 100644 index 000000000..d2b78ab50 --- /dev/null +++ b/src/lib/models/manifests/replicate.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: replicate has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const replicateManifest: ProviderModelManifest = { + defaultContextWindow: 32768, + models: { + _default: { + aliases: [], + contextWindow: 32768, + maxOutputTokens: 32768, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/sagemaker.ts b/src/lib/models/manifests/sagemaker.ts new file mode 100644 index 000000000..e9e0a123a --- /dev/null +++ b/src/lib/models/manifests/sagemaker.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: sagemaker has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const sagemakerManifest: ProviderModelManifest = { + defaultContextWindow: 128000, + models: { + _default: { + aliases: [], + contextWindow: 128000, + maxOutputTokens: 64000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/stability.ts b/src/lib/models/manifests/stability.ts new file mode 100644 index 000000000..2521fd665 --- /dev/null +++ b/src/lib/models/manifests/stability.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: stability has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const stabilityManifest: ProviderModelManifest = { + defaultContextWindow: 2000, + models: { + _default: { + aliases: [], + contextWindow: 2000, + maxOutputTokens: 2000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/together-ai.ts b/src/lib/models/manifests/together-ai.ts new file mode 100644 index 000000000..ede33d6b9 --- /dev/null +++ b/src/lib/models/manifests/together-ai.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: together-ai has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const togetherAiManifest: ProviderModelManifest = { + defaultContextWindow: 128000, + models: { + _default: { + aliases: [], + contextWindow: 128000, + maxOutputTokens: 64000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/vertex.ts b/src/lib/models/manifests/vertex.ts new file mode 100644 index 000000000..f1545d574 --- /dev/null +++ b/src/lib/models/manifests/vertex.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: vertex has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const vertexManifest: ProviderModelManifest = { + defaultContextWindow: 1048576, + models: { + _default: { + aliases: [], + contextWindow: 1048576, + maxOutputTokens: 64000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/voyage.ts b/src/lib/models/manifests/voyage.ts new file mode 100644 index 000000000..4a0845b9c --- /dev/null +++ b/src/lib/models/manifests/voyage.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: voyage has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const voyageManifest: ProviderModelManifest = { + defaultContextWindow: 32000, + models: { + _default: { + aliases: [], + contextWindow: 32000, + maxOutputTokens: 32000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/models/manifests/xai.ts b/src/lib/models/manifests/xai.ts new file mode 100644 index 000000000..5b1b51822 --- /dev/null +++ b/src/lib/models/manifests/xai.ts @@ -0,0 +1,20 @@ +import type { ProviderModelManifest } from "../../types/index.js"; + +/** + * Minimal manifest: xai has no MODEL_REGISTRY entries today, so + * only the provider-wide fallback is known. Named models can be added here + * incrementally without touching any consumer — see Task 5 of the model + * metadata consolidation plan. + */ +export const xaiManifest: ProviderModelManifest = { + defaultContextWindow: 131072, + models: { + _default: { + aliases: [], + contextWindow: 131072, + maxOutputTokens: 64000, + vision: false, + functionCalling: false, + }, + }, +}; diff --git a/src/lib/types/model.ts b/src/lib/types/model.ts index 3772d0e6e..7fa6c0486 100644 --- a/src/lib/types/model.ts +++ b/src/lib/types/model.ts @@ -268,3 +268,83 @@ export type ModelRoutingOptions = { /** Fallback strategy if primary choice fails */ fallbackStrategy?: "fast" | "reasoning" | "auto"; }; + +/** + * A single model's metadata inside a provider's manifest. This is the one + * canonical shape every model-metadata consumer (context windows, pricing, + * MODEL_REGISTRY, vision capability, output-token ceilings) is intended to + * migrate onto — this PR is purely additive and does not yet move any + * consumer over. + * + * `pricingPerMTok` is optional by design: a model with no verified price + * (e.g. a just-announced model pricing.ts hasn't priced yet) must not report + * a fabricated rate. Absence here means "unknown", not "free" — callers that + * need to distinguish "free" from "unknown" already have `hasPricing()` + * (src/lib/utils/pricing.ts) for that. + */ +export type ProviderModelManifestEntry = { + /** Alternate identifiers that resolve to this canonical model id. */ + aliases: string[]; + /** Human-readable name. Falls back to a mechanical id-derived name when absent. */ + displayName?: string; + contextWindow: number; + maxOutputTokens: number; + pricingPerMTok?: { + input: number; + output: number; + cacheRead?: number; + cacheWrite?: number; + }; + vision: boolean; + nativeAudio?: boolean; + functionCalling: boolean; + reasoning?: boolean; + jsonMode?: boolean; + /** + * Whether the model accepts classic sampling parameters (temperature/topP). + * Mirrors ModelCapabilities.samplingParams (src/lib/types/model.ts:131) — + * unset means supported. + */ + samplingParams?: boolean; + /** + * Hand-tuned ModelInfo.performance/useCases/category values, carried + * forward verbatim for the ids that already had a MODEL_REGISTRY entry + * before this migration. Absent for every id that never had one — those + * get performance/useCases/category derived mechanically instead (see + * Task 9's buildModelRegistryFromManifests). Never populate this for a + * genuinely new model: mechanical derivation is the correct default, and + * a fabricated "curated" value would be worse than an honestly-derived one. + */ + curated?: { + performance?: ModelPerformance; + useCases?: UseCaseSuitability; + category?: ModelInfo["category"]; + }; +}; + +/** + * A regex-driven patch applied to an unlisted, gateway-shaped model id that + * matches `pattern` (e.g. "vertex_ai/claude-sonnet-5@20260203"). Generalizes + * the pattern VISION_FAMILY_RULES (src/lib/adapters/providerImageAdapter.ts) + * and SAMPLING_PARAM_REJECTING_FAMILIES (src/lib/models/modelRegistry.ts) + * already use independently, keyed per-provider instead of globally. + */ +export type ManifestFamilyRule = { + pattern: RegExp; + patch: Partial; +}; + +/** + * One provider's complete model manifest: every model NeuroLink knows about + * for that provider, plus the provider-wide fallback used when a caller + * passes a model id the manifest has never seen (a symbolic/local provider + * model, or a brand-new release the manifest hasn't been updated for yet). + */ +export type ProviderModelManifest = { + /** Used for `_default`-key lookups and providers with no named-model list. */ + defaultContextWindow: number; + /** Applied, in order, to the resolved entry (see manifestRegistry.ts). */ + familyRules?: ManifestFamilyRule[]; + /** Keyed by canonical model id (the same id `ModelInfo.id` / AIProvider calls use). */ + models: Record; +}; diff --git a/test/continuous-test-suite-model-manifests.ts b/test/continuous-test-suite-model-manifests.ts new file mode 100644 index 000000000..79885b52b --- /dev/null +++ b/test/continuous-test-suite-model-manifests.ts @@ -0,0 +1,198 @@ +#!/usr/bin/env tsx +import "dotenv/config"; + +/** + * Continuous Test Suite — Model Manifests + * + * Zero-API structural checks over MANIFEST_REGISTRY (src/lib/models/manifestRegistry.ts) + * and its resolver, resolveManifestEntry/resolveManifestEntryExact. Covers the + * resolution cascade (exact id -> alias -> prefix -> family rule -> default) + * and the output-ceiling invariant (maxOutputTokens must never exceed the + * relevant contextWindow) across every registered provider manifest. + * + * ## Why this reaches into `dist/lib/` directly (CLAUDE.md rule 15) + * + * manifestRegistry.ts has no consumer yet, by design: this PR (model-metadata + * consolidation, Task 5) is purely additive and deliberately does not migrate + * any existing code onto the manifest. Confirmed empty blast radius — + * `dist/index.js` contains no reference to manifestRegistry, and no file + * under `src/lib/` outside `models/manifestRegistry.ts` / + * `models/manifests/*` calls `resolveManifestEntry`, `resolveManifestEntryExact`, + * or reads `MANIFEST_REGISTRY`. There is therefore no `generate()`/`stream()`/ + * CLI path that reaches this resolver at all today — the alias-resolution + * bug this suite was written to catch (a bare model name silently losing its + * real contextWindow/vision/functionCalling data to the provider default) is + * unreachable from any public surface until a future PR wires a consumer + * onto the manifest. Once a consumer migrates, this suite should be converted + * (or retired in favor of assertions on that consumer's public output) per + * rule 15's own guidance that internals reachable only from the inside need + * a public surface, not a unit test. + * + * This is not the rule 15 `allow`-list exception (deterministic control a + * live call can't give) — it is the same "drive a specific compiled dist/lib + * module that isn't re-exported from dist/index.js" pattern already used by + * continuous-test-suite.ts (AccountPool, ModelRouter, the cloaking plugins), + * -credentials.ts (ProviderFactory/ProviderRegistry), -provider-structure.ts + * (providerRegistry.js) and others, none of which are on the `allow` list. + * Every import below resolves under `../dist/lib/...` — the compiled + * artifact, not raw TypeScript source — so it stays a single module graph + * per rule 15's "one module graph per suite" mandate, it just isn't the + * top-level public one. + * + * Run: npx tsx test/continuous-test-suite-model-manifests.ts + * pnpm run test:model-manifests + */ + +import { assert, defineSuite } from "./helpers/harness.js"; +import { assertDistFresh } from "./helpers/distFreshness.js"; + +// Fail loudly rather than silently testing a stale build. +assertDistFresh(); + +const { test, runSuite } = defineSuite("Model Manifests"); + +const { + resolveManifestEntry, + resolveManifestEntryExact, + getManifestForProvider, + getAllManifestProviders, +} = await import("../dist/lib/models/manifestRegistry.js"); + +await test("Alias resolves to its canonical entry, not the default", async () => { + // ollama's "llama3.2:latest" entry declares "llama3.2" as a bare-name + // alias. Before the resolver consulted aliases, this bare name missed + // both the exact-match and prefix-match checks (the request string is + // shorter than the tagged key it would need to be a prefix of) and fell + // through to the synthesized `_default` entry, silently discarding the + // model-specific contextWindow/vision/functionCalling data. + const viaAlias = resolveManifestEntry("ollama", "llama3.2"); + const canonical = resolveManifestEntry("ollama", "llama3.2:latest"); + assert(!!viaAlias, "alias lookup returned nothing"); + assert(!!canonical, "canonical lookup returned nothing"); + assert( + JSON.stringify(viaAlias) === JSON.stringify(canonical), + "alias entry does not match its canonical entry", + ); + assert( + viaAlias.contextWindow === 131072, + "alias resolution fell through to the provider default instead of the canonical entry", + ); +}); + +await test("Alias resolution also works via resolveManifestEntryExact", async () => { + // anthropic's "claude-sonnet-5" entry declares "sonnet-5" as an alias. + const viaAlias = resolveManifestEntryExact("anthropic", "sonnet-5"); + assert(!!viaAlias, "exact-path alias lookup returned nothing"); + assert( + viaAlias.contextWindow === 1_000_000, + "exact-path alias resolution did not reach the canonical entry", + ); +}); + +await test("Exact and prefix resolution still take precedence over alias/default", async () => { + const exact = resolveManifestEntryExact("anthropic", "claude-sonnet-5"); + assert(!!exact, "exact id lookup returned nothing"); + assert(exact.contextWindow === 1_000_000, "exact id resolved wrong entry"); + + // A gateway-shaped id ("") not itself a key or + // alias should still resolve via longest-prefix match. + const prefixed = resolveManifestEntryExact( + "anthropic", + "claude-sonnet-5-some-gateway-suffix", + ); + assert(!!prefixed, "prefix lookup returned nothing"); + assert( + prefixed.contextWindow === 1_000_000, + "prefix lookup resolved the wrong entry", + ); +}); + +await test("Unknown model falls back to an honest provider default", async () => { + const fallback = resolveManifestEntry("bedrock", "totally-unknown-model-id"); + assert(!!fallback, "default fallback returned nothing"); + const manifest = getManifestForProvider("bedrock"); + assert(!!manifest, "bedrock manifest missing from registry"); + assert( + fallback.maxOutputTokens <= manifest!.defaultContextWindow, + "synthesized default advertises an output ceiling above the context window", + ); +}); + +await test("Every manifest's synthesized/declared default keeps maxOutputTokens <= contextWindow", async () => { + const providers = getAllManifestProviders(); + assert(providers.length > 0, "no manifest providers registered"); + + const failures: string[] = []; + for (const provider of providers) { + const manifest = getManifestForProvider(provider); + if (!manifest) { + failures.push(`${provider}: manifest missing`); + continue; + } + const resolved = resolveManifestEntry(provider, "__unknown_model_probe__"); + if (!resolved) { + failures.push(`${provider}: default resolution returned nothing`); + continue; + } + if (resolved.maxOutputTokens > resolved.contextWindow) { + failures.push( + `${provider}: default maxOutputTokens (${resolved.maxOutputTokens}) exceeds contextWindow (${resolved.contextWindow})`, + ); + } + } + assert( + failures.length === 0, + `${failures.length} provider(s) with an inverted default output ceiling: ${failures.join("; ")}`, + ); +}); + +await test("Every declared model entry keeps maxOutputTokens <= contextWindow", async () => { + const providers = getAllManifestProviders(); + const failures: string[] = []; + for (const provider of providers) { + const manifest = getManifestForProvider(provider); + if (!manifest) { + continue; + } + for (const [modelId, entry] of Object.entries(manifest.models)) { + if (entry.maxOutputTokens > entry.contextWindow) { + failures.push(`${provider}/${modelId}`); + } + } + } + assert( + failures.length === 0, + `${failures.length} declared entr(ies) with maxOutputTokens > contextWindow: ${failures.join(", ")}`, + ); +}); + +await test("Every alias points at a real key in its own manifest, never dangling", async () => { + // Not a correctness requirement of the resolver (an alias only needs to + // be a string the caller might pass in), but a data-hygiene check: an + // alias that collides with another model's own canonical id would shadow + // that model behind the wrong entry. + const providers = getAllManifestProviders(); + const failures: string[] = []; + for (const provider of providers) { + const manifest = getManifestForProvider(provider); + if (!manifest) { + continue; + } + const modelIds = new Set(Object.keys(manifest.models)); + for (const [modelId, entry] of Object.entries(manifest.models)) { + for (const alias of entry.aliases) { + if (modelIds.has(alias) && alias !== modelId) { + failures.push( + `${provider}: alias "${alias}" on ${modelId} shadows another model's canonical id`, + ); + } + } + } + } + assert( + failures.length === 0, + `${failures.length} alias/canonical-id collision(s) found`, + ); +}); + +await runSuite();