From 443b90a2674d401f89c64264592ee56738689456 Mon Sep 17 00:00:00 2001 From: Sachin Sharma Date: Fri, 12 Dec 2025 12:55:48 +0530 Subject: [PATCH] feat(models): add GPT-5.2 and comprehensive model updates across all providers Add OpenAI GPT-5.2 series (released Dec 11, 2025): - GPT-5.2 (Thinking) - deep reasoning with 100% AIME 2025 - GPT-5.2 Chat Latest (Instant) - fast everyday model - GPT-5.2 Pro - highest quality for science/math (92.4% GPQA Diamond) Updates include: - New enum values in constants/enums.ts - Full model registry entries with pricing, capabilities, and limits - Vision capabilities for all GPT-5.2 variants - Updated USE_CASE_RECOMMENDATIONS with GPT-5.2 at top --- docs/features/index.md | 34 +- docs/features/office-documents.md | 133 +- docs/sdk/api-reference.md | 22 +- src/lib/adapters/providerImageAdapter.ts | 145 +- src/lib/constants/enums.ts | 596 +++++- src/lib/factories/providerRegistry.ts | 19 +- src/lib/models/modelRegistry.ts | 2236 ++++++++++++++++++++-- test/unit/cli/video-flags.test.ts | 16 +- 8 files changed, 2857 insertions(+), 344 deletions(-) diff --git a/docs/features/index.md b/docs/features/index.md index 523942069..9369b398a 100644 --- a/docs/features/index.md +++ b/docs/features/index.md @@ -25,30 +25,30 @@ Comprehensive guides for all NeuroLink features organized by category. Each guid ## Core Features (Q3 2025) -| Feature | Description | -| ------------------------------------------------------------------------------ | ----------------------------------------------------------------------------------------------- | -| :material-image-text: **[Multimodal Chat Experiences](multimodal-chat.md)** | Stream text and images together with automatic provider fallbacks and format conversion. | -| :material-table-large: **[CSV File Support](csv-support.md)** | Process CSV files for data analysis with automatic format conversion. Works with all providers. | -| :material-file-pdf-box: **[PDF File Support](pdf-support.md)** | Process PDF documents for visual analysis and content extraction. Native provider support. | +| Feature | Description | +| ------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------ | +| :material-image-text: **[Multimodal Chat Experiences](multimodal-chat.md)** | Stream text and images together with automatic provider fallbacks and format conversion. | +| :material-table-large: **[CSV File Support](csv-support.md)** | Process CSV files for data analysis with automatic format conversion. Works with all providers. | +| :material-file-pdf-box: **[PDF File Support](pdf-support.md)** | Process PDF documents for visual analysis and content extraction. Native provider support. | | :material-file-word: **[Office Documents](office-documents.md)** | Process DOCX, PPTX, XLSX files for document analysis. Native Bedrock, Vertex, Anthropic support. | -| :material-chart-line: **[Auto Evaluation Engine](auto-evaluation.md)** | Automated quality scoring and metrics export for AI response validation using LLM-as-judge. | -| :material-console: **[CLI Loop Sessions](cli-loop-sessions.md)** | Persistent interactive mode with conversation memory and session state for prompt engineering. | -| :material-earth: **[Regional Streaming Controls](regional-streaming.md)** | Region-specific model deployment and routing for compliance and latency optimization. | -| :material-brain: **[Provider Orchestration Brain](provider-orchestration.md)** | Adaptive provider and model selection with intelligent fallbacks based on task classification. | +| :material-chart-line: **[Auto Evaluation Engine](auto-evaluation.md)** | Automated quality scoring and metrics export for AI response validation using LLM-as-judge. | +| :material-console: **[CLI Loop Sessions](cli-loop-sessions.md)** | Persistent interactive mode with conversation memory and session state for prompt engineering. | +| :material-earth: **[Regional Streaming Controls](regional-streaming.md)** | Region-specific model deployment and routing for compliance and latency optimization. | +| :material-brain: **[Provider Orchestration Brain](provider-orchestration.md)** | Adaptive provider and model selection with intelligent fallbacks based on task classification. | --- ## Platform Capabilities at a Glance -| Category | Features | Documentation | -| ------------------------ | ------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------------------------------------------------------------------- | -| **Provider unification** | 12+ providers with automatic failover, cost-aware routing, provider orchestration (Q3) | [Provider Setup](../getting-started/provider-setup.md) | +| Category | Features | Documentation | +| ------------------------ | ------------------------------------------------------------------------------------------------------------------ | ---------------------------------------------------------------------------------------------------------------------------------------- | +| **Provider unification** | 12+ providers with automatic failover, cost-aware routing, provider orchestration (Q3) | [Provider Setup](../getting-started/provider-setup.md) | | **Multimodal pipeline** | Stream images + CSV data + PDF documents + Office files across providers with auto-detection for mixed file types. | [Multimodal Guide](multimodal-chat.md), [CSV Support](csv-support.md), [PDF Support](pdf-support.md), [Office Docs](office-documents.md) | -| **Quality & governance** | Auto-evaluation engine (Q3), guardrails middleware (Q4), HITL workflows (Q4), audit logging | [Auto Evaluation](auto-evaluation.md), [Guardrails](guardrails.md), [HITL](hitl.md) | -| **Memory & context** | Conversation memory, Mem0 integration, Redis history export (Q4), context summarization (Q4) | [Conversation Memory](../CONVERSATION-MEMORY.md), [Redis Export](conversation-history.md) | -| **CLI tooling** | Loop sessions (Q3), setup wizard, config validation, Redis auto-detect, JSON output | [CLI Loop](cli-loop-sessions.md), [CLI Commands](../cli/commands.md) | -| **Enterprise ops** | Proxy support, regional routing (Q3), telemetry hooks, configuration management | [Enterprise Proxy](../ENTERPRISE-PROXY-SETUP.md), [Telemetry](../TELEMETRY-GUIDE.md) | -| **Tool ecosystem** | MCP auto discovery, LiteLLM hub access, SageMaker custom deployment, web search | [MCP Integration](../advanced/mcp-integration.md), [MCP Catalog](../guides/mcp/server-catalog.md) | +| **Quality & governance** | Auto-evaluation engine (Q3), guardrails middleware (Q4), HITL workflows (Q4), audit logging | [Auto Evaluation](auto-evaluation.md), [Guardrails](guardrails.md), [HITL](hitl.md) | +| **Memory & context** | Conversation memory, Mem0 integration, Redis history export (Q4), context summarization (Q4) | [Conversation Memory](../CONVERSATION-MEMORY.md), [Redis Export](conversation-history.md) | +| **CLI tooling** | Loop sessions (Q3), setup wizard, config validation, Redis auto-detect, JSON output | [CLI Loop](cli-loop-sessions.md), [CLI Commands](../cli/commands.md) | +| **Enterprise ops** | Proxy support, regional routing (Q3), telemetry hooks, configuration management | [Enterprise Proxy](../ENTERPRISE-PROXY-SETUP.md), [Telemetry](../TELEMETRY-GUIDE.md) | +| **Tool ecosystem** | MCP auto discovery, LiteLLM hub access, SageMaker custom deployment, web search | [MCP Integration](../advanced/mcp-integration.md), [MCP Catalog](../guides/mcp/server-catalog.md) | --- diff --git a/docs/features/office-documents.md b/docs/features/office-documents.md index d97e34e98..7ff5cf242 100644 --- a/docs/features/office-documents.md +++ b/docs/features/office-documents.md @@ -16,18 +16,18 @@ Office document support in NeuroLink works as a native multimodal input - the sy ## Supported File Types -| Format | Extension | MIME Type | Description | -|--------|-----------|-----------|-------------| -| **Word Document** | `.docx` | `application/vnd.openxmlformats-officedocument.wordprocessingml.document` | Microsoft Word documents with text, images, tables | -| **PowerPoint** | `.pptx` | `application/vnd.openxmlformats-officedocument.presentationml.presentation` | Presentations with slides, charts, images | -| **Excel Spreadsheet** | `.xlsx` | `application/vnd.openxmlformats-officedocument.spreadsheetml.sheet` | Spreadsheets with data, formulas, charts | +| Format | Extension | MIME Type | Description | +| --------------------- | --------- | --------------------------------------------------------------------------- | -------------------------------------------------- | +| **Word Document** | `.docx` | `application/vnd.openxmlformats-officedocument.wordprocessingml.document` | Microsoft Word documents with text, images, tables | +| **PowerPoint** | `.pptx` | `application/vnd.openxmlformats-officedocument.presentationml.presentation` | Presentations with slides, charts, images | +| **Excel Spreadsheet** | `.xlsx` | `application/vnd.openxmlformats-officedocument.spreadsheetml.sheet` | Spreadsheets with data, formulas, charts | **Legacy Formats:** -| Format | Extension | MIME Type | Support | -|--------|-----------|-----------|---------| -| Word (Legacy) | `.doc` | `application/msword` | Provider-dependent | -| Excel (Legacy) | `.xls` | `application/vnd.ms-excel` | Provider-dependent | +| Format | Extension | MIME Type | Support | +| -------------- | --------- | -------------------------- | ------------------ | +| Word (Legacy) | `.doc` | `application/msword` | Provider-dependent | +| Excel (Legacy) | `.xls` | `application/vnd.ms-excel` | Provider-dependent | ## Quick Start @@ -132,11 +132,11 @@ neurolink batch prompts.txt --office meeting-notes.docx --provider bedrock type GenerateOptions = { input: { text: string; - images?: Array; // Image files - csvFiles?: Array; // CSV files (converted to text) - pdfFiles?: Array; // PDF files (native binary) + images?: Array; // Image files + csvFiles?: Array; // CSV files (converted to text) + pdfFiles?: Array; // PDF files (native binary) officeFiles?: Array; // Office files (native binary) - files?: Array; // Auto-detect file types + files?: Array; // Auto-detect file types }; // Provider selection (REQUIRED for Office files) @@ -218,11 +218,11 @@ officeFiles: ["report.docx", docxBuffer, "./presentation.pptx"]; ### Supported Providers -| Provider | Max Size | DOCX | PPTX | XLSX | DOC | XLS | Notes | -|----------|----------|------|------|------|-----|-----|-------| -| **AWS Bedrock** | 5 MB | ✅ | ✅ | ✅ | ✅ | ✅ | Full native support via Converse API | -| **Google Vertex AI** | 5 MB | ✅ | ⚠️ | ✅ | ⚠️ | ⚠️ | Best for DOCX and XLSX | -| **Anthropic Claude** | 5 MB | ✅ | ⚠️ | ✅ | ⚠️ | ⚠️ | Via document API | +| Provider | Max Size | DOCX | PPTX | XLSX | DOC | XLS | Notes | +| -------------------- | -------- | ---- | ---- | ---- | --- | --- | ------------------------------------ | +| **AWS Bedrock** | 5 MB | ✅ | ✅ | ✅ | ✅ | ✅ | Full native support via Converse API | +| **Google Vertex AI** | 5 MB | ✅ | ⚠️ | ✅ | ⚠️ | ⚠️ | Best for DOCX and XLSX | +| **Anthropic Claude** | 5 MB | ✅ | ⚠️ | ✅ | ⚠️ | ⚠️ | Via document API | ### Unsupported Providers @@ -267,6 +267,7 @@ await neurolink.generate({ ``` **Supported Document Formats in Bedrock Converse API:** + - Office formats: `doc`, `docx`, `xls`, `xlsx` - Other formats: `pdf`, `csv`, `html`, `txt`, `md` @@ -308,11 +309,11 @@ await neurolink.generate({ input: { text: "Analyze all these documents", files: [ - "report.docx", // Auto-detected as Word document - "data.xlsx", // Auto-detected as Excel spreadsheet - "slides.pptx", // Auto-detected as PowerPoint - "summary.pdf", // Auto-detected as PDF - "chart.png", // Auto-detected as image + "report.docx", // Auto-detected as Word document + "data.xlsx", // Auto-detected as Excel spreadsheet + "slides.pptx", // Auto-detected as PowerPoint + "summary.pdf", // Auto-detected as PDF + "chart.png", // Auto-detected as image ], }, provider: "bedrock", @@ -467,11 +468,15 @@ try { }); } catch (error) { if (error instanceof OfficeSizeError) { - console.error(`File too large: ${error.actualSize}MB (max: ${error.maxSize}MB)`); + console.error( + `File too large: ${error.actualSize}MB (max: ${error.maxSize}MB)`, + ); console.error("Try: --provider google-ai-studio for larger files"); } else if (error instanceof OfficeProviderError) { console.error(`Provider ${error.provider} doesn't support Office files`); - console.error(`Supported providers: ${error.supportedProviders.join(", ")}`); + console.error( + `Supported providers: ${error.supportedProviders.join(", ")}`, + ); } else if (error instanceof OfficeValidationError) { console.error(`Invalid Office file: ${error.message}`); console.error(`Validation type: ${error.validationType}`); @@ -487,16 +492,16 @@ try { When processing Office documents, the following metadata is available: -| Field | Type | Description | -|-------|------|-------------| -| `confidence` | `number` | Detection confidence (0-100) | -| `size` | `number` | File size in bytes | -| `filename` | `string` | Original filename | -| `format` | `OfficeFileType` | Detected Office format | -| `provider` | `string` | Provider used for processing | -| `estimatedPages` | `number` | Estimated page/slide/sheet count | -| `hasEmbeddedImages` | `boolean` | Whether document contains images | -| `hasCharts` | `boolean` | Whether document contains charts | +| Field | Type | Description | +| ------------------- | ---------------- | -------------------------------- | +| `confidence` | `number` | Detection confidence (0-100) | +| `size` | `number` | File size in bytes | +| `filename` | `string` | Original filename | +| `format` | `OfficeFileType` | Detected Office format | +| `provider` | `string` | Provider used for processing | +| `estimatedPages` | `number` | Estimated page/slide/sheet count | +| `hasEmbeddedImages` | `boolean` | Whether document contains images | +| `hasCharts` | `boolean` | Whether document contains charts | ### Accessing Metadata @@ -526,13 +531,13 @@ console.log(result.metadata?.officeFiles?.[0]); ```typescript // For comprehensive Office support -provider: "bedrock"; // Best overall Office document support +provider: "bedrock"; // Best overall Office document support // For Word documents primarily -provider: "vertex"; // Good DOCX support +provider: "vertex"; // Good DOCX support // For enterprise deployments -provider: "bedrock"; // AWS infrastructure integration +provider: "bedrock"; // AWS infrastructure integration ``` ### 2. Optimize File Size @@ -544,19 +549,19 @@ import { stat } from "fs/promises"; async function validateOfficeFile(filePath: string, provider: string) { const stats = await stat(filePath); const sizeMB = stats.size / (1024 * 1024); - + const limits: Record = { bedrock: 5, vertex: 5, anthropic: 5, }; - + if (sizeMB > (limits[provider] || 5)) { throw new Error( - `File ${filePath} (${sizeMB.toFixed(2)}MB) exceeds ${limits[provider]}MB limit for ${provider}` + `File ${filePath} (${sizeMB.toFixed(2)}MB) exceeds ${limits[provider]}MB limit for ${provider}`, ); } - + console.log(`✓ File validated: ${sizeMB.toFixed(2)}MB`); } @@ -605,13 +610,13 @@ for await (const chunk of stream) { ### Provider Limitations -| Limitation | Description | Workaround | -|------------|-------------|------------| -| Size limits | Most providers limit to 5MB | Split large documents or convert to PDF | -| Password protection | Not supported | Remove password before processing | -| Macros | VBA macros are ignored | N/A - security feature | -| External links | May not be resolved | Embed content instead | -| Complex formatting | Some formatting may be lost | Focus on content extraction | +| Limitation | Description | Workaround | +| ------------------- | --------------------------- | --------------------------------------- | +| Size limits | Most providers limit to 5MB | Split large documents or convert to PDF | +| Password protection | Not supported | Remove password before processing | +| Macros | VBA macros are ignored | N/A - security feature | +| External links | May not be resolved | Embed content instead | +| Complex formatting | Some formatting may be lost | Focus on content extraction | ### Token Usage @@ -633,7 +638,7 @@ await neurolink.generate({ officeFiles: ["document.docx"], }, provider: "bedrock", - maxTokens: 4000, // Allow enough tokens for response + maxTokens: 4000, // Allow enough tokens for response }); ``` @@ -644,6 +649,7 @@ await neurolink.generate({ **Problem:** Using unsupported provider (OpenAI, Ollama, etc.) **Solution:** + ```bash # Change provider to supported one neurolink generate "Analyze document" --office doc.docx --provider bedrock @@ -657,6 +663,7 @@ neurolink generate "Analyze document" --file doc.docx --provider vertex **Problem:** File too large for provider (>5MB for most providers) **Solution:** + ```bash # Option 1: Split the document into smaller parts # Option 2: Convert to PDF first (may have larger size limits) @@ -668,6 +675,7 @@ neurolink generate "Analyze document" --file doc.docx --provider vertex **Problem:** File is not a valid Office Open XML format or corrupted **Solution:** + ```bash # Verify file is valid Office format file document.docx # Should show "Microsoft Word 2007+" @@ -681,6 +689,7 @@ file document.docx # Should show "Microsoft Word 2007+" **Problem:** No provider selected (Office files require explicit provider) **Solution:** + ```typescript // ❌ Missing provider await neurolink.generate({ @@ -696,7 +705,7 @@ await neurolink.generate({ text: "Analyze", officeFiles: ["doc.docx"], }, - provider: "bedrock", // Required for Office files + provider: "bedrock", // Required for Office files }); ``` @@ -705,23 +714,25 @@ await neurolink.generate({ **Problem:** AI says "I cannot read the document" even though file is attached **Common Causes:** + 1. **Wrong provider**: Make sure using supported provider 2. **File path wrong**: Verify file exists at specified path 3. **Buffer issue**: If using Buffer, ensure it's valid Office data 4. **Format mismatch**: Ensure file extension matches actual format **Debug:** + ```typescript import { readFile, stat } from "fs/promises"; // Verify file exists -await stat("document.docx"); // Throws if not found +await stat("document.docx"); // Throws if not found // Verify it's a valid Office file (DOCX is ZIP-based) const buffer = await readFile("document.docx"); const header = buffer.slice(0, 4); // DOCX files start with ZIP magic bytes: PK\x03\x04 -console.log("Magic bytes:", header.toString("hex")); // Should be "504b0304" +console.log("Magic bytes:", header.toString("hex")); // Should be "504b0304" // Check size const sizeMB = buffer.length / (1024 * 1024); @@ -735,6 +746,7 @@ console.log("Size:", sizeMB.toFixed(2), "MB"); If you were previously using manual document extraction: **Before (Manual Processing):** + ```typescript // Old approach: Extract text manually import { readFileSync } from "fs"; @@ -749,6 +761,7 @@ const result = await provider.generate({ ``` **After (Native Support):** + ```typescript // New approach: Direct document support const result = await neurolink.generate({ @@ -765,6 +778,7 @@ const result = await neurolink.generate({ If you were converting Office files to PDF first: **Before (PDF Conversion):** + ```typescript // Old approach: Convert to PDF first import { convertToPdf } from "some-pdf-library"; @@ -781,12 +795,13 @@ const result = await neurolink.generate({ ``` **After (Direct Office Support):** + ```typescript // New approach: Direct Office document support const result = await neurolink.generate({ input: { text: "Analyze this document", - officeFiles: ["report.docx"], // No conversion needed + officeFiles: ["report.docx"], // No conversion needed }, provider: "bedrock", }); @@ -794,12 +809,12 @@ const result = await neurolink.generate({ ### API Changes Summary -| Previous API | New API | Notes | -|--------------|---------|-------| -| Manual text extraction | `officeFiles: [...]` | Native document support | -| PDF conversion workflow | Direct Office support | No conversion needed | +| Previous API | New API | Notes | +| -------------------------- | --------------------------- | -------------------------------- | +| Manual text extraction | `officeFiles: [...]` | Native document support | +| PDF conversion workflow | Direct Office support | No conversion needed | | Provider-specific handling | Unified `officeFiles` array | Works across supported providers | -| Custom MIME type handling | Auto-detection | Format automatically detected | +| Custom MIME type handling | Auto-detection | Format automatically detected | ## Usage Examples diff --git a/docs/sdk/api-reference.md b/docs/sdk/api-reference.md index e17768984..75497aa97 100644 --- a/docs/sdk/api-reference.md +++ b/docs/sdk/api-reference.md @@ -593,7 +593,7 @@ interface GenerateOptions { toolUsageContext?: string; context?: Record; conversationHistory?: Array<{ role: string; content: string }>; - + // Document processing options officeOptions?: OfficeProcessorOptions; } @@ -1935,13 +1935,13 @@ type FileType = "csv" | "image" | "pdf" | "office" | "text" | "unknown"; interface OfficeProcessorOptions { /** Provider to use for document processing */ provider?: string; - + /** Maximum file size in MB (default: 5) */ maxSizeMB?: number; - + /** Whether to extract embedded images */ extractImages?: boolean; - + /** Whether to preserve document structure in output */ preserveStructure?: boolean; } @@ -1978,13 +1978,13 @@ interface OfficeProviderConfig { **Office Document Provider Support:** -| Provider | DOCX | PPTX | XLSX | DOC | XLS | Notes | -|----------|------|------|------|-----|-----|-------| -| **AWS Bedrock** | ✅ | ✅ | ✅ | ✅ | ✅ | Full native support via Converse API | -| **Google Vertex AI** | ✅ | ⚠️ | ✅ | ⚠️ | ⚠️ | Best for DOCX and XLSX | -| **Anthropic Claude** | ✅ | ⚠️ | ✅ | ⚠️ | ⚠️ | Via document API | -| **OpenAI** | ❌ | ❌ | ❌ | ❌ | ❌ | Not supported | -| **Azure OpenAI** | ❌ | ❌ | ❌ | ❌ | ❌ | Not supported | +| Provider | DOCX | PPTX | XLSX | DOC | XLS | Notes | +| -------------------- | ---- | ---- | ---- | --- | --- | ------------------------------------ | +| **AWS Bedrock** | ✅ | ✅ | ✅ | ✅ | ✅ | Full native support via Converse API | +| **Google Vertex AI** | ✅ | ⚠️ | ✅ | ⚠️ | ⚠️ | Best for DOCX and XLSX | +| **Anthropic Claude** | ✅ | ⚠️ | ✅ | ⚠️ | ⚠️ | Via document API | +| **OpenAI** | ❌ | ❌ | ❌ | ❌ | ❌ | Not supported | +| **Azure OpenAI** | ❌ | ❌ | ❌ | ❌ | ❌ | Not supported | **Example Usage:** diff --git a/src/lib/adapters/providerImageAdapter.ts b/src/lib/adapters/providerImageAdapter.ts index f286914d9..8d77ed492 100644 --- a/src/lib/adapters/providerImageAdapter.ts +++ b/src/lib/adapters/providerImageAdapter.ts @@ -25,6 +25,10 @@ export class MultimodalLogger { */ const VISION_CAPABILITIES = { openai: [ + // GPT-5.2 family (released Dec 11, 2025) - Latest flagship models + "gpt-5.2", + "gpt-5.2-chat-latest", + "gpt-5.2-pro", // GPT-5 family (released Aug 2025) "gpt-5", "gpt-5-2025-08-07", @@ -38,6 +42,7 @@ const VISION_CAPABILITIES = { // o-series reasoning models (released Apr 2025) "o3", "o3-mini", + "o3-pro", "o4", "o4-mini", "o4-mini-deep-research", @@ -52,12 +57,17 @@ const VISION_CAPABILITIES = { "gemini-3-pro-preview", "gemini-3-pro-preview-11-2025", "gemini-3-pro-latest", + "gemini-3-pro-image-preview", // Gemini 2.5 Series "gemini-2.5-pro", "gemini-2.5-flash", + "gemini-2.5-flash-lite", + "gemini-2.5-flash-image", // Gemini 2.0 Series "gemini-2.0-flash", + "gemini-2.0-flash-001", "gemini-2.0-flash-lite", + "gemini-2.0-flash-preview-image-generation", // Gemini 1.5 Series (Legacy) "gemini-1.5-pro", "gemini-1.5-flash", @@ -71,24 +81,44 @@ const VISION_CAPABILITIES = { "claude-opus-4-5-20251101", "claude-haiku-4-5", "claude-haiku-4-5-20251001", + // Claude 4.1 and 4.0 Series + "claude-opus-4-1", + "claude-opus-4-1-20250805", + "claude-opus-4", + "claude-opus-4-20250514", + "claude-sonnet-4", + "claude-sonnet-4-20250514", // Claude 3.7 Series "claude-3-7-sonnet", + "claude-3-7-sonnet-20250219", // Claude 3.5 Series "claude-3-5-sonnet", + "claude-3-5-sonnet-20241022", // Claude 3 Series "claude-3-opus", "claude-3-sonnet", "claude-3-haiku", ], azure: [ + // GPT-5.1 family (December 2025) + "gpt-5.1", + "gpt-5.1-chat", + "gpt-5.1-codex", // GPT-5 family "gpt-5", "gpt-5-pro", + "gpt-5-turbo", + "gpt-5-chat", "gpt-5-mini", // GPT-4.1 family "gpt-4.1", "gpt-4.1-mini", "gpt-4.1-nano", + // O-series + "o3", + "o3-mini", + "o3-pro", + "o4-mini", // Existing GPT-4 "gpt-4o", "gpt-4o-mini", @@ -101,10 +131,12 @@ const VISION_CAPABILITIES = { "gemini-3-pro-preview-11-2025", "gemini-3-pro-latest", "gemini-3-pro-preview", + "gemini-3-pro", // Gemini 2.5 models on Vertex AI "gemini-2.5-pro", "gemini-2.5-flash", "gemini-2.5-flash-lite", + "gemini-2.5-flash-image", // Gemini 2.0 models on Vertex AI "gemini-2.0-flash-001", "gemini-2.0-flash-lite", @@ -150,27 +182,41 @@ const VISION_CAPABILITIES = { litellm: [ // LiteLLM proxies to underlying providers // List models that support vision when going through the proxy - // Gemini models - "gemini-3-pro-preview", - "gemini-3-pro-latest", - "gemini-2.5-pro", - "gemini-2.5-flash", - "gemini-2.0-flash-lite", - // Claude 4.5 models + // OpenAI models via LiteLLM + "openai/gpt-5", + "openai/gpt-4o", + "openai/gpt-4o-mini", + "openai/gpt-4-turbo", + "gpt-5", + "gpt-4o", + "gpt-4.1", + // Anthropic models via LiteLLM + "anthropic/claude-sonnet-4-5-20250929", + "anthropic/claude-opus-4-1-20250805", + "anthropic/claude-3-5-sonnet-20240620", "claude-sonnet-4-5", "claude-sonnet-4-5-20250929", "claude-opus-4-5", "claude-opus-4-5-20251101", "claude-haiku-4-5-20251001", - // Claude 4 models "claude-sonnet-4", "claude-opus-4-1", - // OpenAI models - "gpt-4o", - "gpt-4.1", - "gpt-5", + // Gemini models via LiteLLM + "vertex_ai/gemini-2.5-pro", + "gemini/gemini-2.5-pro", + "gemini/gemini-2.0-flash", + "gemini-3-pro-preview", + "gemini-3-pro-latest", + "gemini-2.5-pro", + "gemini-2.5-flash", + "gemini-2.0-flash-lite", + // Groq models via LiteLLM (vision) + "groq/llama-3.2-11b-vision-preview", ], mistral: [ + // Mistral Large (latest has vision via Pixtral integration) + "mistral-large-latest", + "mistral-large-2512", // Mistral Small 3.2 (vision support for images: PNG, JPEG, WEBP, GIF) "mistral-small", "mistral-small-latest", @@ -180,26 +226,39 @@ const VISION_CAPABILITIES = { "mistral-medium", "mistral-medium-latest", "mistral-medium-3.1", + "mistral-medium-2508", // Magistral models (vision support) "magistral-small", + "magistral-small-latest", "magistral-medium", + "magistral-medium-latest", // Pixtral models (specialized vision models) "pixtral-12b", "pixtral-12b-latest", "pixtral-large", "pixtral-large-latest", + "pixtral-large-2502", ], ollama: [ // Llama 4 family (May 2025 - Best vision + tool calling) "llama4:scout", "llama4:maverick", - // Llama 3.2 vision + "llama4:latest", + "llama4", + // Llama 3.2 vision variants "llama3.2-vision", + "llama3.2-vision:11b", + "llama3.2-vision:90b", // Gemma 3 family (SigLIP vision encoder - supports tool calling + vision) + "gemma3", "gemma3:4b", "gemma3:12b", "gemma3:27b", "gemma3:latest", + // Qwen 2.5 VL (Vision-Language) + "qwen2.5-vl", + "qwen2.5-vl:72b", + "qwen2.5-vl:32b", // Mistral Small family (vision + tool calling) "mistral-small3.1", "mistral-small3.1:large", @@ -207,8 +266,24 @@ const VISION_CAPABILITIES = { "mistral-small3.1:small", // LLaVA (vision-focused) "llava", + "llava:7b", + "llava:13b", + "llava:34b", + "llava-llama3", + "llava-llama3:8b", ], bedrock: [ + // Amazon Nova models (December 2024+) - multimodal vision support + "amazon.nova-premier", + "amazon.nova-premier-v1:0", + "amazon.nova-pro", + "amazon.nova-pro-v1:0", + "amazon.nova-lite", + "amazon.nova-lite-v1:0", + "amazon.nova-2-lite-v1:0", + "nova-premier", + "nova-pro", + "nova-lite", // Claude 4.5 family (supports vision, PDFs, images - September-November 2025) "claude-sonnet-4-5", "claude-sonnet-4.5", @@ -217,7 +292,7 @@ const VISION_CAPABILITIES = { "claude-opus-4-5", "claude-opus-4.5", "anthropic.claude-opus-4-5", - "anthropic.claude-opus-4-5-20251101-v1:0", + "anthropic.claude-opus-4-5-20251124-v1:0", "claude-haiku-4-5", "claude-haiku-4.5", "anthropic.claude-haiku-4-5", @@ -226,18 +301,22 @@ const VISION_CAPABILITIES = { "claude-sonnet-4", "claude-sonnet-4@", "anthropic.claude-sonnet-4", + "anthropic.claude-sonnet-4-20250514-v1:0", "claude-opus-4", "claude-opus-4-1", "claude-opus-4@", "anthropic.claude-opus-4", + "anthropic.claude-opus-4-1-20250805-v1:0", // Claude 3.7 Sonnet "claude-3-7-sonnet", "claude-3.7-sonnet", "anthropic.claude-3-7-sonnet", + "anthropic.claude-3-7-sonnet-20250219-v1:0", // Claude 3.5 Sonnet "claude-3-5-sonnet", "claude-3.5-sonnet", "anthropic.claude-3-5-sonnet", + "anthropic.claude-3-5-sonnet-20241022-v1:0", // Claude 3 Opus "claude-3-opus", "anthropic.claude-3-opus", @@ -247,9 +326,32 @@ const VISION_CAPABILITIES = { // Claude 3 Haiku "claude-3-haiku", "anthropic.claude-3-haiku", + // Meta Llama 4 models (multimodal vision) + "meta.llama4-maverick-17b-instruct-v1:0", + "meta.llama4-scout-17b-instruct-v1:0", + // Meta Llama 3.2 vision models + "meta.llama3-2-90b-instruct-v1:0", + "meta.llama3-2-11b-instruct-v1:0", + // Mistral Pixtral (multimodal vision) + "mistral.pixtral-large-2502-v1:0", // Generic anthropic.claude prefix (catches all Claude models) "anthropic.claude", ], + huggingface: [ + // Qwen 2.5 VL (Vision-Language) + "Qwen/Qwen2.5-VL-32B-Instruct", + "Qwen/Qwen2.5-VL-7B-Instruct", + // Microsoft Phi-3 Vision + "microsoft/Phi-3-vision-128k-instruct", + // LLaVA variants + "llava-hf/llava-1.5-7b-hf", + "llava-hf/llava-v1.6-mistral-7b-hf", + ], + sagemaker: [ + // Meta Llama 4 vision models + "meta-llama-4-maverick-17b-128e-instruct", + "meta-llama-4-scout-17b-16e-instruct", + ], } as const; /** @@ -293,6 +395,21 @@ export class ProviderImageAdapter { case "ollama": adaptedPayload = this.formatForOpenAI(text, images); break; + case "huggingface": + adaptedPayload = this.formatForOpenAI(text, images); + break; + case "sagemaker": + adaptedPayload = this.formatForOpenAI(text, images); + break; + case "litellm": + adaptedPayload = this.formatForOpenAI(text, images); + break; + case "mistral": + adaptedPayload = this.formatForOpenAI(text, images); + break; + case "bedrock": + adaptedPayload = this.formatForAnthropic(text, images); + break; default: throw new Error(`Vision not supported for provider: ${provider}`); } diff --git a/src/lib/constants/enums.ts b/src/lib/constants/enums.ts index 130ee16ec..b66d7faed 100644 --- a/src/lib/constants/enums.ts +++ b/src/lib/constants/enums.ts @@ -25,33 +25,237 @@ export enum AIProviderName { * Supported Models for Amazon Bedrock */ export enum BedrockModels { + // ============================================================================ + // ANTHROPIC CLAUDE MODELS + // ============================================================================ + // Claude 4.5 Series (Latest - September-November 2025) - CLAUDE_4_5_SONNET = "anthropic.claude-sonnet-4-5-20250929-v1:0", CLAUDE_4_5_OPUS = "anthropic.claude-opus-4-5-20251124-v1:0", + CLAUDE_4_5_SONNET = "anthropic.claude-sonnet-4-5-20250929-v1:0", CLAUDE_4_5_HAIKU = "anthropic.claude-haiku-4-5-20251001-v1:0", + // Claude 4 Series (May-August 2025) + CLAUDE_4_1_OPUS = "anthropic.claude-opus-4-1-20250805-v1:0", + CLAUDE_4_SONNET = "anthropic.claude-sonnet-4-20250514-v1:0", + // Claude 3.7 Series CLAUDE_3_7_SONNET = "anthropic.claude-3-7-sonnet-20250219-v1:0", // Claude 3.5 Series CLAUDE_3_5_SONNET = "anthropic.claude-3-5-sonnet-20241022-v1:0", + CLAUDE_3_5_HAIKU = "anthropic.claude-3-5-haiku-20241022-v1:0", // Claude 3 Series (Legacy support) CLAUDE_3_SONNET = "anthropic.claude-3-sonnet-20240229-v1:0", CLAUDE_3_HAIKU = "anthropic.claude-3-haiku-20240307-v1:0", + + // ============================================================================ + // AMAZON NOVA MODELS + // ============================================================================ + + // Nova Generation 1 + NOVA_PREMIER = "amazon.nova-premier-v1:0", + NOVA_PRO = "amazon.nova-pro-v1:0", + NOVA_LITE = "amazon.nova-lite-v1:0", + NOVA_MICRO = "amazon.nova-micro-v1:0", + + // Nova Generation 2 (December 2025) + NOVA_2_LITE = "amazon.nova-2-lite-v1:0", + NOVA_2_SONIC = "amazon.nova-2-sonic-v1:0", + + // Nova Specialized Models + NOVA_SONIC = "amazon.nova-sonic-v1:0", + NOVA_CANVAS = "amazon.nova-canvas-v1:0", + NOVA_REEL = "amazon.nova-reel-v1:0", + NOVA_REEL_V1_1 = "amazon.nova-reel-v1:1", + NOVA_MULTIMODAL_EMBEDDINGS = "amazon.nova-2-multimodal-embeddings-v1:0", + + // ============================================================================ + // AMAZON TITAN MODELS + // ============================================================================ + + // Titan Text Generation + TITAN_TEXT_LARGE = "amazon.titan-tg1-large", + + // Titan Text Embeddings + TITAN_EMBED_TEXT_V2 = "amazon.titan-embed-text-v2:0", + TITAN_EMBED_TEXT_V1 = "amazon.titan-embed-text-v1", + TITAN_EMBED_G1_TEXT_02 = "amazon.titan-embed-g1-text-02", + + // Titan Multimodal Embeddings + TITAN_EMBED_IMAGE_V1 = "amazon.titan-embed-image-v1", + + // Titan Image Generation + TITAN_IMAGE_GENERATOR_V2 = "amazon.titan-image-generator-v2:0", + + // ============================================================================ + // META LLAMA MODELS + // ============================================================================ + + // Llama 4 Series (2025) + LLAMA_4_MAVERICK_17B = "meta.llama4-maverick-17b-instruct-v1:0", + LLAMA_4_SCOUT_17B = "meta.llama4-scout-17b-instruct-v1:0", + + // Llama 3.3 Series + LLAMA_3_3_70B = "meta.llama3-3-70b-instruct-v1:0", + + // Llama 3.2 Series (Multimodal) + LLAMA_3_2_90B = "meta.llama3-2-90b-instruct-v1:0", + LLAMA_3_2_11B = "meta.llama3-2-11b-instruct-v1:0", + LLAMA_3_2_3B = "meta.llama3-2-3b-instruct-v1:0", + LLAMA_3_2_1B = "meta.llama3-2-1b-instruct-v1:0", + + // Llama 3.1 Series + LLAMA_3_1_405B = "meta.llama3-1-405b-instruct-v1:0", + LLAMA_3_1_70B = "meta.llama3-1-70b-instruct-v1:0", + LLAMA_3_1_8B = "meta.llama3-1-8b-instruct-v1:0", + + // Llama 3 Series (Legacy) + LLAMA_3_70B = "meta.llama3-70b-instruct-v1:0", + LLAMA_3_8B = "meta.llama3-8b-instruct-v1:0", + + // ============================================================================ + // MISTRAL AI MODELS + // ============================================================================ + + // Mistral Large Series + MISTRAL_LARGE_3 = "mistral.mistral-large-3-675b-instruct", + MISTRAL_LARGE_2407 = "mistral.mistral-large-2407-v1:0", + MISTRAL_LARGE_2402 = "mistral.mistral-large-2402-v1:0", + + // Magistral & Ministral Series + MAGISTRAL_SMALL_2509 = "mistral.magistral-small-2509", + MINISTRAL_3_14B = "mistral.ministral-3-14b-instruct", + MINISTRAL_3_8B = "mistral.ministral-3-8b-instruct", + MINISTRAL_3_3B = "mistral.ministral-3-3b-instruct", + + // Mistral Base Series + MISTRAL_7B = "mistral.mistral-7b-instruct-v0:2", + MIXTRAL_8x7B = "mistral.mixtral-8x7b-instruct-v0:1", + + // Mistral Multimodal & Audio + PIXTRAL_LARGE_2502 = "mistral.pixtral-large-2502-v1:0", + VOXTRAL_SMALL_24B = "mistral.voxtral-small-24b-2507", + VOXTRAL_MINI_3B = "mistral.voxtral-mini-3b-2507", + + // ============================================================================ + // OTHER MODELS + // ============================================================================ + + // Cohere Models + COHERE_COMMAND_R_PLUS = "cohere.command-r-plus-v1:0", + COHERE_COMMAND_R = "cohere.command-r-v1:0", + + // DeepSeek Models + DEEPSEEK_R1 = "deepseek.r1-v1:0", + DEEPSEEK_V3 = "deepseek.v3-v1:0", + + // Qwen Models + QWEN_3_235B_A22B = "qwen.qwen3-235b-a22b-2507-v1:0", + QWEN_3_CODER_480B_A35B = "qwen.qwen3-coder-480b-a35b-v1:0", + QWEN_3_CODER_30B_A3B = "qwen.qwen3-coder-30b-a3b-v1:0", + QWEN_3_32B = "qwen.qwen3-32b-v1:0", + QWEN_3_NEXT_80B_A3B = "qwen.qwen3-next-80b-a3b", + QWEN_3_VL_235B_A22B = "qwen.qwen3-vl-235b-a22b", + + // Google Gemma + GEMMA_3_27B_IT = "google.gemma-3-27b-it", + GEMMA_3_12B_IT = "google.gemma-3-12b-it", + GEMMA_3_4B_IT = "google.gemma-3-4b-it", + + // AI21 Labs Models + JAMBA_1_5_LARGE = "ai21.jamba-1-5-large-v1:0", + JAMBA_1_5_MINI = "ai21.jamba-1-5-mini-v1:0", } /** * Supported Models for OpenAI */ export enum OpenAIModels { - GPT_4 = "gpt-4", - GPT_4_TURBO = "gpt-4-turbo", + // GPT-5.2 Series (Released December 11, 2025) - Latest flagship models + GPT_5_2 = "gpt-5.2", + GPT_5_2_CHAT_LATEST = "gpt-5.2-chat-latest", + GPT_5_2_PRO = "gpt-5.2-pro", + + // GPT-5 Series (Released August 7, 2025) + GPT_5 = "gpt-5", + GPT_5_MINI = "gpt-5-mini", + GPT_5_NANO = "gpt-5-nano", + + // GPT-4.1 Series (Released April 14, 2025) + GPT_4_1 = "gpt-4.1", + GPT_4_1_MINI = "gpt-4.1-mini", + GPT_4_1_NANO = "gpt-4.1-nano", + + // GPT-4o Series GPT_4O = "gpt-4o", GPT_4O_MINI = "gpt-4o-mini", - GPT_3_5_TURBO = "gpt-3.5-turbo", + + // O-Series Reasoning Models + O3 = "o3", + O3_MINI = "o3-mini", + O3_PRO = "o3-pro", + O4_MINI = "o4-mini", + O1 = "o1", O1_PREVIEW = "o1-preview", O1_MINI = "o1-mini", + + // GPT-4 Series (Legacy) + GPT_4 = "gpt-4", + GPT_4_TURBO = "gpt-4-turbo", + + // Legacy Models + GPT_3_5_TURBO = "gpt-3.5-turbo", +} + +/** + * Supported Models for Azure OpenAI + * Note: Azure uses deployment names, these are model identifiers + */ +export enum AzureOpenAIModels { + // GPT-5.1 Series (Latest - December 2025) + GPT_5_1 = "gpt-5.1", + GPT_5_1_CHAT = "gpt-5.1-chat", + GPT_5_1_CODEX = "gpt-5.1-codex", + GPT_5_1_CODEX_MINI = "gpt-5.1-codex-mini", + GPT_5_1_CODEX_MAX = "gpt-5.1-codex-max", + + // GPT-5.0 Series + GPT_5 = "gpt-5", + GPT_5_MINI = "gpt-5-mini", + GPT_5_NANO = "gpt-5-nano", + GPT_5_CHAT = "gpt-5-chat", + GPT_5_CODEX = "gpt-5-codex", + GPT_5_PRO = "gpt-5-pro", + GPT_5_TURBO = "gpt-5-turbo", + + // O-Series Reasoning Models + O4_MINI = "o4-mini", + O3 = "o3", + O3_MINI = "o3-mini", + O3_PRO = "o3-pro", + O1 = "o1", + O1_MINI = "o1-mini", + O1_PREVIEW = "o1-preview", + CODEX_MINI = "codex-mini", + + // GPT-4.1 Series + GPT_4_1 = "gpt-4.1", + GPT_4_1_NANO = "gpt-4.1-nano", + GPT_4_1_MINI = "gpt-4.1-mini", + + // GPT-4o Series (Multimodal) + GPT_4O = "gpt-4o", + GPT_4O_MINI = "gpt-4o-mini", + + // GPT-4 Turbo & GPT-4 + GPT_4_TURBO = "gpt-4-turbo", + GPT_4 = "gpt-4", + GPT_4_32K = "gpt-4-32k", + + // GPT-3.5 Turbo (Legacy) + GPT_3_5_TURBO = "gpt-35-turbo", + GPT_3_5_TURBO_INSTRUCT = "gpt-35-turbo-instruct", } /** @@ -59,13 +263,17 @@ export enum OpenAIModels { */ export enum VertexModels { // Claude 4.5 Series (Latest - December 2025) - CLAUDE_4_5_SONNET = "claude-sonnet-4-5@20250929", CLAUDE_4_5_OPUS = "claude-opus-4-5@20251124", + CLAUDE_4_5_SONNET = "claude-sonnet-4-5@20250929", + CLAUDE_4_5_HAIKU = "claude-haiku-4-5@20251001", // Claude 4 Series (May 2025) CLAUDE_4_0_SONNET = "claude-sonnet-4@20250514", CLAUDE_4_0_OPUS = "claude-opus-4@20250514", + // Claude 3.7 Series (February 2025) + CLAUDE_3_7_SONNET = "claude-3-7-sonnet@20250219", + // Claude 3.5 Series (Still supported) CLAUDE_3_5_SONNET = "claude-3-5-sonnet-20241022", CLAUDE_3_5_HAIKU = "claude-3-5-haiku-20241022", @@ -76,6 +284,8 @@ export enum VertexModels { CLAUDE_3_HAIKU = "claude-3-haiku-20240307", // Gemini 3 Series (Preview) + /** Gemini 3 Pro - Base model with adaptive thinking */ + GEMINI_3_PRO = "gemini-3-pro", /** Gemini 3 Pro Preview - Versioned preview (November 2025) */ GEMINI_3_PRO_PREVIEW_11_2025 = "gemini-3-pro-preview-11-2025", /** Gemini 3 Pro Latest - Auto-updated alias (always points to latest preview) */ @@ -87,41 +297,47 @@ export enum VertexModels { GEMINI_2_5_PRO = "gemini-2.5-pro", GEMINI_2_5_FLASH = "gemini-2.5-flash", GEMINI_2_5_FLASH_LITE = "gemini-2.5-flash-lite", + GEMINI_2_5_FLASH_IMAGE = "gemini-2.5-flash-image", // Gemini 2.0 Series + GEMINI_2_0_FLASH = "gemini-2.0-flash", GEMINI_2_0_FLASH_001 = "gemini-2.0-flash-001", /** Gemini 2.0 Flash Lite - GA, production-ready, cost-optimized */ GEMINI_2_0_FLASH_LITE = "gemini-2.0-flash-lite", // Gemini 1.5 Series (Legacy support) - GEMINI_1_5_PRO = "gemini-1.5-pro", - GEMINI_1_5_FLASH = "gemini-1.5-flash", + GEMINI_1_5_PRO = "gemini-1.5-pro-002", + GEMINI_1_5_FLASH = "gemini-1.5-flash-002", } /** * Supported Models for Google AI Studio */ export enum GoogleAIModels { - // Gemini 3 Series (Preview) - /** Gemini 3 Pro Preview - Versioned preview (November 2025) */ - GEMINI_3_PRO_PREVIEW_11_2025 = "gemini-3-pro-preview-11-2025", - /** Gemini 3 Pro Latest - Auto-updated alias (always points to latest preview) */ - GEMINI_3_PRO_LATEST = "gemini-3-pro-latest", + // Gemini 3 Series + GEMINI_3_PRO_PREVIEW = "gemini-3-pro-preview", + GEMINI_3_PRO_IMAGE_PREVIEW = "gemini-3-pro-image-preview", - // Gemini 2.5 Series (Latest - 2025) + // Gemini 2.5 Series GEMINI_2_5_PRO = "gemini-2.5-pro", GEMINI_2_5_FLASH = "gemini-2.5-flash", GEMINI_2_5_FLASH_LITE = "gemini-2.5-flash-lite", + GEMINI_2_5_FLASH_IMAGE = "gemini-2.5-flash-image", + GEMINI_2_5_FLASH_LIVE = "gemini-2.5-flash-native-audio-preview-09-2025", // Gemini 2.0 Series + GEMINI_2_0_FLASH = "gemini-2.0-flash", GEMINI_2_0_FLASH_001 = "gemini-2.0-flash-001", - /** Gemini 2.0 Flash Lite - GA, production-ready, cost-optimized */ GEMINI_2_0_FLASH_LITE = "gemini-2.0-flash-lite", + GEMINI_2_0_FLASH_IMAGE = "gemini-2.0-flash-preview-image-generation", - // Gemini 1.5 Series (Legacy support) + // Gemini 1.5 Series (Legacy) GEMINI_1_5_PRO = "gemini-1.5-pro", GEMINI_1_5_FLASH = "gemini-1.5-flash", - GEMINI_1_5_FLASH_LITE = "gemini-1.5-flash-lite", + + // Embedding Models + GEMINI_EMBEDDING = "gemini-embedding-001", + TEXT_EMBEDDING_004 = "text-embedding-004", } /** @@ -129,20 +345,362 @@ export enum GoogleAIModels { */ export enum AnthropicModels { // Claude 4.5 Series (Latest - September-November 2025) + CLAUDE_OPUS_4_5 = "claude-opus-4-5-20251101", CLAUDE_SONNET_4_5 = "claude-sonnet-4-5-20250929", - CLAUDE_OPUS_4_5 = "claude-opus-4-5-20251124", CLAUDE_4_5_HAIKU = "claude-haiku-4-5-20251001", - // Claude 3.5 Series + // Claude 4.1 Series (Legacy) + CLAUDE_OPUS_4_1 = "claude-opus-4-1-20250805", + + // Claude 4.0 Series (Legacy) + CLAUDE_OPUS_4_0 = "claude-opus-4-20250514", + CLAUDE_SONNET_4_0 = "claude-sonnet-4-20250514", + + // Claude 3.7 Series (Legacy) + CLAUDE_SONNET_3_7 = "claude-3-7-sonnet-20250219", + + // Claude 3.5 Series (Legacy) CLAUDE_3_5_SONNET = "claude-3-5-sonnet-20241022", CLAUDE_3_5_HAIKU = "claude-3-5-haiku-20241022", - // Claude 3 Series (Legacy support) + // Claude 3 Series (Legacy - Deprecated) CLAUDE_3_SONNET = "claude-3-sonnet-20240229", CLAUDE_3_OPUS = "claude-3-opus-20240229", CLAUDE_3_HAIKU = "claude-3-haiku-20240307", } +/** + * Supported Models for Mistral AI + */ +export enum MistralModels { + // Mistral Large (Latest) + MISTRAL_LARGE_LATEST = "mistral-large-latest", + MISTRAL_LARGE_2512 = "mistral-large-2512", + + // Mistral Medium + MISTRAL_MEDIUM_LATEST = "mistral-medium-latest", + MISTRAL_MEDIUM_2508 = "mistral-medium-2508", + + // Mistral Small + MISTRAL_SMALL_LATEST = "mistral-small-latest", + MISTRAL_SMALL_2506 = "mistral-small-2506", + + // Magistral (Reasoning) + MAGISTRAL_MEDIUM_LATEST = "magistral-medium-latest", + MAGISTRAL_SMALL_LATEST = "magistral-small-latest", + + // Ministral (Edge Models) + MINISTRAL_14B_2512 = "ministral-14b-2512", + MINISTRAL_8B_2512 = "ministral-8b-2512", + MINISTRAL_3B_2512 = "ministral-3b-2512", + + // Codestral (Code Generation) + CODESTRAL_LATEST = "codestral-latest", + CODESTRAL_2508 = "codestral-2508", + CODESTRAL_EMBED = "codestral-embed", + + // Devstral (Software Development) + DEVSTRAL_MEDIUM_LATEST = "devstral-medium-latest", + DEVSTRAL_SMALL_LATEST = "devstral-small-latest", + + // Pixtral (Multimodal/Vision) + PIXTRAL_LARGE = "pixtral-large", + PIXTRAL_12B = "pixtral-12b", + + // Voxtral (Audio) + VOXTRAL_SMALL_LATEST = "voxtral-small-latest", + VOXTRAL_MINI_LATEST = "voxtral-mini-latest", + + // Specialized Models + MISTRAL_NEMO = "mistral-nemo", + MISTRAL_EMBED = "mistral-embed", + MISTRAL_MODERATION_LATEST = "mistral-moderation-latest", +} + +/** + * Supported Models for Ollama (Local) + * All models can be run locally without requiring API keys or cloud services + */ +export enum OllamaModels { + // Llama 4 Series - Multimodal with vision and tool capabilities + LLAMA4_SCOUT = "llama4:scout", + LLAMA4_MAVERICK = "llama4:maverick", + LLAMA4_LATEST = "llama4:latest", + + // Llama 3.3 Series - High-performance models + LLAMA3_3_LATEST = "llama3.3:latest", + LLAMA3_3_70B = "llama3.3:70b", + + // Llama 3.2 Series - Optimized for edge and mobile deployment + LLAMA3_2_LATEST = "llama3.2:latest", + LLAMA3_2_3B = "llama3.2:3b", + LLAMA3_2_1B = "llama3.2:1b", + + // Llama 3.1 Series - Open models rivaling proprietary models + LLAMA3_1_8B = "llama3.1:8b", + LLAMA3_1_70B = "llama3.1:70b", + LLAMA3_1_405B = "llama3.1:405b", + + // Qwen 3 Series - Advanced reasoning and multilingual support + QWEN3_4B = "qwen3:4b", + QWEN3_8B = "qwen3:8b", + QWEN3_14B = "qwen3:14b", + QWEN3_32B = "qwen3:32b", + QWEN3_72B = "qwen3:72b", + + // Qwen 2.5 Series - Enhanced coding and mathematics + QWEN2_5_3B = "qwen2.5:3b", + QWEN2_5_7B = "qwen2.5:7b", + QWEN2_5_14B = "qwen2.5:14b", + QWEN2_5_32B = "qwen2.5:32b", + QWEN2_5_72B = "qwen2.5:72b", + + // Qwen Reasoning Model + QWQ_32B = "qwq:32b", + QWQ_LATEST = "qwq:latest", + + // DeepSeek-R1 Series - State-of-the-art reasoning models + DEEPSEEK_R1_1_5B = "deepseek-r1:1.5b", + DEEPSEEK_R1_7B = "deepseek-r1:7b", + DEEPSEEK_R1_8B = "deepseek-r1:8b", + DEEPSEEK_R1_14B = "deepseek-r1:14b", + DEEPSEEK_R1_32B = "deepseek-r1:32b", + DEEPSEEK_R1_70B = "deepseek-r1:70b", + + // DeepSeek-V3 Series - Mixture of Experts model + DEEPSEEK_V3_671B = "deepseek-v3:671b", + DEEPSEEK_V3_LATEST = "deepseek-v3:latest", + + // Mistral AI Series - Efficient general-purpose models + MISTRAL_LATEST = "mistral:latest", + MISTRAL_7B = "mistral:7b", + MISTRAL_SMALL_LATEST = "mistral-small:latest", + MISTRAL_NEMO_LATEST = "mistral-nemo:latest", + MISTRAL_LARGE_LATEST = "mistral-large:latest", + + // Google Gemma Series - Efficient edge and cloud models + GEMMA3_LATEST = "gemma3:latest", + GEMMA2_2B = "gemma2:2b", + GEMMA2_9B = "gemma2:9b", + GEMMA2_27B = "gemma2:27b", + + // Microsoft Phi Series - Compact, efficient models + PHI4_LATEST = "phi4:latest", + PHI4_14B = "phi4:14b", + PHI3_MINI = "phi3:mini", + PHI3_3_8B = "phi3:3.8b", + PHI3_MEDIUM = "phi3:medium", + PHI3_14B = "phi3:14b", + + // Vision-Language Models + LLAVA_7B = "llava:7b", + LLAVA_13B = "llava:13b", + LLAVA_34B = "llava:34b", + LLAVA_LLAMA3_8B = "llava-llama3:8b", + + // Code-Specialized Models + CODELLAMA_7B = "codellama:7b", + CODELLAMA_13B = "codellama:13b", + CODELLAMA_34B = "codellama:34b", + CODELLAMA_70B = "codellama:70b", + QWEN2_5_CODER_7B = "qwen2.5-coder:7b", + QWEN2_5_CODER_32B = "qwen2.5-coder:32b", + STARCODER2_3B = "starcoder2:3b", + STARCODER2_7B = "starcoder2:7b", + STARCODER2_15B = "starcoder2:15b", + + // Mixture of Experts Models + MIXTRAL_8X7B = "mixtral:8x7b", + MIXTRAL_8X22B = "mixtral:8x22b", + + // Enterprise Models + COMMAND_R_PLUS = "command-r-plus:104b", +} + +/** + * Common Models for LiteLLM Proxy + * LiteLLM supports 100+ models through unified proxy interface + * Models use provider-specific prefixes (e.g., "openai/", "anthropic/") + */ +export enum LiteLLMModels { + // OpenAI via LiteLLM + OPENAI_GPT_5 = "openai/gpt-5", + OPENAI_GPT_4O = "openai/gpt-4o", + OPENAI_GPT_4O_MINI = "openai/gpt-4o-mini", + OPENAI_GPT_4_TURBO = "openai/gpt-4-turbo", + OPENAI_GPT_4 = "openai/gpt-4", + OPENAI_GPT_3_5_TURBO = "openai/gpt-3.5-turbo", + + // Anthropic via LiteLLM + ANTHROPIC_CLAUDE_SONNET_4_5 = "anthropic/claude-sonnet-4-5-20250929", + ANTHROPIC_CLAUDE_OPUS_4_1 = "anthropic/claude-opus-4-1-20250805", + ANTHROPIC_CLAUDE_3_5_SONNET = "anthropic/claude-3-5-sonnet-20240620", + ANTHROPIC_CLAUDE_3_HAIKU = "anthropic/claude-3-haiku-20240307", + + // Google Vertex AI via LiteLLM + VERTEX_GEMINI_2_5_PRO = "vertex_ai/gemini-2.5-pro", + VERTEX_GEMINI_1_5_PRO = "vertex_ai/gemini-1.5-pro", + VERTEX_GEMINI_1_5_FLASH = "vertex_ai/gemini-1.5-flash", + + // Google AI Studio (Gemini) via LiteLLM + GEMINI_2_5_PRO = "gemini/gemini-2.5-pro", + GEMINI_2_0_FLASH = "gemini/gemini-2.0-flash", + GEMINI_1_5_PRO = "gemini/gemini-1.5-pro", + GEMINI_1_5_FLASH = "gemini/gemini-1.5-flash", + + // Groq via LiteLLM + GROQ_LLAMA_3_1_70B_VERSATILE = "groq/llama-3.1-70b-versatile", + GROQ_LLAMA_3_1_8B_INSTANT = "groq/llama-3.1-8b-instant", + GROQ_LLAMA_3_2_11B_VISION = "groq/llama-3.2-11b-vision-preview", + GROQ_MIXTRAL_8X7B = "groq/mixtral-8x7b-32768", + + // Together AI via LiteLLM + TOGETHER_LLAMA_2_70B_CHAT = "together_ai/togethercomputer/llama-2-70b-chat", + TOGETHER_MIXTRAL_8X7B = "together_ai/mistralai/Mixtral-8x7B-Instruct-v0.1", + TOGETHER_CODELLAMA_34B = "together_ai/codellama/CodeLlama-34b-Instruct-hf", + + // DeepInfra via LiteLLM + DEEPINFRA_LLAMA_3_70B = "deepinfra/meta-llama/Meta-Llama-3-70B-Instruct", + DEEPINFRA_LLAMA_2_70B = "deepinfra/meta-llama/Llama-2-70b-chat-hf", + DEEPINFRA_MISTRAL_7B = "deepinfra/mistralai/Mistral-7B-Instruct-v0.1", + + // Mistral AI via LiteLLM + MISTRAL_LARGE = "mistral/mistral-large-latest", + MISTRAL_SMALL = "mistral/mistral-small-latest", + MISTRAL_MAGISTRAL_MEDIUM = "mistral/magistral-medium-2506", + + // AWS Bedrock via LiteLLM + BEDROCK_CLAUDE_3_5_SONNET = "bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", + BEDROCK_CLAUDE_3_HAIKU = "bedrock/anthropic.claude-3-haiku-20240307-v1:0", + + // Perplexity AI via LiteLLM + PERPLEXITY_SONAR_PRO = "perplexity/sonar-pro", + PERPLEXITY_SONAR_REASONING_PRO = "perplexity/sonar-reasoning-pro", +} + +/** + * Supported Models for Hugging Face Inference API + */ +export enum HuggingFaceModels { + // Meta Llama 3.3 + LLAMA_3_3_70B_INSTRUCT = "meta-llama/Llama-3.3-70B-Instruct", + + // Meta Llama 3.2 + LLAMA_3_2_1B = "meta-llama/Llama-3.2-1B", + LLAMA_3_2_3B_INSTRUCT = "meta-llama/Llama-3.2-3B-Instruct", + + // Meta Llama 3.1 + LLAMA_3_1_8B = "meta-llama/Llama-3.1-8B", + LLAMA_3_1_70B_INSTRUCT = "meta-llama/Llama-3.1-70B-Instruct", + LLAMA_3_1_405B_INSTRUCT = "meta-llama/Llama-3.1-405B-Instruct", + + // Meta Llama 3.0 + LLAMA_3_8B_INSTRUCT = "meta-llama/Meta-Llama-3-8B-Instruct", + LLAMA_3_70B_INSTRUCT = "meta-llama/Meta-Llama-3-70B-Instruct", + + // Mistral Large + MISTRAL_LARGE_3_675B = "mistralai/Mistral-Large-3-675B-Instruct-2512", + + // Mistral Small + MISTRAL_SMALL_3_1_24B = "mistralai/Mistral-Small-3.1-24B-Instruct-2503", + MISTRAL_SMALL_24B = "mistralai/Mistral-Small-24B-Instruct-2501", + + // Mistral + MISTRAL_7B_INSTRUCT = "mistralai/Mistral-7B-Instruct-v0.2", + MIXTRAL_8X7B_INSTRUCT = "mistralai/Mixtral-8x7B-Instruct-v0.1", + + // Mistral Devstral + DEVSTRAL_2 = "mistralai/Devstral-2", + + // Qwen 2.5 + QWEN_2_5_7B = "Qwen/Qwen2.5-7B", + QWEN_2_5_32B = "Qwen/Qwen2.5-32B", + QWEN_2_5_72B_INSTRUCT = "Qwen/Qwen2.5-72B-Instruct", + + // Qwen 2.5 Coder + QWEN_2_5_CODER_7B = "Qwen/Qwen2.5-Coder-7B", + QWEN_2_5_CODER_32B_INSTRUCT = "Qwen/Qwen2.5-Coder-32B-Instruct", + + // Qwen QwQ + QWQ_32B = "Qwen/QwQ-32B", + + // Qwen 2.5 VL (Multimodal) + QWEN_2_5_VL_32B = "Qwen/Qwen2.5-VL-32B-Instruct", + + // DeepSeek + DEEPSEEK_R1 = "deepseek-ai/DeepSeek-R1", + DEEPSEEK_V3 = "deepseek-ai/DeepSeek-V3", + DEEPSEEK_V3_1 = "deepseek-ai/DeepSeek-V3.1", + DEEPSEEK_V3_2_EXP = "deepseek-ai/DeepSeek-V3.2-Exp", + + // Microsoft Phi + PHI_4 = "microsoft/phi-4", + PHI_4_REASONING = "microsoft/Phi-4-reasoning", + PHI_4_MINI_INSTRUCT = "microsoft/Phi-4-mini-instruct", + PHI_4_MINI_REASONING = "microsoft/Phi-4-mini-reasoning", + PHI_3_MINI_128K_INSTRUCT = "microsoft/Phi-3-mini-128k-instruct", + PHI_3_VISION_128K_INSTRUCT = "microsoft/Phi-3-vision-128k-instruct", + + // Google Gemma 3 + GEMMA_3_270M = "google/gemma-3-270m", + GEMMA_3_1B_IT = "google/gemma-3-1b-it", + GEMMA_3_4B_IT = "google/gemma-3-4b-it", + GEMMA_3_12B_IT = "google/gemma-3-12b-it", + GEMMA_3_27B_IT = "google/gemma-3-27b-it", + + // Google Gemma 2 + GEMMA_2_9B = "google/gemma-2-9b", + GEMMA_2_27B = "google/gemma-2-27b", + + // Google Gemma 1 + GEMMA_2B = "google/gemma-2b", + GEMMA_7B = "google/gemma-7b", + + // Falcon + FALCON_40B_INSTRUCT = "tiiuae/falcon-40b-instruct", + FALCON_180B_CHAT = "tiiuae/falcon-180B-chat", + + // Code Models + STARCODER2_15B = "bigcode/starcoder2-15b", + CODELLAMA_34B_INSTRUCT = "codellama/CodeLlama-34b-Instruct-hf", + + // BLOOM + BLOOM_7B1 = "bigscience/bloom-7b1", + BLOOM_1B3 = "bigscience/bloom-1b3", +} + +/** + * Supported Models for AWS SageMaker JumpStart + * https://docs.aws.amazon.com/sagemaker/latest/dg/jumpstart-foundation-models-latest.html + */ +export enum SageMakerModels { + // Meta Llama 4 Series (Latest - 2025) + LLAMA_4_SCOUT_17B_16E = "meta-llama-4-scout-17b-16e-instruct", + LLAMA_4_MAVERICK_17B_128E = "meta-llama-4-maverick-17b-128e-instruct", + LLAMA_4_MAVERICK_17B_128E_FP8 = "meta-llama-4-maverick-17b-128e-instruct-fp8", + + // Meta Llama 3 Series + LLAMA_3_8B = "meta-llama-3-8b-instruct", + LLAMA_3_70B = "meta-llama-3-70b-instruct", + + // Meta Code Llama Series + CODE_LLAMA_7B = "meta-code-llama-7b", + CODE_LLAMA_13B = "meta-code-llama-13b", + CODE_LLAMA_34B = "meta-code-llama-34b", + + // Mistral AI Models + MISTRAL_SMALL_24B = "mistral-small-24b-instruct-2501", + MISTRAL_7B_INSTRUCT = "mistral-7b-instruct-v0.3", + MIXTRAL_8X7B = "mistral-mixtral-8x7b-instruct-v0.1", + MIXTRAL_8X22B = "mistral-mixtral-8x22b-instruct-v0.1", + + // Falcon Models + FALCON_3_7B = "tii-falcon-3-7b-instruct", + FALCON_3_10B = "tii-falcon-3-10b-instruct", + FALCON_40B = "tii-falcon-40b-instruct", + FALCON_180B = "tii-falcon-180b", +} + /** * API Versions for various providers */ diff --git a/src/lib/factories/providerRegistry.ts b/src/lib/factories/providerRegistry.ts index dbdfc5311..2cecadc0d 100644 --- a/src/lib/factories/providerRegistry.ts +++ b/src/lib/factories/providerRegistry.ts @@ -11,6 +11,12 @@ import { AIProviderName, GoogleAIModels, OpenAIModels, + AnthropicModels, + VertexModels, + MistralModels, + OllamaModels, + LiteLLMModels, + HuggingFaceModels, } from "../constants/enums.js"; /** @@ -83,7 +89,7 @@ export class ProviderRegistry { ); return new AnthropicProvider(modelName, sdk as NeuroLink | undefined); }, - "claude-3-5-sonnet-20241022", + AnthropicModels.CLAUDE_SONNET_4_0, ["claude", "anthropic"], ); @@ -152,7 +158,7 @@ export class ProviderRegistry { region, ); }, - "claude-sonnet-4@20250514", + VertexModels.CLAUDE_4_0_SONNET, ["vertex", "googleVertex"], ); @@ -165,7 +171,8 @@ export class ProviderRegistry { ); return new HuggingFaceProvider(modelName); }, - process.env.HUGGINGFACE_MODEL || "microsoft/DialoGPT-medium", + process.env.HUGGINGFACE_MODEL || + HuggingFaceModels.QWEN_2_5_72B_INSTRUCT, ["huggingface", "hf"], ); @@ -183,7 +190,7 @@ export class ProviderRegistry { sdk as MistralProviderType | undefined, ); }, - "mistral-large-latest", + MistralModels.MISTRAL_LARGE_LATEST, ["mistral"], ); @@ -194,7 +201,7 @@ export class ProviderRegistry { const { OllamaProvider } = await import("../providers/ollama.js"); return new OllamaProvider(modelName); }, - process.env.OLLAMA_MODEL || "llama3.1:8b", + process.env.OLLAMA_MODEL || OllamaModels.LLAMA3_2_LATEST, ["ollama", "local"], ); @@ -209,7 +216,7 @@ export class ProviderRegistry { const { LiteLLMProvider } = await import("../providers/litellm.js"); return new LiteLLMProvider(modelName, sdk as NeuroLink | undefined); }, - process.env.LITELLM_MODEL || "openai/gpt-4o-mini", + process.env.LITELLM_MODEL || LiteLLMModels.OPENAI_GPT_4O_MINI, ["litellm"], ); diff --git a/src/lib/models/modelRegistry.ts b/src/lib/models/modelRegistry.ts index 369fb73e1..ba7606b23 100644 --- a/src/lib/models/modelRegistry.ts +++ b/src/lib/models/modelRegistry.ts @@ -8,8 +8,12 @@ import { DEFAULT_MODEL_ALIASES } from "../types/providers.js"; import { AIProviderName, OpenAIModels, + AzureOpenAIModels, GoogleAIModels, AnthropicModels, + BedrockModels, + MistralModels, + OllamaModels, } from "../constants/enums.js"; import type { JsonValue } from "../types/common.js"; import type { ModelInfo } from "../types/modelTypes.js"; @@ -109,13 +113,1666 @@ export const MODEL_REGISTRY: Record = { category: "general", }, + // OpenAI GPT-5 Series + [OpenAIModels.GPT_5]: { + id: OpenAIModels.GPT_5, + name: "GPT-5", + provider: AIProviderName.OPENAI, + description: + "OpenAI's most advanced model with breakthrough reasoning and multimodal capabilities", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.01, + outputCostPer1K: 0.03, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 256000, + maxOutputTokens: 32768, + maxRequestsPerMinute: 200, + }, + useCases: { + coding: 10, + creative: 10, + analysis: 10, + conversation: 10, + reasoning: 10, + translation: 9, + summarization: 9, + }, + aliases: ["gpt5", "gpt-5-flagship", "openai-latest"], + deprecated: false, + isLocal: false, + releaseDate: "2025-08-07", + category: "reasoning", + }, + + [OpenAIModels.GPT_5_MINI]: { + id: OpenAIModels.GPT_5_MINI, + name: "GPT-5 Mini", + provider: AIProviderName.OPENAI, + description: "Fast and efficient GPT-5 variant for everyday tasks", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.002, + outputCostPer1K: 0.006, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 128000, + maxOutputTokens: 16384, + maxRequestsPerMinute: 500, + }, + useCases: { + coding: 8, + creative: 8, + analysis: 8, + conversation: 9, + reasoning: 8, + translation: 8, + summarization: 9, + }, + aliases: ["gpt5-mini", "gpt-5-fast"], + deprecated: false, + isLocal: false, + releaseDate: "2025-08-07", + category: "general", + }, + + // OpenAI O-Series Reasoning Models + [OpenAIModels.O3]: { + id: OpenAIModels.O3, + name: "O3", + provider: AIProviderName.OPENAI, + description: + "Advanced reasoning model with extended thinking capabilities for complex tasks", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.015, + outputCostPer1K: 0.06, + currency: "USD", + }, + performance: { + speed: "slow", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 200000, + maxOutputTokens: 100000, + maxRequestsPerMinute: 100, + }, + useCases: { + coding: 10, + creative: 8, + analysis: 10, + conversation: 7, + reasoning: 10, + translation: 7, + summarization: 8, + }, + aliases: ["o3-reasoning", "o3-thinking"], + deprecated: false, + isLocal: false, + releaseDate: "2025-01-31", + category: "reasoning", + }, + + [OpenAIModels.O3_MINI]: { + id: OpenAIModels.O3_MINI, + name: "O3 Mini", + provider: AIProviderName.OPENAI, + description: + "Cost-effective reasoning model with strong logical capabilities", + capabilities: { + vision: false, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: false, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.003, + outputCostPer1K: 0.012, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 200000, + maxOutputTokens: 65536, + maxRequestsPerMinute: 200, + }, + useCases: { + coding: 9, + creative: 6, + analysis: 9, + conversation: 7, + reasoning: 9, + translation: 6, + summarization: 7, + }, + aliases: ["o3-mini-reasoning"], + deprecated: false, + isLocal: false, + releaseDate: "2025-01-31", + category: "reasoning", + }, + + [OpenAIModels.GPT_5_NANO]: { + id: OpenAIModels.GPT_5_NANO, + name: "GPT-5 Nano", + provider: AIProviderName.OPENAI, + description: + "Fastest and most cost-effective GPT-5 variant for simple tasks", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.00005, + outputCostPer1K: 0.0004, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "medium", + accuracy: "medium", + }, + limits: { + maxContextTokens: 272000, + maxOutputTokens: 128000, + maxRequestsPerMinute: 2000, + }, + useCases: { + coding: 6, + creative: 6, + analysis: 6, + conversation: 8, + reasoning: 6, + translation: 7, + summarization: 8, + }, + aliases: ["gpt5-nano", "gpt-5-cheapest"], + deprecated: false, + isLocal: false, + releaseDate: "2025-08-07", + category: "general", + }, + + // OpenAI GPT-5.2 Series (Released December 11, 2025) - Latest flagship models + [OpenAIModels.GPT_5_2]: { + id: OpenAIModels.GPT_5_2, + name: "GPT-5.2 Thinking", + provider: AIProviderName.OPENAI, + description: + "OpenAI's latest flagship model with deep reasoning capabilities, 100% on AIME 2025, 80% SWE-bench Verified", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.00175, + outputCostPer1K: 0.014, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 256000, + maxOutputTokens: 64000, + maxRequestsPerMinute: 150, + }, + useCases: { + coding: 10, + creative: 10, + analysis: 10, + conversation: 9, + reasoning: 10, + translation: 9, + summarization: 9, + }, + aliases: ["gpt52", "gpt-5.2-thinking", "openai-latest-reasoning"], + deprecated: false, + isLocal: false, + releaseDate: "2025-12-11", + category: "reasoning", + }, + + [OpenAIModels.GPT_5_2_CHAT_LATEST]: { + id: OpenAIModels.GPT_5_2_CHAT_LATEST, + name: "GPT-5.2 Instant", + provider: AIProviderName.OPENAI, + description: + "Fast everyday model for quick tasks with excellent performance across all domains", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.00175, + outputCostPer1K: 0.014, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 256000, + maxOutputTokens: 32000, + maxRequestsPerMinute: 300, + }, + useCases: { + coding: 9, + creative: 9, + analysis: 9, + conversation: 10, + reasoning: 9, + translation: 9, + summarization: 9, + }, + aliases: ["gpt52-chat", "gpt-5.2-instant", "gpt52-fast"], + deprecated: false, + isLocal: false, + releaseDate: "2025-12-11", + category: "general", + }, + + [OpenAIModels.GPT_5_2_PRO]: { + id: OpenAIModels.GPT_5_2_PRO, + name: "GPT-5.2 Pro", + provider: AIProviderName.OPENAI, + description: + "Highest quality model for science, math, and complex problem-solving with 92.4% GPQA Diamond performance", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.021, + outputCostPer1K: 0.168, + currency: "USD", + }, + performance: { + speed: "slow", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 256000, + maxOutputTokens: 128000, + maxRequestsPerMinute: 50, + }, + useCases: { + coding: 10, + creative: 9, + analysis: 10, + conversation: 8, + reasoning: 10, + translation: 9, + summarization: 9, + }, + aliases: ["gpt52-pro", "gpt-5.2-professional", "openai-science"], + deprecated: false, + isLocal: false, + releaseDate: "2025-12-11", + category: "reasoning", + }, + + // OpenAI GPT-4.1 Series (1M context window) + [OpenAIModels.GPT_4_1]: { + id: OpenAIModels.GPT_4_1, + name: "GPT-4.1", + provider: AIProviderName.OPENAI, + description: "Advanced coding model with 1 million token context window", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.002, + outputCostPer1K: 0.008, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 1000000, + maxOutputTokens: 128000, + maxRequestsPerMinute: 200, + }, + useCases: { + coding: 10, + creative: 8, + analysis: 9, + conversation: 8, + reasoning: 9, + translation: 8, + summarization: 9, + }, + aliases: ["gpt-4.1", "gpt41", "million-context"], + deprecated: false, + isLocal: false, + releaseDate: "2025-04-14", + category: "coding", + }, + + [OpenAIModels.GPT_4_1_MINI]: { + id: OpenAIModels.GPT_4_1_MINI, + name: "GPT-4.1 Mini", + provider: AIProviderName.OPENAI, + description: "Fast GPT-4.1 variant with 1M context for efficient coding", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.0004, + outputCostPer1K: 0.0016, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 1000000, + maxOutputTokens: 128000, + maxRequestsPerMinute: 500, + }, + useCases: { + coding: 9, + creative: 7, + analysis: 8, + conversation: 8, + reasoning: 8, + translation: 8, + summarization: 9, + }, + aliases: ["gpt-4.1-mini", "gpt41-mini"], + deprecated: false, + isLocal: false, + releaseDate: "2025-04-14", + category: "coding", + }, + + [OpenAIModels.GPT_4_1_NANO]: { + id: OpenAIModels.GPT_4_1_NANO, + name: "GPT-4.1 Nano", + provider: AIProviderName.OPENAI, + description: "Most cost-effective GPT-4.1 variant with 1M context", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.0001, + outputCostPer1K: 0.0004, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "medium", + accuracy: "medium", + }, + limits: { + maxContextTokens: 1000000, + maxOutputTokens: 128000, + maxRequestsPerMinute: 1000, + }, + useCases: { + coding: 7, + creative: 6, + analysis: 7, + conversation: 7, + reasoning: 7, + translation: 7, + summarization: 8, + }, + aliases: ["gpt-4.1-nano", "gpt41-nano"], + deprecated: false, + isLocal: false, + releaseDate: "2025-04-14", + category: "coding", + }, + + // OpenAI O-Series Additional Models + [OpenAIModels.O3_PRO]: { + id: OpenAIModels.O3_PRO, + name: "O3 Pro", + provider: AIProviderName.OPENAI, + description: + "Most powerful reasoning model for complex scientific and coding tasks", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.03, + outputCostPer1K: 0.12, + currency: "USD", + }, + performance: { + speed: "slow", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 200000, + maxOutputTokens: 100000, + maxRequestsPerMinute: 50, + }, + useCases: { + coding: 10, + creative: 7, + analysis: 10, + conversation: 6, + reasoning: 10, + translation: 6, + summarization: 7, + }, + aliases: ["o3-pro", "o3-professional"], + deprecated: false, + isLocal: false, + releaseDate: "2025-04-16", + category: "reasoning", + }, + + [OpenAIModels.O4_MINI]: { + id: OpenAIModels.O4_MINI, + name: "O4 Mini", + provider: AIProviderName.OPENAI, + description: + "Fast reasoning model optimized for math, coding, and visual tasks", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.003, + outputCostPer1K: 0.012, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 200000, + maxOutputTokens: 100000, + maxRequestsPerMinute: 200, + }, + useCases: { + coding: 9, + creative: 6, + analysis: 9, + conversation: 7, + reasoning: 10, + translation: 6, + summarization: 7, + }, + aliases: ["o4-mini", "o4-fast"], + deprecated: false, + isLocal: false, + releaseDate: "2025-04-16", + category: "reasoning", + }, + + [OpenAIModels.O1]: { + id: OpenAIModels.O1, + name: "O1", + provider: AIProviderName.OPENAI, + description: + "Premium reasoning model with highest capability for mission-critical tasks", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.15, + outputCostPer1K: 0.6, + currency: "USD", + }, + performance: { + speed: "slow", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 128000, + maxOutputTokens: 32768, + maxRequestsPerMinute: 50, + }, + useCases: { + coding: 10, + creative: 7, + analysis: 10, + conversation: 6, + reasoning: 10, + translation: 6, + summarization: 7, + }, + aliases: ["o1-full", "o1-premium"], + deprecated: false, + isLocal: false, + releaseDate: "2024-09-12", + category: "reasoning", + }, + + [OpenAIModels.O1_PREVIEW]: { + id: OpenAIModels.O1_PREVIEW, + name: "O1 Preview", + provider: AIProviderName.OPENAI, + description: "Preview version of O1 reasoning model", + capabilities: { + vision: false, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: false, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.015, + outputCostPer1K: 0.06, + currency: "USD", + }, + performance: { + speed: "slow", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 128000, + maxOutputTokens: 32768, + maxRequestsPerMinute: 100, + }, + useCases: { + coding: 9, + creative: 6, + analysis: 9, + conversation: 6, + reasoning: 9, + translation: 5, + summarization: 6, + }, + aliases: ["o1-preview"], + deprecated: false, + isLocal: false, + releaseDate: "2024-09-12", + category: "reasoning", + }, + + [OpenAIModels.O1_MINI]: { + id: OpenAIModels.O1_MINI, + name: "O1 Mini", + provider: AIProviderName.OPENAI, + description: "Cost-effective O1 variant with strong reasoning capabilities", + capabilities: { + vision: false, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: false, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.003, + outputCostPer1K: 0.012, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 128000, + maxOutputTokens: 65536, + maxRequestsPerMinute: 200, + }, + useCases: { + coding: 8, + creative: 5, + analysis: 8, + conversation: 6, + reasoning: 8, + translation: 5, + summarization: 6, + }, + aliases: ["o1-mini", "o1-budget"], + deprecated: false, + isLocal: false, + releaseDate: "2024-09-12", + category: "reasoning", + }, + + // OpenAI Legacy Models + [OpenAIModels.GPT_4]: { + id: OpenAIModels.GPT_4, + name: "GPT-4", + provider: AIProviderName.OPENAI, + description: "Previous generation flagship model (legacy)", + capabilities: { + vision: false, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: false, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.03, + outputCostPer1K: 0.06, + currency: "USD", + }, + performance: { + speed: "slow", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 8192, + maxOutputTokens: 4096, + maxRequestsPerMinute: 200, + }, + useCases: { + coding: 8, + creative: 8, + analysis: 8, + conversation: 8, + reasoning: 8, + translation: 8, + summarization: 8, + }, + aliases: ["gpt4", "gpt-4-base"], + deprecated: true, + isLocal: false, + releaseDate: "2023-03-14", + category: "general", + }, + + [OpenAIModels.GPT_4_TURBO]: { + id: OpenAIModels.GPT_4_TURBO, + name: "GPT-4 Turbo", + provider: AIProviderName.OPENAI, + description: "Faster GPT-4 variant with extended context (legacy)", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.01, + outputCostPer1K: 0.03, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 128000, + maxOutputTokens: 4096, + maxRequestsPerMinute: 500, + }, + useCases: { + coding: 8, + creative: 8, + analysis: 9, + conversation: 8, + reasoning: 8, + translation: 8, + summarization: 8, + }, + aliases: ["gpt4-turbo", "gpt-4-turbo-preview"], + deprecated: true, + isLocal: false, + releaseDate: "2024-04-09", + category: "general", + }, + + [OpenAIModels.GPT_3_5_TURBO]: { + id: OpenAIModels.GPT_3_5_TURBO, + name: "GPT-3.5 Turbo", + provider: AIProviderName.OPENAI, + description: "Fast and cost-effective model for simpler tasks (legacy)", + capabilities: { + vision: false, + functionCalling: true, + codeGeneration: true, + reasoning: false, + multimodal: false, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.0005, + outputCostPer1K: 0.0015, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "medium", + accuracy: "medium", + }, + limits: { + maxContextTokens: 16385, + maxOutputTokens: 4096, + maxRequestsPerMinute: 3500, + }, + useCases: { + coding: 6, + creative: 6, + analysis: 6, + conversation: 7, + reasoning: 5, + translation: 7, + summarization: 7, + }, + aliases: ["gpt35", "gpt-3.5", "chatgpt"], + deprecated: true, + isLocal: false, + releaseDate: "2023-03-01", + category: "general", + }, + // Google AI Studio Models [GoogleAIModels.GEMINI_2_5_PRO]: { id: GoogleAIModels.GEMINI_2_5_PRO, name: "Gemini 2.5 Pro", provider: AIProviderName.GOOGLE_AI, description: - "Google's most capable multimodal model with large context window", + "Google's most capable multimodal model with large context window", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.00125, + outputCostPer1K: 0.005, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 2097152, // 2M tokens + maxOutputTokens: 8192, + maxRequestsPerMinute: 360, + }, + useCases: { + coding: 9, + creative: 8, + analysis: 10, + conversation: 8, + reasoning: 9, + translation: 9, + summarization: 9, + }, + aliases: ["gemini-pro", "google-flagship", "best-analysis"], + deprecated: false, + isLocal: false, // Cloud-based model + releaseDate: "2024-12-11", + category: "reasoning", + }, + + [GoogleAIModels.GEMINI_2_5_FLASH]: { + id: GoogleAIModels.GEMINI_2_5_FLASH, + name: "Gemini 2.5 Flash", + provider: AIProviderName.GOOGLE_AI, + description: "Fast and efficient multimodal model with large context", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.000075, + outputCostPer1K: 0.0003, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 1048576, // 1M tokens + maxOutputTokens: 8192, + maxRequestsPerMinute: 1000, + }, + useCases: { + coding: 8, + creative: 7, + analysis: 9, + conversation: 8, + reasoning: 8, + translation: 8, + summarization: 9, + }, + aliases: ["gemini-flash", "google-fast", "best-value"], + deprecated: false, + isLocal: false, // Cloud-based model + releaseDate: "2024-12-11", + category: "general", + }, + + // Anthropic Models + [AnthropicModels.CLAUDE_OPUS_4_5]: { + id: AnthropicModels.CLAUDE_OPUS_4_5, + name: "Claude Opus 4.5", + provider: AIProviderName.ANTHROPIC, + description: + "Anthropic's most capable model with exceptional reasoning, coding, and multimodal capabilities", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: false, + }, + pricing: { + inputCostPer1K: 0.015, + outputCostPer1K: 0.075, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 200000, + maxOutputTokens: 64000, + maxRequestsPerMinute: 50, + }, + useCases: { + coding: 10, + creative: 10, + analysis: 10, + conversation: 9, + reasoning: 10, + translation: 9, + summarization: 9, + }, + aliases: [ + "claude-4.5-opus", + "claude-opus-latest", + "opus-4.5", + "anthropic-flagship", + ], + deprecated: false, + isLocal: false, + releaseDate: "2025-11-24", + category: "reasoning", + }, + + [AnthropicModels.CLAUDE_SONNET_4_5]: { + id: AnthropicModels.CLAUDE_SONNET_4_5, + name: "Claude Sonnet 4.5", + provider: AIProviderName.ANTHROPIC, + description: + "Balanced Claude model with excellent performance across all tasks including vision and reasoning", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: false, + }, + pricing: { + inputCostPer1K: 0.003, + outputCostPer1K: 0.015, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 200000, + maxOutputTokens: 64000, + maxRequestsPerMinute: 100, + }, + useCases: { + coding: 10, + creative: 9, + analysis: 9, + conversation: 9, + reasoning: 10, + translation: 8, + summarization: 8, + }, + aliases: ["claude-4.5-sonnet", "claude-sonnet-latest", "sonnet-4.5"], + deprecated: false, + isLocal: false, + releaseDate: "2025-09-29", + category: "coding", + }, + + [AnthropicModels.CLAUDE_4_5_HAIKU]: { + id: AnthropicModels.CLAUDE_4_5_HAIKU, + name: "Claude 4.5 Haiku", + provider: AIProviderName.ANTHROPIC, + description: "Latest fast and efficient Claude model with vision support", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: false, + }, + pricing: { + inputCostPer1K: 0.001, + outputCostPer1K: 0.005, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 200000, + maxOutputTokens: 64000, + maxRequestsPerMinute: 100, + }, + useCases: { + coding: 8, + creative: 8, + analysis: 8, + conversation: 9, + reasoning: 8, + translation: 8, + summarization: 9, + }, + aliases: ["claude-4.5-haiku", "claude-haiku-latest", "haiku-4.5"], + deprecated: false, + isLocal: false, + releaseDate: "2025-10-15", + category: "general", + }, + + [AnthropicModels.CLAUDE_3_5_SONNET]: { + id: AnthropicModels.CLAUDE_3_5_SONNET, + name: "Claude 3.5 Sonnet", + provider: AIProviderName.ANTHROPIC, + description: + "Anthropic's most capable model with excellent reasoning and coding", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: false, + }, + pricing: { + inputCostPer1K: 0.003, + outputCostPer1K: 0.015, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 200000, + maxOutputTokens: 8192, + maxRequestsPerMinute: 50, + }, + useCases: { + coding: 10, + creative: 9, + analysis: 9, + conversation: 9, + reasoning: 10, + translation: 8, + summarization: 8, + }, + aliases: [ + "claude-3.5-sonnet", + "claude-sonnet", + "best-coding", + "claude-latest", + ], + deprecated: false, + isLocal: false, // Cloud-based model + releaseDate: "2024-10-22", + category: "coding", + }, + + [AnthropicModels.CLAUDE_3_5_HAIKU]: { + id: AnthropicModels.CLAUDE_3_5_HAIKU, + name: "Claude 3.5 Haiku", + provider: AIProviderName.ANTHROPIC, + description: "Fast and efficient Claude model for quick tasks", + capabilities: { + vision: false, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: false, + streaming: true, + jsonMode: false, + }, + pricing: { + inputCostPer1K: 0.001, + outputCostPer1K: 0.005, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 200000, + maxOutputTokens: 8192, + maxRequestsPerMinute: 100, + }, + useCases: { + coding: 8, + creative: 7, + analysis: 8, + conversation: 8, + reasoning: 8, + translation: 8, + summarization: 9, + }, + aliases: ["claude-3.5-haiku", "claude-haiku", "claude-fast"], + deprecated: false, + isLocal: false, // Cloud-based model + releaseDate: "2024-10-22", + category: "general", + }, + + // Mistral Models + [MistralModels.MISTRAL_LARGE_LATEST]: { + id: MistralModels.MISTRAL_LARGE_LATEST, + name: "Mistral Large", + provider: AIProviderName.MISTRAL, + description: + "Mistral's flagship model with excellent reasoning and multilingual capabilities", + capabilities: { + vision: false, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: false, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.002, + outputCostPer1K: 0.006, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 131072, + maxOutputTokens: 8192, + maxRequestsPerMinute: 100, + }, + useCases: { + coding: 9, + creative: 8, + analysis: 9, + conversation: 8, + reasoning: 9, + translation: 9, + summarization: 8, + }, + aliases: ["mistral-large", "mistral-flagship"], + deprecated: false, + isLocal: false, + releaseDate: "2025-12-01", + category: "reasoning", + }, + + [MistralModels.MISTRAL_SMALL_LATEST]: { + id: MistralModels.MISTRAL_SMALL_LATEST, + name: "Mistral Small", + provider: AIProviderName.MISTRAL, + description: + "Efficient model for simple tasks and cost-sensitive applications", + capabilities: { + vision: false, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: false, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.001, + outputCostPer1K: 0.003, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "medium", + accuracy: "medium", + }, + limits: { + maxContextTokens: 32768, + maxOutputTokens: 8192, + maxRequestsPerMinute: 200, + }, + useCases: { + coding: 6, + creative: 6, + analysis: 7, + conversation: 7, + reasoning: 6, + translation: 7, + summarization: 7, + }, + aliases: ["mistral-small", "mistral-cheap"], + deprecated: false, + isLocal: false, + releaseDate: "2024-02-26", + category: "general", + }, + + [MistralModels.CODESTRAL_LATEST]: { + id: MistralModels.CODESTRAL_LATEST, + name: "Codestral", + provider: AIProviderName.MISTRAL, + description: + "Specialized code generation model trained on 80+ programming languages", + capabilities: { + vision: false, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: false, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.001, + outputCostPer1K: 0.003, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 32768, + maxOutputTokens: 8192, + maxRequestsPerMinute: 200, + }, + useCases: { + coding: 10, + creative: 5, + analysis: 7, + conversation: 5, + reasoning: 8, + translation: 5, + summarization: 6, + }, + aliases: ["codestral", "mistral-code"], + deprecated: false, + isLocal: false, + releaseDate: "2024-05-29", + category: "coding", + }, + + [MistralModels.PIXTRAL_LARGE]: { + id: MistralModels.PIXTRAL_LARGE, + name: "Pixtral Large", + provider: AIProviderName.MISTRAL, + description: "Multimodal vision-language model for image understanding", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.002, + outputCostPer1K: 0.006, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 131072, + maxOutputTokens: 8192, + maxRequestsPerMinute: 100, + }, + useCases: { + coding: 8, + creative: 8, + analysis: 9, + conversation: 7, + reasoning: 8, + translation: 7, + summarization: 8, + }, + aliases: ["pixtral", "mistral-vision"], + deprecated: false, + isLocal: false, + releaseDate: "2024-09-01", + category: "vision", + }, + + // Ollama Models (local) + [OllamaModels.LLAMA4_LATEST]: { + id: OllamaModels.LLAMA4_LATEST, + name: "Llama 4", + provider: AIProviderName.OLLAMA, + description: + "Latest Llama 4 with multimodal vision and tool capabilities, runs locally", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0, + outputCostPer1K: 0, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 131072, + maxOutputTokens: 8192, + }, + useCases: { + coding: 9, + creative: 8, + analysis: 9, + conversation: 8, + reasoning: 9, + translation: 8, + summarization: 8, + }, + aliases: ["llama4", "llama4-local"], + deprecated: false, + isLocal: true, + releaseDate: "2025-04-01", + category: "reasoning", + }, + + [OllamaModels.LLAMA3_3_LATEST]: { + id: OllamaModels.LLAMA3_3_LATEST, + name: "Llama 3.3", + provider: AIProviderName.OLLAMA, + description: "High-performance Llama 3.3 for local inference", + capabilities: { + vision: false, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: false, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0, + outputCostPer1K: 0, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 131072, + maxOutputTokens: 8192, + }, + useCases: { + coding: 8, + creative: 8, + analysis: 8, + conversation: 8, + reasoning: 8, + translation: 8, + summarization: 8, + }, + aliases: ["llama3.3", "llama3.3-local"], + deprecated: false, + isLocal: true, + releaseDate: "2024-12-01", + category: "general", + }, + + [OllamaModels.LLAMA3_2_LATEST]: { + id: OllamaModels.LLAMA3_2_LATEST, + name: "Llama 3.2 Latest", + provider: AIProviderName.OLLAMA, + description: "Local Llama model for private, offline AI generation", + capabilities: { + vision: false, + functionCalling: false, + codeGeneration: true, + reasoning: true, + multimodal: false, + streaming: true, + jsonMode: false, + }, + pricing: { + inputCostPer1K: 0, + outputCostPer1K: 0, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "medium", + accuracy: "medium", + }, + limits: { + maxContextTokens: 131072, + maxOutputTokens: 8192, + }, + useCases: { + coding: 6, + creative: 7, + analysis: 6, + conversation: 7, + reasoning: 6, + translation: 6, + summarization: 6, + }, + aliases: ["llama3.2", "llama", "local", "offline"], + deprecated: false, + isLocal: true, + releaseDate: "2024-09-25", + category: "general", + }, + + [OllamaModels.DEEPSEEK_R1_70B]: { + id: OllamaModels.DEEPSEEK_R1_70B, + name: "DeepSeek-R1 70B", + provider: AIProviderName.OLLAMA, + description: + "State-of-the-art reasoning model rivaling OpenAI O1, runs locally", + capabilities: { + vision: false, + functionCalling: false, + codeGeneration: true, + reasoning: true, + multimodal: false, + streaming: true, + jsonMode: false, + }, + pricing: { + inputCostPer1K: 0, + outputCostPer1K: 0, + currency: "USD", + }, + performance: { + speed: "slow", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 65536, + maxOutputTokens: 8192, + }, + useCases: { + coding: 10, + creative: 7, + analysis: 10, + conversation: 6, + reasoning: 10, + translation: 7, + summarization: 7, + }, + aliases: ["deepseek-r1", "deepseek-reasoning", "local-reasoning"], + deprecated: false, + isLocal: true, + releaseDate: "2025-01-20", + category: "reasoning", + }, + + [OllamaModels.QWEN3_72B]: { + id: OllamaModels.QWEN3_72B, + name: "Qwen 3 72B", + provider: AIProviderName.OLLAMA, + description: "Advanced reasoning and multilingual model from Alibaba", + capabilities: { + vision: false, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: false, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0, + outputCostPer1K: 0, + currency: "USD", + }, + performance: { + speed: "slow", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 131072, + maxOutputTokens: 8192, + }, + useCases: { + coding: 9, + creative: 8, + analysis: 9, + conversation: 8, + reasoning: 9, + translation: 9, + summarization: 8, + }, + aliases: ["qwen3", "qwen3-72b-local"], + deprecated: false, + isLocal: true, + releaseDate: "2025-04-01", + category: "reasoning", + }, + + [OllamaModels.MISTRAL_LARGE_LATEST]: { + id: OllamaModels.MISTRAL_LARGE_LATEST, + name: "Mistral Large (Local)", + provider: AIProviderName.OLLAMA, + description: "Mistral Large model for local inference", + capabilities: { + vision: false, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: false, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0, + outputCostPer1K: 0, + currency: "USD", + }, + performance: { + speed: "slow", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 131072, + maxOutputTokens: 8192, + }, + useCases: { + coding: 8, + creative: 8, + analysis: 8, + conversation: 8, + reasoning: 8, + translation: 9, + summarization: 8, + }, + aliases: ["mistral-large-local"], + deprecated: false, + isLocal: true, + releaseDate: "2024-02-26", + category: "general", + }, + + // Bedrock Models + [BedrockModels.NOVA_PREMIER]: { + id: BedrockModels.NOVA_PREMIER, + name: "Amazon Nova Premier", + provider: AIProviderName.BEDROCK, + description: + "Amazon's most capable foundation model with advanced multimodal capabilities", capabilities: { vision: true, functionCalling: true, @@ -126,8 +1783,8 @@ export const MODEL_REGISTRY: Record = { jsonMode: true, }, pricing: { - inputCostPer1K: 0.00125, - outputCostPer1K: 0.005, + inputCostPer1K: 0.0025, + outputCostPer1K: 0.0125, currency: "USD", }, performance: { @@ -136,31 +1793,168 @@ export const MODEL_REGISTRY: Record = { accuracy: "high", }, limits: { - maxContextTokens: 2097152, // 2M tokens - maxOutputTokens: 8192, - maxRequestsPerMinute: 360, + maxContextTokens: 300000, + maxOutputTokens: 5000, + maxRequestsPerMinute: 100, }, useCases: { coding: 9, - creative: 8, + creative: 9, analysis: 10, conversation: 8, reasoning: 9, + translation: 8, + summarization: 9, + }, + aliases: ["nova-premier", "aws-flagship"], + deprecated: false, + isLocal: false, + releaseDate: "2025-01-01", + category: "reasoning", + }, + + [BedrockModels.NOVA_PRO]: { + id: BedrockModels.NOVA_PRO, + name: "Amazon Nova Pro", + provider: AIProviderName.BEDROCK, + description: "Highly capable multimodal model balancing accuracy and speed", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.0008, + outputCostPer1K: 0.0032, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 300000, + maxOutputTokens: 5000, + maxRequestsPerMinute: 200, + }, + useCases: { + coding: 8, + creative: 8, + analysis: 9, + conversation: 8, + reasoning: 8, + translation: 8, + summarization: 9, + }, + aliases: ["nova-pro", "aws-balanced"], + deprecated: false, + isLocal: false, + releaseDate: "2024-12-03", + category: "general", + }, + + [BedrockModels.NOVA_LITE]: { + id: BedrockModels.NOVA_LITE, + name: "Amazon Nova Lite", + provider: AIProviderName.BEDROCK, + description: + "Fast and cost-effective multimodal model optimized for everyday tasks", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: true, + }, + pricing: { + inputCostPer1K: 0.00006, + outputCostPer1K: 0.00024, + currency: "USD", + }, + performance: { + speed: "fast", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 300000, + maxOutputTokens: 5000, + maxRequestsPerMinute: 500, + }, + useCases: { + coding: 7, + creative: 7, + analysis: 8, + conversation: 8, + reasoning: 7, + translation: 8, + summarization: 9, + }, + aliases: ["nova-lite", "aws-lite", "aws-cheap"], + deprecated: false, + isLocal: false, + releaseDate: "2024-12-03", + category: "general", + }, + + [BedrockModels.CLAUDE_4_5_OPUS]: { + id: BedrockModels.CLAUDE_4_5_OPUS, + name: "Claude 4.5 Opus (Bedrock)", + provider: AIProviderName.BEDROCK, + description: + "Anthropic's most capable model available on Bedrock for enterprise workloads", + capabilities: { + vision: true, + functionCalling: true, + codeGeneration: true, + reasoning: true, + multimodal: true, + streaming: true, + jsonMode: false, + }, + pricing: { + inputCostPer1K: 0.015, + outputCostPer1K: 0.075, + currency: "USD", + }, + performance: { + speed: "medium", + quality: "high", + accuracy: "high", + }, + limits: { + maxContextTokens: 200000, + maxOutputTokens: 64000, + maxRequestsPerMinute: 50, + }, + useCases: { + coding: 10, + creative: 10, + analysis: 10, + conversation: 9, + reasoning: 10, translation: 9, summarization: 9, }, - aliases: ["gemini-pro", "google-flagship", "best-analysis"], + aliases: ["bedrock-claude-4.5-opus", "bedrock-claude-flagship"], deprecated: false, - isLocal: false, // Cloud-based model - releaseDate: "2024-12-11", + isLocal: false, + releaseDate: "2025-11-24", category: "reasoning", }, - [GoogleAIModels.GEMINI_2_5_FLASH]: { - id: GoogleAIModels.GEMINI_2_5_FLASH, - name: "Gemini 2.5 Flash", - provider: AIProviderName.GOOGLE_AI, - description: "Fast and efficient multimodal model with large context", + [BedrockModels.LLAMA_4_MAVERICK_17B]: { + id: BedrockModels.LLAMA_4_MAVERICK_17B, + name: "Llama 4 Maverick (Bedrock)", + provider: AIProviderName.BEDROCK, + description: "Meta's latest Llama 4 model with vision on Bedrock", capabilities: { vision: true, functionCalling: true, @@ -171,8 +1965,8 @@ export const MODEL_REGISTRY: Record = { jsonMode: true, }, pricing: { - inputCostPer1K: 0.000075, - outputCostPer1K: 0.0003, + inputCostPer1K: 0.00019, + outputCostPer1K: 0.00055, currency: "USD", }, performance: { @@ -181,33 +1975,33 @@ export const MODEL_REGISTRY: Record = { accuracy: "high", }, limits: { - maxContextTokens: 1048576, // 1M tokens + maxContextTokens: 131072, maxOutputTokens: 8192, - maxRequestsPerMinute: 1000, + maxRequestsPerMinute: 200, }, useCases: { coding: 8, - creative: 7, - analysis: 9, + creative: 8, + analysis: 8, conversation: 8, reasoning: 8, translation: 8, - summarization: 9, + summarization: 8, }, - aliases: ["gemini-flash", "google-fast", "best-value"], + aliases: ["bedrock-llama4", "bedrock-llama-maverick"], deprecated: false, - isLocal: false, // Cloud-based model - releaseDate: "2024-12-11", + isLocal: false, + releaseDate: "2025-04-01", category: "general", }, - // Anthropic Models - [AnthropicModels.CLAUDE_OPUS_4_5]: { - id: AnthropicModels.CLAUDE_OPUS_4_5, - name: "Claude Opus 4.5", - provider: AIProviderName.ANTHROPIC, + // Azure OpenAI GPT-5.1 Series (Latest - December 2025) + [AzureOpenAIModels.GPT_5_1]: { + id: AzureOpenAIModels.GPT_5_1, + name: "GPT-5.1 (Azure)", + provider: AIProviderName.AZURE, description: - "Anthropic's most capable model with exceptional reasoning, coding, and multimodal capabilities", + "Azure's latest GPT-5.1 flagship model with enhanced reasoning and multimodal capabilities", capabilities: { vision: true, functionCalling: true, @@ -215,11 +2009,11 @@ export const MODEL_REGISTRY: Record = { reasoning: true, multimodal: true, streaming: true, - jsonMode: false, + jsonMode: true, }, pricing: { inputCostPer1K: 0.015, - outputCostPer1K: 0.075, + outputCostPer1K: 0.045, currency: "USD", }, performance: { @@ -228,37 +2022,31 @@ export const MODEL_REGISTRY: Record = { accuracy: "high", }, limits: { - maxContextTokens: 200000, + maxContextTokens: 300000, maxOutputTokens: 64000, - maxRequestsPerMinute: 50, + maxRequestsPerMinute: 100, }, useCases: { coding: 10, creative: 10, analysis: 10, - conversation: 9, + conversation: 10, reasoning: 10, translation: 9, summarization: 9, }, - aliases: [ - "claude-4.5-opus", - "claude-opus-latest", - "opus-4.5", - "anthropic-flagship", - ], + aliases: ["azure-gpt-5.1", "gpt51-azure", "azure-flagship"], deprecated: false, isLocal: false, - releaseDate: "2025-11-24", + releaseDate: "2025-12-01", category: "reasoning", }, - [AnthropicModels.CLAUDE_SONNET_4_5]: { - id: AnthropicModels.CLAUDE_SONNET_4_5, - name: "Claude Sonnet 4.5", - provider: AIProviderName.ANTHROPIC, - description: - "Balanced Claude model with excellent performance across all tasks including vision and reasoning", + [AzureOpenAIModels.GPT_5_1_CHAT]: { + id: AzureOpenAIModels.GPT_5_1_CHAT, + name: "GPT-5.1 Chat (Azure)", + provider: AIProviderName.AZURE, + description: "Azure GPT-5.1 optimized for conversational interactions", capabilities: { vision: true, functionCalling: true, @@ -266,44 +2054,45 @@ export const MODEL_REGISTRY: Record = { reasoning: true, multimodal: true, streaming: true, - jsonMode: false, + jsonMode: true, }, pricing: { - inputCostPer1K: 0.003, - outputCostPer1K: 0.015, + inputCostPer1K: 0.012, + outputCostPer1K: 0.036, currency: "USD", }, performance: { - speed: "medium", + speed: "fast", quality: "high", accuracy: "high", }, limits: { - maxContextTokens: 200000, - maxOutputTokens: 64000, - maxRequestsPerMinute: 100, + maxContextTokens: 300000, + maxOutputTokens: 32000, + maxRequestsPerMinute: 150, }, useCases: { - coding: 10, + coding: 8, creative: 9, analysis: 9, - conversation: 9, - reasoning: 10, - translation: 8, - summarization: 8, + conversation: 10, + reasoning: 9, + translation: 9, + summarization: 9, }, - aliases: ["claude-4.5-sonnet", "claude-sonnet-latest", "sonnet-4.5"], + aliases: ["azure-gpt-5.1-chat", "gpt51-chat-azure"], deprecated: false, isLocal: false, - releaseDate: "2025-09-29", - category: "coding", + releaseDate: "2025-12-01", + category: "general", }, - [AnthropicModels.CLAUDE_4_5_HAIKU]: { - id: AnthropicModels.CLAUDE_4_5_HAIKU, - name: "Claude 4.5 Haiku", - provider: AIProviderName.ANTHROPIC, - description: "Latest fast and efficient Claude model with vision support", + [AzureOpenAIModels.GPT_5_1_CODEX]: { + id: AzureOpenAIModels.GPT_5_1_CODEX, + name: "GPT-5.1 Codex (Azure)", + provider: AIProviderName.AZURE, + description: + "Azure GPT-5.1 specialized for code generation and software development", capabilities: { vision: true, functionCalling: true, @@ -311,45 +2100,45 @@ export const MODEL_REGISTRY: Record = { reasoning: true, multimodal: true, streaming: true, - jsonMode: false, + jsonMode: true, }, pricing: { - inputCostPer1K: 0.001, - outputCostPer1K: 0.005, + inputCostPer1K: 0.012, + outputCostPer1K: 0.036, currency: "USD", }, performance: { - speed: "fast", + speed: "medium", quality: "high", accuracy: "high", }, limits: { - maxContextTokens: 200000, + maxContextTokens: 300000, maxOutputTokens: 64000, maxRequestsPerMinute: 100, }, useCases: { - coding: 8, - creative: 8, - analysis: 8, - conversation: 9, - reasoning: 8, - translation: 8, - summarization: 9, + coding: 10, + creative: 7, + analysis: 9, + conversation: 7, + reasoning: 10, + translation: 7, + summarization: 8, }, - aliases: ["claude-4.5-haiku", "claude-haiku-latest", "haiku-4.5"], + aliases: ["azure-gpt-5.1-codex", "gpt51-codex-azure", "azure-code"], deprecated: false, isLocal: false, - releaseDate: "2025-10-15", - category: "general", + releaseDate: "2025-12-01", + category: "coding", }, - [AnthropicModels.CLAUDE_3_5_SONNET]: { - id: AnthropicModels.CLAUDE_3_5_SONNET, - name: "Claude 3.5 Sonnet", - provider: AIProviderName.ANTHROPIC, + [AzureOpenAIModels.GPT_5_1_CODEX_MINI]: { + id: AzureOpenAIModels.GPT_5_1_CODEX_MINI, + name: "GPT-5.1 Codex Mini (Azure)", + provider: AIProviderName.AZURE, description: - "Anthropic's most capable model with excellent reasoning and coding", + "Fast and efficient Azure code model for quick development tasks", capabilities: { vision: true, functionCalling: true, @@ -357,180 +2146,180 @@ export const MODEL_REGISTRY: Record = { reasoning: true, multimodal: true, streaming: true, - jsonMode: false, + jsonMode: true, }, pricing: { inputCostPer1K: 0.003, - outputCostPer1K: 0.015, + outputCostPer1K: 0.009, currency: "USD", }, performance: { - speed: "medium", + speed: "fast", quality: "high", accuracy: "high", }, limits: { maxContextTokens: 200000, - maxOutputTokens: 8192, - maxRequestsPerMinute: 50, + maxOutputTokens: 32000, + maxRequestsPerMinute: 300, }, useCases: { - coding: 10, - creative: 9, - analysis: 9, - conversation: 9, - reasoning: 10, - translation: 8, - summarization: 8, + coding: 9, + creative: 6, + analysis: 8, + conversation: 7, + reasoning: 8, + translation: 6, + summarization: 7, }, - aliases: [ - "claude-3.5-sonnet", - "claude-sonnet", - "best-coding", - "claude-latest", - ], + aliases: ["azure-gpt-5.1-codex-mini", "gpt51-codex-mini-azure"], deprecated: false, - isLocal: false, // Cloud-based model - releaseDate: "2024-10-22", + isLocal: false, + releaseDate: "2025-12-01", category: "coding", }, - [AnthropicModels.CLAUDE_3_5_HAIKU]: { - id: AnthropicModels.CLAUDE_3_5_HAIKU, - name: "Claude 3.5 Haiku", - provider: AIProviderName.ANTHROPIC, - description: "Fast and efficient Claude model for quick tasks", + [AzureOpenAIModels.GPT_5_1_CODEX_MAX]: { + id: AzureOpenAIModels.GPT_5_1_CODEX_MAX, + name: "GPT-5.1 Codex Max (Azure)", + provider: AIProviderName.AZURE, + description: + "Azure's most powerful code model for complex enterprise development", capabilities: { - vision: false, + vision: true, functionCalling: true, codeGeneration: true, reasoning: true, - multimodal: false, + multimodal: true, streaming: true, - jsonMode: false, + jsonMode: true, }, pricing: { - inputCostPer1K: 0.001, - outputCostPer1K: 0.005, + inputCostPer1K: 0.025, + outputCostPer1K: 0.075, currency: "USD", }, performance: { - speed: "fast", + speed: "slow", quality: "high", accuracy: "high", }, limits: { - maxContextTokens: 200000, - maxOutputTokens: 8192, - maxRequestsPerMinute: 100, + maxContextTokens: 500000, + maxOutputTokens: 128000, + maxRequestsPerMinute: 50, }, useCases: { - coding: 8, - creative: 7, - analysis: 8, - conversation: 8, - reasoning: 8, - translation: 8, - summarization: 9, + coding: 10, + creative: 8, + analysis: 10, + conversation: 7, + reasoning: 10, + translation: 7, + summarization: 8, }, - aliases: ["claude-3.5-haiku", "claude-haiku", "claude-fast"], + aliases: [ + "azure-gpt-5.1-codex-max", + "gpt51-codex-max-azure", + "azure-enterprise", + ], deprecated: false, - isLocal: false, // Cloud-based model - releaseDate: "2024-10-22", - category: "general", + isLocal: false, + releaseDate: "2025-12-01", + category: "coding", }, - // Mistral Models - "mistral-small-latest": { - id: "mistral-small-latest", - name: "Mistral Small", - provider: AIProviderName.MISTRAL, - description: - "Efficient model for simple tasks and cost-sensitive applications", + // Azure OpenAI GPT-5.0 Series (Azure-unique variants only) + [AzureOpenAIModels.GPT_5_PRO]: { + id: AzureOpenAIModels.GPT_5_PRO, + name: "GPT-5 Pro (Azure)", + provider: AIProviderName.AZURE, + description: "Azure GPT-5 Pro with enhanced enterprise features", capabilities: { - vision: false, + vision: true, functionCalling: true, codeGeneration: true, reasoning: true, - multimodal: false, + multimodal: true, streaming: true, jsonMode: true, }, pricing: { - inputCostPer1K: 0.001, - outputCostPer1K: 0.003, + inputCostPer1K: 0.02, + outputCostPer1K: 0.06, currency: "USD", }, performance: { - speed: "fast", - quality: "medium", - accuracy: "medium", + speed: "medium", + quality: "high", + accuracy: "high", }, limits: { - maxContextTokens: 32768, - maxOutputTokens: 8192, - maxRequestsPerMinute: 200, + maxContextTokens: 256000, + maxOutputTokens: 64000, + maxRequestsPerMinute: 100, }, useCases: { - coding: 6, - creative: 6, - analysis: 7, - conversation: 7, - reasoning: 6, - translation: 7, - summarization: 7, + coding: 10, + creative: 10, + analysis: 10, + conversation: 9, + reasoning: 10, + translation: 9, + summarization: 9, }, - aliases: ["mistral-small", "mistral-cheap"], + aliases: ["azure-gpt-5-pro", "gpt5-pro-azure"], deprecated: false, - isLocal: false, // Cloud-based model - releaseDate: "2024-02-26", - category: "general", + isLocal: false, + releaseDate: "2025-08-07", + category: "reasoning", }, - // Ollama Models (local) - "llama3.2:latest": { - id: "llama3.2:latest", - name: "Llama 3.2 Latest", - provider: AIProviderName.OLLAMA, - description: "Local Llama model for private, offline AI generation", + [AzureOpenAIModels.GPT_5_TURBO]: { + id: AzureOpenAIModels.GPT_5_TURBO, + name: "GPT-5 Turbo (Azure)", + provider: AIProviderName.AZURE, + description: "Azure GPT-5 Turbo optimized for fast responses", capabilities: { - vision: false, - functionCalling: false, + vision: true, + functionCalling: true, codeGeneration: true, reasoning: true, - multimodal: false, + multimodal: true, streaming: true, - jsonMode: false, + jsonMode: true, }, pricing: { - inputCostPer1K: 0, // Local execution - outputCostPer1K: 0, + inputCostPer1K: 0.008, + outputCostPer1K: 0.024, currency: "USD", }, performance: { - speed: "slow", // Depends on hardware - quality: "medium", - accuracy: "medium", + speed: "fast", + quality: "high", + accuracy: "high", }, limits: { - maxContextTokens: 4096, - maxOutputTokens: 2048, + maxContextTokens: 200000, + maxOutputTokens: 32768, + maxRequestsPerMinute: 300, }, useCases: { - coding: 6, - creative: 7, - analysis: 6, - conversation: 7, - reasoning: 6, - translation: 6, - summarization: 6, + coding: 9, + creative: 9, + analysis: 9, + conversation: 9, + reasoning: 9, + translation: 9, + summarization: 9, }, - aliases: ["llama3.2", "llama", "local", "offline"], + aliases: ["azure-gpt-5-turbo", "gpt5-turbo-azure"], deprecated: false, - isLocal: true, // Ollama runs locally - releaseDate: "2024-09-25", + isLocal: false, + releaseDate: "2025-08-07", category: "general", }, + // Note: Azure models like O3, O4-mini, GPT-4o share IDs with OpenAI and use the OpenAI registry entries }; /** @@ -550,61 +2339,88 @@ Object.entries(DEFAULT_MODEL_ALIASES).forEach(([k, v]) => { MODEL_ALIASES[k.toLowerCase().replace(/_/g, "-")] = v; }); -MODEL_ALIASES.local = "llama3.2:latest"; +MODEL_ALIASES.local = OllamaModels.LLAMA3_2_LATEST; /** * Use case to model mappings */ export const USE_CASE_RECOMMENDATIONS: Record = { coding: [ - AnthropicModels.CLAUDE_3_5_SONNET, - OpenAIModels.GPT_4O, - GoogleAIModels.GEMINI_2_5_PRO, + OpenAIModels.GPT_5_2_PRO, + AnthropicModels.CLAUDE_OPUS_4_5, + OpenAIModels.GPT_5_2, + MistralModels.CODESTRAL_LATEST, + AnthropicModels.CLAUDE_SONNET_4_5, ], creative: [ - AnthropicModels.CLAUDE_3_5_SONNET, - OpenAIModels.GPT_4O, + OpenAIModels.GPT_5_2, + AnthropicModels.CLAUDE_OPUS_4_5, + OpenAIModels.GPT_5, GoogleAIModels.GEMINI_2_5_PRO, ], analysis: [ + OpenAIModels.GPT_5_2_PRO, GoogleAIModels.GEMINI_2_5_PRO, - AnthropicModels.CLAUDE_3_5_SONNET, - OpenAIModels.GPT_4O, + AnthropicModels.CLAUDE_OPUS_4_5, + OpenAIModels.O3, + BedrockModels.NOVA_PREMIER, ], conversation: [ + OpenAIModels.GPT_5_2_CHAT_LATEST, + OpenAIModels.GPT_5, + AnthropicModels.CLAUDE_SONNET_4_5, OpenAIModels.GPT_4O, - AnthropicModels.CLAUDE_3_5_SONNET, - AnthropicModels.CLAUDE_3_5_HAIKU, ], reasoning: [ - AnthropicModels.CLAUDE_3_5_SONNET, + OpenAIModels.GPT_5_2_PRO, + OpenAIModels.GPT_5_2, + OpenAIModels.O3, + AnthropicModels.CLAUDE_OPUS_4_5, GoogleAIModels.GEMINI_2_5_PRO, - OpenAIModels.GPT_4O, + OllamaModels.DEEPSEEK_R1_70B, ], translation: [ GoogleAIModels.GEMINI_2_5_PRO, - OpenAIModels.GPT_4O, - AnthropicModels.CLAUDE_3_5_HAIKU, + MistralModels.MISTRAL_LARGE_LATEST, + OpenAIModels.GPT_5, ], summarization: [ GoogleAIModels.GEMINI_2_5_FLASH, - OpenAIModels.GPT_4O_MINI, - AnthropicModels.CLAUDE_3_5_HAIKU, + OpenAIModels.GPT_5_MINI, + AnthropicModels.CLAUDE_4_5_HAIKU, ], "cost-effective": [ GoogleAIModels.GEMINI_2_5_FLASH, OpenAIModels.GPT_4O_MINI, - "mistral-small-latest", + MistralModels.MISTRAL_SMALL_LATEST, + BedrockModels.NOVA_LITE, ], "high-quality": [ - AnthropicModels.CLAUDE_3_5_SONNET, - OpenAIModels.GPT_4O, + OpenAIModels.GPT_5_2_PRO, + OpenAIModels.GPT_5_2, + AnthropicModels.CLAUDE_OPUS_4_5, GoogleAIModels.GEMINI_2_5_PRO, ], fast: [ - OpenAIModels.GPT_4O_MINI, + OpenAIModels.GPT_5_2_CHAT_LATEST, + OpenAIModels.GPT_5_MINI, GoogleAIModels.GEMINI_2_5_FLASH, - AnthropicModels.CLAUDE_3_5_HAIKU, + AnthropicModels.CLAUDE_4_5_HAIKU, + OpenAIModels.O3_MINI, + ], + local: [ + OllamaModels.LLAMA4_LATEST, + OllamaModels.DEEPSEEK_R1_70B, + OllamaModels.QWEN3_72B, + OllamaModels.LLAMA3_3_LATEST, + ], + multimodal: [ + OpenAIModels.GPT_5_2, + OpenAIModels.GPT_5_2_PRO, + AnthropicModels.CLAUDE_OPUS_4_5, + GoogleAIModels.GEMINI_2_5_PRO, + MistralModels.PIXTRAL_LARGE, + BedrockModels.NOVA_PREMIER, ], }; diff --git a/test/unit/cli/video-flags.test.ts b/test/unit/cli/video-flags.test.ts index 417bfdc29..f1fb6dc8d 100644 --- a/test/unit/cli/video-flags.test.ts +++ b/test/unit/cli/video-flags.test.ts @@ -3,7 +3,7 @@ import { describe, it, expect } from "vitest"; /** * Test suite for video CLI flags * These tests verify the expected configuration and behavior of video flags. - * + * * Note: The actual CLI flags are defined in commandFactory.ts commonOptions (private), * so these tests document the expected values and validate the configuration matches * the implementation. For full integration testing, run CLI commands with video flags. @@ -20,10 +20,10 @@ describe("Video CLI Flags Configuration", () => { // These flag names should be available in commandFactory.ts commonOptions // Using kebab-case format as defined in CLI const expectedVideoFlags = [ - "video", // Path to video file - "video-frames", // Number of frames to extract (default: 8) - "video-quality", // Frame quality 0-100 (default: 85) - "video-format", // Frame format (jpeg|png, default: jpeg) + "video", // Path to video file + "video-frames", // Number of frames to extract (default: 8) + "video-quality", // Frame quality 0-100 (default: 85) + "video-format", // Frame format (jpeg|png, default: jpeg) "transcribe-audio", // Extract and transcribe audio from video ]; @@ -39,9 +39,9 @@ describe("Video CLI Flags Configuration", () => { // Yargs converts kebab-case CLI flags to camelCase for argv access const yargsPropertyNames = [ "video", - "videoFrames", // --video-frames - "videoQuality", // --video-quality - "videoFormat", // --video-format + "videoFrames", // --video-frames + "videoQuality", // --video-quality + "videoFormat", // --video-format "transcribeAudio", // --transcribe-audio ];