diff --git a/package.json b/package.json index b8741d9ec..0a82c9c37 100644 --- a/package.json +++ b/package.json @@ -108,6 +108,7 @@ "test:error-classification-e2e": "pnpm exec tsx test/continuous-test-suite-error-classification-e2e.ts", "test:error-classifier-contract": "pnpm exec tsx test/continuous-test-suite-error-classifier-contract.ts", "test:bedrock-inference-profile": "pnpm exec tsx test/continuous-test-suite-bedrock-inference-profile.ts", + "test:retired-model-defaults": "pnpm exec tsx test/continuous-test-suite-retired-model-defaults.ts", "test:loop-engine": "pnpm exec tsx test/continuous-test-suite-loop-engine.ts", "test:middleware": "pnpm exec tsx test/continuous-test-suite-middleware.ts", "test:stream-middleware": "pnpm exec tsx test/continuous-test-suite-stream-middleware.ts", @@ -189,7 +190,7 @@ "// CI tier — fast, no live AI calls, safe for every commit (test:unit; also see the separate provider-safety-net CI job, which runs build + test:providers-mocked + test:provider-structure + test:error-classifier-contract on every PR)": "", "test:tool-routing": "pnpm exec tsx test/continuous-test-suite-tool-routing.ts", "test:tool-routing-semantic": "pnpm exec tsx test/continuous-test-suite-tool-routing-semantic.ts", - "test:unit": "pnpm run test:bugfixes && pnpm run test:mcp:infra && pnpm run test:mcp:spans && pnpm run test:tool-routing && pnpm run test:tool-routing-cli && pnpm run test:tool-dedup && pnpm run test:model-pool && pnpm run test:classifier-router && pnpm run test:tool-routing-semantic && pnpm run test:mcp-result-cache && pnpm run test:mcp-direct-name-repair && pnpm run test:mcp-breaker-resolved-errors && pnpm run test:model-not-found-retryable && pnpm run test:archive:security && pnpm run test:office:security && pnpm run test:image-exif && pnpm run test:vector-chroma && pnpm run test:vector-pgvector && pnpm run test:vector-pinecone && pnpm run test:provider-wiring && pnpm run test:docs-mcp", + "test:unit": "pnpm run test:bugfixes && pnpm run test:mcp:infra && pnpm run test:mcp:spans && pnpm run test:tool-routing && pnpm run test:tool-routing-cli && pnpm run test:tool-dedup && pnpm run test:model-pool && pnpm run test:classifier-router && pnpm run test:tool-routing-semantic && pnpm run test:mcp-result-cache && pnpm run test:mcp-direct-name-repair && pnpm run test:mcp-breaker-resolved-errors && pnpm run test:model-not-found-retryable && pnpm run test:archive:security && pnpm run test:office:security && pnpm run test:image-exif && pnpm run test:vector-chroma && pnpm run test:vector-pgvector && pnpm run test:vector-pinecone && pnpm run test:provider-wiring && pnpm run test:docs-mcp && pnpm run test:retired-model-defaults", "// CI tier — live providers, runs only when API keys are present (test:credentials and test:dynamic make real provider calls when keys are set, so they live here, not in test:unit; test:matrix, a different suite covering the full provider capability matrix, runs nightly via .github/workflows/live-matrix.yml — test:providers itself is still only wired into test:live, not any GitHub Actions workflow)": "", "test:live": "pnpm run test:providers && pnpm run test:mcp:http && pnpm run test:mcp:sdk && pnpm run test:mcp:cli && pnpm run test:observability && pnpm run test:context && pnpm run test:memory && pnpm run test:tool-reliability && pnpm run test:evaluation && pnpm run test:autoresearch && pnpm run test:credentials && pnpm run test:dynamic", "// CI tier — product output (image/video/TTS/PPT) — costs $$ per run (not wired into any GitHub Actions workflow as of this comment; run manually or add to live-matrix.yml if nightly coverage is needed)": "", diff --git a/src/lib/evaluation/EvaluatorFactory.ts b/src/lib/evaluation/EvaluatorFactory.ts index ef19163b2..d5ea7c67c 100644 --- a/src/lib/evaluation/EvaluatorFactory.ts +++ b/src/lib/evaluation/EvaluatorFactory.ts @@ -67,7 +67,7 @@ export class EvaluatorFactory extends BaseFactory { threshold: 7, evaluationStrategy: "ragas", evaluationModel: - process.env.NEUROLINK_RAGAS_EVALUATION_MODEL || "gemini-1.5-flash", + process.env.NEUROLINK_RAGAS_EVALUATION_MODEL || "gemini-2.5-flash", provider: process.env.NEUROLINK_RAGAS_EVALUATION_PROVIDER || "vertex", ...config, }; @@ -118,7 +118,7 @@ export class EvaluatorFactory extends BaseFactory { threshold: 5, evaluationStrategy: "ragas", evaluationModel: - process.env.NEUROLINK_RAGAS_EVALUATION_MODEL || "gemini-1.5-flash", + process.env.NEUROLINK_RAGAS_EVALUATION_MODEL || "gemini-2.5-flash", provider: process.env.NEUROLINK_RAGAS_EVALUATION_PROVIDER || "vertex", ...config, }; @@ -144,7 +144,7 @@ export class EvaluatorFactory extends BaseFactory { const mergedConfig: EvaluationConfig = { threshold: 6, evaluationStrategy: "ragas", - evaluationModel: "gemini-1.5-flash", + evaluationModel: "gemini-2.5-flash", provider: "vertex", ...config, }; diff --git a/src/lib/evaluation/EvaluatorRegistry.ts b/src/lib/evaluation/EvaluatorRegistry.ts index e629a56b8..a5549fc40 100644 --- a/src/lib/evaluation/EvaluatorRegistry.ts +++ b/src/lib/evaluation/EvaluatorRegistry.ts @@ -113,7 +113,7 @@ export class EvaluatorRegistry extends BaseRegistry< description: "RAGAS-style LLM-as-judge evaluation with relevance, accuracy, and completeness metrics", requiresLLM: true, - defaultModel: "gemini-1.5-flash", + defaultModel: "gemini-2.5-flash", defaultProvider: "vertex", version: "1.0.0", features: [ diff --git a/src/lib/evaluation/ragasEvaluator.ts b/src/lib/evaluation/ragasEvaluator.ts index 441f4a509..f89b537cb 100644 --- a/src/lib/evaluation/ragasEvaluator.ts +++ b/src/lib/evaluation/ragasEvaluator.ts @@ -35,7 +35,7 @@ export class RAGASEvaluator { this.evaluationModel = evaluationModel || process.env.NEUROLINK_RAGAS_EVALUATION_MODEL || - "gemini-1.5-flash"; + "gemini-2.5-flash"; this.providerName = providerName || process.env.NEUROLINK_RAGAS_EVALUATION_PROVIDER || diff --git a/src/lib/middleware/utils/guardrailsUtils.ts b/src/lib/middleware/utils/guardrailsUtils.ts index 998a4460f..098bea8c3 100644 --- a/src/lib/middleware/utils/guardrailsUtils.ts +++ b/src/lib/middleware/utils/guardrailsUtils.ts @@ -160,7 +160,7 @@ export async function performPrecallEvaluation( try { const provider = await AIProviderFactory.createProvider( config.provider || "google-ai", - config.evaluationModel || "gemini-1.5-flash", + config.evaluationModel || "gemini-2.5-flash", ); const evaluationPrompt = diff --git a/src/lib/providers/googleVertex/client.ts b/src/lib/providers/googleVertex/client.ts index 1a8ef6f59..3bd371fcf 100644 --- a/src/lib/providers/googleVertex/client.ts +++ b/src/lib/providers/googleVertex/client.ts @@ -8413,8 +8413,6 @@ export class GoogleVertexProvider extends BaseProvider { "gemini-2.5-flash-lite", "gemini-2.0-flash-001", "gemini-2.0-flash-lite", - "gemini-1.5-pro", - "gemini-1.5-flash", ], claude: [ "claude-sonnet-4-5@20250929", diff --git a/src/lib/providers/litellm/client.ts b/src/lib/providers/litellm/client.ts index 9b5181ca5..c15d388b8 100644 --- a/src/lib/providers/litellm/client.ts +++ b/src/lib/providers/litellm/client.ts @@ -38,7 +38,7 @@ import { OpenAIChatCompletionsProvider } from "../openaiChatCompletionsBase.js"; const streamTracer = trace.getTracer("neurolink.provider.litellm"); -const FALLBACK_LITELLM_MODEL = "openai/gpt-4o-mini"; +const FALLBACK_LITELLM_MODEL = "openai/gpt-5.4-mini"; const getLiteLLMConfig = () => ({ baseURL: process.env.LITELLM_BASE_URL || "http://localhost:4000", @@ -333,7 +333,7 @@ export class LiteLLMProvider extends OpenAIChatCompletionsProvider { .map((m) => m.trim()) .filter((m) => m.length > 0) || [ "openai/gpt-4o", - "anthropic/claude-3-haiku", + "anthropic/claude-haiku-4-5-20251001", "meta-llama/llama-3.1-8b-instruct", "google/gemini-2.5-flash", ] diff --git a/src/lib/providers/openAI/client.ts b/src/lib/providers/openAI/client.ts index e51610e53..bcf994c35 100644 --- a/src/lib/providers/openAI/client.ts +++ b/src/lib/providers/openAI/client.ts @@ -74,7 +74,8 @@ const resolveOpenAIBaseURL = ( const getOpenAIApiKey = (): string => validateApiKey(createOpenAIConfig()); -const getOpenAIModel = (): string => getProviderModel("OPENAI_MODEL", "gpt-4o"); +const getOpenAIModel = (): string => + getProviderModel("OPENAI_MODEL", "gpt-5.4"); const streamTracer = trace.getTracer("neurolink.provider.openai"); diff --git a/src/lib/providers/openRouter/client.ts b/src/lib/providers/openRouter/client.ts index 44799fd42..91ab83be8 100644 --- a/src/lib/providers/openRouter/client.ts +++ b/src/lib/providers/openRouter/client.ts @@ -358,7 +358,7 @@ export class OpenRouterProvider extends OpenAIChatCompletionsProvider { "openai/gpt-4-turbo", // Google models "google/gemini-2.0-flash", - "google/gemini-1.5-pro", + "google/gemini-2.5-pro", // Meta Llama models "meta-llama/llama-3.1-70b-instruct", "meta-llama/llama-3.1-8b-instruct", diff --git a/src/lib/utils/modelChoices.ts b/src/lib/utils/modelChoices.ts index c222c6cf4..bc42c2653 100644 --- a/src/lib/utils/modelChoices.ts +++ b/src/lib/utils/modelChoices.ts @@ -141,14 +141,6 @@ const TOP_MODELS_CONFIG: Record< model: GoogleAIModels.GEMINI_3_PRO_PREVIEW, description: "Latest preview", }, - { - model: GoogleAIModels.GEMINI_1_5_PRO, - description: "Previous generation", - }, - { - model: GoogleAIModels.GEMINI_1_5_FLASH, - description: "Legacy fast model", - }, ], [AIProviderName.VERTEX]: [ { @@ -164,7 +156,6 @@ const TOP_MODELS_CONFIG: Record< model: VertexModels.GEMINI_2_0_FLASH, description: "Stable production model", }, - { model: VertexModels.GEMINI_1_5_PRO, description: "Previous generation" }, { model: VertexModels.CLAUDE_3_5_SONNET, description: "Claude 3.5 on Vertex", diff --git a/src/lib/utils/providerHealth.ts b/src/lib/utils/providerHealth.ts index c00800b5d..1cd2e7391 100644 --- a/src/lib/utils/providerHealth.ts +++ b/src/lib/utils/providerHealth.ts @@ -586,7 +586,7 @@ export class ProviderHealthChecker { `Available models for ${providerName} (using dual provider architecture):\n` + ` Google Models (via vertex provider):\n` + ` • gemini-2.5-pro, gemini-2.5-flash, gemini-2.5-flash-lite\n` + - ` • gemini-2.0-flash-001, gemini-1.5-pro, gemini-1.5-flash\n` + + ` • gemini-2.0-flash-001, gemini-2.0-flash-lite\n` + ` Anthropic Models (via vertexAnthropic provider):\n` + ` • claude-sonnet-4@20250514, claude-opus-4@20250514\n` + ` • claude-3-5-sonnet-20241022, claude-3-5-haiku-20241022\n` + @@ -1037,7 +1037,7 @@ export class ProviderHealthChecker { } private static getConfiguredLiteLLMModel(): string { - return process.env.LITELLM_MODEL || "openai/gpt-4o-mini"; + return process.env.LITELLM_MODEL || "openai/gpt-5.4-mini"; } private static getOllamaBaseUrl(): string { @@ -1237,7 +1237,7 @@ export class ProviderHealthChecker { } // Only pin the availability check to a specific model when the user - // explicitly configured one. The fallback default ("openai/gpt-4o-mini") + // explicitly configured one. The fallback default ("openai/gpt-5.4-mini") // is a guess, not configuration — proxies that serve a different model // set (every self-hosted gateway) were reported "Not configured" here // while generate/stream against them worked fine with explicit models. @@ -1349,9 +1349,9 @@ export class ProviderHealthChecker { ]; case AIProviderName.GOOGLE_AI: return [ - GoogleAIModels.GEMINI_1_5_PRO, - GoogleAIModels.GEMINI_1_5_FLASH, GoogleAIModels.GEMINI_2_5_PRO, + GoogleAIModels.GEMINI_2_5_FLASH, + GoogleAIModels.GEMINI_2_0_FLASH_001, ]; case AIProviderName.VERTEX: return [ @@ -1360,8 +1360,6 @@ export class ProviderHealthChecker { GoogleAIModels.GEMINI_2_5_FLASH, GoogleAIModels.GEMINI_2_5_FLASH_LITE, GoogleAIModels.GEMINI_2_0_FLASH_001, - GoogleAIModels.GEMINI_1_5_PRO, - GoogleAIModels.GEMINI_1_5_FLASH, // Anthropic models (via vertexAnthropic provider) "claude-sonnet-4@20250514", "claude-opus-4@20250514", @@ -1377,8 +1375,8 @@ export class ProviderHealthChecker { return [OpenAIModels.GPT_4O, OpenAIModels.GPT_4O_MINI, "gpt-35-turbo"]; case AIProviderName.LITELLM: return [ - "openai/gpt-4o-mini", - "anthropic/claude-3-haiku", + "openai/gpt-5.4-mini", + "anthropic/claude-haiku-4-5-20251001", "google/gemini-2.5-flash", ]; case AIProviderName.OLLAMA: { diff --git a/src/lib/workflow/workflows/multiJudgeWorkflow.ts b/src/lib/workflow/workflows/multiJudgeWorkflow.ts index a286e5772..469e1009d 100644 --- a/src/lib/workflow/workflows/multiJudgeWorkflow.ts +++ b/src/lib/workflow/workflows/multiJudgeWorkflow.ts @@ -294,8 +294,8 @@ export function createMultiJudgeWorkflow( }, { provider: AIProviderName.GOOGLE_AI, - model: "gemini-1.5-flash", - label: "Gemini 1.5 Flash", + model: "gemini-2.5-flash", + label: "Gemini 2.5 Flash", weight: 0.8, temperature: 0.7, }, diff --git a/test/continuous-test-suite-retired-model-defaults.ts b/test/continuous-test-suite-retired-model-defaults.ts new file mode 100644 index 000000000..fa19729a1 --- /dev/null +++ b/test/continuous-test-suite-retired-model-defaults.ts @@ -0,0 +1,225 @@ +#!/usr/bin/env tsx +import "dotenv/config"; + +/** + * Continuous Test Suite — retired/stale model runtime defaults + * + * Several silent runtime defaults (RAGAS/guardrails judge models, the + * LiteLLM and OpenAI provider defaults, health-check "try this instead" + * recommendations, and static model-suggestion/fallback lists) pointed at + * models that are either hard-retired (Google's gemini-1.5-pro/flash: + * "SHUT DOWN. Returns 404." per `src/lib/constants/enums.ts`, and + * Anthropic's claude-3-haiku, retired from the Anthropic API) or stale + * (OpenAI's gpt-4o / gpt-4o-mini, superseded per the model-currency map). + * A caller who never sets the relevant env var, or who hits a "model not + * found" error and follows the SDK's own suggestion, would be handed a + * model that 404s or is materially out of date. + * + * This is a static source-text guard rather than a live-call test: every + * fix here is a string-literal default, the fix is "use a different + * literal," and asserting on the literal in situ is exact, offline, and + * immune to provider flakiness/API-key availability. It intentionally does + * NOT touch `constants/tokens.ts`, `constants/contextWindows.ts`, + * `utils/pricing.ts` (capability/limit/pricing tables keyed by model id, + * not defaults) or `adapters/providerImageAdapter.ts` (vision-capability + * allowlist) — those legitimately still list retired ids as table keys. + * + * Source: src/lib/evaluation/EvaluatorFactory.ts, + * src/lib/evaluation/ragasEvaluator.ts, + * src/lib/evaluation/EvaluatorRegistry.ts, + * src/lib/middleware/utils/guardrailsUtils.ts, + * src/lib/workflow/workflows/multiJudgeWorkflow.ts, + * src/lib/providers/litellm/client.ts, + * src/lib/providers/openAI/client.ts, + * src/lib/providers/openRouter/client.ts, + * src/lib/providers/googleVertex/client.ts, + * src/lib/utils/modelChoices.ts, + * src/lib/utils/providerHealth.ts (recommendation lists AND the + * standalone getConfiguredLiteLLMModel() availability-probe guess) + * + * Run: npx tsx test/continuous-test-suite-retired-model-defaults.ts + * pnpm run test:retired-model-defaults + */ + +import * as fs from "node:fs"; +import * as path from "node:path"; +import { fileURLToPath } from "node:url"; +import { defineSuite, assert } from "./helpers/harness.js"; + +const __filename = fileURLToPath(import.meta.url); +const __dirname = path.dirname(__filename); +const SRC = path.join(__dirname, "..", "src", "lib"); + +const { test, runSuite } = defineSuite("Retired Model Defaults Regression", { + offline: true, +}); + +function read(relPath: string): string { + return fs.readFileSync(path.join(SRC, relPath), "utf8"); +} + +await test("RAGAS evaluator presets no longer default to retired gemini-1.5-flash", async () => { + const content = read("evaluation/EvaluatorFactory.ts"); + assert( + !content.includes("gemini-1.5-flash"), + "EvaluatorFactory.ts must not reference retired gemini-1.5-flash (SHUT DOWN, returns 404) in any preset", + ); + const currentCount = (content.match(/gemini-2\.5-flash/g) ?? []).length; + assert( + currentCount >= 3, + `expected the default/lenient/fast presets to all default to gemini-2.5-flash, found ${currentCount} occurrences`, + ); +}); + +await test("RAGASEvaluator class constructor default is not retired gemini-1.5-flash", async () => { + const content = read("evaluation/ragasEvaluator.ts"); + assert( + !content.includes("gemini-1.5-flash"), + "ragasEvaluator.ts must not default evaluationModel to retired gemini-1.5-flash", + ); + assert( + content.includes('"gemini-2.5-flash"'), + "ragasEvaluator.ts must default evaluationModel to current gemini-2.5-flash", + ); +}); + +await test("EvaluatorRegistry RAGAS metadata default is not retired gemini-1.5-flash", async () => { + const content = read("evaluation/EvaluatorRegistry.ts"); + assert( + !content.includes("gemini-1.5-flash"), + "EvaluatorRegistry.ts must not advertise retired gemini-1.5-flash as the RAGAS evaluator's defaultModel", + ); + assert( + content.includes('defaultModel: "gemini-2.5-flash"'), + "EvaluatorRegistry.ts must advertise gemini-2.5-flash as the RAGAS evaluator's defaultModel", + ); +}); + +await test("Guardrails precall evaluation default is not retired gemini-1.5-flash", async () => { + const content = read("middleware/utils/guardrailsUtils.ts"); + assert( + !content.includes("gemini-1.5-flash"), + "guardrailsUtils.ts must not default performPrecallEvaluation's model to retired gemini-1.5-flash", + ); + assert( + content.includes('config.evaluationModel || "gemini-2.5-flash"'), + "guardrailsUtils.ts must default performPrecallEvaluation's model to gemini-2.5-flash", + ); +}); + +await test("Multi-judge workflow ensemble no longer includes retired gemini-1.5-flash", async () => { + const content = read("workflow/workflows/multiJudgeWorkflow.ts"); + assert( + !content.includes("gemini-1.5-flash"), + "multiJudgeWorkflow.ts must not add retired gemini-1.5-flash as an ensemble judge/model", + ); + assert( + content.includes('model: "gemini-2.5-flash"'), + "multiJudgeWorkflow.ts must use gemini-2.5-flash for the model it previously pinned to gemini-1.5-flash", + ); +}); + +await test("LiteLLM provider default and fallback list are current, not stale/retired", async () => { + const content = read("providers/litellm/client.ts"); + assert( + /const FALLBACK_LITELLM_MODEL = "openai\/gpt-5\.4-mini";/.test(content), + "litellm/client.ts FALLBACK_LITELLM_MODEL must be openai/gpt-5.4-mini (was stale openai/gpt-4o-mini)", + ); + assert( + !content.includes('"openai/gpt-4o-mini"'), + "litellm/client.ts must not contain the stale openai/gpt-4o-mini literal anywhere", + ); + assert( + !content.includes('"anthropic/claude-3-haiku"'), + "litellm/client.ts getFallbackModels() must not offer retired anthropic/claude-3-haiku", + ); + assert( + content.includes('"anthropic/claude-haiku-4-5-20251001"'), + "litellm/client.ts getFallbackModels() must offer current anthropic/claude-haiku-4-5-20251001", + ); +}); + +await test("OpenAI provider default is not stale gpt-4o", async () => { + const content = read("providers/openAI/client.ts"); + assert( + /getProviderModel\("OPENAI_MODEL", "gpt-5\.4"\)/.test(content), + 'openAI/client.ts getOpenAIModel() must default to "gpt-5.4" (was stale "gpt-4o")', + ); +}); + +await test("OpenRouter fallback model list no longer offers retired google/gemini-1.5-pro", async () => { + const content = read("providers/openRouter/client.ts"); + assert( + !content.includes('"google/gemini-1.5-pro"'), + "openRouter/client.ts hardcoded fallback list must not offer retired google/gemini-1.5-pro", + ); + assert( + content.includes('"google/gemini-2.5-pro"'), + "openRouter/client.ts hardcoded fallback list must offer current google/gemini-2.5-pro", + ); +}); + +await test("Vertex 'model not available' suggestions drop retired gemini-1.5 ids", async () => { + const content = read("providers/googleVertex/client.ts"); + assert( + !content.includes("gemini-1.5-pro") && + !content.includes("gemini-1.5-flash"), + "googleVertex/client.ts getModelSuggestions() must not suggest retired gemini-1.5-pro/flash to a caller recovering from a model-not-found error", + ); +}); + +await test("CLI model choice lists drop retired gemini-1.5 entries", async () => { + const content = read("utils/modelChoices.ts"); + assert( + !content.includes("GoogleAIModels.GEMINI_1_5_PRO") && + !content.includes("GoogleAIModels.GEMINI_1_5_FLASH"), + "modelChoices.ts GOOGLE_AI TOP_MODELS_CONFIG must not list retired GEMINI_1_5_PRO/GEMINI_1_5_FLASH", + ); + assert( + !content.includes("VertexModels.GEMINI_1_5_PRO"), + "modelChoices.ts VERTEX TOP_MODELS_CONFIG must not list retired GEMINI_1_5_PRO", + ); +}); + +await test("Provider health checks recommend current models, not retired gemini-1.5 ids", async () => { + const content = read("utils/providerHealth.ts"); + assert( + !content.includes("GEMINI_1_5_PRO") && + !content.includes("GEMINI_1_5_FLASH"), + "providerHealth.ts getCommonModelsForProvider() must not recommend retired GEMINI_1_5_PRO/GEMINI_1_5_FLASH for GOOGLE_AI or VERTEX", + ); + assert( + !content.includes("gemini-1.5-pro") && + !content.includes("gemini-1.5-flash"), + "providerHealth.ts VERTEX dual-architecture health text must not list retired gemini-1.5-pro/gemini-1.5-flash", + ); +}); + +await test("Provider health LiteLLM recommendations are current, not stale/retired", async () => { + const content = read("utils/providerHealth.ts"); + assert( + !content.includes('"openai/gpt-4o-mini"'), + "providerHealth.ts LITELLM case must not recommend stale openai/gpt-4o-mini", + ); + assert( + !content.includes('"anthropic/claude-3-haiku"'), + "providerHealth.ts LITELLM case must not recommend retired anthropic/claude-3-haiku", + ); + assert( + content.includes('"openai/gpt-5.4-mini"') && + content.includes('"anthropic/claude-haiku-4-5-20251001"'), + "providerHealth.ts LITELLM case must recommend openai/gpt-5.4-mini and anthropic/claude-haiku-4-5-20251001", + ); +}); + +await test("Provider health LiteLLM availability probe default is not stale gpt-4o-mini", async () => { + const content = read("utils/providerHealth.ts"); + assert( + /getConfiguredLiteLLMModel\(\): string \{\s*return process\.env\.LITELLM_MODEL \|\| "openai\/gpt-5\.4-mini";/.test( + content, + ), + "providerHealth.ts getConfiguredLiteLLMModel() must guess openai/gpt-5.4-mini, not stale openai/gpt-4o-mini, when LITELLM_MODEL is unset", + ); +}); + +await runSuite();