Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion package.json
Original file line number Diff line number Diff line change
Expand Up @@ -108,6 +108,7 @@
"test:error-classification-e2e": "pnpm exec tsx test/continuous-test-suite-error-classification-e2e.ts",
"test:error-classifier-contract": "pnpm exec tsx test/continuous-test-suite-error-classifier-contract.ts",
"test:bedrock-inference-profile": "pnpm exec tsx test/continuous-test-suite-bedrock-inference-profile.ts",
"test:retired-model-defaults": "pnpm exec tsx test/continuous-test-suite-retired-model-defaults.ts",
"test:loop-engine": "pnpm exec tsx test/continuous-test-suite-loop-engine.ts",
"test:middleware": "pnpm exec tsx test/continuous-test-suite-middleware.ts",
"test:stream-middleware": "pnpm exec tsx test/continuous-test-suite-stream-middleware.ts",
Expand Down Expand Up @@ -189,7 +190,7 @@
"// CI tier — fast, no live AI calls, safe for every commit (test:unit; also see the separate provider-safety-net CI job, which runs build + test:providers-mocked + test:provider-structure + test:error-classifier-contract on every PR)": "",
"test:tool-routing": "pnpm exec tsx test/continuous-test-suite-tool-routing.ts",
"test:tool-routing-semantic": "pnpm exec tsx test/continuous-test-suite-tool-routing-semantic.ts",
"test:unit": "pnpm run test:bugfixes && pnpm run test:mcp:infra && pnpm run test:mcp:spans && pnpm run test:tool-routing && pnpm run test:tool-routing-cli && pnpm run test:tool-dedup && pnpm run test:model-pool && pnpm run test:classifier-router && pnpm run test:tool-routing-semantic && pnpm run test:mcp-result-cache && pnpm run test:mcp-direct-name-repair && pnpm run test:mcp-breaker-resolved-errors && pnpm run test:model-not-found-retryable && pnpm run test:archive:security && pnpm run test:office:security && pnpm run test:image-exif && pnpm run test:vector-chroma && pnpm run test:vector-pgvector && pnpm run test:vector-pinecone && pnpm run test:provider-wiring && pnpm run test:docs-mcp",
"test:unit": "pnpm run test:bugfixes && pnpm run test:mcp:infra && pnpm run test:mcp:spans && pnpm run test:tool-routing && pnpm run test:tool-routing-cli && pnpm run test:tool-dedup && pnpm run test:model-pool && pnpm run test:classifier-router && pnpm run test:tool-routing-semantic && pnpm run test:mcp-result-cache && pnpm run test:mcp-direct-name-repair && pnpm run test:mcp-breaker-resolved-errors && pnpm run test:model-not-found-retryable && pnpm run test:archive:security && pnpm run test:office:security && pnpm run test:image-exif && pnpm run test:vector-chroma && pnpm run test:vector-pgvector && pnpm run test:vector-pinecone && pnpm run test:provider-wiring && pnpm run test:docs-mcp && pnpm run test:retired-model-defaults",
"// CI tier — live providers, runs only when API keys are present (test:credentials and test:dynamic make real provider calls when keys are set, so they live here, not in test:unit; test:matrix, a different suite covering the full provider capability matrix, runs nightly via .github/workflows/live-matrix.yml — test:providers itself is still only wired into test:live, not any GitHub Actions workflow)": "",
"test:live": "pnpm run test:providers && pnpm run test:mcp:http && pnpm run test:mcp:sdk && pnpm run test:mcp:cli && pnpm run test:observability && pnpm run test:context && pnpm run test:memory && pnpm run test:tool-reliability && pnpm run test:evaluation && pnpm run test:autoresearch && pnpm run test:credentials && pnpm run test:dynamic",
"// CI tier — product output (image/video/TTS/PPT) — costs $$ per run (not wired into any GitHub Actions workflow as of this comment; run manually or add to live-matrix.yml if nightly coverage is needed)": "",
Expand Down
6 changes: 3 additions & 3 deletions src/lib/evaluation/EvaluatorFactory.ts
Original file line number Diff line number Diff line change
Expand Up @@ -67,7 +67,7 @@ export class EvaluatorFactory extends BaseFactory<Evaluator, EvaluationConfig> {
threshold: 7,
evaluationStrategy: "ragas",
evaluationModel:
process.env.NEUROLINK_RAGAS_EVALUATION_MODEL || "gemini-1.5-flash",
process.env.NEUROLINK_RAGAS_EVALUATION_MODEL || "gemini-2.5-flash",
provider: process.env.NEUROLINK_RAGAS_EVALUATION_PROVIDER || "vertex",
...config,
};
Expand Down Expand Up @@ -118,7 +118,7 @@ export class EvaluatorFactory extends BaseFactory<Evaluator, EvaluationConfig> {
threshold: 5,
evaluationStrategy: "ragas",
evaluationModel:
process.env.NEUROLINK_RAGAS_EVALUATION_MODEL || "gemini-1.5-flash",
process.env.NEUROLINK_RAGAS_EVALUATION_MODEL || "gemini-2.5-flash",
provider: process.env.NEUROLINK_RAGAS_EVALUATION_PROVIDER || "vertex",
...config,
};
Expand All @@ -144,7 +144,7 @@ export class EvaluatorFactory extends BaseFactory<Evaluator, EvaluationConfig> {
const mergedConfig: EvaluationConfig = {
threshold: 6,
evaluationStrategy: "ragas",
evaluationModel: "gemini-1.5-flash",
evaluationModel: "gemini-2.5-flash",
provider: "vertex",
...config,
};
Expand Down
2 changes: 1 addition & 1 deletion src/lib/evaluation/EvaluatorRegistry.ts
Original file line number Diff line number Diff line change
Expand Up @@ -113,7 +113,7 @@ export class EvaluatorRegistry extends BaseRegistry<
description:
"RAGAS-style LLM-as-judge evaluation with relevance, accuracy, and completeness metrics",
requiresLLM: true,
defaultModel: "gemini-1.5-flash",
defaultModel: "gemini-2.5-flash",
defaultProvider: "vertex",
version: "1.0.0",
features: [
Expand Down
2 changes: 1 addition & 1 deletion src/lib/evaluation/ragasEvaluator.ts
Original file line number Diff line number Diff line change
Expand Up @@ -35,7 +35,7 @@ export class RAGASEvaluator {
this.evaluationModel =
evaluationModel ||
process.env.NEUROLINK_RAGAS_EVALUATION_MODEL ||
"gemini-1.5-flash";
"gemini-2.5-flash";
this.providerName =
providerName ||
process.env.NEUROLINK_RAGAS_EVALUATION_PROVIDER ||
Expand Down
2 changes: 1 addition & 1 deletion src/lib/middleware/utils/guardrailsUtils.ts
Original file line number Diff line number Diff line change
Expand Up @@ -160,7 +160,7 @@ export async function performPrecallEvaluation(
try {
const provider = await AIProviderFactory.createProvider(
config.provider || "google-ai",
config.evaluationModel || "gemini-1.5-flash",
config.evaluationModel || "gemini-2.5-flash",
);

const evaluationPrompt =
Expand Down
2 changes: 0 additions & 2 deletions src/lib/providers/googleVertex/client.ts
Original file line number Diff line number Diff line change
Expand Up @@ -8413,8 +8413,6 @@ export class GoogleVertexProvider extends BaseProvider {
"gemini-2.5-flash-lite",
"gemini-2.0-flash-001",
"gemini-2.0-flash-lite",
"gemini-1.5-pro",
"gemini-1.5-flash",
],
claude: [
"claude-sonnet-4-5@20250929",
Expand Down
4 changes: 2 additions & 2 deletions src/lib/providers/litellm/client.ts
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@ import { OpenAIChatCompletionsProvider } from "../openaiChatCompletionsBase.js";

const streamTracer = trace.getTracer("neurolink.provider.litellm");

const FALLBACK_LITELLM_MODEL = "openai/gpt-4o-mini";
const FALLBACK_LITELLM_MODEL = "openai/gpt-5.4-mini";

const getLiteLLMConfig = () => ({
baseURL: process.env.LITELLM_BASE_URL || "http://localhost:4000",
Expand Down Expand Up @@ -333,7 +333,7 @@ export class LiteLLMProvider extends OpenAIChatCompletionsProvider {
.map((m) => m.trim())
.filter((m) => m.length > 0) || [
"openai/gpt-4o",
"anthropic/claude-3-haiku",
"anthropic/claude-haiku-4-5-20251001",
Comment on lines 335 to +336

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

MAJOR — getFallbackModels() still advertises stale openai/gpt-4o as its first fallback.

This PR's whole purpose is retiring the stale OpenAI defaults: FALLBACK_LITELLM_MODEL moved to openai/gpt-5.4-mini, and the health-check "try this instead" probe moved off gpt-4o-mini. But the first entry of this fallback list — the model handed to any caller whose /v1/models fetch fails and falls back to the static suggestion list — is still "openai/gpt-4o" (line 335, left unchanged two lines above the edited line). A caller who exhausts the model-not-found path is handed the exact stale/retired default this PR claims to remove, contradicting the other edits in the same function.

The new test also misses this (see the test-file finding). Suggest aligning the head of the list with FALLBACK_LITELLM_MODEL:

Suggested change
"openai/gpt-4o",
"anthropic/claude-3-haiku",
"anthropic/claude-haiku-4-5-20251001",
"openai/gpt-5.4-mini",
"anthropic/claude-haiku-4-5-20251001",

"meta-llama/llama-3.1-8b-instruct",
"google/gemini-2.5-flash",
]
Expand Down
3 changes: 2 additions & 1 deletion src/lib/providers/openAI/client.ts
Original file line number Diff line number Diff line change
Expand Up @@ -74,7 +74,8 @@ const resolveOpenAIBaseURL = (

const getOpenAIApiKey = (): string => validateApiKey(createOpenAIConfig());

const getOpenAIModel = (): string => getProviderModel("OPENAI_MODEL", "gpt-4o");
const getOpenAIModel = (): string =>

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

MAJOR — OpenAI runtime default diverges from the public default surface.

You moved the provider runtime default here to "gpt-5.4", but the sibling OpenAI default surfaces were left pointing at gpt-4o:

Surface Current value
getOpenAIModel() (this line, fixed) gpt-5.4
DEFAULT_MODELS[OPENAI] → getDefaultModel("openai") (src/lib/utils/modelChoices.ts) GPT_4O
TOP_MODELS_CONFIG[OPENAI] headline, labelled "Recommended - Latest multimodal model" GPT_4O
providerHealth getCommonModelsForProvider OPENAI case ("try this instead") GPT_4O, GPT_4O_MINI first

So getDefaultModel(OPENAI) and the health-check recommendation now contradict the actual model a bare new OpenAIProvider() runs (rule 5 surface: unmodified callers reading the default get a different answer than the runtime produces). This is exactly the divergence the PR set out to close, but applied only on one surface. Align modelChoices.ts and providerHealth.ts to GPT_5_4/GPT_5_4_MINI (or whichever current id the currency map uses) for consistency.

getProviderModel("OPENAI_MODEL", "gpt-5.4");

const streamTracer = trace.getTracer("neurolink.provider.openai");

Expand Down
2 changes: 1 addition & 1 deletion src/lib/providers/openRouter/client.ts
Original file line number Diff line number Diff line change
Expand Up @@ -358,7 +358,7 @@ export class OpenRouterProvider extends OpenAIChatCompletionsProvider {
"openai/gpt-4-turbo",
// Google models
"google/gemini-2.0-flash",
"google/gemini-1.5-pro",
"google/gemini-2.5-pro",
// Meta Llama models
"meta-llama/llama-3.1-70b-instruct",
"meta-llama/llama-3.1-8b-instruct",
Expand Down
9 changes: 0 additions & 9 deletions src/lib/utils/modelChoices.ts
Original file line number Diff line number Diff line change
Expand Up @@ -141,14 +141,6 @@ const TOP_MODELS_CONFIG: Record<
model: GoogleAIModels.GEMINI_3_PRO_PREVIEW,
description: "Latest preview",
},
{
model: GoogleAIModels.GEMINI_1_5_PRO,
description: "Previous generation",
},
{
model: GoogleAIModels.GEMINI_1_5_FLASH,
description: "Legacy fast model",
},
],
[AIProviderName.VERTEX]: [
{
Expand All @@ -164,7 +156,6 @@ const TOP_MODELS_CONFIG: Record<
model: VertexModels.GEMINI_2_0_FLASH,
description: "Stable production model",
},
{ model: VertexModels.GEMINI_1_5_PRO, description: "Previous generation" },
{
model: VertexModels.CLAUDE_3_5_SONNET,
description: "Claude 3.5 on Vertex",
Expand Down
16 changes: 7 additions & 9 deletions src/lib/utils/providerHealth.ts
Original file line number Diff line number Diff line change
Expand Up @@ -586,7 +586,7 @@ export class ProviderHealthChecker {
`Available models for ${providerName} (using dual provider architecture):\n` +
` Google Models (via vertex provider):\n` +
` • gemini-2.5-pro, gemini-2.5-flash, gemini-2.5-flash-lite\n` +
` • gemini-2.0-flash-001, gemini-1.5-pro, gemini-1.5-flash\n` +
` • gemini-2.0-flash-001, gemini-2.0-flash-lite\n` +
` Anthropic Models (via vertexAnthropic provider):\n` +
` • claude-sonnet-4@20250514, claude-opus-4@20250514\n` +
` • claude-3-5-sonnet-20241022, claude-3-5-haiku-20241022\n` +
Expand Down Expand Up @@ -1037,7 +1037,7 @@ export class ProviderHealthChecker {
}

private static getConfiguredLiteLLMModel(): string {
return process.env.LITELLM_MODEL || "openai/gpt-4o-mini";
return process.env.LITELLM_MODEL || "openai/gpt-5.4-mini";
}

private static getOllamaBaseUrl(): string {
Expand Down Expand Up @@ -1237,7 +1237,7 @@ export class ProviderHealthChecker {
}

// Only pin the availability check to a specific model when the user
// explicitly configured one. The fallback default ("openai/gpt-4o-mini")
// explicitly configured one. The fallback default ("openai/gpt-5.4-mini")
// is a guess, not configuration — proxies that serve a different model
// set (every self-hosted gateway) were reported "Not configured" here
// while generate/stream against them worked fine with explicit models.
Expand Down Expand Up @@ -1349,9 +1349,9 @@ export class ProviderHealthChecker {
];
case AIProviderName.GOOGLE_AI:
return [
GoogleAIModels.GEMINI_1_5_PRO,
GoogleAIModels.GEMINI_1_5_FLASH,
GoogleAIModels.GEMINI_2_5_PRO,
GoogleAIModels.GEMINI_2_5_FLASH,
GoogleAIModels.GEMINI_2_0_FLASH_001,
];
case AIProviderName.VERTEX:
return [
Expand All @@ -1360,8 +1360,6 @@ export class ProviderHealthChecker {
GoogleAIModels.GEMINI_2_5_FLASH,
GoogleAIModels.GEMINI_2_5_FLASH_LITE,
GoogleAIModels.GEMINI_2_0_FLASH_001,
GoogleAIModels.GEMINI_1_5_PRO,
GoogleAIModels.GEMINI_1_5_FLASH,
// Anthropic models (via vertexAnthropic provider)
"claude-sonnet-4@20250514",
"claude-opus-4@20250514",
Expand All @@ -1377,8 +1375,8 @@ export class ProviderHealthChecker {
return [OpenAIModels.GPT_4O, OpenAIModels.GPT_4O_MINI, "gpt-35-turbo"];
case AIProviderName.LITELLM:
return [
"openai/gpt-4o-mini",
"anthropic/claude-3-haiku",
"openai/gpt-5.4-mini",
"anthropic/claude-haiku-4-5-20251001",
"google/gemini-2.5-flash",
];
case AIProviderName.OLLAMA: {
Expand Down
4 changes: 2 additions & 2 deletions src/lib/workflow/workflows/multiJudgeWorkflow.ts
Original file line number Diff line number Diff line change
Expand Up @@ -294,8 +294,8 @@ export function createMultiJudgeWorkflow(
},
{
provider: AIProviderName.GOOGLE_AI,
model: "gemini-1.5-flash",
label: "Gemini 1.5 Flash",
model: "gemini-2.5-flash",
label: "Gemini 2.5 Flash",
weight: 0.8,
temperature: 0.7,
},
Expand Down
Loading
Loading