From 33396f26ff9895cbd8939091b394294af52180bc Mon Sep 17 00:00:00 2001 From: Herjarsa Date: Sun, 7 Jun 2026 16:58:23 +0200 Subject: [PATCH 1/3] feat(vision-bridge): auto-route to fastest vision model - Add visionBridgeRouter.ts with auto-selection of fastest vision-capable model - Selection based on latency, priority (local > API > free), and success rate - Automatic fallback chain if primary model fails - Latency recording for future decisions - Configurable via VisionBridgeRouterConfig (fixedModel, maxFallbackAttempts, etc.) - Integrate auto-router into visionBridgeHelpers.ts callVisionModel() - Add 7 unit tests for the router Fixes #2484 --- src/lib/guardrails/visionBridgeHelpers.ts | 43 ++- src/lib/guardrails/visionBridgeRouter.ts | 268 ++++++++++++++++++ .../guardrails/visionBridgeRouter.test.tsx | 82 ++++++ 3 files changed, 392 insertions(+), 1 deletion(-) create mode 100644 src/lib/guardrails/visionBridgeRouter.ts create mode 100644 tests/unit/guardrails/visionBridgeRouter.test.tsx diff --git a/src/lib/guardrails/visionBridgeHelpers.ts b/src/lib/guardrails/visionBridgeHelpers.ts index 549ca5ab8fe..475e6356ec6 100644 --- a/src/lib/guardrails/visionBridgeHelpers.ts +++ b/src/lib/guardrails/visionBridgeHelpers.ts @@ -3,7 +3,11 @@ */ import { fetchRemoteImage } from "@/shared/network/remoteImageFetch"; import { getRuntimePorts } from "@/lib/runtime/ports"; - +import { + getBestVisionModel, + getFallbackModels, + recordLatency, +} from "./visionBridgeRouter"; /** * Provider to environment variable mapping for API key resolution. */ @@ -204,11 +208,48 @@ export interface VisionModelConfig { /** * Call the vision model to get an image description. * Supports both OpenAI-compatible and Anthropic API formats. + * Uses auto-routing to select the fastest available model. */ export async function callVisionModel( imageDataUri: string, config: VisionModelConfig, apiKey?: string +): Promise { + // Auto-select the best vision model if not explicitly configured + const modelToUse = getBestVisionModel({ fixedModel: config.model }); + const startTime = Date.now(); + let lastError: Error | null = null; + + // Try primary model + fallbacks + const modelsToTry = [modelToUse, ...getFallbackModels(modelToUse)]; + const maxAttempts = Math.min(modelsToTry.length, 3); + + for (let attempt = 0; attempt < maxAttempts; attempt++) { + const currentModel = modelsToTry[attempt]; + try { + const result = await callVisionModelSingle(imageDataUri, { ...config, model: currentModel }, apiKey); + const latency = Date.now() - startTime; + recordLatency(currentModel, latency, true); + return result; + } catch (error) { + const latency = Date.now() - startTime; + recordLatency(currentModel, latency, false); + lastError = error instanceof Error ? error : new Error(String(error)); + // Continue to next model on failure + } + } + + // All models failed + throw lastError || new Error("All vision models failed"); +} + +/** + * Internal function to call a single vision model. + */ +async function callVisionModelSingle( + imageDataUri: string, + config: VisionModelConfig, + apiKey?: string ): Promise { const controller = new AbortController(); const timeoutId = setTimeout(() => controller.abort(), config.timeoutMs); diff --git a/src/lib/guardrails/visionBridgeRouter.ts b/src/lib/guardrails/visionBridgeRouter.ts new file mode 100644 index 00000000000..60a2c02c4dc --- /dev/null +++ b/src/lib/guardrails/visionBridgeRouter.ts @@ -0,0 +1,268 @@ +/** + * Vision Bridge Auto-Router + * Automatically selects the fastest vision-capable model from available models. + */ + +import { getResolvedModelCapabilities } from "@/lib/modelCapabilities"; +import { PROVIDER_MODELS, PROVIDER_ID_TO_ALIAS } from "@omniroute/open-sse/config/providerModels"; + +export interface VisionModelCandidate { + modelId: string; + fullName: string; // provider/model format + priority: number; // lower = better (local models first) + averageLatencyMs: number; + lastUsedAt: number; + successRate: number; +} + +export interface LatencyRecord { + modelId: string; + latencyMs: number; + timestamp: number; + success: boolean; +} + +export interface VisionBridgeRouterConfig { + /** Fixed model to use (overrides auto-routing) */ + fixedModel?: string; + /** Maximum number of fallback attempts */ + maxFallbackAttempts: number; + /** Cache TTL for selection decisions (ms) */ + selectionCacheTtlMs: number; + /** Minimum number of latency samples before trusting average */ + minLatencySamples: number; + /** Models to exclude from auto-routing */ + excludedModels: string[]; +} + +const DEFAULT_ROUTER_CONFIG: VisionBridgeRouterConfig = { + maxFallbackAttempts: 3, + selectionCacheTtlMs: 60_000, // 1 minute + minLatencySamples: 5, + excludedModels: [], +}; + +// In-memory latency tracker (would be Redis in production) +const latencyStore = new Map(); +const selectionCache = new Map(); + +/** + * Record a latency measurement for a model. + */ +export function recordLatency(modelId: string, latencyMs: number, success: boolean): void { + const records = latencyStore.get(modelId) || []; + records.push({ + modelId, + latencyMs, + timestamp: Date.now(), + success, + }); + + // Keep only last 100 records per model + if (records.length > 100) { + records.splice(0, records.length - 100); + } + + latencyStore.set(modelId, records); +} + +/** + * Calculate average latency for a model, considering only recent records. + */ +function calculateAverageLatency(modelId: string, windowMs: number = 300_000): number { + const records = latencyStore.get(modelId) || []; + const cutoff = Date.now() - windowMs; + const recentRecords = records.filter((r) => r.timestamp > cutoff && r.success); + + if (recentRecords.length === 0) { + return Infinity; // No data = assume slow + } + + const sum = recentRecords.reduce((acc, r) => acc + r.latencyMs, 0); + return sum / recentRecords.length; +} + +/** + * Calculate success rate for a model. + */ +function calculateSuccessRate(modelId: string): number { + const records = latencyStore.get(modelId) || []; + if (records.length === 0) return 1.0; // No data = assume good + + const recentRecords = records.slice(-50); // Last 50 attempts + const successes = recentRecords.filter((r) => r.success).length; + return successes / recentRecords.length; +} + +/** + * Get all vision-capable models from the registry. + */ +function getVisionCapableModels(): VisionModelCandidate[] { + const candidates: VisionModelCandidate[] = []; + + for (const [providerAlias, models] of Object.entries(PROVIDER_MODELS)) { + if (!Array.isArray(models)) continue; + + for (const model of models) { + if (!model?.id) continue; + + const fullModelId = `${providerAlias}/${model.id}`; + const caps = getResolvedModelCapabilities(fullModelId); + + if (caps.supportsVision === true) { + // Determine priority based on provider type + let priority = 100; + if (providerAlias.startsWith("opencode-")) { + priority = 0; // Local/free models first + } else if (providerAlias === "openai" || providerAlias === "anthropic") { + priority = 50; // Major providers + } else { + priority = 75; // Other providers + } + + candidates.push({ + modelId: model.id, + fullName: fullModelId, + priority, + averageLatencyMs: calculateAverageLatency(fullModelId), + lastUsedAt: 0, + successRate: calculateSuccessRate(fullModelId), + }); + } + } + } + + return candidates; +} + +/** + * Select the best vision model based on latency, priority, and success rate. + */ +function selectBestModel( + candidates: VisionModelCandidate[], + config: VisionBridgeRouterConfig +): VisionModelCandidate | null { + const filtered = candidates.filter((c) => { + // Exclude explicitly excluded models + if (config.excludedModels.includes(c.fullName)) return false; + if (config.excludedModels.includes(c.modelId)) return false; + + // Exclude models with poor success rate (< 50%) + if (c.successRate < 0.5) return false; + + return true; + }); + + if (filtered.length === 0) return null; + + // Score each candidate: lower is better + // Score = priority * 1000 + averageLatencyMs + // This prioritizes local models, then fastest latency + const scored = filtered.map((c) => ({ + ...c, + score: c.priority * 1000 + (c.averageLatencyMs === Infinity ? 10000 : c.averageLatencyMs), + })); + + scored.sort((a, b) => a.score - b.score); + + return scored[0]; +} + +/** + * Get the best vision model for image description. + * Respects fixed model override if configured. + */ +export function getBestVisionModel( + config: Partial = {} +): string { + const fullConfig = { ...DEFAULT_ROUTER_CONFIG, ...config }; + + // If fixed model is configured, use it + if (fullConfig.fixedModel) { + return fullConfig.fixedModel; + } + + // Check selection cache + const cacheKey = "default"; + const cached = selectionCache.get(cacheKey); + if (cached && cached.expiresAt > Date.now()) { + return cached.modelId; + } + + // Get all vision-capable candidates + const candidates = getVisionCapableModels(); + + // Select best model + const best = selectBestModel(candidates, fullConfig); + + if (!best) { + // Fallback to default + return "openai/gpt-4o-mini"; + } + + // Cache the selection + selectionCache.set(cacheKey, { + modelId: best.fullName, + expiresAt: Date.now() + fullConfig.selectionCacheTtlMs, + }); + + return best.fullName; +} + +/** + * Get fallback models for retry logic. + */ +export function getFallbackModels( + excludeModel: string, + config: Partial = {} +): string[] { + const fullConfig = { ...DEFAULT_ROUTER_CONFIG, ...config }; + const candidates = getVisionCapableModels(); + + const filtered = candidates.filter( + (c) => + c.fullName !== excludeModel && + !fullConfig.excludedModels.includes(c.fullName) && + c.successRate >= 0.5 + ); + + // Sort by score + const scored = filtered.map((c) => ({ + ...c, + score: c.priority * 1000 + (c.averageLatencyMs === Infinity ? 10000 : c.averageLatencyMs), + })); + + scored.sort((a, b) => a.score - b.score); + + return scored.slice(0, fullConfig.maxFallbackAttempts - 1).map((c) => c.fullName); +} + +/** + * Clear the selection cache (e.g., after config change). + */ +export function clearSelectionCache(): void { + selectionCache.clear(); +} + +/** + * Get latency statistics for debugging. + */ +export function getLatencyStats(): Record { + const stats: Record = {}; + + for (const [modelId, records] of latencyStore.entries()) { + const recentRecords = records.filter((r) => r.timestamp > Date.now() - 300_000); + if (recentRecords.length === 0) continue; + + const avg = recentRecords.reduce((acc, r) => acc + r.latencyMs, 0) / recentRecords.length; + const successRate = recentRecords.filter((r) => r.success).length / recentRecords.length; + + stats[modelId] = { + avg: Math.round(avg), + samples: recentRecords.length, + successRate: Math.round(successRate * 100) / 100, + }; + } + + return stats; +} diff --git a/tests/unit/guardrails/visionBridgeRouter.test.tsx b/tests/unit/guardrails/visionBridgeRouter.test.tsx new file mode 100644 index 00000000000..85ae71dd7c6 --- /dev/null +++ b/tests/unit/guardrails/visionBridgeRouter.test.tsx @@ -0,0 +1,82 @@ +/** + * Vision Bridge Auto-Router Tests + */ + +import { describe, it, expect, beforeEach, vi } from "vitest"; +import { + getBestVisionModel, + getFallbackModels, + recordLatency, + clearSelectionCache, + getLatencyStats, +} from "@/lib/guardrails/visionBridgeRouter"; + +describe("Vision Bridge Auto-Router", () => { + beforeEach(() => { + clearSelectionCache(); + }); + + describe("getBestVisionModel", () => { + it("should return a vision-capable model", () => { + const model = getBestVisionModel(); + expect(model).toBeTruthy(); + expect(typeof model).toBe("string"); + }); + + it("should respect fixed model override", () => { + const fixedModel = "openai/gpt-4o-mini"; + const model = getBestVisionModel({ fixedModel }); + expect(model).toBe(fixedModel); + }); + + it("should exclude specified models", () => { + const model = getBestVisionModel({ + excludedModels: ["openai/gpt-4o-mini", "openai/gpt-4o"], + }); + expect(model).not.toBe("openai/gpt-4o-mini"); + expect(model).not.toBe("openai/gpt-4o"); + }); + }); + + describe("getFallbackModels", () => { + it("should return fallback models excluding the primary", () => { + const primary = "openai/gpt-4o-mini"; + const fallbacks = getFallbackModels(primary); + expect(fallbacks).not.toContain(primary); + expect(fallbacks.length).toBeGreaterThan(0); + }); + + it("should respect max fallback attempts", () => { + const fallbacks = getFallbackModels("openai/gpt-4o-mini", { + maxFallbackAttempts: 2, + }); + expect(fallbacks.length).toBeLessThanOrEqual(2); + }); + }); + + describe("recordLatency", () => { + it("should record latency measurements", () => { + recordLatency("test-model", 100, true); + recordLatency("test-model", 150, true); + recordLatency("test-model", 200, false); + + const stats = getLatencyStats(); + expect(stats["test-model"]).toBeTruthy(); + expect(stats["test-model"].samples).toBe(3); + }); + }); + + describe("getLatencyStats", () => { + it("should return latency statistics", () => { + recordLatency("model-a", 100, true); + recordLatency("model-a", 120, true); + recordLatency("model-b", 200, true); + + const stats = getLatencyStats(); + expect(stats["model-a"]).toBeTruthy(); + expect(stats["model-b"]).toBeTruthy(); + expect(stats["model-a"].avg).toBe(110); + expect(stats["model-a"].successRate).toBe(1); + }); + }); +}); From 2d586d12e9746100f5e716a4b524dde7ff64d061 Mon Sep 17 00:00:00 2001 From: Herjarsa Date: Sun, 7 Jun 2026 17:09:48 +0200 Subject: [PATCH 2/3] fix(vision-bridge): address Gemini review feedback - Measure latency per-attempt instead of cumulative across retries - Make selection cache key dynamic to prevent pollution across configs - Respect maxFallbackAttempts from routerConfig instead of hardcoding 3 - Add routerConfig parameter to callVisionModel for full configurability --- src/lib/guardrails/visionBridgeHelpers.ts | 26 ++++++++++++++--------- src/lib/guardrails/visionBridgeRouter.ts | 7 ++++-- 2 files changed, 21 insertions(+), 12 deletions(-) diff --git a/src/lib/guardrails/visionBridgeHelpers.ts b/src/lib/guardrails/visionBridgeHelpers.ts index 475e6356ec6..d2c9d4239d3 100644 --- a/src/lib/guardrails/visionBridgeHelpers.ts +++ b/src/lib/guardrails/visionBridgeHelpers.ts @@ -213,27 +213,33 @@ export interface VisionModelConfig { export async function callVisionModel( imageDataUri: string, config: VisionModelConfig, - apiKey?: string + apiKey?: string, + routerConfig?: Partial ): Promise { // Auto-select the best vision model if not explicitly configured - const modelToUse = getBestVisionModel({ fixedModel: config.model }); - const startTime = Date.now(); + const modelToUse = getBestVisionModel({ + fixedModel: config.model, + ...routerConfig, + }); let lastError: Error | null = null; // Try primary model + fallbacks - const modelsToTry = [modelToUse, ...getFallbackModels(modelToUse)]; - const maxAttempts = Math.min(modelsToTry.length, 3); + const modelsToTry = [modelToUse, ...getFallbackModels(modelToUse, routerConfig)]; + const maxAttempts = Math.min(modelsToTry.length, routerConfig?.maxFallbackAttempts ?? 3); for (let attempt = 0; attempt < maxAttempts; attempt++) { const currentModel = modelsToTry[attempt]; + const attemptStart = Date.now(); try { - const result = await callVisionModelSingle(imageDataUri, { ...config, model: currentModel }, apiKey); - const latency = Date.now() - startTime; - recordLatency(currentModel, latency, true); + const result = await callVisionModelSingle( + imageDataUri, + { ...config, model: currentModel }, + apiKey + ); + recordLatency(currentModel, Date.now() - attemptStart, true); return result; } catch (error) { - const latency = Date.now() - startTime; - recordLatency(currentModel, latency, false); + recordLatency(currentModel, Date.now() - attemptStart, false); lastError = error instanceof Error ? error : new Error(String(error)); // Continue to next model on failure } diff --git a/src/lib/guardrails/visionBridgeRouter.ts b/src/lib/guardrails/visionBridgeRouter.ts index 60a2c02c4dc..76b1331d050 100644 --- a/src/lib/guardrails/visionBridgeRouter.ts +++ b/src/lib/guardrails/visionBridgeRouter.ts @@ -182,8 +182,11 @@ export function getBestVisionModel( return fullConfig.fixedModel; } - // Check selection cache - const cacheKey = "default"; + // Check selection cache — key includes excluded models to prevent cache pollution + // across different configurations + const cacheKey = fullConfig.excludedModels.length > 0 + ? `excl:${[...fullConfig.excludedModels].sort().join(",")}` + : "default"; const cached = selectionCache.get(cacheKey); if (cached && cached.expiresAt > Date.now()) { return cached.modelId; From 83680e5d6d4db11c9a4876b6ec0dcc73986a5ba3 Mon Sep 17 00:00:00 2001 From: herjarsa Date: Mon, 8 Jun 2026 00:51:16 -0300 Subject: [PATCH 3/3] test(vision-bridge): add integration tests for callVisionModel fallback (Rule #18) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Proves that when the primary vision model fails, callVisionModel iterates through the fallback list (primary → fallback), and throws the last error when all attempts fail. Co-authored-by: diegosouzapw --- .../vision-bridge-callmodel.test.ts | 106 ++++++++++++++++++ 1 file changed, 106 insertions(+) create mode 100644 tests/unit/guardrails/vision-bridge-callmodel.test.ts diff --git a/tests/unit/guardrails/vision-bridge-callmodel.test.ts b/tests/unit/guardrails/vision-bridge-callmodel.test.ts new file mode 100644 index 00000000000..74b9ee65319 --- /dev/null +++ b/tests/unit/guardrails/vision-bridge-callmodel.test.ts @@ -0,0 +1,106 @@ +/** + * callVisionModel fallback behavior — Integration test (PR #3377, Rule #18) + * + * Verifies that when the primary vision model fails, callVisionModel falls + * through to the next model in the fallback list, and that when ALL models + * fail it throws the last error (not a silent empty result). + * + * Run: node --import tsx/esm --test tests/unit/guardrails/vision-bridge-callmodel.test.ts + */ + +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync( + path.join(os.tmpdir(), "omniroute-vision-bridge-") +); +process.env.DATA_DIR = TEST_DATA_DIR; +// Prevent vision bridge from routing through a real API +process.env.VISION_BRIDGE_ENABLED = "false"; + +const { callVisionModel } = await import( + "../../../src/lib/guardrails/visionBridgeHelpers.ts" +); + +const originalFetch = globalThis.fetch; + +test.after(() => { + globalThis.fetch = originalFetch; + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +test.afterEach(() => { + globalThis.fetch = originalFetch; +}); + +// Helper: build a minimal OpenAI-compat image data URI +const TINY_PNG = "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNk+M9QDwADhgGAWjR9awAAAABJRU5ErkJggg=="; + +test("callVisionModel falls through to next model when primary fails", async () => { + let fetchCallCount = 0; + const FALLBACK_RESPONSE = JSON.stringify({ + choices: [{ message: { content: "fallback model description" } }], + }); + + globalThis.fetch = async (url: RequestInfo | URL, _init?: RequestInit) => { + fetchCallCount++; + if (fetchCallCount === 1) { + // First call (primary model) — simulate API error + throw new Error("mock: primary model unavailable"); + } + // Second call (fallback model) — return valid response + return new Response(FALLBACK_RESPONSE, { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + }; + + const result = await callVisionModel( + TINY_PNG, + { model: "openai/gpt-4o-mini", prompt: "Describe this image." }, + "sk-test-key", + { fixedModel: "openai/gpt-4o-mini", maxFallbackAttempts: 2 } + ); + + assert.equal( + fetchCallCount, + 2, + "must have attempted exactly 2 models (primary + 1 fallback)" + ); + assert.equal( + result, + "fallback model description", + "must return the fallback model's response" + ); +}); + +test("callVisionModel throws when ALL models fail", async () => { + let fetchCallCount = 0; + + globalThis.fetch = async () => { + fetchCallCount++; + throw new Error(`mock: model-${fetchCallCount} unavailable`); + }; + + await assert.rejects( + () => + callVisionModel( + TINY_PNG, + { model: "openai/gpt-4o-mini", prompt: "Describe this image." }, + "sk-test-key", + { fixedModel: "openai/gpt-4o-mini", maxFallbackAttempts: 2 } + ), + (err: Error) => { + assert.ok( + err.message.includes("unavailable") || err.message.includes("All vision models failed"), + `error should indicate failure, got: ${err.message}` + ); + return true; + } + ); + + assert.ok(fetchCallCount >= 1, "must have attempted at least 1 model"); +});