diff --git a/changelog.d/fixes/vertex-express-xai-models.md b/changelog.d/fixes/vertex-express-xai-models.md new file mode 100644 index 00000000000..5752468279a --- /dev/null +++ b/changelog.d/fixes/vertex-express-xai-models.md @@ -0,0 +1 @@ +- **fix(vertex):** prefer live model discovery for OAuth, Service Account, and service-account-bound authorization-key credentials while falling back cleanly to the Gemini-only Express catalog for standard API keys. Model Garden resources now route by publisher protocol: Claude and Mistral use their native `rawPredict` APIs, and Grok plus current or future open MaaS publishers use Vertex's OpenAI-compatible endpoint with normalized request IDs such as `xai/grok-4.6`. Distinguish intentional API-key catalog rejection from transient HTTP or network failures. ([#12471](https://github.com/diegosouzapw/OmniRoute/pull/12471)) — thanks @JxnLexn diff --git a/next.config.mjs b/next.config.mjs index f4d4622c90b..15085232807 100644 --- a/next.config.mjs +++ b/next.config.mjs @@ -349,6 +349,10 @@ const nextConfig = { "keytar", "wreq-js", "zod", + // jsdom relies on Node class relationships that Turbopack's server-chunk transform can break + // (observed as "Class extends value undefined" during Vertex metadata sync). Keep the native + // package boundary; standalone file tracing still copies the runtime dependency. + "jsdom", "@ngrok/ngrok", "@huggingface/transformers", // The ESM entry imports tiktoken_bg.wasm as a module. Turbopack can compile diff --git a/open-sse/config/freeModelCatalog.data.ts b/open-sse/config/freeModelCatalog.data.ts index 3d7ebed5950..cc934744c32 100644 --- a/open-sse/config/freeModelCatalog.data.ts +++ b/open-sse/config/freeModelCatalog.data.ts @@ -442,11 +442,17 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [ { provider: "vertex", modelId: "gemini-3.1-pro-preview", displayName: "Gemini 3.1 Pro Preview (Vertex)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, { provider: "vertex", modelId: "gemini-3.1-flash-lite", displayName: "Gemini 3.1 Flash Lite (Vertex)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, { provider: "vertex", modelId: "gemini-3-flash-preview", displayName: "Gemini 3 Flash Preview (Vertex)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, - { provider: "vertex", modelId: "gemma-4-31b-it", displayName: "Gemma 4 31B (Vertex)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, - { provider: "vertex", modelId: "DeepSeek-V4-Flash", displayName: "DeepSeek V4 Flash (Vertex Partner)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, - { provider: "vertex", modelId: "DeepSeek-V4-Pro", displayName: "DeepSeek V4 Pro (Vertex Partner)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, - { provider: "vertex", modelId: "Qwen3.6-35B-A3B", displayName: "Qwen3.6 35B A3B (Vertex Partner)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, - { provider: "vertex", modelId: "GLM-5.1-FP8", displayName: "GLM-5.1 (Vertex Partner)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, + { provider: "vertex", modelId: "gemini-3.7-flash", displayName: "Gemini 3.7 Flash (Vertex)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, + { provider: "vertex", modelId: "gemini-3.6-flash", displayName: "Gemini 3.6 Flash (Vertex)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, + { provider: "vertex", modelId: "gemini-3.5-flash", displayName: "Gemini 3.5 Flash (Vertex)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, + { provider: "vertex", modelId: "gemini-3.5-flash-lite", displayName: "Gemini 3.5 Flash-Lite (Vertex)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, + { provider: "vertex", modelId: "gemini-2.5-pro", displayName: "Gemini 2.5 Pro (Vertex)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, + { provider: "vertex", modelId: "gemini-2.5-flash", displayName: "Gemini 2.5 Flash (Vertex)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, + { provider: "vertex", modelId: "gemini-2.5-flash-lite", displayName: "Gemini 2.5 Flash-Lite (Vertex)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, + { provider: "vertex", modelId: "deepseek-ai/deepseek-v3.2-maas", displayName: "DeepSeek V3.2 (Vertex MaaS)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, + { provider: "vertex", modelId: "deepseek-ai/deepseek-v3.1-maas", displayName: "DeepSeek V3.1 (Vertex MaaS)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, + { provider: "vertex", modelId: "qwen/qwen3-next-80b-a3b-instruct-maas", displayName: "Qwen3 Next 80B Instruct (Vertex MaaS)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, + { provider: "vertex", modelId: "zai-org/glm-5-maas", displayName: "GLM 5 (Vertex MaaS)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, { provider: "vertex", modelId: "claude-opus-4-7", displayName: "Claude Opus 4.7 (Vertex)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, { provider: "vertex", modelId: "claude-sonnet-4-6", displayName: "Claude Sonnet 4.6 (Vertex)", monthlyTokens: 0, creditTokens: 300000000, freeType: "one-time-initial", poolKey: "vertex", tos: "caution" }, { provider: "requesty", modelId: "openai/gpt-oss-120b", displayName: "GPT-OSS 120B (Requesty free)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "requesty-free", tos: "ok" }, diff --git a/open-sse/config/providerModels.ts b/open-sse/config/providerModels.ts index 3e6dffbf99e..5f6eb189d98 100644 --- a/open-sse/config/providerModels.ts +++ b/open-sse/config/providerModels.ts @@ -1,4 +1,5 @@ import { generateModels, generateAliasMap, type RegistryModel } from "./providerRegistry.ts"; +import { getVertexModelTargetFormat } from "./vertexModels.ts"; // Lazy PROVIDER_MODELS: deferred until first property access to speed up startup. // The Proxy defers `generateModels()` from module-evaluation time to the first read. @@ -228,9 +229,10 @@ export function getModelTargetFormat(aliasOrId: string, modelId: string): string // executor's /codex/i routing, 9router#102). Scoped to the openai alias so other // providers shipping *-pro ids keep their own endpoint semantics. if (alias === "openai" && /-pro$/i.test(bareModelId)) return "openai-responses"; - // ponytail: Claude models on Vertex use rawPredict with Anthropic Messages format, - // not the Gemini generateContent format. Mirrors executor isClaudeModel() check. - if ((alias === "vertex" || alias === "vp") && /^claude-/i.test(bareModelId)) return "claude"; + // Vertex uses three protocol families: Gemini generateContent, Anthropic Messages rawPredict, + // and OpenAI-shaped Mistral/Open-MaaS requests. Resource names retain enough publisher data to + // route future dynamically-synced models without adding another pinned prefix here. + if (alias === "vertex" || alias === "vp") return getVertexModelTargetFormat(bareModelId); // Model-level targetFormat is provider-scoped: a catalog entry declares how THIS // provider's endpoint serves the model — do NOT import another provider's tag. // #9994 scoped this for providers WITH a catalog; #10072 extends it to catalogless diff --git a/open-sse/config/providers/registry/vertex/index.ts b/open-sse/config/providers/registry/vertex/index.ts index 4b72b9f7c45..1a8975e2e30 100644 --- a/open-sse/config/providers/registry/vertex/index.ts +++ b/open-sse/config/providers/registry/vertex/index.ts @@ -1,4 +1,5 @@ import type { RegistryEntry } from "../../shared.ts"; +import { VERTEX_XAI_MODELS } from "../../../vertexModels.ts"; export const vertexProvider: RegistryEntry = { id: "vertex", @@ -9,7 +10,7 @@ export const vertexProvider: RegistryEntry = { // URL uses {project_id} and {region} from providerSpecificData — handled by custom executor or fallback // Default to us-central1 / generic endpoint; users configure project via providerSpecificData baseUrl: "https://us-central1-aiplatform.googleapis.com/v1/projects", - urlBuilder: (base, model, stream) => { + urlBuilder: (_base, model, stream) => { // Full URL: {base}/{project}/locations/{region}/publishers/google/models/{model}:{action} // For a generic fallback, we build a Gemini-compatible URL // The actual project/region are configured via providerSpecificData in the DB connection @@ -19,14 +20,33 @@ export const vertexProvider: RegistryEntry = { authType: "apikey", authHeader: "bearer", models: [ + { id: "gemini-3.7-flash", name: "Gemini 3.7 Flash (Vertex)" }, + { id: "gemini-3.6-flash", name: "Gemini 3.6 Flash (Vertex)" }, + { id: "gemini-3.5-flash", name: "Gemini 3.5 Flash (Vertex)" }, + { id: "gemini-3.5-flash-lite", name: "Gemini 3.5 Flash-Lite (Vertex)" }, { id: "gemini-3.1-pro-preview", name: "Gemini 3.1 Pro Preview (Vertex)" }, { id: "gemini-3.1-flash-lite", name: "Gemini 3.1 Flash Lite (Vertex)" }, { id: "gemini-3-flash-preview", name: "Gemini 3 Flash Preview (Vertex)" }, - { id: "gemma-4-31b-it", name: "Gemma 4 31B (Vertex)" }, - { id: "DeepSeek-V4-Flash", name: "DeepSeek V4 Flash (Vertex Partner)" }, - { id: "DeepSeek-V4-Pro", name: "DeepSeek V4 Pro (Vertex Partner)" }, - { id: "Qwen3.6-35B-A3B", name: "Qwen3.6 35B A3B (Vertex Partner)" }, - { id: "GLM-5.1-FP8", name: "GLM-5.1 (Vertex Partner)" }, + { id: "gemini-2.5-pro", name: "Gemini 2.5 Pro (Vertex)" }, + { id: "gemini-2.5-flash", name: "Gemini 2.5 Flash (Vertex)" }, + { id: "gemini-2.5-flash-lite", name: "Gemini 2.5 Flash-Lite (Vertex)" }, + { + id: "deepseek-ai/deepseek-v3.2-maas", + name: "DeepSeek V3.2 (Vertex MaaS)", + targetFormat: "openai", + }, + { + id: "deepseek-ai/deepseek-v3.1-maas", + name: "DeepSeek V3.1 (Vertex MaaS)", + targetFormat: "openai", + }, + { + id: "qwen/qwen3-next-80b-a3b-instruct-maas", + name: "Qwen3 Next 80B Instruct (Vertex MaaS)", + targetFormat: "openai", + }, + { id: "zai-org/glm-5-maas", name: "GLM 5 (Vertex MaaS)", targetFormat: "openai" }, + ...VERTEX_XAI_MODELS, { id: "claude-fable-5-1", name: "Claude Fable 5.1 (Vertex)", targetFormat: "claude" }, { id: "claude-fable-5", name: "Claude Fable 5 (Vertex)", targetFormat: "claude" }, { id: "claude-opus-5", name: "Claude Opus 5 (Vertex)", targetFormat: "claude" }, @@ -41,4 +61,7 @@ export const vertexProvider: RegistryEntry = { { id: "claude-haiku-4-5", name: "Claude Haiku 4.5 (Vertex)", targetFormat: "claude" }, ], passthroughModels: true, + // Gemini + publisher discovery are independent APIs. A partial successful response must not + // hide curated partner models omitted by another publisher catalog. + liveCatalogAuthoritative: false, }; diff --git a/open-sse/config/providers/registry/vertex/partner/index.ts b/open-sse/config/providers/registry/vertex/partner/index.ts index 6524cd6a2b2..0ce2be94ac6 100644 --- a/open-sse/config/providers/registry/vertex/partner/index.ts +++ b/open-sse/config/providers/registry/vertex/partner/index.ts @@ -1,4 +1,5 @@ import type { RegistryEntry } from "../../../shared.ts"; +import { VERTEX_XAI_MODELS } from "../../../../vertexModels.ts"; export const vertex_partnerProvider: RegistryEntry = { id: "vertex-partner", @@ -9,10 +10,15 @@ export const vertex_partnerProvider: RegistryEntry = { authType: "apikey", authHeader: "bearer", models: [ - { id: "DeepSeek-V4-Flash", name: "DeepSeek V4 Flash" }, - { id: "DeepSeek-V4-Pro", name: "DeepSeek V4 Pro" }, - { id: "Qwen3.6-35B-A3B", name: "Qwen 3.6 35B A3B" }, - { id: "GLM-5.1-FP8", name: "GLM 5.1" }, + { id: "deepseek-ai/deepseek-v3.2-maas", name: "DeepSeek V3.2", targetFormat: "openai" }, + { id: "deepseek-ai/deepseek-v3.1-maas", name: "DeepSeek V3.1", targetFormat: "openai" }, + { + id: "qwen/qwen3-next-80b-a3b-instruct-maas", + name: "Qwen3 Next 80B Instruct", + targetFormat: "openai", + }, + { id: "zai-org/glm-5-maas", name: "GLM 5", targetFormat: "openai" }, + ...VERTEX_XAI_MODELS, { id: "claude-fable-5-1", name: "Claude Fable 5.1", targetFormat: "claude" }, { id: "claude-fable-5", name: "Claude Fable 5", targetFormat: "claude" }, { id: "claude-opus-5", name: "Claude Opus 5", targetFormat: "claude" }, @@ -27,4 +33,5 @@ export const vertex_partnerProvider: RegistryEntry = { { id: "claude-opus-4-5", name: "Claude Opus 4.5", targetFormat: "claude" }, { id: "claude-haiku-4-5", name: "Claude Haiku 4.5", targetFormat: "claude" }, ], + liveCatalogAuthoritative: false, }; diff --git a/open-sse/config/vertexModels.ts b/open-sse/config/vertexModels.ts new file mode 100644 index 00000000000..448c23eb525 --- /dev/null +++ b/open-sse/config/vertexModels.ts @@ -0,0 +1,106 @@ +import type { RegistryModel } from "./providers/shared.ts"; + +const VERTEX_PUBLISHER_RESOURCE_PATTERN = + /^(?:(?:projects\/[^/]+\/locations\/[^/]+\/)?publishers\/([^/]+)\/models\/|([^/]+)\/models\/)(.+)$/i; +const NATIVE_PUBLISHERS = new Set(["google", "anthropic", "mistralai"]); + +const LEGACY_OPENAI_MAAS_PUBLISHERS = [ + { prefix: "deepseek-", publisher: "deepseek-ai" }, + { prefix: "qwen", publisher: "qwen" }, + { prefix: "llama-", publisher: "meta" }, + { prefix: "glm-", publisher: "zai-org" }, + { prefix: "gpt-oss-", publisher: "openai" }, + { prefix: "kimi-", publisher: "moonshotai" }, + { prefix: "minimax-", publisher: "minimaxai" }, +] as const; + +export type VertexModelTransport = "gemini" | "anthropic" | "mistral" | "openai"; + +function normalizePublisherModel(publisher: string, model: string): string { + const normalizedPublisher = publisher.toLowerCase(); + if (NATIVE_PUBLISHERS.has(normalizedPublisher)) return model; + return `${normalizedPublisher}/${model}`; +} + +/** + * Convert Model Garden resource/version names to the model value accepted by its inference API. + * Native Google, Anthropic, and Mistral endpoints encode the publisher in the URL and take a bare + * model id. OpenAI-compatible MaaS requests keep the publisher namespace in the request body. + */ +export function normalizeVertexModelId(model: string): string { + const trimmed = model.trim().replace(/^(?:vertex|vp)\//i, ""); + const resourceMatch = trimmed.match(VERTEX_PUBLISHER_RESOURCE_PATTERN); + if (resourceMatch) { + const publisher = resourceMatch[1] || resourceMatch[2]; + const modelId = resourceMatch[3]; + if (publisher && modelId) return normalizePublisherModel(publisher, modelId); + } + + const namespacedMatch = trimmed.match(/^([^/]+)\/(.+)$/); + if (namespacedMatch && NATIVE_PUBLISHERS.has(namespacedMatch[1].toLowerCase())) { + return namespacedMatch[2]; + } + + // Keep manually-added historical Grok ids working while emitting Google's documented MaaS + // request form (`xai/grok-*`) for the OpenAI-compatible endpoint. + if (/^grok-/i.test(trimmed)) return `xai/${trimmed}`; + const legacyMaaS = LEGACY_OPENAI_MAAS_PUBLISHERS.find(({ prefix }) => + trimmed.toLowerCase().startsWith(prefix) + ); + if (legacyMaaS) return `${legacyMaaS.publisher}/${trimmed}`; + return trimmed; +} + +/** Resolve the Vertex transport from either a request id or a Model Garden resource name. */ +export function getVertexModelTransport(model: string): VertexModelTransport { + const normalized = normalizeVertexModelId(model).toLowerCase(); + if (normalized.startsWith("claude-")) return "anthropic"; + if (normalized.startsWith("mistral-")) return "mistral"; + if (normalized.includes("/")) return "openai"; + return "gemini"; +} + +/** Wire format used before the request reaches the provider-specific Vertex executor. */ +export function getVertexModelTargetFormat(model: string): "claude" | "openai" | null { + const transport = getVertexModelTransport(model); + if (transport === "anthropic") return "claude"; + if (transport === "mistral" || transport === "openai") return "openai"; + return null; +} + +/** True for xAI models served through Vertex's OpenAI-compatible MaaS endpoint. */ +export function isVertexXaiModel(model: string): boolean { + return /^xai\/grok-/i.test(normalizeVertexModelId(model)); +} + +/** True only for models exposed by the project-less Vertex Express API. */ +export function isVertexExpressModel(model: string): boolean { + return ( + getVertexModelTransport(model) === "gemini" && /^gemini-/i.test(normalizeVertexModelId(model)) + ); +} + +/** Curated fallback metadata for Vertex xAI MaaS models. */ +export const VERTEX_XAI_MODELS = [ + { + id: "xai/grok-4.6", + name: "Grok 4.6", + contextLength: 524288, + supportsReasoning: true, + supportsVision: true, + toolCalling: true, + targetFormat: "openai", + }, + { id: "xai/grok-4.3", name: "Grok 4.3", targetFormat: "openai" }, + { + id: "xai/grok-4.20-reasoning", + name: "Grok 4.20 Reasoning", + supportsReasoning: true, + targetFormat: "openai", + }, + { + id: "xai/grok-4.20-non-reasoning", + name: "Grok 4.20 Non-Reasoning", + targetFormat: "openai", + }, +] as const satisfies readonly RegistryModel[]; diff --git a/open-sse/executors/vertex.ts b/open-sse/executors/vertex.ts index 9d9ed6502f2..80fc0c78791 100644 --- a/open-sse/executors/vertex.ts +++ b/open-sse/executors/vertex.ts @@ -1,6 +1,7 @@ import { SignJWT, importPKCS8 } from "jose"; import { BaseExecutor, ExecuteInput } from "./base.ts"; import { PROVIDERS } from "../config/constants.ts"; +import { getVertexModelTransport, normalizeVertexModelId } from "../config/vertexModels.ts"; interface ServiceAccount { type: string; @@ -26,18 +27,25 @@ export const VERTEX_OAUTH_SCOPES = [ ] as const; export function parseSAFromApiKey(apiKey: string): ServiceAccount { + let parsed: ServiceAccount & { web?: unknown; installed?: unknown }; try { - return JSON.parse(apiKey); + parsed = JSON.parse(apiKey); } catch { throw new Error("Vertex AI requires a valid Service Account JSON as the API key"); } + if (parsed && typeof parsed === "object" && ("web" in parsed || "installed" in parsed)) { + throw new Error( + "OAuth client JSON is not a Service Account key; create a Service Account JSON key or provide an OAuth access token" + ); + } + return parsed; } /** * A Service Account credential is a JSON object (type/client_email/private_key). A Vertex AI * Express-mode API key is an opaque non-JSON string. Distinguishing them lets the executor * support BOTH: Service Account JSON (JWT → OAuth → project-scoped endpoint + Bearer auth) and - * Express keys (project-less publisher endpoint + x-goog-api-key auth), instead of failing every + * Express keys (project-less publisher endpoint + query-key auth), instead of failing every * Express key with "requires a valid Service Account JSON". */ export function looksLikeServiceAccountJson(apiKey: string): boolean { @@ -52,7 +60,9 @@ export function looksLikeServiceAccountJson(apiKey: string): boolean { /** True for a Vertex AI Express-mode API key (a non-empty, non-JSON, non-OAuth credential). */ export function isExpressApiKey(apiKey?: string | null): boolean { - return typeof apiKey === "string" && apiKey.trim().length > 0 && !looksLikeServiceAccountJson(apiKey); + return ( + typeof apiKey === "string" && apiKey.trim().length > 0 && !looksLikeServiceAccountJson(apiKey) + ); } export async function getAccessToken(sa: ServiceAccount): Promise { @@ -115,35 +125,97 @@ export async function getAccessToken(sa: ServiceAccount): Promise { return accessToken; } -const PARTNER_MODELS = new Set([ - // Generic prefix, not pinned to a version: every Claude model on Vertex is an Anthropic - // partner model, never a Google-publisher one. Pinned prefixes (e.g. "claude-3-5-sonnet") - // silently break every time Anthropic ships a new generation — see issue #1985. - "claude-", - "deepseek-v3", - "deepseek-v3.2", - "deepseek-v4", - "deepseek-deepseek-r1", - "qwen3-next-80b", - "qwen3.6-", - "llama-3.1", - "mistral-", - "glm-5", - "glm-5.1", - "meta/llama", -]); - function isPartnerModel(model: string) { - const normalizedModel = model.toLowerCase(); - return [...PARTNER_MODELS].some((prefix) => normalizedModel.startsWith(prefix)); + return getVertexModelTransport(model) === "openai"; } // Anthropic models need their own branch: they use Vertex's native Anthropic Messages API -// (publishers/anthropic/.../rawPredict), not the generic OpenAI-compatible partner endpoint the -// other PARTNER_MODELS entries (DeepSeek, Qwen, Llama, Mistral, GLM) go through — the OpenAI-shaped -// endpoint 404s/"malformed argument"s for Claude models on at least some projects. +// (publishers/anthropic/.../rawPredict), not the generic OpenAI-compatible MaaS endpoint used by +// xAI and open models. Mistral has a separate native rawPredict branch below. function isClaudeModel(model: string) { - return model.toLowerCase().startsWith("claude-"); + return getVertexModelTransport(model) === "anthropic"; +} + +function isMistralModel(model: string) { + return getVertexModelTransport(model) === "mistral"; +} + +interface VertexUrlCredentials { + apiKey?: string | null; + accessToken?: string | null; + projectId?: string; + providerSpecificData?: { + projectId?: string; + project?: string; + region?: string; + }; +} + +function configuredVertexProjectId(credentials?: VertexUrlCredentials | null): string | undefined { + return ( + credentials?.projectId || + credentials?.providerSpecificData?.projectId || + credentials?.providerSpecificData?.project + ); +} + +function projectIdFromServiceAccount( + credentials?: VertexUrlCredentials | null +): string | undefined { + if (!credentials?.apiKey) return undefined; + try { + const sa = parseSAFromApiKey(credentials.apiKey); + if (sa.project_id) return sa.project_id; + } catch { + // Ignored, handled in execute + } + return undefined; +} + +function encodeOpaqueApiKey(credentials?: VertexUrlCredentials | null): string | null { + if (!isExpressApiKey(credentials?.apiKey) || credentials?.accessToken) return null; + return encodeURIComponent(String(credentials.apiKey).trim()); +} + +function buildExpressGeminiUrl( + canonicalModel: string, + stream: boolean, + expressKey: string +): string { + if (getVertexModelTransport(canonicalModel) !== "gemini") { + throw new Error( + "Vertex partner models require project-scoped credentials; Express API keys support Gemini models only" + ); + } + const op = stream ? "streamGenerateContent?alt=sse&" : "generateContent?"; + return `https://aiplatform.googleapis.com/v1/publishers/google/models/${canonicalModel}:${op}key=${expressKey}`; +} + +function buildProjectScopedVertexUrl( + canonicalModel: string, + stream: boolean, + project: string, + region: string, + opaqueApiKey: string | null +): string { + const apiKeySuffix = opaqueApiKey ? `?key=${opaqueApiKey}` : ""; + if (isClaudeModel(canonicalModel)) { + // streamRawPredict?alt=sse was verified to return a single plain JSON body (not real SSE + // framing) rather than actual chunked events, which breaks the SSE parser upstream + // ("stream ended before producing a non-ping SSE event"). rawPredict is confirmed reliable + // for both streaming and non-streaming requests; always use it here. + return `https://aiplatform.googleapis.com/v1/projects/${project}/locations/${region}/publishers/anthropic/models/${canonicalModel}:rawPredict${apiKeySuffix}`; + } + if (isMistralModel(canonicalModel)) { + const operation = stream ? "streamRawPredict" : "rawPredict"; + return `https://aiplatform.googleapis.com/v1/projects/${project}/locations/${region}/publishers/mistralai/models/${canonicalModel}:${operation}${apiKeySuffix}`; + } + if (isPartnerModel(canonicalModel)) { + return `https://aiplatform.googleapis.com/v1/projects/${project}/locations/global/endpoints/openapi/chat/completions${apiKeySuffix}`; + } + const operation = stream ? "streamGenerateContent?alt=sse" : "generateContent"; + const querySeparator = opaqueApiKey ? (stream ? "&" : "?") : ""; + return `https://aiplatform.googleapis.com/v1/projects/${project}/locations/${region}/publishers/google/models/${canonicalModel}:${operation}${querySeparator}${opaqueApiKey ? `key=${opaqueApiKey}` : ""}`; } // Defensive normalizer: target-format resolution for manually-added custom Claude models under @@ -153,7 +225,8 @@ function isClaudeModel(model: string) { // converts a Gemini-shaped body to Anthropic Messages shape so the executor works either way, // independent of that unresolved upstream resolution gap. function toAnthropicBody(body: Record): Record { - const contents = body.contents as Array<{ role?: string; parts?: Array<{ text?: string }> }> | undefined; + const contents = body.contents as + Array<{ role?: string; parts?: Array<{ text?: string }> }> | undefined; if (!Array.isArray(contents)) return body; const messages = contents.map((c) => ({ @@ -161,7 +234,8 @@ function toAnthropicBody(body: Record): Record content: (c.parts || []).map((p) => p.text || "").join(""), })); const generationConfig = body.generationConfig as { maxOutputTokens?: number } | undefined; - const systemInstruction = body.systemInstruction as { parts?: Array<{ text?: string }> } | undefined; + const systemInstruction = body.systemInstruction as + { parts?: Array<{ text?: string }> } | undefined; const converted: Record = { messages, @@ -244,7 +318,11 @@ function synthesizeClaudeSse(response: Record): string { } else if (block.type === "thinking") { events.push({ event: "content_block_start", - data: { type: "content_block_start", index, content_block: { type: "thinking", thinking: "" } }, + data: { + type: "content_block_start", + index, + content_block: { type: "thinking", thinking: "" }, + }, }); if (block.thinking) { events.push({ @@ -280,6 +358,21 @@ export class VertexExecutor extends BaseExecutor { } async execute(input: ExecuteInput) { + const canonicalModel = normalizeVertexModelId(input.model); + let canonicalBody = input.body; + if (canonicalBody && typeof canonicalBody === "object" && !Array.isArray(canonicalBody)) { + const bodyRecord = canonicalBody as Record; + if (typeof bodyRecord.model === "string") { + const canonicalBodyModel = normalizeVertexModelId(bodyRecord.model); + if (canonicalBodyModel !== bodyRecord.model) { + canonicalBody = { ...bodyRecord, model: canonicalBodyModel }; + } + } + } + if (canonicalModel !== input.model || canonicalBody !== input.body) { + input = { ...input, model: canonicalModel, body: canonicalBody }; + } + const { credentials, log, model, stream } = input; // Defensive: trim stray surrounding whitespace from a pasted credential. if (typeof credentials.apiKey === "string") { @@ -287,12 +380,17 @@ export class VertexExecutor extends BaseExecutor { } // Service Account JSON → mint a short-lived OAuth token (Bearer). An Express-mode API key is // sent as-is via x-goog-api-key (see buildHeaders), so no token exchange is needed for it. - if (credentials.apiKey && !credentials.accessToken && looksLikeServiceAccountJson(credentials.apiKey)) { + if ( + credentials.apiKey && + !credentials.accessToken && + looksLikeServiceAccountJson(credentials.apiKey) + ) { try { const sa = parseSAFromApiKey(credentials.apiKey); credentials.accessToken = await getAccessToken(sa); - } catch (err: any) { - log?.error?.("VERTEX", `Failed to generate JWT token: ${err.message}`); + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + log?.error?.("VERTEX", `Failed to generate JWT token: ${message}`); throw err; } } @@ -318,7 +416,10 @@ export class VertexExecutor extends BaseExecutor { const response = result instanceof Response ? result : result?.response; if (response?.ok) { const contentType = response.headers.get("content-type") || ""; - if (contentType.includes("application/json") && !contentType.includes("text/event-stream")) { + if ( + contentType.includes("application/json") && + !contentType.includes("text/event-stream") + ) { const jsonText = await response.text(); let newBody = jsonText; let newContentType = contentType; @@ -345,46 +446,28 @@ export class VertexExecutor extends BaseExecutor { return result; } - buildUrl(model: string, stream: boolean, urlIndex = 0, credentials: any = null) { + buildUrl( + model: string, + stream: boolean, + _urlIndex = 0, + credentials: VertexUrlCredentials | null = null + ) { + const canonicalModel = normalizeVertexModelId(model); + const configuredProject = configuredVertexProjectId(credentials); + const opaqueApiKey = encodeOpaqueApiKey(credentials); // Vertex AI Express mode: project-less v1 publisher endpoint with the API key passed as a - // ?key= query parameter (verified working contract — same as the CaptionAI GeminiClient). The - // Express key is NOT accepted as a Bearer/OAuth credential or via x-goog-api-key on this API. - if (isExpressApiKey(credentials?.apiKey) && !credentials?.accessToken) { - const expressKey = encodeURIComponent(String(credentials.apiKey).trim()); - if (isPartnerModel(model)) { - // Partner (Anthropic/etc.) models are not available via Express keys; best-effort. - return `https://aiplatform.googleapis.com/v1/publishers/openapi/chat/completions?key=${expressKey}`; - } - const op = stream ? "streamGenerateContent?alt=sse&" : "generateContent?"; - return `https://aiplatform.googleapis.com/v1/publishers/google/models/${model}:${op}key=${expressKey}`; + // ?key= query parameter. Express currently exposes Gemini models only; opaque Authorization + // Keys can use project-scoped APIs when a projectId is configured on the connection. + if (opaqueApiKey && !configuredProject) { + return buildExpressGeminiUrl(canonicalModel, stream, opaqueApiKey); } - + const project = + projectIdFromServiceAccount(credentials) || configuredProject || "unknown-project"; const region = credentials?.providerSpecificData?.region || "us-central1"; - let project = "unknown-project"; - - if (credentials?.apiKey) { - try { - const sa = parseSAFromApiKey(credentials.apiKey); - if (sa.project_id) project = sa.project_id; - } catch { - // Ignored, handled in execute - } - } - - if (isClaudeModel(model)) { - // streamRawPredict?alt=sse was verified to return a single plain JSON body (not real SSE - // framing) rather than actual chunked events, which breaks the SSE parser upstream - // ("stream ended before producing a non-ping SSE event"). rawPredict is confirmed reliable - // for both streaming and non-streaming requests; always use it here. - return `https://aiplatform.googleapis.com/v1/projects/${project}/locations/${region}/publishers/anthropic/models/${model}:rawPredict`; - } - if (isPartnerModel(model)) { - return `https://aiplatform.googleapis.com/v1/projects/${project}/locations/global/endpoints/openapi/chat/completions`; - } - return `https://aiplatform.googleapis.com/v1/projects/${project}/locations/${region}/publishers/google/models/${model}:${stream ? "streamGenerateContent?alt=sse" : "generateContent"}`; + return buildProjectScopedVertexUrl(canonicalModel, stream, project, region, opaqueApiKey); } - buildHeaders(credentials: any, stream = true) { + buildHeaders(credentials: VertexUrlCredentials, stream = true) { const headers: Record = { "Content-Type": "application/json" }; if (credentials.accessToken) { headers["Authorization"] = `Bearer ${credentials.accessToken}`; diff --git a/src/app/(dashboard)/dashboard/providers/[id]/__tests__/useModelImportHandlers.test.tsx b/src/app/(dashboard)/dashboard/providers/[id]/__tests__/useModelImportHandlers.test.tsx index f9909c92f95..f6c4c056472 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/__tests__/useModelImportHandlers.test.tsx +++ b/src/app/(dashboard)/dashboard/providers/[id]/__tests__/useModelImportHandlers.test.tsx @@ -294,3 +294,47 @@ describe("useModelImportHandlers — upstream model auto-fetch", () => { expect(fetchConnections).toHaveBeenCalled(); }); }); + +describe("useModelImportHandlers — imported model metadata", () => { + it("preserves a discovered per-model targetFormat", async () => { + const hook = renderHook( + buildParams({ + providerId: "vertex", + connections: [conn("vertex-connection", true)], + }) + ); + const fetchMock = vi.mocked(fetch); + fetchMock + .mockResolvedValueOnce({ + ok: true, + json: async () => ({ + models: [ + { + id: "grok-4.6", + name: "Grok 4.6", + apiFormat: "chat-completions", + supportedEndpoints: ["chat"], + targetFormat: "openai", + }, + ], + }), + } as Response) + .mockResolvedValue({ ok: true } as Response); + + await act(async () => { + await hook.get().handleImportModels(); + }); + + const createCall = fetchMock.mock.calls.find( + ([url, init]) => url === "/api/provider-models" && init?.method === "POST" + ); + expect(createCall).toBeDefined(); + expect(JSON.parse(String(createCall?.[1]?.body))).toEqual( + expect.objectContaining({ + provider: "vertex", + modelId: "grok-4.6", + targetFormat: "openai", + }) + ); + }); +}); diff --git a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx index 3611433feb1..9fd9439a658 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx +++ b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/AddApiKeyModal.tsx @@ -184,13 +184,15 @@ export default function AddApiKeyModal({ ? providerText(t, "modalTokenIdLabel", "Token ID") : isAwsPolly ? providerText(t, "awsPollySecretAccessKeyLabel", "AWS Secret Access Key") - : isQoder - ? t("personalAccessTokenLabel") - : webSessionCredential - ? getWebSessionCredentialLabel(t, webSessionCredential, apiKeyOptional) - : apiKeyOptional - ? `${t("apiKeyLabel")} (${t("optional").toLowerCase()})` - : t("apiKeyLabel"); + : isVertex + ? providerText(t, "vertexCredentialLabel", "API Key or Service Account JSON") + : isQoder + ? t("personalAccessTokenLabel") + : webSessionCredential + ? getWebSessionCredentialLabel(t, webSessionCredential, apiKeyOptional) + : apiKeyOptional + ? `${t("apiKeyLabel")} (${t("optional").toLowerCase()})` + : t("apiKeyLabel"); const apiCredentialPlaceholder = isModal ? "ak-xxxxxxxxxxxxxxxx" : isVertex @@ -210,19 +212,25 @@ export default function AddApiKeyModal({ "modalTokenIdHint", "Modal auth uses a Token ID + Token Secret pair. Create one at https://modal.com/settings → API Tokens." ) - : isQoder - ? t("qoderPatHint") - : isFreebuff - ? "Freebuff uses an authentic CLI auth token obtained via codebuff CLI login or automated harvester." - : isWebSessionCredential - ? getWebSessionCredentialHint(t, webSessionCredential, providerDisplayName, false) - : isLocalSelfHostedProvider - ? t("localProviderApiKeyOptionalHint", { - provider: localProviderMetadata?.name || providerName || provider || "", - }) - : apiKeyOptional - ? t("apiKeyOptionalHint") - : undefined; + : isVertex + ? providerText( + t, + "vertexCredentialHint", + "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." + ) + : isQoder + ? t("qoderPatHint") + : isFreebuff + ? "Freebuff uses an authentic CLI auth token obtained via codebuff CLI login or automated harvester." + : isWebSessionCredential + ? getWebSessionCredentialHint(t, webSessionCredential, providerDisplayName, false) + : isLocalSelfHostedProvider + ? t("localProviderApiKeyOptionalHint", { + provider: localProviderMetadata?.name || providerName || provider || "", + }) + : apiKeyOptional + ? t("apiKeyOptionalHint") + : undefined; const credentialValidationFailedMessage = isWebSessionCredential ? providerText( t, diff --git a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx index 59d7328fbda..347c5ec3e54 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx +++ b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx @@ -57,6 +57,7 @@ import { assignEditApiKeyProviderSpecificData } from "./connectionProviderSpecif import { isM365TierCapableProvider, normalizeM365TierValue, type M365TierValue } from "./m365Tier"; import ProviderTierField from "./ProviderTierField"; import AgentrouterConsoleFields from "./AgentrouterConsoleFields"; +import { getVertexCredentialCopy } from "./vertexCredentialCopy"; import QuotaScrapingFields, { EMPTY_QUOTA_SCRAPING_FIELDS } from "./QuotaScrapingFields"; import GlmTeamQuotaFields, { EMPTY_GLM_TEAM_QUOTA_FIELDS } from "./GlmTeamQuotaFields"; import ProviderRegionField, { getProviderRegionConfig } from "./AlibabaProviderRegionField"; @@ -245,27 +246,25 @@ export default function EditConnectionModal({ const isCcCompatible = isClaudeCodeCompatibleProvider(provider); const isCompatible = isOpenAICompatibleProvider(provider) || isAnthropicCompatibleProvider(provider); + const vertexCopy = isVertex ? getVertexCredentialCopy(t) : null; const apiCredentialLabel = webSessionCredential ? getWebSessionCredentialLabel(t, webSessionCredential, apiKeyOptional) : isAwsPolly ? providerText(t, "awsPollySecretAccessKeyLabel", "AWS Secret Access Key") - : apiKeyOptional - ? t("apiKeyOptionalLabel") - : t("apiKeyLabel"); + : (vertexCopy?.label ?? (apiKeyOptional ? t("apiKeyOptionalLabel") : t("apiKeyLabel"))); const apiCredentialPlaceholder = isWebSessionCredential ? webSessionCredential.placeholder - : isVertex - ? t("vertexServiceAccountPlaceholder") - : t("enterNewApiKey"); + : (vertexCopy?.placeholder ?? t("enterNewApiKey")); const apiCredentialHint = isWebSessionCredential ? getWebSessionCredentialHint(t, webSessionCredential, providerDisplayName, true) - : isLocalSelfHostedProvider - ? t("localProviderApiKeyOptionalHint", { - provider: localProviderMetadata?.name || provider || "", - }) - : apiKeyOptional - ? t("apiKeyOptionalHint") - : t("leaveBlankKeepCurrentApiKey"); + : (vertexCopy?.hint ?? + (isLocalSelfHostedProvider + ? t("localProviderApiKeyOptionalHint", { + provider: localProviderMetadata?.name || provider || "", + }) + : apiKeyOptional + ? t("apiKeyOptionalHint") + : t("leaveBlankKeepCurrentApiKey"))); // Modal-open form initialization from the loaded connection — applied as a // render-phase adjustment guarded by the previously initialized connection // (react.dev "adjusting state when a prop changes") instead of the former diff --git a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/vertexCredentialCopy.ts b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/vertexCredentialCopy.ts new file mode 100644 index 00000000000..70af8e72b35 --- /dev/null +++ b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/vertexCredentialCopy.ts @@ -0,0 +1,14 @@ +import { providerText, type ProviderMessageTranslator } from "../../providerPageHelpers"; + +/** Vertex-specific credential label/placeholder/hint for the connection modals. */ +export function getVertexCredentialCopy(t: ProviderMessageTranslator) { + return { + label: providerText(t, "vertexCredentialLabel", "API Key or Service Account JSON"), + placeholder: t("vertexServiceAccountPlaceholder"), + hint: providerText( + t, + "vertexCredentialHint", + "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." + ), + }; +} diff --git a/src/app/(dashboard)/dashboard/providers/[id]/hooks/useModelImportHandlers.ts b/src/app/(dashboard)/dashboard/providers/[id]/hooks/useModelImportHandlers.ts index 99370e606e0..88121833bcd 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/hooks/useModelImportHandlers.ts +++ b/src/app/(dashboard)/dashboard/providers/[id]/hooks/useModelImportHandlers.ts @@ -230,6 +230,7 @@ export function useModelImportHandlers({ ...(Array.isArray(model.supportedEndpoints) ? { supportedEndpoints: model.supportedEndpoints } : {}), + ...(typeof model.targetFormat === "string" ? { targetFormat: model.targetFormat } : {}), }), }); if (!modelAliases[baseAlias]) { diff --git a/src/app/api/providers/[id]/models/route.ts b/src/app/api/providers/[id]/models/route.ts index c09a2636b7d..da35a319459 100755 --- a/src/app/api/providers/[id]/models/route.ts +++ b/src/app/api/providers/[id]/models/route.ts @@ -87,7 +87,6 @@ import { } from "@/lib/providerModels/modelDiscovery"; import { buildProviderModelsUrl, getDiscoveryClientVersionOptions } from "./discoveryClientVersion"; import { getAdobeModels } from "./adobeFireflyDiscovery"; -import { parseGeminiModelsList } from "@/lib/providerModels/geminiModelsParser"; import { getSyncedAvailableModels, getCustomModels, getModelIsHidden } from "@/lib/db/models"; import { isConnectionUnavailableToAuxiliaryActivity } from "@/lib/exclusiveLeaseIsolation"; import { fetchCursorAgentModels } from "@/lib/providerModels/cursorAgent"; @@ -129,6 +128,7 @@ import { fetchCodexGithubCatalogModels, } from "./discovery/codex"; import { maybeHandleConolModelDiscovery } from "./conolDiscovery"; +import { maybeHandleVertexModelDiscovery } from "./vertexDiscovery"; import { buildNoAuthModelsResponse, filterModelsForRoute } from "./modelRouteProjection"; /** @@ -616,9 +616,7 @@ export async function GET( try { const discovery = await discoverMaxaiModels({ providerSpecificData: connection.providerSpecificData as - | Record - | null - | undefined, + Record | null | undefined, accessToken: apiKey || accessToken, fetchImpl: (url, init) => safeOutboundFetch(url, { @@ -1788,169 +1786,22 @@ export async function GET( }); } - if (provider === "vertex" || provider === "vertex-partner") { - const cachedResponse = maybeReturnCachedDiscovery(); - if (cachedResponse) return cachedResponse; - - const autoFetchDisabledResponse = maybeReturnAutoFetchDisabled(); - if (autoFetchDisabledResponse) return autoFetchDisabledResponse; - - // Vertex AI lists models from the Generative Language `v1beta/models` endpoint, which both - // Express-mode API keys (via ?key=) and Service Account JSON (via a minted OAuth Bearer - // token) can reach. This surfaces the live catalog, including gemini-*-image models - // absent from the static registry list. - const credential = (apiKey || "").trim(); - let queryKey: string | null = null; - let bearerToken: string | null = null; - try { - const { parseSAFromApiKey, getAccessToken } = - await import("@omniroute/open-sse/executors/vertex.ts"); - if (accessToken) { - bearerToken = accessToken; - } else if (credential) { - // A Service Account credential is a JSON object; a Vertex AI Express-mode API key is an - // opaque (non-JSON) string. Detect locally so this branch has no dependency on optional - // executor helpers. - let isServiceAccountJson = false; - try { - const parsed = JSON.parse(credential); - isServiceAccountJson = !!parsed && typeof parsed === "object" && !Array.isArray(parsed); - } catch { - isServiceAccountJson = false; - } - - if (isServiceAccountJson) { - bearerToken = await getAccessToken(parseSAFromApiKey(credential)); - } else { - queryKey = credential; - } - } - } catch (error) { - // Couldn't resolve a usable credential (e.g. malformed Service Account JSON). - const fallback = buildDiscoveryErrorFallbackResponse(error, { - cacheWarning: "Vertex credential unavailable — using cached catalog", - localWarning: "Vertex credential unavailable — using local catalog", - }); - if (fallback) return fallback; - } - - if (!queryKey && !bearerToken) { - const fallback = buildDiscoveryFallbackResponse({ - cacheWarning: "No usable Vertex credential — using cached catalog", - localWarning: "No usable Vertex credential — using local catalog", - }); - if (fallback) return fallback; - return NextResponse.json( - { error: "No usable Vertex AI credential configured for model discovery." }, - { status: 400 } - ); - } - - const baseUrl = "https://generativelanguage.googleapis.com/v1beta/models?pageSize=1000"; - const headers: Record = { "Content-Type": "application/json" }; - if (bearerToken) headers["Authorization"] = `Bearer ${bearerToken}`; - - const allModels: any[] = []; - let pageUrl = queryKey ? `${baseUrl}&key=${encodeURIComponent(queryKey)}` : baseUrl; - let pageCount = 0; - const MAX_PAGES = 20; - const seenTokens = new Set(); - - try { - while (pageUrl && pageCount < MAX_PAGES) { - pageCount++; - const response = await safeOutboundFetch(pageUrl, { - ...SAFE_OUTBOUND_FETCH_PRESETS.modelsPagination, - guard: getProviderOutboundGuard(), - proxyConfig: proxy, - method: "GET", - headers, - }); - - if (!response.ok) { - // Avoid logging the raw upstream body (may contain sensitive data); status is enough. - console.log("[models] Vertex model discovery failed", { - provider, - status: response.status, - }); - const fallback = buildDiscoveryFallbackResponse(); - if (fallback) return fallback; - return NextResponse.json( - { error: `Failed to fetch Vertex models: ${response.status}` }, - { status: response.status } - ); - } - - const data = await response.json(); - allModels.push(...parseGeminiModelsList(data)); - - const nextPageToken = data.nextPageToken; - if (!nextPageToken || seenTokens.has(nextPageToken)) break; - seenTokens.add(nextPageToken); - pageUrl = `${baseUrl}&pageToken=${encodeURIComponent(nextPageToken)}`; - if (queryKey) pageUrl += `&key=${encodeURIComponent(queryKey)}`; - } - } catch (error) { - const fallback = buildDiscoveryErrorFallbackResponse(error); - if (fallback) return fallback; - throw error; - } - - // Anthropic partner models via Model Garden publisher endpoint (Bearer only). - // - // Model Garden's publisher-model LIST is served by the v1beta1 API — the v1 - // API does not support list operations (every /v1/.../publishers/anthropic/models - // path 404s at the Google Front End). The list is also global: it returns the - // full Anthropic Claude catalog regardless of the connection's project or - // region, so no project/region scoping is applied here (execution region is - // handled separately by the vertex executor at request time). - if (bearerToken) { - const anthropicModelsUrl = - "https://aiplatform.googleapis.com/v1beta1/publishers/anthropic/models"; - - try { - const anthropicResponse = await safeOutboundFetch(anthropicModelsUrl, { - ...SAFE_OUTBOUND_FETCH_PRESETS.modelsDiscovery, - guard: getProviderOutboundGuard(), - proxyConfig: proxy, - method: "GET", - headers: { - "Content-Type": "application/json", - Authorization: `Bearer ${bearerToken}`, - }, - }); - if (anthropicResponse.ok) { - const anthropicData = await anthropicResponse.json(); - const { parseVertexAnthropicModels } = - await import("@/lib/providerModels/vertexAnthropicModelsParser"); - allModels.push(...parseVertexAnthropicModels(anthropicData)); - } else { - console.log("[models] Vertex Anthropic partner discovery failed", { - provider, - status: anthropicResponse.status, - }); - } - } catch (err) { - console.log("[models] Vertex Anthropic partner discovery error", { - provider, - error: err instanceof Error ? err.message : String(err), - }); - } - } - - if (allModels.length > 0) { - return buildApiDiscoveryResponse(allModels); - } - - const fallback = buildDiscoveryFallbackResponse(); - if (fallback) return fallback; - return buildResponse({ - provider, - connectionId, - models: [], - source: "api", - }); - } + const vertexResponse = await maybeHandleVertexModelDiscovery({ + provider, + connectionId, + connection, + apiKey, + accessToken, + proxy, + cachedDiscoveryModels, + maybeReturnCachedDiscovery, + maybeReturnAutoFetchDisabled, + buildDiscoveryFallbackResponse, + buildDiscoveryErrorFallbackResponse, + buildResponse, + buildApiDiscoveryResponse, + }); + if (vertexResponse) return vertexResponse; if (isAnthropicCompatibleProvider(provider)) { // CC providers never support models listing — this check must precede diff --git a/src/app/api/providers/[id]/models/vertexDiscovery.ts b/src/app/api/providers/[id]/models/vertexDiscovery.ts new file mode 100644 index 00000000000..c1173668182 --- /dev/null +++ b/src/app/api/providers/[id]/models/vertexDiscovery.ts @@ -0,0 +1,260 @@ +import { NextResponse } from "next/server"; +import { getModelsByProviderId } from "@/shared/constants/models"; +import { getStaticModelsForProvider } from "@/lib/providers/staticModels"; +import { updateProviderConnection } from "@/lib/db/providers"; +import { SAFE_OUTBOUND_FETCH_PRESETS, safeOutboundFetch } from "@/shared/network/safeOutboundFetch"; +import { getProviderOutboundGuard } from "@/shared/network/outboundUrlGuardPolicy"; +import type { VertexMetadataModel } from "@/lib/providerModels/vertexModelMetadata"; +import { asRecord, mergeLocalCatalogModels, toNonEmptyString } from "./discovery/helpers"; + +interface DiscoveryWarnings { + cacheWarning?: string; + localWarning?: string; +} + +interface VertexDiscoveryRouteOptions { + provider: string; + connectionId: string; + connection: { projectId?: unknown; providerSpecificData?: unknown }; + apiKey: unknown; + accessToken: unknown; + proxy: unknown; + cachedDiscoveryModels: VertexMetadataModel[]; + maybeReturnCachedDiscovery: () => Response | null; + maybeReturnAutoFetchDisabled: () => Response | null; + buildDiscoveryFallbackResponse: (warnings?: DiscoveryWarnings) => Response | null; + buildDiscoveryErrorFallbackResponse: ( + error: unknown, + warnings?: DiscoveryWarnings + ) => Response | null; + buildResponse: (payload: Record) => Response; + buildApiDiscoveryResponse: ( + models: unknown[], + warning?: string, + extraPayload?: Record + ) => Promise; +} + +type CuratedVertexModel = { id: string; name: string; owned_by: string }; + +function asNamedModels(models: unknown[]): Array<{ id: string; name?: string }> { + return models.flatMap((model) => + model && typeof model === "object" && "id" in model && typeof model.id === "string" + ? [model as { id: string; name?: string }] + : [] + ); +} + +function vertexFetch(proxy: unknown): (url: string, init: RequestInit) => Promise { + return (url, init) => + safeOutboundFetch(url, { + ...SAFE_OUTBOUND_FETCH_PRESETS.modelsPagination, + guard: getProviderOutboundGuard(), + proxyConfig: proxy, + ...init, + }); +} + +function vertexMetadataFetch( + proxy: unknown +): (url: string, init: RequestInit) => Promise { + return (url, init) => + safeOutboundFetch(url, { + ...SAFE_OUTBOUND_FETCH_PRESETS.modelsDiscovery, + guard: "public-only", + proxyConfig: proxy, + ...init, + }); +} + +function curatedVertexCatalog(provider: string): CuratedVertexModel[] { + return mergeLocalCatalogModels( + getModelsByProviderId(provider) || [], + getStaticModelsForProvider(provider) || [] + ).map((model) => ({ + ...model, + name: model.name || model.id, + owned_by: provider, + })); +} + +async function resolveVertexDiscoveryAuth(options: { + apiKey: unknown; + accessToken: unknown; +}): Promise<{ queryKey: string | null; bearerToken: string | null }> { + const { parseSAFromApiKey, getAccessToken, looksLikeServiceAccountJson } = + await import("@omniroute/open-sse/executors/vertex.ts"); + const credential = (typeof options.apiKey === "string" ? options.apiKey : "").trim(); + if (typeof options.accessToken === "string" && options.accessToken) { + return { queryKey: null, bearerToken: options.accessToken }; + } + if (!credential) return { queryKey: null, bearerToken: null }; + if (looksLikeServiceAccountJson(credential)) { + return { + queryKey: null, + bearerToken: await getAccessToken(parseSAFromApiKey(credential)), + }; + } + return { queryKey: credential, bearerToken: null }; +} + +async function handleVertexApiKeyCatalog( + options: VertexDiscoveryRouteOptions, + queryKey: string, + catalog: CuratedVertexModel[] +): Promise { + const { discoverVertexModelsWithApiKey } = + await import("@/lib/providerModels/vertexModelDiscovery"); + const discovery = await discoverVertexModelsWithApiKey({ + apiKey: queryKey, + fetchImpl: vertexFetch(options.proxy), + }); + + if (discovery.models.length === 0 && discovery.unavailable) { + const warning = "Vertex model discovery temporarily unavailable — using local catalog"; + return ( + options.buildDiscoveryFallbackResponse({ + cacheWarning: "Vertex model discovery temporarily unavailable — using cached catalog", + localWarning: warning, + }) ?? + options.buildResponse({ + provider: options.provider, + connectionId: options.connectionId, + models: [], + source: "local_catalog", + warning, + }) + ); + } + + const providerData = asRecord(options.connection.providerSpecificData); + const configuredProjectId = + toNonEmptyString(options.connection.projectId) || + toNonEmptyString(providerData.projectId) || + toNonEmptyString(providerData.project); + const projectId = configuredProjectId || discovery.projectId || null; + const projectIdAutoDetected = !configuredProjectId && !!discovery.projectId; + if (projectIdAutoDetected && projectId) { + await updateProviderConnection(options.connectionId, { projectId }); + } + + const { isVertexExpressModel } = await import("@omniroute/open-sse/config/vertexModels.ts"); + if (projectId) { + const liveGeminiModels = asNamedModels(discovery.models); + const projectCatalog = mergeLocalCatalogModels(liveGeminiModels, catalog); + if (liveGeminiModels.length > 0) { + return options.buildApiDiscoveryResponse(projectCatalog, undefined, { + projectIdAutoDetected, + catalogMode: "live_gemini_curated_project", + }); + } + return options.buildResponse({ + provider: options.provider, + connectionId: options.connectionId, + models: projectCatalog, + source: "local_catalog", + intentional: true, + projectIdAutoDetected, + catalogMode: "curated_project", + }); + } + + if (discovery.models.length > 0) { + return options.buildApiDiscoveryResponse(discovery.models, discovery.warning); + } + + return options.buildResponse({ + provider: options.provider, + connectionId: options.connectionId, + models: catalog.filter((model) => isVertexExpressModel(model.id)), + source: "local_catalog", + intentional: true, + warning: + (discovery.failureStatus + ? `Generative Language model listing rejected the API key (HTTP ${discovery.failureStatus}). ` + : "") + "No live catalog available for this API key — using curated Express catalog", + }); +} + +export async function maybeHandleVertexModelDiscovery( + options: VertexDiscoveryRouteOptions +): Promise { + if (options.provider !== "vertex" && options.provider !== "vertex-partner") return null; + + const cachedResponse = options.maybeReturnCachedDiscovery(); + if (cachedResponse) return cachedResponse; + + const autoFetchDisabledResponse = options.maybeReturnAutoFetchDisabled(); + if (autoFetchDisabledResponse) return autoFetchDisabledResponse; + + let queryKey: string | null = null; + let bearerToken: string | null = null; + try { + const resolved = await resolveVertexDiscoveryAuth({ + apiKey: options.apiKey, + accessToken: options.accessToken, + }); + queryKey = resolved.queryKey; + bearerToken = resolved.bearerToken; + } catch (error) { + const fallback = options.buildDiscoveryErrorFallbackResponse(error, { + cacheWarning: "Vertex credential unavailable — using cached catalog", + localWarning: "Vertex credential unavailable — using local catalog", + }); + if (fallback) return fallback; + } + + if (!queryKey && !bearerToken) { + const fallback = options.buildDiscoveryFallbackResponse({ + cacheWarning: "No usable Vertex credential — using cached catalog", + localWarning: "No usable Vertex credential — using local catalog", + }); + if (fallback) return fallback; + return NextResponse.json( + { error: "No usable Vertex AI credential configured for model discovery." }, + { status: 400 } + ); + } + + const catalog = curatedVertexCatalog(options.provider); + if (queryKey) return handleVertexApiKeyCatalog(options, queryKey, catalog); + + const { discoverVertexModelsWithBearer } = + await import("@/lib/providerModels/vertexModelDiscovery"); + const discovery = await discoverVertexModelsWithBearer({ + bearerToken: bearerToken as string, + fetchImpl: vertexFetch(options.proxy), + }); + + if (discovery.models.length > 0) { + const liveModels = asNamedModels(discovery.models); + let enrichedLiveModels = liveModels; + try { + const { enrichVertexModelsWithMetadata } = + await import("@/lib/providerModels/vertexModelMetadata"); + enrichedLiveModels = await enrichVertexModelsWithMetadata({ + models: liveModels, + staleModels: options.cachedDiscoveryModels, + fetchImpl: vertexMetadataFetch(options.proxy), + }); + } catch { + // Metadata is optional. A documentation or parser failure must never turn an authenticated + // live Vertex catalog into a failed model sync. + } + return options.buildApiDiscoveryResponse(enrichedLiveModels, discovery.warning, { + catalogMode: "live_vertex_catalog", + }); + } + + const fallback = options.buildDiscoveryFallbackResponse({ + cacheWarning: "Vertex model catalogs unavailable — using cached catalog", + localWarning: "Vertex model catalogs unavailable — using local catalog", + }); + if (fallback) return fallback; + return options.buildResponse({ + provider: options.provider, + connectionId: options.connectionId, + models: [], + source: "api", + }); +} diff --git a/src/app/api/v1beta/models/route.ts b/src/app/api/v1beta/models/route.ts index bf9ddec175d..d9233b4f433 100644 --- a/src/app/api/v1beta/models/route.ts +++ b/src/app/api/v1beta/models/route.ts @@ -99,7 +99,12 @@ export async function GET() { displayName: m.name || m.id, ...(typeof m.description === "string" ? { description: m.description } : {}), supportedGenerationMethods: ["generateContent"], - inputTokenLimit: typeof m.inputTokenLimit === "number" ? m.inputTokenLimit : 128000, + inputTokenLimit: + typeof m.inputTokenLimit === "number" + ? m.inputTokenLimit + : typeof m.contextWindow === "number" + ? m.contextWindow + : 128000, outputTokenLimit: typeof m.outputTokenLimit === "number" ? m.outputTokenLimit : 8192, ...(m.supportsThinking === true ? { thinking: true } : {}), }); @@ -131,7 +136,9 @@ export async function GET() { inputTokenLimit: typeof m.inputTokenLimit === "number" ? m.inputTokenLimit - : resolved.maxInputTokens || resolved.contextWindow || 128000, + : typeof m.contextWindow === "number" + ? m.contextWindow + : resolved.maxInputTokens || resolved.contextWindow || 128000, outputTokenLimit: typeof m.outputTokenLimit === "number" ? m.outputTokenLimit @@ -173,7 +180,9 @@ export async function GET() { inputTokenLimit: typeof m.inputTokenLimit === "number" ? m.inputTokenLimit - : resolved.maxInputTokens || resolved.contextWindow || 128000, + : typeof m.contextWindow === "number" + ? m.contextWindow + : resolved.maxInputTokens || resolved.contextWindow || 128000, outputTokenLimit: typeof m.outputTokenLimit === "number" ? m.outputTokenLimit diff --git a/src/i18n/messages/ar.json b/src/i18n/messages/ar.json index 1d7e80a0c06..fc632b54871 100644 --- a/src/i18n/messages/ar.json +++ b/src/i18n/messages/ar.json @@ -5963,6 +5963,8 @@ "validationModelIdHint": "النموذج المستخدم للتحقق من الاتصال", "validationModelIdLabel": "معرّف نموذج التحقق", "validationModelIdPlaceholder": "مثال: gpt-4o-mini", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery.", "vertexServiceAccountPlaceholder": "ألصق JSON لحساب الخدمة ({\"type\":\"service_account\",\"project_id\":\"…\",\"client_email\":\"…\",\"private_key\":\"…\"}) أو OAuth access_token", "webCookieProviders": "مزودو ملفات تعريف ارتباط الويب", "weeklyShort": "أسبوعي", diff --git a/src/i18n/messages/az.json b/src/i18n/messages/az.json index f6ca25914a7..128cc27d288 100644 --- a/src/i18n/messages/az.json +++ b/src/i18n/messages/az.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Saxlanmış model son nöqtəsi parametrləri", "searchByModelAria": "Model üzrə axtarış edin", "selectSupportedEndpoint": "Ən azı bir dəstəklənən son nöqtəni seçin", - "antigravityClientProfileHarness": "Test sistemi / CLI" + "antigravityClientProfileHarness": "Test sistemi / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Parametrlər", diff --git a/src/i18n/messages/bg.json b/src/i18n/messages/bg.json index 5ea8ef1e4c8..ebca6eaeab6 100644 --- a/src/i18n/messages/bg.json +++ b/src/i18n/messages/bg.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Настройки на крайна точка на запазен модел", "searchByModelAria": "Търсене по модел", "selectSupportedEndpoint": "Изберете поне една поддържана крайна точка", - "antigravityClientProfileHarness": "Тестова среда / CLI" + "antigravityClientProfileHarness": "Тестова среда / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Настройки", diff --git a/src/i18n/messages/bn.json b/src/i18n/messages/bn.json index 33d59055656..509b86ed1c4 100644 --- a/src/i18n/messages/bn.json +++ b/src/i18n/messages/bn.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "সংরক্ষিত মডেল এন্ডপয়েন্ট সেটিংস", "searchByModelAria": "মডেল দ্বারা অনুসন্ধান করুন", "selectSupportedEndpoint": "কমপক্ষে একটি সমর্থিত এন্ডপয়েন্ট নির্বাচন করুন", - "antigravityClientProfileHarness": "Harness / CLI" + "antigravityClientProfileHarness": "Harness / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "সেটিংস", diff --git a/src/i18n/messages/cs.json b/src/i18n/messages/cs.json index a6d823ccdc0..ddf8bdcbb8b 100644 --- a/src/i18n/messages/cs.json +++ b/src/i18n/messages/cs.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Nastavení koncového bodu uloženého modelu", "searchByModelAria": "Hledat podle modelu", "selectSupportedEndpoint": "Vyberte alespoň jeden podporovaný koncový bod", - "antigravityClientProfileHarness": "Testovací rozhraní / CLI" + "antigravityClientProfileHarness": "Testovací rozhraní / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Nastavení", diff --git a/src/i18n/messages/da.json b/src/i18n/messages/da.json index f345293d50f..9004e921e13 100644 --- a/src/i18n/messages/da.json +++ b/src/i18n/messages/da.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Indstillinger for gemt model endpoint", "searchByModelAria": "Søg efter model", "selectSupportedEndpoint": "Vælg mindst én understøttet endpoint", - "antigravityClientProfileHarness": "Harness / CLI" + "antigravityClientProfileHarness": "Harness / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Indstillinger", diff --git a/src/i18n/messages/de.json b/src/i18n/messages/de.json index e4d6ac4ce5c..a3bbf4ca96e 100644 --- a/src/i18n/messages/de.json +++ b/src/i18n/messages/de.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Einstellungen für den gespeicherten Modell-Endpunkt", "searchByModelAria": "Nach Modell suchen", "selectSupportedEndpoint": "Wählen Sie mindestens einen unterstützten Endpunkt aus", - "antigravityClientProfileHarness": "Testsystem / CLI" + "antigravityClientProfileHarness": "Testsystem / CLI", + "vertexCredentialLabel": "API-Key oder Service-Account-JSON", + "vertexCredentialHint": "API-Keys verwenden den kuratierten Projektkatalog. Ein Service-Account-JSON aktiviert die Live-Erkennung im Model Garden." }, "settings": { "title": "Einstellungen", diff --git a/src/i18n/messages/en.json b/src/i18n/messages/en.json index 0a685beed5a..e8cb19eb44a 100644 --- a/src/i18n/messages/en.json +++ b/src/i18n/messages/en.json @@ -5963,7 +5963,9 @@ "validationModelIdHint": "Model used to verify the API key. Leave blank to use the provider’s first available model.", "validationModelIdLabel": "Validation Model", "validationModelIdPlaceholder": "e.g. meta-llama/llama-3.1-8b-instruct", - "vertexServiceAccountPlaceholder": "Paste your Service Account JSON ({\"type\":\"service_account\",\"project_id\":\"…\",\"client_email\":\"…\",\"private_key\":\"…\"}) or an OAuth access_token", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery.", + "vertexServiceAccountPlaceholder": "Paste an API key, Service Account JSON ({\"type\":\"service_account\",\"project_id\":\"…\",\"client_email\":\"…\",\"private_key\":\"…\"}), or OAuth access token", "webCookieProviders": "Web Cookie Providers", "weeklyShort": "Weekly Short", "xiaomiMimoBaseUrlHint": "Xiaomi Mimo Base Url Hint", diff --git a/src/i18n/messages/es.json b/src/i18n/messages/es.json index 5588aabd008..a234d4afaf7 100644 --- a/src/i18n/messages/es.json +++ b/src/i18n/messages/es.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Configuración del punto final del modelo guardado", "searchByModelAria": "Buscar por modelo", "selectSupportedEndpoint": "Seleccione al menos un endpoint compatible", - "antigravityClientProfileHarness": "Entorno / CLI" + "antigravityClientProfileHarness": "Entorno / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Configuración", diff --git a/src/i18n/messages/fa.json b/src/i18n/messages/fa.json index d8742159bd6..e4f54135888 100644 --- a/src/i18n/messages/fa.json +++ b/src/i18n/messages/fa.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "تنظیمات نقطه پایانی مدل ذخیره شده", "searchByModelAria": "جستجو بر اساس مدل", "selectSupportedEndpoint": "حداقل یک نقطه پایانی پشتیبانی شده را انتخاب کنید", - "antigravityClientProfileHarness": "هارنس / CLI" + "antigravityClientProfileHarness": "هارنس / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "تنظیمات", diff --git a/src/i18n/messages/fi.json b/src/i18n/messages/fi.json index b5b5c054da8..003b0b62a30 100644 --- a/src/i18n/messages/fi.json +++ b/src/i18n/messages/fi.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Tallennetun mallin päätepisteen asetukset", "searchByModelAria": "Hae mallin mukaan", "selectSupportedEndpoint": "Valitse vähintään yksi tuettu päätepiste", - "antigravityClientProfileHarness": "Testauskehys / CLI" + "antigravityClientProfileHarness": "Testauskehys / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Asetukset", diff --git a/src/i18n/messages/fr.json b/src/i18n/messages/fr.json index af56b66d8c5..d48a1b7f194 100644 --- a/src/i18n/messages/fr.json +++ b/src/i18n/messages/fr.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Saved modèles endpoint paramètres", "searchByModelAria": "Rechercher un modèle", "selectSupportedEndpoint": "Sélectionnez au moins un endpoint pris en charge", - "antigravityClientProfileHarness": "Harnais / CLI" + "antigravityClientProfileHarness": "Harnais / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Paramètres", diff --git a/src/i18n/messages/gu.json b/src/i18n/messages/gu.json index 92fc0f0ede9..734ab7194d2 100644 --- a/src/i18n/messages/gu.json +++ b/src/i18n/messages/gu.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "સાચવેલ મોડેલ અંતિમ બિંદુની સેટિંગ્સ", "searchByModelAria": "મોડલ દ્વારા શોધો", "selectSupportedEndpoint": "કમથી કમ એક સમર્થિત અંતિમ બિંદુ પસંદ કરો", - "antigravityClientProfileHarness": "હાર્નેસ / CLI" + "antigravityClientProfileHarness": "હાર્નેસ / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "સેટિંગ્સ", diff --git a/src/i18n/messages/he.json b/src/i18n/messages/he.json index 034fbd6991a..586097d48a8 100644 --- a/src/i18n/messages/he.json +++ b/src/i18n/messages/he.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "הגדרות נקודת הקצה של המודל השמור", "searchByModelAria": "חפש לפי דגם", "selectSupportedEndpoint": "בחר לפחות נקודת קצה אחת נתמכת", - "antigravityClientProfileHarness": "Harness / CLI" + "antigravityClientProfileHarness": "Harness / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "הגדרות", diff --git a/src/i18n/messages/hi.json b/src/i18n/messages/hi.json index 473f9f45e86..d385465c89f 100644 --- a/src/i18n/messages/hi.json +++ b/src/i18n/messages/hi.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "सहेजे गए मॉडल एंडपॉइंट सेटिंग्स", "searchByModelAria": "मॉडल द्वारा खोजें", "selectSupportedEndpoint": "कम से कम एक समर्थित एंडपॉइंट चुनें", - "antigravityClientProfileHarness": "हार्नेस / CLI" + "antigravityClientProfileHarness": "हार्नेस / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "सेटिंग्स", diff --git a/src/i18n/messages/hu.json b/src/i18n/messages/hu.json index c8943186f16..2d586907916 100644 --- a/src/i18n/messages/hu.json +++ b/src/i18n/messages/hu.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Mentett modell végpont beállításai", "searchByModelAria": "Keresés modell szerint", "selectSupportedEndpoint": "Válasszon ki legalább egy támogatott végpontot", - "antigravityClientProfileHarness": "Tesztkeretrendszer / CLI" + "antigravityClientProfileHarness": "Tesztkeretrendszer / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Beállítások elemre", diff --git a/src/i18n/messages/id.json b/src/i18n/messages/id.json index 5976b326491..2c7bba2540f 100644 --- a/src/i18n/messages/id.json +++ b/src/i18n/messages/id.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Pengaturan endpoint model yang disimpan", "searchByModelAria": "Cari berdasarkan model", "selectSupportedEndpoint": "Pilih setidaknya satu endpoint yang didukung", - "antigravityClientProfileHarness": "Harness / CLI" + "antigravityClientProfileHarness": "Harness / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Pengaturan", diff --git a/src/i18n/messages/it.json b/src/i18n/messages/it.json index 39dfe2aeef8..a6985019012 100644 --- a/src/i18n/messages/it.json +++ b/src/i18n/messages/it.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Impostazioni dell'endpoint del modello salvato", "searchByModelAria": "Cerca per modello", "selectSupportedEndpoint": "Seleziona almeno un endpoint supportato", - "antigravityClientProfileHarness": "Harness / CLI" + "antigravityClientProfileHarness": "Harness / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Impostazioni", diff --git a/src/i18n/messages/ja.json b/src/i18n/messages/ja.json index edad2e2f6b2..b7620f50673 100644 --- a/src/i18n/messages/ja.json +++ b/src/i18n/messages/ja.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "保存されたモデルエンドポイント設定", "searchByModelAria": "モデルで検索", "selectSupportedEndpoint": "サポートされているエンドポイントを少なくとも1つ選択してください", - "antigravityClientProfileHarness": "Harness / CLI" + "antigravityClientProfileHarness": "Harness / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "設定", diff --git a/src/i18n/messages/ko.json b/src/i18n/messages/ko.json index 6d4047278f1..c3c3ec7f510 100644 --- a/src/i18n/messages/ko.json +++ b/src/i18n/messages/ko.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "저장된 모델 엔드포인트 설정", "searchByModelAria": "모델로 검색", "selectSupportedEndpoint": "지원되는 엔드포인트를 최소한 하나 선택하세요.", - "antigravityClientProfileHarness": "하네스 / CLI" + "antigravityClientProfileHarness": "하네스 / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "설정", diff --git a/src/i18n/messages/mr.json b/src/i18n/messages/mr.json index 036a6215c69..1dedfc2e32a 100644 --- a/src/i18n/messages/mr.json +++ b/src/i18n/messages/mr.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "सुरक्षित केलेल्या मॉडेल एंडपॉइंट सेटिंग्ज", "searchByModelAria": "मॉडेलद्वारे शोधा", "selectSupportedEndpoint": "किमान एक समर्थित एंडपॉइंट निवडा", - "antigravityClientProfileHarness": "हार्नेस / CLI" + "antigravityClientProfileHarness": "हार्नेस / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "सेटिंग्ज", diff --git a/src/i18n/messages/ms.json b/src/i18n/messages/ms.json index 82d40caada4..1524c32ca8c 100644 --- a/src/i18n/messages/ms.json +++ b/src/i18n/messages/ms.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Tetapan titik akhir model yang disimpan", "searchByModelAria": "Cari mengikut model", "selectSupportedEndpoint": "Pilih sekurang-kurangnya satu titik akhir yang disokong", - "antigravityClientProfileHarness": "Harness / CLI" + "antigravityClientProfileHarness": "Harness / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "tetapan", diff --git a/src/i18n/messages/nl.json b/src/i18n/messages/nl.json index 95eb77fb447..4dfd626472e 100644 --- a/src/i18n/messages/nl.json +++ b/src/i18n/messages/nl.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Instellingen voor opgeslagen model-eindpunt", "searchByModelAria": "Zoeken op model", "selectSupportedEndpoint": "Selecteer ten minste één ondersteunde eindpunt", - "antigravityClientProfileHarness": "Testomgeving / CLI" + "antigravityClientProfileHarness": "Testomgeving / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Instellingen", diff --git a/src/i18n/messages/no.json b/src/i18n/messages/no.json index 1b561286bf3..6f099cd72a3 100644 --- a/src/i18n/messages/no.json +++ b/src/i18n/messages/no.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Innstillinger for lagrede modellendepunkter", "searchByModelAria": "Søk etter modell", "selectSupportedEndpoint": "Velg minst ett støttet endepunkt", - "antigravityClientProfileHarness": "Testverktøy / CLI" + "antigravityClientProfileHarness": "Testverktøy / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Innstillinger", diff --git a/src/i18n/messages/phi.json b/src/i18n/messages/phi.json index 4f7ea295922..4e70b90ac73 100644 --- a/src/i18n/messages/phi.json +++ b/src/i18n/messages/phi.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Naka-save na mga setting ng endpoint ng modelo", "searchByModelAria": "Maghanap ayon sa modelo", "selectSupportedEndpoint": "Pumili ng hindi bababa sa isang sinusuportahang endpoint", - "antigravityClientProfileHarness": "Harness / CLI" + "antigravityClientProfileHarness": "Harness / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Mga setting", diff --git a/src/i18n/messages/pl.json b/src/i18n/messages/pl.json index 454cbb4cd2e..f2a72f1523d 100644 --- a/src/i18n/messages/pl.json +++ b/src/i18n/messages/pl.json @@ -5963,6 +5963,8 @@ "validationModelIdHint": "Model używany do weryfikacji klucza API. Pozostaw puste, aby użyć pierwszego dostępnego model dla provider.", "validationModelIdLabel": "Model weryfikacyjny", "validationModelIdPlaceholder": "np. meta-llama/llama-3.1-8b-instruct", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery.", "vertexServiceAccountPlaceholder": "Wklej Service Account JSON ({\"type\":\"service_account\",\"project_id\":\"…\",\"client_email\":\"…\",\"private_key\":\"…\"}) lub OAuth access_token", "webCookieProviders": "Dostawcy korzystający z ciasteczek WWW", "weeklyShort": "Tygodniowo", diff --git a/src/i18n/messages/pt-BR.json b/src/i18n/messages/pt-BR.json index 5d83758096a..8f79ebef64d 100644 --- a/src/i18n/messages/pt-BR.json +++ b/src/i18n/messages/pt-BR.json @@ -5963,6 +5963,8 @@ "validationModelIdHint": "Usado como fallback se a listagem de modelos não estiver disponível.", "validationModelIdLabel": "ID do Modelo (opcional)", "validationModelIdPlaceholder": "ex.: grok-3 ou meta-llama/Llama-3.1-8B-Instruct", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery.", "vertexServiceAccountPlaceholder": "Cole o JSON da Service Account aqui", "webCookieProviders": "Provedores Web / Cookie", "weeklyShort": "Semanal", diff --git a/src/i18n/messages/pt.json b/src/i18n/messages/pt.json index 136e9920861..9c70c9333c8 100644 --- a/src/i18n/messages/pt.json +++ b/src/i18n/messages/pt.json @@ -5963,6 +5963,8 @@ "validationModelIdHint": "Validation Model Id Hint", "validationModelIdLabel": "Validation Model Id Label", "validationModelIdPlaceholder": "Validation Model Id Placeholder", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery.", "vertexServiceAccountPlaceholder": "Cola o teu JSON de Service Account ({\"type\":\"service_account\",\"project_id\":\"…\",\"client_email\":\"…\",\"private_key\":\"…\"})", "webCookieProviders": "Fornecedores de Cookies Web", "weeklyShort": "Semanal", diff --git a/src/i18n/messages/ro.json b/src/i18n/messages/ro.json index 9c103d9ba8e..933d501d2ff 100644 --- a/src/i18n/messages/ro.json +++ b/src/i18n/messages/ro.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Setările punctului final al modelului salvat", "searchByModelAria": "Caută după model", "selectSupportedEndpoint": "Selectați cel puțin un punct final acceptat", - "antigravityClientProfileHarness": "Cadru / CLI" + "antigravityClientProfileHarness": "Cadru / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Setări", diff --git a/src/i18n/messages/ru.json b/src/i18n/messages/ru.json index 697938d8dfb..b7f8b341539 100644 --- a/src/i18n/messages/ru.json +++ b/src/i18n/messages/ru.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Настройки конечной точки сохраненной модели", "searchByModelAria": "Поиск по модели", "selectSupportedEndpoint": "Выберите хотя бы одну поддерживаемую конечную точку", - "antigravityClientProfileHarness": "Среда / CLI" + "antigravityClientProfileHarness": "Среда / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Настройки", diff --git a/src/i18n/messages/sk.json b/src/i18n/messages/sk.json index 7e1a6d557af..c611a1e736d 100644 --- a/src/i18n/messages/sk.json +++ b/src/i18n/messages/sk.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Nastavenia koncového bodu uloženého modelu", "searchByModelAria": "Hľadať podľa modelu", "selectSupportedEndpoint": "Vyberte aspoň jeden podporovaný koncový bod", - "antigravityClientProfileHarness": "Testovacie prostredie / CLI" + "antigravityClientProfileHarness": "Testovacie prostredie / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Nastavenia", diff --git a/src/i18n/messages/sv.json b/src/i18n/messages/sv.json index 5b0bfe83500..f98d3aadeb1 100644 --- a/src/i18n/messages/sv.json +++ b/src/i18n/messages/sv.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Inställningar för sparad modellslutpunkt", "searchByModelAria": "Sök efter modell", "selectSupportedEndpoint": "Välj minst en stödd slutpunkt", - "antigravityClientProfileHarness": "Testmiljö / CLI" + "antigravityClientProfileHarness": "Testmiljö / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Inställningar", diff --git a/src/i18n/messages/sw.json b/src/i18n/messages/sw.json index b3f6e4fb7c0..80de72824ea 100644 --- a/src/i18n/messages/sw.json +++ b/src/i18n/messages/sw.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Mipangilio ya mwisho wa mfano uliohifadhiwa", "searchByModelAria": "Tafuta kwa mfano", "selectSupportedEndpoint": "Chagua angalau kiunganishi kimoja kinachoungwa mkono", - "antigravityClientProfileHarness": "Kiunzi / CLI" + "antigravityClientProfileHarness": "Kiunzi / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Mipangilio", diff --git a/src/i18n/messages/ta.json b/src/i18n/messages/ta.json index b4de89c2a46..c501b9174d0 100644 --- a/src/i18n/messages/ta.json +++ b/src/i18n/messages/ta.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "சேமிக்கப்பட்ட மாதிரி முடிவுறுப்பு அமைப்புகள்", "searchByModelAria": "மாதிரியில் தேடு", "selectSupportedEndpoint": "குறைந்தது ஒரு ஆதரிக்கப்படும் முடிவுகளைத் தேர்ந்தெடுக்கவும்", - "antigravityClientProfileHarness": "ஹார்னஸ் / CLI" + "antigravityClientProfileHarness": "ஹார்னஸ் / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "அமைப்புகள்", diff --git a/src/i18n/messages/te.json b/src/i18n/messages/te.json index ab48cf6f812..c95427b21df 100644 --- a/src/i18n/messages/te.json +++ b/src/i18n/messages/te.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "సేవ్ చేసిన మోడల్ ఎండ్‌పాయింట్ సెట్టింగ్స్", "searchByModelAria": "మోడల్ ద్వారా శోధించండి", "selectSupportedEndpoint": "కమిషన్ చేయబడిన కనెక్ట్ చేయబడిన ఎండ్‌పాయింట్‌లలో కనీసం ఒకటి ఎంచుకోండి", - "antigravityClientProfileHarness": "Harness / CLI" + "antigravityClientProfileHarness": "Harness / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "సెట్టింగ్లు", diff --git a/src/i18n/messages/th.json b/src/i18n/messages/th.json index 8d512b93d43..ede97a216be 100644 --- a/src/i18n/messages/th.json +++ b/src/i18n/messages/th.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "การตั้งค่า endpoint ของโมเดลที่บันทึกไว้", "searchByModelAria": "ค้นหาตามรุ่น", "selectSupportedEndpoint": "เลือกจุดสิ้นสุดที่รองรับอย่างน้อยหนึ่งจุด", - "antigravityClientProfileHarness": "Harness / CLI" + "antigravityClientProfileHarness": "Harness / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "การตั้งค่า", diff --git a/src/i18n/messages/tr.json b/src/i18n/messages/tr.json index 8eef16e5131..5ab58cb192d 100644 --- a/src/i18n/messages/tr.json +++ b/src/i18n/messages/tr.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Kaydedilmiş model uç noktası ayarları", "searchByModelAria": "Model ile ara", "selectSupportedEndpoint": "En az bir desteklenen uç noktayı seçin", - "antigravityClientProfileHarness": "Harness / CLI" + "antigravityClientProfileHarness": "Harness / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Ayarlar", diff --git a/src/i18n/messages/uk-UA.json b/src/i18n/messages/uk-UA.json index 5b95cd89dde..17d64b60aef 100644 --- a/src/i18n/messages/uk-UA.json +++ b/src/i18n/messages/uk-UA.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "Налаштування кінцевої точки збереженої моделі", "searchByModelAria": "Пошук за моделлю", "selectSupportedEndpoint": "Виберіть принаймні одну підтримувану точку доступу", - "antigravityClientProfileHarness": "Harness / CLI" + "antigravityClientProfileHarness": "Harness / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "Налаштування", diff --git a/src/i18n/messages/ur.json b/src/i18n/messages/ur.json index af87e321f1d..009a15ef547 100644 --- a/src/i18n/messages/ur.json +++ b/src/i18n/messages/ur.json @@ -6478,7 +6478,9 @@ "savedModelEndpointSettings": "محفوظ شدہ ماڈل اینڈپوائنٹ کی ترتیبات", "searchByModelAria": "ماڈل کے ذریعے تلاش کریں", "selectSupportedEndpoint": "کم از کم ایک سپورٹ کردہ اینڈپوائنٹ منتخب کریں", - "antigravityClientProfileHarness": "Harness / CLI" + "antigravityClientProfileHarness": "Harness / CLI", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery." }, "settings": { "title": "ترتیبات", diff --git a/src/i18n/messages/vi.json b/src/i18n/messages/vi.json index f1673cbc487..5908b33964f 100644 --- a/src/i18n/messages/vi.json +++ b/src/i18n/messages/vi.json @@ -5963,6 +5963,8 @@ "validationModelIdHint": "Mô hình được sử dụng để xác minh khóa API. Để trống để sử dụng mô hình khả dụng đầu tiên của nhà cung cấp.", "validationModelIdLabel": "Mô hình xác thực", "validationModelIdPlaceholder": "Ví dụ: meta-llama/llama-3.1-8b-instruct", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery.", "vertexServiceAccountPlaceholder": "Dán JSON Tài khoản dịch vụ của bạn ({\"type\":\"service_account\",\"project_id\":\"…\",\"client_email\":\"…\",\"private_key\":\"…\"}) hoặc access_token OAuth", "webCookieProviders": "Nhà cung cấp Web Cookie", "weeklyShort": "Hàng tuần", diff --git a/src/i18n/messages/zh-CN.json b/src/i18n/messages/zh-CN.json index d1b93cb084b..05bd00fa510 100644 --- a/src/i18n/messages/zh-CN.json +++ b/src/i18n/messages/zh-CN.json @@ -5963,6 +5963,8 @@ "validationModelIdHint": "用于验证此提供者连接的模型 ID。", "validationModelIdLabel": "验证模型 ID", "validationModelIdPlaceholder": "输入用于测试的模型 ID", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery.", "vertexServiceAccountPlaceholder": "粘贴 Service Account JSON({\"type\":\"service_account\"...})或 OAuth access_token", "webCookieProviders": "Web Cookie 提供者", "weeklyShort": "每周", diff --git a/src/i18n/messages/zh-TW.json b/src/i18n/messages/zh-TW.json index 4ed7fbecceb..4eee594e494 100644 --- a/src/i18n/messages/zh-TW.json +++ b/src/i18n/messages/zh-TW.json @@ -5963,6 +5963,8 @@ "validationModelIdHint": "用於驗證此提供者連線的模型 ID。", "validationModelIdLabel": "驗證模型 ID", "validationModelIdPlaceholder": "輸入用於測試的模型 ID", + "vertexCredentialLabel": "API Key or Service Account JSON", + "vertexCredentialHint": "API keys use the curated project catalog. Service Account JSON enables live Model Garden discovery.", "vertexServiceAccountPlaceholder": "貼上 Service Account JSON({\"type\":\"service_account\"...})或 OAuth access_token", "webCookieProviders": "Web Cookie Provider", "weeklyShort": "每週", diff --git a/src/lib/combos/builderOptions.ts b/src/lib/combos/builderOptions.ts index 0f1cb529cca..945479b2c9a 100644 --- a/src/lib/combos/builderOptions.ts +++ b/src/lib/combos/builderOptions.ts @@ -30,6 +30,7 @@ type CustomModelLike = { apiFormat?: string; supportedEndpoints?: string[]; inputTokenLimit?: number; + contextWindow?: number; outputTokenLimit?: number; supportsThinking?: boolean; isHidden?: boolean; @@ -41,6 +42,7 @@ type SyncedModelLike = { source?: string; supportedEndpoints?: string[]; inputTokenLimit?: number; + contextWindow?: number; outputTokenLimit?: number; description?: string; supportsThinking?: boolean; @@ -384,7 +386,10 @@ function buildModelOptions( name: toStringOrNull(model.name), source: "imported", supportedEndpoints: toStringArray(model.supportedEndpoints), - contextLength: toNumberOrNull(model.inputTokenLimit) ?? resolved.contextWindow, + contextLength: + toNumberOrNull(model.contextWindow) ?? + toNumberOrNull(model.inputTokenLimit) ?? + resolved.contextWindow, outputTokenLimit: toNumberOrNull(model.outputTokenLimit) ?? resolved.maxOutputTokens, supportsThinking: typeof model.supportsThinking === "boolean" @@ -528,7 +533,7 @@ function buildModelOptions( source, supportedEndpoints: toStringArray(model.supportedEndpoints), apiFormat: toStringOrNull(model.apiFormat), - contextLength: toNumberOrNull(model.inputTokenLimit), + contextLength: toNumberOrNull(model.contextWindow) ?? toNumberOrNull(model.inputTokenLimit), outputTokenLimit: toNumberOrNull(model.outputTokenLimit), supportsThinking: typeof model.supportsThinking === "boolean" ? model.supportsThinking : undefined, diff --git a/src/lib/contextWindowResolver.ts b/src/lib/contextWindowResolver.ts index 8e7124ea85f..fc8c9895ba7 100644 --- a/src/lib/contextWindowResolver.ts +++ b/src/lib/contextWindowResolver.ts @@ -75,12 +75,19 @@ export function reconcileContextWindows( /** Flatten the per-provider discovery map into the reconcile input. */ function toDiscoveredWindows( - byProvider: Record> + byProvider: Record< + string, + Array<{ id: string; contextWindow?: number; inputTokenLimit?: number }> + > ): DiscoveredWindow[] { const out: DiscoveredWindow[] = []; for (const [provider, models] of Object.entries(byProvider)) { for (const m of models) { - out.push({ provider, modelId: m.id, window: m.inputTokenLimit ?? null }); + out.push({ + provider, + modelId: m.id, + window: m.contextWindow ?? m.inputTokenLimit ?? null, + }); } } return out; diff --git a/src/lib/db/models/synced.ts b/src/lib/db/models/synced.ts index 92f2c4eb7ac..1e282265a80 100644 --- a/src/lib/db/models/synced.ts +++ b/src/lib/db/models/synced.ts @@ -1,4 +1,5 @@ import { isRetiredGitHubCopilotModelId } from "@omniroute/open-sse/config/providers/registry/github/retiredModels.ts"; +import type { VertexModelMetadataProvenance } from "@/lib/providerModels/vertexModelMetadata"; import { asRecord, toNonEmptyString } from "./shared"; @@ -13,7 +14,9 @@ export interface SyncedAvailableModel { supportedThinkingEfforts?: string[]; defaultThinkingEffort?: string; inputTokenLimit?: number; + contextWindow?: number; outputTokenLimit?: number; + metadataProvenance?: VertexModelMetadataProvenance; description?: string; supportsThinking?: boolean; alwaysThinking?: boolean; @@ -76,9 +79,15 @@ function normalizeSyncedAvailableModel(model: unknown): SyncedAvailableModel | n ...(typeof record.inputTokenLimit === "number" ? { inputTokenLimit: record.inputTokenLimit } : {}), + ...(typeof record.contextWindow === "number" ? { contextWindow: record.contextWindow } : {}), ...(typeof record.outputTokenLimit === "number" ? { outputTokenLimit: record.outputTokenLimit } : {}), + ...(record.metadataProvenance && typeof record.metadataProvenance === "object" + ? { + metadataProvenance: record.metadataProvenance as VertexModelMetadataProvenance, + } + : {}), ...(typeof record.description === "string" ? { description: record.description } : {}), ...(typeof record.supportsThinking === "boolean" ? { supportsThinking: record.supportsThinking } diff --git a/src/lib/providerModels/managedModelImport.ts b/src/lib/providerModels/managedModelImport.ts index 8faff3613d7..173ae910710 100644 --- a/src/lib/providerModels/managedModelImport.ts +++ b/src/lib/providerModels/managedModelImport.ts @@ -26,6 +26,7 @@ import { isDiscoverableAgyModelId } from "@omniroute/open-sse/config/agyModels.t import { filterChatSelectableModels } from "@omniroute/open-sse/services/modelEndpointPolicy.ts"; import { filterSelectableModels } from "@omniroute/open-sse/services/modelLifecycle.ts"; import { isSelfHostedChatProvider } from "@/shared/constants/providers"; +import type { VertexModelMetadataProvenance } from "@/lib/providerModels/vertexModelMetadata"; type JsonRecord = Record; @@ -42,7 +43,9 @@ export type ManagedImportedModel = { supportedThinkingEfforts?: string[]; defaultThinkingEffort?: string; inputTokenLimit?: number; + contextWindow?: number; outputTokenLimit?: number; + metadataProvenance?: VertexModelMetadataProvenance; description?: string; supportsThinking?: boolean; alwaysThinking?: boolean; @@ -77,9 +80,13 @@ function copyImportedModelMetadata(target: ManagedImportedModel, model: JsonReco target.defaultThinkingEffort = model.defaultThinkingEffort as string; } if (typeof model.inputTokenLimit === "number") target.inputTokenLimit = model.inputTokenLimit; + if (typeof model.contextWindow === "number") target.contextWindow = model.contextWindow; if (typeof model.outputTokenLimit === "number") { target.outputTokenLimit = model.outputTokenLimit; } + if (model.metadataProvenance && typeof model.metadataProvenance === "object") { + target.metadataProvenance = model.metadataProvenance as VertexModelMetadataProvenance; + } if (typeof model.description === "string") target.description = model.description; if (typeof model.supportsThinking === "boolean") { target.supportsThinking = model.supportsThinking; @@ -124,6 +131,7 @@ function copyComparableModelMetadata(target: JsonRecord, model: JsonRecord): voi target.defaultThinkingEffort = model.defaultThinkingEffort; } if (typeof model.inputTokenLimit === "number") target.inputTokenLimit = model.inputTokenLimit; + if (typeof model.contextWindow === "number") target.contextWindow = model.contextWindow; if (typeof model.outputTokenLimit === "number") { target.outputTokenLimit = model.outputTokenLimit; } @@ -307,7 +315,9 @@ export async function importManagedModels({ supportedThinkingEfforts?: string[]; defaultThinkingEffort?: string; inputTokenLimit?: number; + contextWindow?: number; outputTokenLimit?: number; + metadataProvenance?: VertexModelMetadataProvenance; description?: string; supportsThinking?: boolean; alwaysThinking?: boolean; diff --git a/src/lib/providerModels/modelDiscovery.ts b/src/lib/providerModels/modelDiscovery.ts index 9a5697c54a6..4e5cf1c3005 100644 --- a/src/lib/providerModels/modelDiscovery.ts +++ b/src/lib/providerModels/modelDiscovery.ts @@ -306,16 +306,20 @@ export function normalizeDiscoveredModels( const topProvider = asRecord(record.top_provider); - // OpenRouter (and similar passthrough catalogs) report the context window as - // `context_length` / `top_provider.context_length`, not `inputTokenLimit`. - // Fall back across those names so synced models carry a real window instead - // of the provider default (128K). Explicit `inputTokenLimit` still wins. #3202 - const inputTokenLimit = firstPositiveNumber( - record.inputTokenLimit, + // Keep the total context window distinct from an explicit maximum-input limit. Existing + // providers historically stored context_length as inputTokenLimit, so retain that compatibility + // outside Vertex while persisting the separate contextWindow field for new consumers. + const contextWindow = firstPositiveNumber( record.context_length, record.contextLength, + record.contextWindow, topProvider.context_length ); + const isVertexProvider = providerId === "vertex" || providerId === "vertex-partner"; + const inputTokenLimit = firstPositiveNumber( + record.inputTokenLimit, + ...(isVertexProvider ? [] : [contextWindow]) + ); const outputTokenLimit = firstPositiveNumber( record.outputTokenLimit, topProvider.max_completion_tokens @@ -346,7 +350,11 @@ export function normalizeDiscoveredModels( ...(supportedThinkingEfforts !== undefined ? { supportedThinkingEfforts } : {}), ...(defaultThinkingEffort !== undefined ? { defaultThinkingEffort } : {}), ...(typeof inputTokenLimit === "number" ? { inputTokenLimit } : {}), + ...(isVertexProvider && typeof contextWindow === "number" ? { contextWindow } : {}), ...(typeof outputTokenLimit === "number" ? { outputTokenLimit } : {}), + ...(record.metadataProvenance && typeof record.metadataProvenance === "object" + ? { metadataProvenance: record.metadataProvenance } + : {}), ...(typeof record.description === "string" ? { description: record.description } : {}), ...(typeof record.supportsThinking === "boolean" ? { supportsThinking: record.supportsThinking } diff --git a/src/lib/providerModels/vertexModelDiscovery.ts b/src/lib/providerModels/vertexModelDiscovery.ts new file mode 100644 index 00000000000..a7d99fac8ad --- /dev/null +++ b/src/lib/providerModels/vertexModelDiscovery.ts @@ -0,0 +1,239 @@ +import { parseGeminiModelsList } from "@/lib/providerModels/geminiModelsParser"; +import { parseVertexPublisherModels } from "@/lib/providerModels/vertexPublisherModelsParser"; + +const GOOGLE_MODELS_URL = "https://generativelanguage.googleapis.com/v1beta/models?pageSize=1000"; +const VERTEX_PUBLISHER_PAGE_SIZE = 300; +const MAX_CATALOG_PAGES = 20; + +/** + * Serverless chat publishers currently documented by Vertex Model Garden. The parser and executor + * remain publisher-generic, so newly returned model versions need no source change. + */ +export const VERTEX_MODEL_GARDEN_PUBLISHERS = [ + "google", + "anthropic", + "xai", + "mistralai", + "meta", + "deepseek-ai", + "qwen", + "moonshotai", + "minimaxai", + "openai", + "zai-org", +] as const; + +export type VertexModelDiscoveryFetch = (url: string, init: RequestInit) => Promise; + +export interface VertexModelDiscoveryResult { + models: unknown[]; + warning?: string; + projectId?: string; + failureStatus?: number; + unavailable?: boolean; +} + +interface DiscoveryAuth { + headers: Record; +} + +function asObject(value: unknown): Record | null { + return value && typeof value === "object" && !Array.isArray(value) + ? (value as Record) + : null; +} + +function mergeDiscoveryModelsById(models: unknown[]): unknown[] { + const merged = new Map>(); + const unkeyed: unknown[] = []; + + for (const candidate of models) { + const model = asObject(candidate); + if (!model) { + unkeyed.push(candidate); + continue; + } + const id = typeof model.id === "string" ? model.id : null; + if (!id) { + unkeyed.push(candidate); + continue; + } + const existing = merged.get(id); + // Earlier sources have higher precedence. The Generative Language API is queried before the + // publisher catalog, so its structured token limits survive while publisher transport fields + // fill gaps. + merged.set(id, existing ? { ...model, ...existing } : model); + } + + return [...merged.values(), ...unkeyed]; +} + +function readNextPageToken(data: unknown): string | null { + const token = asObject(data)?.nextPageToken; + return typeof token === "string" && token.length > 0 ? token : null; +} + +function readConsumerProjectId(consumer: unknown): string | null { + if (typeof consumer !== "string") return null; + const match = consumer.match(/^projects\/([^/]+)$/); + return match?.[1] ?? null; +} + +function readApiKeyConsumerProjectId(data: unknown): string | null { + const details = asObject(asObject(data)?.error)?.details; + if (!Array.isArray(details)) return null; + for (const detail of details) { + const projectId = readConsumerProjectId(asObject(asObject(detail)?.metadata)?.consumer); + if (projectId) return projectId; + } + return null; +} + +async function discoverVertexModels(options: { + auth: DiscoveryAuth; + fetchImpl: VertexModelDiscoveryFetch; +}): Promise { + const { auth, fetchImpl } = options; + const models: unknown[] = []; + let publisherFailureCount = 0; + + try { + let pageUrl = GOOGLE_MODELS_URL; + let pageCount = 0; + const seenTokens = new Set(); + + while (pageUrl && pageCount < MAX_CATALOG_PAGES) { + pageCount += 1; + const response = await fetchImpl(pageUrl, { method: "GET", headers: auth.headers }); + if (!response.ok) { + break; + } + + const data = await response.json(); + models.push(...parseGeminiModelsList(data)); + const nextPageToken = readNextPageToken(data); + if (!nextPageToken || seenTokens.has(nextPageToken)) break; + seenTokens.add(nextPageToken); + pageUrl = `${GOOGLE_MODELS_URL}&pageToken=${encodeURIComponent(nextPageToken)}`; + } + } catch { + // The Generative Language models API intentionally rejects Service Accounts. Publisher + // discovery below is independent and remains authoritative for the live partner catalog. + } + + const publisherResults = await Promise.all( + VERTEX_MODEL_GARDEN_PUBLISHERS.map(async (publisher) => { + const publisherModels: unknown[] = []; + let pageUrl = + `https://aiplatform.googleapis.com/v1beta1/publishers/${publisher}/models` + + `?pageSize=${VERTEX_PUBLISHER_PAGE_SIZE}`; + let pageCount = 0; + const seenTokens = new Set(); + + try { + while (pageUrl && pageCount < MAX_CATALOG_PAGES) { + pageCount += 1; + const response = await fetchImpl(pageUrl, { method: "GET", headers: auth.headers }); + if (!response.ok) { + publisherFailureCount += 1; + break; + } + + const data = await response.json(); + publisherModels.push(...parseVertexPublisherModels(data, publisher)); + const nextPageToken = readNextPageToken(data); + if (!nextPageToken || seenTokens.has(nextPageToken)) break; + seenTokens.add(nextPageToken); + pageUrl = + `https://aiplatform.googleapis.com/v1beta1/publishers/${publisher}/models` + + `?pageSize=${VERTEX_PUBLISHER_PAGE_SIZE}&pageToken=${encodeURIComponent(nextPageToken)}`; + } + } catch { + publisherFailureCount += 1; + } + + return publisherModels; + }) + ); + for (const publisherModels of publisherResults) models.push(...publisherModels); + + return { + models: mergeDiscoveryModelsById(models), + ...(publisherFailureCount > 0 && models.length > 0 + ? { warning: "Some Vertex catalogs were unavailable — imported available models" } + : {}), + }; +} + +/** Discover live Gemini and Model Garden catalogs with OAuth or Service Account credentials. */ +export function discoverVertexModelsWithBearer(options: { + bearerToken: string; + fetchImpl: VertexModelDiscoveryFetch; +}): Promise { + return discoverVertexModels({ + auth: { + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${options.bearerToken}`, + }, + }, + fetchImpl: options.fetchImpl, + }); +} + +/** + * Discover the Gemini catalog exposed to an API key and recover its Google Cloud consumer project + * from structured error metadata when API restrictions block models.list. Vertex Model Garden's + * publishers.models.list rejects API-key authentication even when the same key is authorized for + * project-scoped partner inference, so publisher discovery is deliberately Bearer-only. + */ +export async function discoverVertexModelsWithApiKey(options: { + apiKey: string; + fetchImpl: VertexModelDiscoveryFetch; +}): Promise { + const models: unknown[] = []; + const headers = { + "Content-Type": "application/json", + // Keep the secret out of URLs and any URL-bearing error/log path. + "x-goog-api-key": options.apiKey, + }; + let pageUrl = GOOGLE_MODELS_URL; + let pageCount = 0; + const seenTokens = new Set(); + + try { + while (pageUrl && pageCount < MAX_CATALOG_PAGES) { + pageCount += 1; + const response = await options.fetchImpl(pageUrl, { method: "GET", headers }); + const data = await response.json().catch(() => null); + if (!response.ok) { + const projectId = readApiKeyConsumerProjectId(data); + return { + models, + failureStatus: response.status, + unavailable: ![400, 401, 403].includes(response.status), + ...(projectId ? { projectId } : {}), + ...(models.length > 0 + ? { warning: "Some Vertex Gemini catalog pages were unavailable" } + : {}), + }; + } + + models.push(...parseGeminiModelsList(data)); + const nextPageToken = readNextPageToken(data); + if (!nextPageToken || seenTokens.has(nextPageToken)) break; + seenTokens.add(nextPageToken); + pageUrl = `${GOOGLE_MODELS_URL}&pageToken=${encodeURIComponent(nextPageToken)}`; + } + } catch { + return { + models, + unavailable: true, + ...(models.length > 0 + ? { warning: "Some Vertex Gemini catalog pages were unavailable" } + : {}), + }; + } + + return { models }; +} diff --git a/src/lib/providerModels/vertexModelMetadata.ts b/src/lib/providerModels/vertexModelMetadata.ts new file mode 100644 index 00000000000..53409dfb9ba --- /dev/null +++ b/src/lib/providerModels/vertexModelMetadata.ts @@ -0,0 +1,379 @@ +const VERTEX_DOCS_BASE = "https://docs.cloud.google.com/gemini-enterprise-agent-platform/models"; +const VERTEX_DOCS_PARSER_VERSION = "vertex-docs-v1"; +const FRESH_CACHE_TTL_MS = 24 * 60 * 60 * 1000; +const STALE_CACHE_TTL_MS = 7 * 24 * 60 * 60 * 1000; +const MAX_CONCURRENT_FETCHES = 6; + +export type VertexMetadataField = "contextWindow" | "inputTokenLimit" | "outputTokenLimit"; + +export interface VertexDocsMetadataProvenance { + source: "google-cloud-docs"; + sourceUrl: string; + fetchedAt: string; + lastModified?: string; + parserVersion: typeof VERTEX_DOCS_PARSER_VERSION; + confidence: "verified"; + fields: VertexMetadataField[]; +} + +export interface VertexModelMetadataProvenance { + vertexDocs: VertexDocsMetadataProvenance; +} + +export interface VertexModelMetadata { + contextWindow?: number; + inputTokenLimit?: number; + outputTokenLimit?: number; + metadataProvenance?: VertexModelMetadataProvenance; +} + +export interface VertexMetadataModel { + id: string; + name?: string; + contextWindow?: number; + inputTokenLimit?: number; + outputTokenLimit?: number; + metadataProvenance?: VertexModelMetadataProvenance; +} + +export type VertexMetadataFetch = (url: string, init: RequestInit) => Promise; + +interface ParsedDocsMetadata { + contextWindow?: number; + inputTokenLimit?: number; + outputTokenLimit?: number; +} + +interface CachedDocsPage { + fetchedAtMs: number; + rows: string[][]; + sourceUrl: string; + lastModified?: string; +} + +const docsCache = new Map(); +const docsInflight = new Map>(); + +function terminalModelId(modelId: string): string { + const segments = modelId.trim().split("/"); + return segments[segments.length - 1] || modelId.trim(); +} + +function docsSlug(value: string): string { + return value.toLowerCase().replace(/[._]/g, "-"); +} + +function unique(values: string[]): string[] { + return Array.from(new Set(values)); +} + +/** Resolve official Google Cloud documentation candidates without using user-controlled hosts. */ +export function resolveVertexModelDocsUrls(modelId: string): string[] { + const normalized = modelId.trim().toLowerCase(); + const bareId = terminalModelId(normalized); + const slug = docsSlug(bareId); + + if (/^gemini-/.test(bareId)) { + return [`${VERTEX_DOCS_BASE}/gemini/${slug.replace(/^gemini-/, "")}?hl=en`]; + } + + if (/^claude-/.test(bareId)) { + return [`${VERTEX_DOCS_BASE}/partner-models/claude/${slug.replace(/^claude-/, "")}?hl=en`]; + } + + if (/^mistral-/.test(bareId)) { + return [`${VERTEX_DOCS_BASE}/partner-models/mistral/${slug}?hl=en`]; + } + + if (normalized.startsWith("xai/") && /^grok-/.test(bareId)) { + const familySlug = slug.replace(/-(?:non-)?reasoning$/, ""); + return [`${VERTEX_DOCS_BASE}/partner-models/grok/${familySlug}?hl=en`]; + } + + const namespaceSeparator = normalized.indexOf("/"); + if (namespaceSeparator > 0) { + const publisher = docsSlug(normalized.slice(0, namespaceSeparator)); + const withoutMaasSuffix = slug.replace(/-maas$/, ""); + return unique([ + `${VERTEX_DOCS_BASE}/maas/${publisher}/${slug}?hl=en`, + `${VERTEX_DOCS_BASE}/maas/${publisher}/${withoutMaasSuffix}?hl=en`, + ]); + } + + return []; +} + +function parsePositiveInteger(value: string | undefined): number | undefined { + if (!value) return undefined; + const normalized = value.replace(/[\s,]/g, ""); + if (!/^\d+$/.test(normalized)) return undefined; + const parsed = Number(normalized); + return Number.isSafeInteger(parsed) && parsed > 0 ? parsed : undefined; +} + +function extractLabeledNumber(text: string, label: RegExp): number | undefined { + const match = text.match(new RegExp(`${label.source}\\s*:?\\s*([0-9][0-9,]*)`, "i")); + return parsePositiveInteger(match?.[1]); +} + +function modelIdAppearsInRow(rowText: string, modelId: string): boolean { + const escaped = modelId.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); + return new RegExp(`(?:^|[^a-z0-9._/-])${escaped}(?:$|[^a-z0-9._/-])`, "i").test(rowText); +} + +function decodeHtmlEntities(value: string): string { + const namedEntities: Record = { + amp: "&", + apos: "'", + gt: ">", + lt: "<", + nbsp: " ", + quot: '"', + }; + + return value.replace(/&(#(?:x[0-9a-f]+|\d+)|[a-z]+);/gi, (entity, token: string) => { + if (token.startsWith("#")) { + const hexadecimal = token[1]?.toLowerCase() === "x"; + const parsed = Number.parseInt(token.slice(hexadecimal ? 2 : 1), hexadecimal ? 16 : 10); + if (Number.isInteger(parsed) && parsed >= 0 && parsed <= 0x10ffff) { + return String.fromCodePoint(parsed); + } + return entity; + } + return namedEntities[token.toLowerCase()] ?? entity; + }); +} + +function htmlFragmentToText(fragment: string): string { + return decodeHtmlEntities( + fragment + .replace(/<(?:script|style)\b[^>]*>[\s\S]*?<\/(?:script|style)\s*>/gi, " ") + .replace(/<(?:br|hr)\b[^>]*\/?\s*>/gi, " ") + .replace(/<[^>]+>/g, " ") + ) + .replace(/\s+/g, " ") + .trim(); +} + +/** + * Extract only the table cells needed for metadata matching. Google Cloud model pages are around + * 400 KB each; constructing full JSDOM windows for the live catalog expanded those pages to + * hundreds of megabytes. Keeping compact row text avoids both that transient heap spike and + * retaining raw HTML in the 24-hour cache. + */ +function extractTableRows(html: string): string[][] { + const rows: string[][] = []; + const rowPattern = /]*>([\s\S]*?)<\/tr\s*>/gi; + let rowMatch: RegExpExecArray | null; + while ((rowMatch = rowPattern.exec(html))) { + const cells: string[] = []; + const cellPattern = /<(th|td)\b[^>]*>([\s\S]*?)<\/\1\s*>/gi; + let cellMatch: RegExpExecArray | null; + while ((cellMatch = cellPattern.exec(rowMatch[1]))) { + cells.push(htmlFragmentToText(cellMatch[2])); + } + if (cells.length > 0) rows.push(cells); + } + return rows; +} + +function parseVertexModelDocsRows( + rows: string[][], + expectedModelId: string +): ParsedDocsMetadata | null { + const bareId = terminalModelId(expectedModelId); + const hasExactModelId = rows.some((cells) => { + const label = cells[0]?.toLowerCase(); + const value = cells.slice(1).join(" "); + return label === "model id" && modelIdAppearsInRow(value, bareId); + }); + if (!hasExactModelId) return null; + + const parsed: ParsedDocsMetadata = {}; + for (const cells of rows) { + const text = cells.join(" "); + + const contextWindow = + extractLabeledNumber(text, /context window/i) ?? + extractLabeledNumber(text, /context length/i); + const inputTokenLimit = extractLabeledNumber(text, /maximum input tokens/i); + const outputTokenLimit = extractLabeledNumber(text, /maximum output tokens/i); + if (parsed.contextWindow === undefined && contextWindow !== undefined) { + parsed.contextWindow = contextWindow; + } + if (parsed.inputTokenLimit === undefined && inputTokenLimit !== undefined) { + parsed.inputTokenLimit = inputTokenLimit; + } + if (parsed.outputTokenLimit === undefined && outputTokenLimit !== undefined) { + parsed.outputTokenLimit = outputTokenLimit; + } + } + + return Object.keys(parsed).length > 0 ? parsed : null; +} + +/** + * Parse token metadata only when an official semantic "Model ID" row contains the exact live id. + * This prevents a family, navigation, or similar-model page from lending limits to another model. + */ +export function parseVertexModelDocsHtml( + html: string, + expectedModelId: string +): ParsedDocsMetadata | null { + return parseVertexModelDocsRows(extractTableRows(html), expectedModelId); +} + +function getCachedPage(url: string, nowMs: number, maxAgeMs: number): CachedDocsPage | null { + const cached = docsCache.get(url); + return cached && nowMs - cached.fetchedAtMs <= maxAgeMs ? cached : null; +} + +async function fetchDocsPage(options: { + url: string; + fetchImpl: VertexMetadataFetch; + nowMs: number; +}): Promise { + const { url, fetchImpl, nowMs } = options; + const fresh = getCachedPage(url, nowMs, FRESH_CACHE_TTL_MS); + if (fresh) return fresh; + + const existingInflight = docsInflight.get(url); + if (existingInflight) return existingInflight; + + const request = (async () => { + try { + const response = await fetchImpl(url, { + method: "GET", + headers: { Accept: "text/html", "Accept-Language": "en" }, + }); + if (!response.ok) return getCachedPage(url, nowMs, STALE_CACHE_TTL_MS); + + const html = await response.text(); + const page: CachedDocsPage = { + fetchedAtMs: nowMs, + rows: extractTableRows(html), + sourceUrl: url, + ...(response.headers.get("last-modified") + ? { lastModified: response.headers.get("last-modified")! } + : {}), + }; + docsCache.set(url, page); + return page; + } catch { + return getCachedPage(url, nowMs, STALE_CACHE_TTL_MS); + } + })().finally(() => docsInflight.delete(url)); + + docsInflight.set(url, request); + return request; +} + +function metadataFields(metadata: ParsedDocsMetadata): VertexMetadataField[] { + return (["contextWindow", "inputTokenLimit", "outputTokenLimit"] as const).filter( + (field) => typeof metadata[field] === "number" + ); +} + +function readPersistedDocsMetadata( + model: VertexMetadataModel | undefined, + nowMs: number +): VertexModelMetadata | null { + if (!model) return null; + const provenance = model.metadataProvenance as VertexModelMetadataProvenance | undefined; + const fetchedAtMs = Date.parse(provenance?.vertexDocs?.fetchedAt || ""); + if (!Number.isFinite(fetchedAtMs) || nowMs - fetchedAtMs > STALE_CACHE_TTL_MS) return null; + const verifiedFields = new Set(provenance?.vertexDocs?.fields || []); + + const metadata: VertexModelMetadata = { + ...(verifiedFields.has("contextWindow") && typeof model.contextWindow === "number" + ? { contextWindow: model.contextWindow } + : {}), + ...(verifiedFields.has("inputTokenLimit") && typeof model.inputTokenLimit === "number" + ? { inputTokenLimit: model.inputTokenLimit } + : {}), + ...(verifiedFields.has("outputTokenLimit") && typeof model.outputTokenLimit === "number" + ? { outputTokenLimit: model.outputTokenLimit } + : {}), + metadataProvenance: provenance, + }; + return metadataFields(metadata).length > 0 ? metadata : null; +} + +async function mapWithConcurrency( + values: T[], + concurrency: number, + mapper: (value: T) => Promise +): Promise { + const results = new Array(values.length); + let nextIndex = 0; + const workers = Array.from({ length: Math.min(concurrency, values.length) }, async () => { + while (nextIndex < values.length) { + const index = nextIndex++; + results[index] = await mapper(values[index]); + } + }); + await Promise.all(workers); + return results; +} + +/** + * Enrich live Vertex models from official Google Cloud documentation. Existing structured fields + * win; docs fill only gaps. Failures return the live models unchanged (or verified stale values). + */ +export async function enrichVertexModelsWithMetadata(options: { + models: T[]; + fetchImpl: VertexMetadataFetch; + staleModels?: VertexMetadataModel[]; + now?: Date; +}): Promise { + const now = options.now || new Date(); + const nowMs = now.getTime(); + const staleById = new Map((options.staleModels || []).map((model) => [model.id, model])); + + return mapWithConcurrency(options.models, MAX_CONCURRENT_FETCHES, async (model) => { + const expectedModelId = terminalModelId(model.id); + let docsMetadata: ParsedDocsMetadata | null = null; + let page: CachedDocsPage | null = null; + + for (const url of resolveVertexModelDocsUrls(model.id)) { + page = await fetchDocsPage({ + url, + fetchImpl: options.fetchImpl, + nowMs, + }); + docsMetadata = page ? parseVertexModelDocsRows(page.rows, expectedModelId) : null; + if (docsMetadata) break; + } + + const staleMetadata = readPersistedDocsMetadata(staleById.get(model.id), nowMs); + const metadata: VertexModelMetadata | null = docsMetadata + ? { + ...docsMetadata, + metadataProvenance: { + vertexDocs: { + source: "google-cloud-docs", + sourceUrl: page!.sourceUrl, + fetchedAt: new Date(page!.fetchedAtMs).toISOString(), + ...(page!.lastModified ? { lastModified: page!.lastModified } : {}), + parserVersion: VERTEX_DOCS_PARSER_VERSION, + confidence: "verified", + fields: metadataFields(docsMetadata), + }, + }, + } + : staleMetadata; + if (!metadata) return model; + + return { + ...metadata, + ...model, + metadataProvenance: metadata.metadataProvenance, + } as T; + }); +} + +/** Test seam for deterministic cache and stale-fallback assertions. */ +export function clearVertexModelMetadataCache(): void { + docsCache.clear(); + docsInflight.clear(); +} diff --git a/src/lib/providerModels/vertexPublisherModelsParser.ts b/src/lib/providerModels/vertexPublisherModelsParser.ts new file mode 100644 index 00000000000..ea23e77472c --- /dev/null +++ b/src/lib/providerModels/vertexPublisherModelsParser.ts @@ -0,0 +1,114 @@ +import { + getVertexModelTargetFormat, + normalizeVertexModelId, +} from "@omniroute/open-sse/config/vertexModels.ts"; + +interface VertexPublisherModel { + name?: string; + id?: string; + displayName?: string; + description?: string; + supportedActions?: Record; +} + +export interface VertexPublisherDiscoveryModel { + id: string; + name: string; + supportedEndpoints: ["chat"]; + targetFormat?: "claude" | "openai"; + owned_by: string; + description?: string; +} + +function isCurrentGeminiChatModel(id: string): boolean { + const match = id.match(/^gemini-(\d+(?:\.\d+)?)-(flash-lite|flash|pro)(?:-preview)?$/i); + return !!match && Number(match[1]) >= 2.5; +} + +function readPublisherModelId(model: VertexPublisherModel, publisher: string): string | null { + const rawId = + (typeof model.name === "string" && model.name) || + (typeof model.id === "string" && model.id) || + ""; + if (!rawId) return null; + return rawId.includes("/") ? rawId : `publishers/${publisher}/models/${rawId}`; +} + +function isPublisherChatAction(actions: Record | undefined): boolean { + if (!actions) return true; + return !!( + actions.viewRestApi || + actions.openGenerationAiStudio || + actions.openGenie || + actions.requestAccess + ); +} + +function toDiscoveryModel( + model: VertexPublisherModel, + id: string, + publisher: string, + targetFormat?: "claude" | "openai" +): VertexPublisherDiscoveryModel { + return { + id, + name: (typeof model.displayName === "string" && model.displayName) || id, + supportedEndpoints: ["chat"], + ...(targetFormat ? { targetFormat } : {}), + owned_by: publisher, + ...(typeof model.description === "string" ? { description: model.description } : {}), + }; +} + +function toGoogleChatModel( + model: VertexPublisherModel, + routableId: string, + publisher: string +): VertexPublisherDiscoveryModel | null { + const id = normalizeVertexModelId(routableId); + if (!isCurrentGeminiChatModel(id)) return null; + return toDiscoveryModel(model, id, publisher); +} + +function toPartnerChatModel( + model: VertexPublisherModel, + routableId: string, + publisher: string +): VertexPublisherDiscoveryModel | null { + if (!isPublisherChatAction(model.supportedActions)) return null; + const id = normalizeVertexModelId(routableId); + const targetFormat = getVertexModelTargetFormat(routableId); + if (!id || !targetFormat || /(?:^|[-/])ocr(?:-|$)/i.test(id)) return null; + return toDiscoveryModel(model, id, publisher, targetFormat); +} + +function toPublisherDiscoveryModel( + candidate: unknown, + publisher: string +): VertexPublisherDiscoveryModel[] { + if (!candidate || typeof candidate !== "object" || Array.isArray(candidate)) return []; + const model = candidate as VertexPublisherModel; + const routableId = readPublisherModelId(model, publisher); + if (!routableId) return []; + const parsed = + publisher.toLowerCase() === "google" + ? toGoogleChatModel(model, routableId, publisher) + : toPartnerChatModel(model, routableId, publisher); + return parsed ? [parsed] : []; +} + +/** Parse one Model Garden publisher-list envelope into model ids accepted by Vertex inference. */ +export function parseVertexPublisherModels( + data: unknown, + publisher: string +): VertexPublisherDiscoveryModel[] { + if (!data || typeof data !== "object" || Array.isArray(data)) return []; + const record = data as { models?: unknown[]; publisherModels?: unknown[] }; + const models = Array.isArray(record.publisherModels) + ? record.publisherModels + : Array.isArray(record.models) + ? record.models + : []; + + return models.flatMap((candidate) => toPublisherDiscoveryModel(candidate, publisher)); +} diff --git a/src/lib/providerModels/vertexXaiModelsParser.ts b/src/lib/providerModels/vertexXaiModelsParser.ts new file mode 100644 index 00000000000..e1929674633 --- /dev/null +++ b/src/lib/providerModels/vertexXaiModelsParser.ts @@ -0,0 +1,28 @@ +import { isVertexXaiModel } from "@omniroute/open-sse/config/vertexModels.ts"; + +import { parseVertexPublisherModels } from "./vertexPublisherModelsParser"; + +export interface VertexXaiDiscoveryModel { + id: string; + name: string; + supportedEndpoints: string[]; + targetFormat: "openai"; + owned_by: "xai"; + description?: string; +} + +/** Parse the v1beta1 Model Garden xAI publisher list into routable MaaS model rows. */ +export function parseVertexXaiModels(data: unknown): VertexXaiDiscoveryModel[] { + return parseVertexPublisherModels(data, "xai").flatMap((model) => + isVertexXaiModel(model.id) + ? [ + { + ...model, + supportedEndpoints: ["chat"], + targetFormat: "openai" as const, + owned_by: "xai" as const, + }, + ] + : [] + ); +} diff --git a/src/shared/constants/providers/apikey/enterprise-cloud.ts b/src/shared/constants/providers/apikey/enterprise-cloud.ts index 49831b3c240..9d65a55636b 100644 --- a/src/shared/constants/providers/apikey/enterprise-cloud.ts +++ b/src/shared/constants/providers/apikey/enterprise-cloud.ts @@ -119,7 +119,8 @@ export const APIKEY_PROVIDERS_ENTERPRISE = { textIcon: "VA", website: "https://cloud.google.com/vertex-ai", hasFree: true, - authHint: "Provide Service Account JSON or OAuth access_token", + authHint: + "Provide Service Account JSON, an OAuth access token, a Vertex Express API key, or a service-account-bound authorization key. Express mode supports Gemini only; partner models require project-scoped credentials.", }, "vertex-partner": { id: "vertex-partner", @@ -130,7 +131,8 @@ export const APIKEY_PROVIDERS_ENTERPRISE = { color: "#34A853", textIcon: "VP", website: "https://cloud.google.com/vertex-ai", - authHint: "Provide the same Service Account JSON used for Vertex AI partner models.", + authHint: + "Provide Service Account JSON or OAuth credentials. A service-account-bound authorization key also supports discovery, but partner inference additionally requires its Google Cloud project ID. Standard Express keys support Gemini only.", }, "cloudflare-ai": { id: "cloudflare-ai", diff --git a/tests/unit/credential-hint-copy-3180-3091.test.ts b/tests/unit/credential-hint-copy-3180-3091.test.ts index e3ae205fd9e..1b502c2de22 100644 --- a/tests/unit/credential-hint-copy-3180-3091.test.ts +++ b/tests/unit/credential-hint-copy-3180-3091.test.ts @@ -25,8 +25,12 @@ test("#3180 grok-web credential hint names both sso and sso-rw", () => { test("#3091 vertex Service Account placeholder is real instructional text, not the stub", () => { const en = JSON.parse(readFileSync(path.join(MESSAGES_DIR, "en.json"), "utf8")); + const de = JSON.parse(readFileSync(path.join(MESSAGES_DIR, "de.json"), "utf8")); const placeholder = en.providers?.vertexServiceAccountPlaceholder; assert.equal(typeof placeholder, "string"); assert.notEqual(placeholder, "Vertex Service Account Placeholder"); assert.match(placeholder, /service_account/); + assert.match(en.providers?.vertexCredentialHint || "", /live Model Garden discovery/i); + assert.match(de.providers?.vertexCredentialLabel || "", /Service-Account-JSON/); + assert.match(de.providers?.vertexCredentialHint || "", /Live-Erkennung/); }); diff --git a/tests/unit/executor-vertex-extended.test.ts b/tests/unit/executor-vertex-extended.test.ts index d0093911b29..4efc1cac998 100644 --- a/tests/unit/executor-vertex-extended.test.ts +++ b/tests/unit/executor-vertex-extended.test.ts @@ -68,7 +68,87 @@ test("VertexExecutor.buildUrl routes a non-JSON Express API key to the project-l expressUrl, "https://aiplatform.googleapis.com/v1/publishers/google/models/gemini-2.5-flash:generateContent?key=express-key-abc" ); - assert.ok(!expressUrl.includes("/projects/"), "Express key URL must not route through a project path"); + assert.ok( + !expressUrl.includes("/projects/"), + "Express key URL must not route through a project path" + ); +}); + +test("VertexExecutor.buildUrl rejects partner models for project-less Express credentials", () => { + const executor = new VertexExecutor(); + const ids = ["grok-4.6", "xai/grok-4.6", "xai/models/grok-4.6", "publishers/xai/models/grok-4.6"]; + + for (const modelId of ids) { + assert.throws( + () => executor.buildUrl(modelId, false, 0, { apiKey: "k-express" }), + /partner models require project-scoped credentials/i, + modelId + ); + } +}); + +test("VertexExecutor.execute canonicalizes an xAI resource id in URL and request body", async () => { + const executor = new VertexExecutor(); + const originalFetch = globalThis.fetch; + const calls: Array<{ url: string; body: string }> = []; + + globalThis.fetch = async (url: RequestInfo | URL, init?: RequestInit) => { + calls.push({ url: String(url), body: String(init?.body || "") }); + return Response.json({ choices: [] }); + }; + + try { + await executor.execute({ + model: "publishers/xai/models/grok-4.6", + body: { + model: "publishers/xai/models/grok-4.6", + messages: [{ role: "user", content: "hi" }], + }, + stream: false, + credentials: { + apiKey: "k-authorization", + projectId: "proj-xai", + }, + }); + + assert.equal(calls.length, 1); + assert.equal( + calls[0].url, + "https://aiplatform.googleapis.com/v1/projects/proj-xai/locations/global/endpoints/openapi/chat/completions?key=k-authorization" + ); + assert.equal(JSON.parse(calls[0].body).model, "xai/grok-4.6"); + } finally { + globalThis.fetch = originalFetch; + } +}); + +test("VertexExecutor.buildUrl routes Mistral Model Garden ids to native rawPredict", () => { + const executor = new VertexExecutor(); + const credentials = { + apiKey: createServiceAccountJson({ projectId: "proj-mistral" }), + providerSpecificData: { region: "europe-west4" }, + }; + + assert.equal( + executor.buildUrl("publishers/mistralai/models/mistral-medium-3", false, 0, credentials), + "https://aiplatform.googleapis.com/v1/projects/proj-mistral/locations/europe-west4/publishers/mistralai/models/mistral-medium-3:rawPredict" + ); + assert.equal( + executor.buildUrl("mistralai/mistral-medium-3", true, 0, credentials), + "https://aiplatform.googleapis.com/v1/projects/proj-mistral/locations/europe-west4/publishers/mistralai/models/mistral-medium-3:streamRawPredict" + ); +}); + +test("VertexExecutor.buildUrl generically routes future publisher resources to OpenAI MaaS", () => { + const executor = new VertexExecutor(); + const url = executor.buildUrl("publishers/future-vendor/models/future-chat-maas", false, 0, { + apiKey: createServiceAccountJson({ projectId: "proj-future" }), + }); + + assert.equal( + url, + "https://aiplatform.googleapis.com/v1/projects/proj-future/locations/global/endpoints/openapi/chat/completions" + ); }); test("VertexExecutor.buildUrl routes partner and org-prefixed models to the global partner endpoint", () => { @@ -79,6 +159,9 @@ test("VertexExecutor.buildUrl routes partner and org-prefixed models to the glob const metaLlama = executor.buildUrl("meta/llama-3.1-405b-instruct-maas", true, 0, { apiKey: createServiceAccountJson({ projectId: "proj-llama" }), }); + const grok = executor.buildUrl("publishers/xai/models/grok-4.6", true, 0, { + apiKey: createServiceAccountJson({ projectId: "proj-xai" }), + }); assert.equal( deepseek, @@ -88,6 +171,36 @@ test("VertexExecutor.buildUrl routes partner and org-prefixed models to the glob metaLlama, "https://aiplatform.googleapis.com/v1/projects/proj-llama/locations/global/endpoints/openapi/chat/completions" ); + assert.equal( + grok, + "https://aiplatform.googleapis.com/v1/projects/proj-xai/locations/global/endpoints/openapi/chat/completions" + ); +}); + +test("VertexExecutor.execute namespaces legacy bare open-MaaS model ids", async () => { + const executor = new VertexExecutor(); + const originalFetch = globalThis.fetch; + let sentModel: string | undefined; + + globalThis.fetch = async (_url, init) => { + sentModel = JSON.parse(String(init?.body)).model; + return Response.json({ choices: [] }); + }; + + try { + await executor.execute({ + model: "DeepSeek-V4-Pro", + body: { model: "DeepSeek-V4-Pro", messages: [{ role: "user", content: "hi" }] }, + stream: false, + credentials: { + apiKey: createServiceAccountJson({ projectId: "proj-deepseek" }), + accessToken: "ya29.deepseek", + }, + }); + assert.equal(sentModel, "deepseek-ai/DeepSeek-V4-Pro"); + } finally { + globalThis.fetch = originalFetch; + } }); test("VertexExecutor.buildUrl routes current-generation Claude models to the native Anthropic rawPredict endpoint (#1985, #8994)", () => { diff --git a/tests/unit/t28-model-catalog-updates.test.ts b/tests/unit/t28-model-catalog-updates.test.ts index 5a3c8c129bb..84a22e85867 100644 --- a/tests/unit/t28-model-catalog-updates.test.ts +++ b/tests/unit/t28-model-catalog-updates.test.ts @@ -113,11 +113,35 @@ test("T28: lmarena registry seeds Direct-chat Text/search; image models in IMAGE test("T28: vertex catalog includes partner models when vertex executor is available", () => { const vertexIds = REGISTRY.vertex.models.map((m) => m.id); + const vertexBudgetIds = FREE_MODEL_BUDGETS.filter((m) => m.provider === "vertex").map( + (m) => m.modelId + ); - assert.ok(vertexIds.includes("DeepSeek-V4-Flash")); - assert.ok(vertexIds.includes("DeepSeek-V4-Pro")); - assert.ok(vertexIds.includes("Qwen3.6-35B-A3B")); - assert.ok(vertexIds.includes("GLM-5.1-FP8")); + assert.ok(vertexIds.includes("gemini-3.7-flash")); + assert.ok(vertexIds.includes("gemini-3.6-flash")); + assert.ok(vertexIds.includes("gemini-3.5-flash")); + assert.ok(vertexIds.includes("gemini-3.5-flash-lite")); + assert.ok(vertexIds.includes("gemini-2.5-pro")); + assert.ok(vertexIds.includes("gemini-2.5-flash")); + assert.ok(vertexIds.includes("gemini-2.5-flash-lite")); + assert.ok(!vertexIds.includes("gemma-4-31b-it")); + assert.ok(vertexIds.includes("deepseek-ai/deepseek-v3.2-maas")); + assert.ok(vertexIds.includes("deepseek-ai/deepseek-v3.1-maas")); + assert.ok(vertexIds.includes("qwen/qwen3-next-80b-a3b-instruct-maas")); + assert.ok(vertexIds.includes("zai-org/glm-5-maas")); + assert.ok(!vertexIds.includes("DeepSeek-V4-Flash")); + assert.ok(!vertexIds.includes("DeepSeek-V4-Pro")); + assert.ok(!vertexIds.includes("Qwen3.6-35B-A3B")); + assert.ok(!vertexIds.includes("GLM-5.1-FP8")); + assert.ok(vertexBudgetIds.includes("deepseek-ai/deepseek-v3.2-maas")); + assert.ok(vertexBudgetIds.includes("gemini-3.7-flash")); + assert.ok(vertexBudgetIds.includes("gemini-3.6-flash")); + assert.ok(!vertexBudgetIds.includes("gemma-4-31b-it")); + assert.ok(vertexBudgetIds.includes("qwen/qwen3-next-80b-a3b-instruct-maas")); + assert.ok(vertexBudgetIds.includes("zai-org/glm-5-maas")); + assert.ok(!vertexBudgetIds.includes("DeepSeek-V4-Flash")); + assert.ok(!vertexBudgetIds.includes("Qwen3.6-35B-A3B")); + assert.ok(!vertexBudgetIds.includes("GLM-5.1-FP8")); }); test("T28: volcengine (Ark) catalog includes DeepSeek V4 models", () => { @@ -151,7 +175,7 @@ test("T28: new catalog models resolve through getModelInfoCore", async () => { assert.equal(flashPreview.provider, "gemini"); assert.equal(flashPreview.model, "gemini-3-flash-preview"); - const vertexPartner = await getModelInfoCore("vertex/Qwen3.6-35B-A3B", {}); + const vertexPartner = await getModelInfoCore("vertex/qwen/qwen3-next-80b-a3b-instruct-maas", {}); assert.equal(vertexPartner.provider, "vertex"); - assert.equal(vertexPartner.model, "Qwen3.6-35B-A3B"); + assert.equal(vertexPartner.model, "qwen/qwen3-next-80b-a3b-instruct-maas"); }); diff --git a/tests/unit/t29-vertex-sa-json-executor.test.ts b/tests/unit/t29-vertex-sa-json-executor.test.ts index f367718084e..ffa0dc25413 100644 --- a/tests/unit/t29-vertex-sa-json-executor.test.ts +++ b/tests/unit/t29-vertex-sa-json-executor.test.ts @@ -1,7 +1,7 @@ import test from "node:test"; import assert from "node:assert/strict"; -const { VertexExecutor } = await import("../../open-sse/executors/vertex.ts"); +const { VertexExecutor, parseSAFromApiKey } = await import("../../open-sse/executors/vertex.ts"); const MIN_SA_JSON = JSON.stringify({ project_id: "vertex-project-123", @@ -22,7 +22,7 @@ test("T29: Vertex executor builds regional Gemini URL from Service Account proje test("T29: Vertex executor routes partner models to global openapi endpoint", () => { const executor = new VertexExecutor(); - const url = executor.buildUrl("DeepSeek-V4-Pro", false, 0, { + const url = executor.buildUrl("deepseek-ai/deepseek-v3.2-maas", false, 0, { apiKey: MIN_SA_JSON, providerSpecificData: { region: "us-central1" }, }); @@ -72,6 +72,16 @@ test("T29: Vertex executor rejects incomplete Service Account JSON clearly", asy ); }); +test("T29: Vertex distinguishes OAuth client JSON from a Service Account key", () => { + assert.throws( + () => + parseSAFromApiKey( + JSON.stringify({ web: { client_id: "example.apps.googleusercontent.com" } }) + ), + /OAuth client JSON is not a Service Account key/i + ); +}); + test("T29: Vertex executor routes a non-JSON Express API key to the project-less publisher endpoint", () => { const executor = new VertexExecutor(); const stream = executor.buildUrl("gemini-2.5-flash", true, 0, { apiKey: "express-key-123" }); diff --git a/tests/unit/validation-specialty-inline-split.test.ts b/tests/unit/validation-specialty-inline-split.test.ts index 39c734c961e..6576a9463de 100644 --- a/tests/unit/validation-specialty-inline-split.test.ts +++ b/tests/unit/validation-specialty-inline-split.test.ts @@ -67,3 +67,11 @@ test("validateVertexProvider: malformed Service Account JSON is rejected", async assert.equal(result.valid, false); assert.match(result.error || "", /Invalid Service Account JSON/i); }); + +test("validateVertexProvider: OAuth client JSON explains the required credential type", async () => { + const result = await M.validateVertexProvider({ + apiKey: JSON.stringify({ web: { client_id: "example.apps.googleusercontent.com" } }), + }); + assert.equal(result.valid, false); + assert.match(result.error || "", /OAuth client JSON is not a Service Account key/i); +}); diff --git a/tests/unit/vertex-model-metadata.test.ts b/tests/unit/vertex-model-metadata.test.ts new file mode 100644 index 00000000000..5752d945561 --- /dev/null +++ b/tests/unit/vertex-model-metadata.test.ts @@ -0,0 +1,165 @@ +import assert from "node:assert/strict"; +import test from "node:test"; + +import { normalizeDiscoveredModels } from "../../src/lib/providerModels/modelDiscovery.ts"; +import { + clearVertexModelMetadataCache, + enrichVertexModelsWithMetadata, + parseVertexModelDocsHtml, + resolveVertexModelDocsUrls, +} from "../../src/lib/providerModels/vertexModelMetadata.ts"; + +function modelDocsHtml(rows: string): string { + return `${rows}
`; +} + +test.beforeEach(() => clearVertexModelMetadataCache()); + +test("Vertex docs resolver maps native and MaaS model IDs only to the official docs host", () => { + assert.deepEqual(resolveVertexModelDocsUrls("gemini-3.7-flash"), [ + "https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/gemini/3-7-flash?hl=en", + ]); + assert.deepEqual(resolveVertexModelDocsUrls("claude-opus-4-1"), [ + "https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/partner-models/claude/opus-4-1?hl=en", + ]); + assert.deepEqual(resolveVertexModelDocsUrls("xai/grok-4.20-reasoning"), [ + "https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/partner-models/grok/grok-4-20?hl=en", + ]); + assert.deepEqual(resolveVertexModelDocsUrls("deepseek-ai/deepseek-v3.2-maas"), [ + "https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/maas/deepseek-ai/deepseek-v3-2-maas?hl=en", + "https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/maas/deepseek-ai/deepseek-v3-2?hl=en", + ]); + assert.ok( + resolveVertexModelDocsUrls("xai/grok-4.6").every((url) => + url.startsWith("https://docs.cloud.google.com/") + ) + ); +}); + +test("Vertex docs parser extracts distinct context, input, and output limits", () => { + const gemini = parseVertexModelDocsHtml( + modelDocsHtml(` + Model IDgemini-3.7-flash + Token limitsContext window1,048,576 + Maximum output tokens65,536 + `), + "gemini-3.7-flash" + ); + assert.deepEqual(gemini, { contextWindow: 1048576, outputTokenLimit: 65536 }); + + const claude = parseVertexModelDocsHtml( + modelDocsHtml(` + Model IDclaude-opus-4-1 + Token limits + Maximum input tokens: 200,000 Maximum output tokens: 32,000 + + Quota limitsContext length: 200,000 + `), + "claude-opus-4-1" + ); + assert.deepEqual(claude, { + contextWindow: 200000, + inputTokenLimit: 200000, + outputTokenLimit: 32000, + }); +}); + +test("Vertex docs parser rejects a page without the exact live model ID", () => { + const html = modelDocsHtml(` + Model IDgemini-3.7-flash-lite + Token limitsContext window1,048,576 + `); + assert.equal(parseVertexModelDocsHtml(html, "gemini-3.7-flash"), null); +}); + +test("Vertex enrichment deduplicates shared docs fetches and preserves structured API limits", async () => { + const html = modelDocsHtml(` + Model IDgrok-4.20-reasoning + Model IDgrok-4.20-non-reasoning + Quota limitsContext length: 2,000,000 + Maximum output tokens80,000 + `); + let fetchCalls = 0; + const models = await enrichVertexModelsWithMetadata({ + models: [ + { id: "xai/grok-4.20-reasoning", name: "Reasoning", outputTokenLimit: 12345 }, + { id: "xai/grok-4.20-non-reasoning", name: "Non-Reasoning" }, + ], + now: new Date("2026-09-04T12:00:00.000Z"), + fetchImpl: async () => { + fetchCalls += 1; + return new Response(html, { + status: 200, + headers: { "Last-Modified": "Thu, 03 Sep 2026 10:00:00 GMT" }, + }); + }, + }); + + assert.equal(fetchCalls, 1); + assert.equal(models[0]?.contextWindow, 2000000); + assert.equal(models[0]?.outputTokenLimit, 12345); + assert.equal(models[1]?.outputTokenLimit, 80000); + assert.deepEqual(models[1]?.metadataProvenance?.vertexDocs.fields, [ + "contextWindow", + "outputTokenLimit", + ]); + assert.equal( + models[1]?.metadataProvenance?.vertexDocs.sourceUrl, + "https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/partner-models/grok/grok-4-20?hl=en" + ); +}); + +test("Vertex enrichment reuses only recently verified persisted metadata on docs failure", async () => { + const staleModel = { + id: "gemini-3.7-flash", + contextWindow: 1048576, + outputTokenLimit: 65536, + metadataProvenance: { + vertexDocs: { + source: "google-cloud-docs" as const, + sourceUrl: + "https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/gemini/3-7-flash?hl=en", + fetchedAt: "2026-09-01T12:00:00.000Z", + parserVersion: "vertex-docs-v1" as const, + confidence: "verified" as const, + fields: ["contextWindow", "outputTokenLimit"] as const, + }, + }, + }; + const [withinTtl] = await enrichVertexModelsWithMetadata({ + models: [{ id: staleModel.id }], + staleModels: [staleModel], + now: new Date("2026-09-04T12:00:00.000Z"), + fetchImpl: async () => new Response("unavailable", { status: 503 }), + }); + assert.equal(withinTtl?.contextWindow, 1048576); + assert.equal(withinTtl?.outputTokenLimit, 65536); + + clearVertexModelMetadataCache(); + const [expired] = await enrichVertexModelsWithMetadata({ + models: [{ id: staleModel.id }], + staleModels: [staleModel], + now: new Date("2026-09-10T12:00:00.001Z"), + fetchImpl: async () => new Response("unavailable", { status: 503 }), + }); + assert.equal(expired?.contextWindow, undefined); + assert.equal(expired?.outputTokenLimit, undefined); +}); + +test("Vertex normalization persists context separately while legacy providers retain compatibility", () => { + const metadata = { + id: "gemini-3.7-flash", + contextWindow: 1048576, + outputTokenLimit: 65536, + }; + assert.deepEqual(normalizeDiscoveredModels([metadata], "vertex"), [ + { + id: "gemini-3.7-flash", + name: "gemini-3.7-flash", + source: "imported", + contextWindow: 1048576, + outputTokenLimit: 65536, + }, + ]); + assert.equal(normalizeDiscoveredModels([metadata], "openrouter")[0]?.inputTokenLimit, 1048576); +}); diff --git a/tests/unit/vertex-models-route.test.ts b/tests/unit/vertex-models-route.test.ts new file mode 100644 index 00000000000..e89d041bbb5 --- /dev/null +++ b/tests/unit/vertex-models-route.test.ts @@ -0,0 +1,278 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-vertex-model-routes-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const providersDb = await import("../../src/lib/db/providers.ts"); +const providerModelsRoute = await import("../../src/app/api/providers/[id]/models/route.ts"); + +const originalFetch = globalThis.fetch; + +async function resetStorage(): Promise { + globalThis.fetch = originalFetch; + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); + fs.mkdirSync(TEST_DATA_DIR, { recursive: true }); +} + +async function seedVertexConnection(options: { + apiKey?: string; + accessToken?: string; + projectId?: string; +}): Promise<{ id: string }> { + return providersDb.createProviderConnection({ + provider: "vertex", + authType: "apikey", + name: `vertex-${Math.random().toString(16).slice(2, 8)}`, + apiKey: options.apiKey, + accessToken: options.accessToken, + projectId: options.projectId, + isActive: true, + testStatus: "active", + providerSpecificData: { autoFetchModels: true }, + }); +} + +async function callRoute(connectionId: string): Promise { + return providerModelsRoute.GET( + new Request(`http://localhost/api/providers/${connectionId}/models?refresh=true&chatOnly=true`), + { params: { id: connectionId } } + ); +} + +test.beforeEach(async () => { + await resetStorage(); +}); + +test.after(async () => { + globalThis.fetch = originalFetch; + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }); +}); + +test("Vertex Express probes only Gemini discovery and then uses the intentional Gemini catalog", async () => { + const connection = await seedVertexConnection({ apiKey: "vertex-express-key" }); + let fetchCalls = 0; + const headers: HeadersInit[] = []; + globalThis.fetch = async (_url, init) => { + fetchCalls += 1; + headers.push(init?.headers || {}); + return Response.json({ error: { status: "UNAUTHENTICATED" } }, { status: 401 }); + }; + + const response = await callRoute(connection.id); + const body = await response.json(); + + assert.equal(response.status, 200); + assert.equal(body.source, "local_catalog"); + assert.equal(body.intentional, true); + assert.match(body.warning, /live catalog.*API key.*curated Express catalog/i); + assert.equal(fetchCalls, 1); + assert.ok( + headers.every((header) => new Headers(header).get("x-goog-api-key") === "vertex-express-key") + ); + assert.ok(body.models.some((model: { id?: string }) => model.id.startsWith("gemini-"))); + assert.ok(!body.models.some((model: { id?: string }) => model.id.includes("grok"))); +}); + +test("Vertex API-key endpoint rejection remains intentional and names the HTTP failure", async () => { + for (const status of [400, 403]) { + const connection = await seedVertexConnection({ apiKey: "vertex-express-key" }); + globalThis.fetch = async () => Response.json({ error: { message: "Rejected" } }, { status }); + const response = await callRoute(connection.id); + const body = await response.json(); + assert.equal(response.status, 200); + assert.equal(body.intentional, true); + assert.match(body.warning, new RegExp("HTTP " + status)); + } +}); + +test("Vertex transient HTTP and network failures do not become intentional catalogs", async () => { + for (const status of [429, 500, 503, 0]) { + const connection = await seedVertexConnection({ apiKey: "vertex-express-key" }); + globalThis.fetch = async () => { + if (!status) throw new Error("Fixture network failure"); + return Response.json({ error: { message: "Unavailable" } }, { status }); + }; + const response = await callRoute(connection.id); + const body = await response.json(); + assert.equal(response.status, 200); + assert.equal(body.intentional, undefined); + assert.match(body.warning, /temporarily unavailable/); + } +}); + +test("Vertex authorization API key detects and persists its project for curated partner models", async () => { + const connection = await seedVertexConnection({ apiKey: "vertex-authorization-key" }); + const calledUrls: string[] = []; + globalThis.fetch = async (url, init) => { + const calledUrl = String(url); + calledUrls.push(calledUrl); + assert.equal(new Headers(init?.headers).get("x-goog-api-key"), "vertex-authorization-key"); + return Response.json( + { + error: { + status: "PERMISSION_DENIED", + details: [ + { + "@type": "type.googleapis.com/google.rpc.ErrorInfo", + reason: "API_KEY_SERVICE_BLOCKED", + metadata: { consumer: "projects/316081256616" }, + }, + ], + }, + }, + { status: 403 } + ); + }; + + const response = await callRoute(connection.id); + const body = await response.json(); + const byId = new Map( + body.models.map((model: { id: string; targetFormat?: string }) => [model.id, model]) + ); + assert.equal(response.status, 200); + assert.equal(body.source, "local_catalog"); + assert.equal(body.intentional, true); + assert.equal(body.projectIdAutoDetected, true); + assert.equal(body.catalogMode, "curated_project"); + assert.equal(body.warning, undefined); + assert.equal(byId.get("xai/grok-4.6")?.targetFormat, "openai"); + assert.equal(byId.get("zai-org/glm-5-maas")?.targetFormat, "openai"); + assert.equal(byId.get("qwen/qwen3-next-80b-a3b-instruct-maas")?.targetFormat, "openai"); + assert.ok(!byId.has("GLM-5.1-FP8")); + assert.ok(!byId.has("Qwen3.6-35B-A3B")); + assert.equal(calledUrls.length, 1); + assert.ok(calledUrls[0].includes("generativelanguage.googleapis.com")); + assert.ok(!calledUrls[0].includes("vertex-authorization-key")); + + const saved = await providersDb.getProviderConnectionById(connection.id); + assert.equal(saved?.projectId, "316081256616"); +}); + +test("Vertex Service Account discovery merges Gemini and all partner transport catalogs", async () => { + const connection = await seedVertexConnection({ + accessToken: "ya29.vertex-model-discovery", + }); + const calledUrls: string[] = []; + globalThis.fetch = async (url) => { + const calledUrl = String(url); + calledUrls.push(calledUrl); + if (calledUrl.includes("generativelanguage.googleapis.com")) { + return Response.json({ + models: [ + { + name: "models/gemini-3.1-pro-preview", + displayName: "Gemini 3.1 Pro Preview", + supportedGenerationMethods: ["generateContent"], + }, + { + name: "models/gemini-3.7-flash", + displayName: "Gemini 3.7 Flash", + supportedGenerationMethods: ["generateContent"], + inputTokenLimit: 900000, + outputTokenLimit: 60000, + }, + ], + }); + } + if (calledUrl.includes("/models/gemini/3-7-flash?hl=en")) { + return new Response( + ` + + + +
Model IDgemini-3.7-flash
Token limitsContext window1,048,576
Maximum output tokens65,536
`, + { headers: { "Last-Modified": "Thu, 03 Sep 2026 10:00:00 GMT" } } + ); + } + if (calledUrl.includes("/publishers/google/models")) { + return Response.json({ + publisherModels: [ + { name: "publishers/google/models/gemini-3.7-flash" }, + { name: "publishers/google/models/gemini-3.1-flash-image" }, + ], + }); + } + if (calledUrl.includes("/publishers/anthropic/models")) { + return Response.json({ + publisherModels: [{ name: "publishers/anthropic/models/claude-sonnet-4-6" }], + }); + } + if (calledUrl.includes("/publishers/xai/models")) { + return Response.json({ + publisherModels: [{ name: "publishers/xai/models/grok-4.6" }], + }); + } + if (calledUrl.includes("/publishers/mistralai/models")) { + return Response.json({ + publisherModels: [{ name: "publishers/mistralai/models/mistral-medium-3" }], + }); + } + return new Response("unexpected discovery URL", { status: 500 }); + }; + + const response = await callRoute(connection.id); + const body = await response.json(); + const byId = new Map( + body.models.map((model: { id: string; targetFormat?: string }) => [model.id, model]) + ); + const gemini37 = byId.get("gemini-3.7-flash") as + | { + contextWindow?: number; + inputTokenLimit?: number; + outputTokenLimit?: number; + metadataProvenance?: { vertexDocs?: { source?: string } }; + } + | undefined; + + assert.equal(response.status, 200); + assert.equal(body.source, "api"); + assert.ok(byId.has("gemini-3.1-pro-preview")); + assert.ok(byId.has("gemini-3.7-flash")); + assert.equal(gemini37?.contextWindow, 1048576); + assert.equal(gemini37?.inputTokenLimit, 900000); + assert.equal(gemini37?.outputTokenLimit, 60000); + assert.equal(gemini37?.metadataProvenance?.vertexDocs?.source, "google-cloud-docs"); + assert.ok(!byId.has("gemini-3.1-flash-image")); + assert.equal(byId.get("claude-sonnet-4-6")?.targetFormat, "claude"); + assert.equal(byId.get("xai/grok-4.6")?.targetFormat, "openai"); + assert.equal(byId.get("mistral-medium-3")?.targetFormat, "openai"); + assert.ok(calledUrls.some((url) => url.includes("/v1beta1/publishers/xai/models"))); + assert.ok(calledUrls.some((url) => url.includes("/v1beta1/publishers/google/models"))); + assert.ok( + calledUrls + .filter((url) => url.includes("/v1beta1/publishers/")) + .every((url) => url.includes("pageSize=300")) + ); +}); + +test("Vertex Service Account keeps usable xAI results when Gemini listing is blocked", async () => { + const connection = await seedVertexConnection({ accessToken: "ya29.vertex-xai-only" }); + globalThis.fetch = async (url) => { + const calledUrl = String(url); + if (calledUrl.includes("generativelanguage.googleapis.com")) { + return Response.json({ error: { status: "PERMISSION_DENIED" } }, { status: 403 }); + } + if (calledUrl.includes("/publishers/xai/models")) { + return Response.json({ + publisherModels: [{ name: "publishers/xai/models/grok-4.6" }], + }); + } + return Response.json({ publisherModels: [] }); + }; + + const response = await callRoute(connection.id); + const body = await response.json(); + + assert.equal(response.status, 200); + assert.equal(body.source, "api"); + assert.ok(body.models.some((model: { id?: string }) => model.id === "xai/grok-4.6")); + assert.equal(body.catalogMode, "live_vertex_catalog"); + assert.equal(body.warning, undefined); +}); diff --git a/tests/unit/vertex-publisher-models-parser.test.ts b/tests/unit/vertex-publisher-models-parser.test.ts new file mode 100644 index 00000000000..04c22e3c7e4 --- /dev/null +++ b/tests/unit/vertex-publisher-models-parser.test.ts @@ -0,0 +1,157 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { discoverVertexModelsWithApiKey } from "../../src/lib/providerModels/vertexModelDiscovery.ts"; +import { parseVertexPublisherModels } from "../../src/lib/providerModels/vertexPublisherModelsParser.ts"; + +test("generic Vertex publisher parser produces routable IDs for every transport family", () => { + const cases = [ + ["anthropic", "claude-sonnet-5", "claude", "claude-sonnet-5"], + ["mistralai", "mistral-medium-3", "openai", "mistral-medium-3"], + ["xai", "grok-4.6", "openai", "xai/grok-4.6"], + [ + "meta", + "llama-4-maverick-17b-128e-instruct-maas", + "openai", + "meta/llama-4-maverick-17b-128e-instruct-maas", + ], + ["deepseek-ai", "deepseek-v3.2-maas", "openai", "deepseek-ai/deepseek-v3.2-maas"], + ["future-vendor", "future-chat-maas", "openai", "future-vendor/future-chat-maas"], + ] as const; + + for (const [publisher, rawId, targetFormat, expectedId] of cases) { + const models = parseVertexPublisherModels( + { + publisherModels: [ + { + name: `publishers/${publisher}/models/${rawId}`, + displayName: `Display ${rawId}`, + }, + ], + }, + publisher + ); + + assert.deepEqual(models, [ + { + id: expectedId, + name: `Display ${rawId}`, + supportedEndpoints: ["chat"], + targetFormat, + owned_by: publisher, + }, + ]); + } +}); + +test("generic Vertex publisher parser accepts models and publisherModels envelopes", () => { + assert.equal( + parseVertexPublisherModels( + { models: [{ id: "publishers/qwen/models/qwen3-next-80b-a3b-instruct-maas" }] }, + "qwen" + )[0]?.id, + "qwen/qwen3-next-80b-a3b-instruct-maas" + ); + assert.deepEqual(parseVertexPublisherModels(null, "xai"), []); + assert.deepEqual(parseVertexPublisherModels({ publisherModels: [null, {}] }, "xai"), []); + assert.equal( + parseVertexPublisherModels({ publisherModels: [{ id: "grok-4.6" }] }, "xai")[0]?.id, + "xai/grok-4.6" + ); + assert.deepEqual( + parseVertexPublisherModels( + { + publisherModels: [ + { + id: "self-deploy-only", + supportedActions: { deploy: {} }, + }, + ], + }, + "future-vendor" + ), + [] + ); + assert.equal( + parseVertexPublisherModels( + { + publisherModels: [ + { + name: "publishers/mistralai/models/mistral-medium-3", + supportedActions: { requestAccess: {} }, + }, + ], + }, + "mistralai" + )[0]?.id, + "mistral-medium-3" + ); + assert.deepEqual( + parseVertexPublisherModels( + { + publisherModels: [ + { + name: "publishers/mistralai/models/mistral-ocr-2505", + supportedActions: { requestAccess: {} }, + }, + ], + }, + "mistralai" + ), + [] + ); +}); + +test("Google publisher discovery keeps current Gemini chat models and filters media or retired IDs", () => { + const models = parseVertexPublisherModels( + { + publisherModels: [ + { name: "publishers/google/models/gemini-3.7-flash" }, + { name: "publishers/google/models/gemini-3.8-flash" }, + { name: "publishers/google/models/gemini-3.1-pro-preview" }, + { name: "publishers/google/models/gemini-3.1-flash-image" }, + { name: "publishers/google/models/gemini-2.5-pro-tts" }, + { name: "publishers/google/models/gemini-1.5-pro-002" }, + ], + }, + "google" + ); + + assert.deepEqual( + models.map((model) => model.id), + ["gemini-3.7-flash", "gemini-3.8-flash", "gemini-3.1-pro-preview"] + ); + assert.ok(models.every((model) => model.targetFormat === undefined)); +}); + +test("Vertex API-key discovery extracts its consumer project without probing Model Garden", async () => { + const urls: string[] = []; + const result = await discoverVertexModelsWithApiKey({ + apiKey: "authorization-key", + fetchImpl: async (url, init) => { + urls.push(url); + assert.equal(new Headers(init.headers).get("x-goog-api-key"), "authorization-key"); + assert.ok(!url.includes("authorization-key")); + + return Response.json( + { + error: { + details: [ + { + "@type": "type.googleapis.com/google.rpc.ErrorInfo", + metadata: { consumer: "projects/project-from-key" }, + }, + ], + }, + }, + { status: 403 } + ); + }, + }); + + assert.deepEqual(result.models, []); + assert.equal(result.projectId, "project-from-key"); + assert.equal(urls.length, 1); + assert.ok(urls[0].includes("generativelanguage.googleapis.com")); + assert.ok(!urls[0].includes("/publishers/")); +}); diff --git a/tests/unit/vertex-xai-models.test.ts b/tests/unit/vertex-xai-models.test.ts new file mode 100644 index 00000000000..f6cedf80bcb --- /dev/null +++ b/tests/unit/vertex-xai-models.test.ts @@ -0,0 +1,142 @@ +import test from "node:test"; +import assert from "node:assert/strict"; + +import { getModelTargetFormat } from "../../open-sse/config/providerModels.ts"; +import { getRegistryEntry } from "../../open-sse/config/providerRegistry.ts"; +import { + getVertexModelTargetFormat, + getVertexModelTransport, + isVertexXaiModel, + normalizeVertexModelId, +} from "../../open-sse/config/vertexModels.ts"; +import { parseVertexXaiModels } from "../../src/lib/providerModels/vertexXaiModelsParser.ts"; + +test("normalizeVertexModelId accepts Model Garden resource names and request names", () => { + const cases = [ + "grok-4.6", + "xai/grok-4.6", + "xai/models/grok-4.6", + "publishers/xai/models/grok-4.6", + "projects/demo/locations/global/publishers/xai/models/grok-4.6", + "vertex/publishers/xai/models/grok-4.6", + "vp/xai/grok-4.6", + ]; + + for (const modelId of cases) { + assert.equal(normalizeVertexModelId(modelId), "xai/grok-4.6", modelId); + assert.equal(isVertexXaiModel(modelId), true, modelId); + } + + assert.equal(normalizeVertexModelId("gemini-3.1-pro-preview"), "gemini-3.1-pro-preview"); + assert.equal(isVertexXaiModel("gemini-3.1-pro-preview"), false); +}); + +test("parseVertexXaiModels maps the Model Garden publisher envelope to routable models", () => { + const models = parseVertexXaiModels({ + publisherModels: [ + { + name: "publishers/xai/models/grok-4.6", + displayName: "Grok 4.6", + description: "xAI partner model", + }, + { + name: "projects/demo/locations/global/publishers/xai/models/grok-4.20-reasoning", + }, + ], + }); + + assert.deepEqual(models, [ + { + id: "xai/grok-4.6", + name: "Grok 4.6", + supportedEndpoints: ["chat"], + targetFormat: "openai", + description: "xAI partner model", + owned_by: "xai", + }, + { + id: "xai/grok-4.20-reasoning", + name: "xai/grok-4.20-reasoning", + supportedEndpoints: ["chat"], + targetFormat: "openai", + owned_by: "xai", + }, + ]); +}); + +test("Vertex xAI models always resolve to the OpenAI wire format", () => { + const ids = ["grok-4.6", "xai/grok-4.6", "xai/models/grok-4.6", "publishers/xai/models/grok-4.6"]; + + for (const provider of ["vertex", "vertex-partner", "vp"]) { + for (const modelId of ids) { + assert.equal(getModelTargetFormat(provider, modelId), "openai", `${provider}/${modelId}`); + } + } +}); + +test("all curated and generic Vertex partner IDs resolve away from Gemini format", () => { + const openAiIds = [ + "deepseek-ai/deepseek-v3.2-maas", + "qwen/qwen3-next-80b-a3b-instruct-maas", + "zai-org/glm-5-maas", + "mistralai/mistral-medium-3", + "publishers/meta/models/llama-4-maverick-17b-128e-instruct-maas", + "publishers/future-vendor/models/future-chat-maas", + ]; + + for (const provider of ["vertex", "vertex-partner", "vp"]) { + for (const modelId of openAiIds) { + assert.equal(getModelTargetFormat(provider, modelId), "openai", `${provider}/${modelId}`); + } + assert.equal( + getModelTargetFormat(provider, "publishers/anthropic/models/claude-sonnet-5"), + "claude", + provider + ); + } +}); + +test("Vertex routing covers native partner APIs and arbitrary OpenAI-compatible MaaS publishers", () => { + assert.equal( + normalizeVertexModelId("publishers/anthropic/models/claude-sonnet-5"), + "claude-sonnet-5" + ); + assert.equal( + normalizeVertexModelId("publishers/mistralai/models/mistral-medium-3"), + "mistral-medium-3" + ); + assert.equal( + normalizeVertexModelId("publishers/meta/models/llama-4-maverick-17b-128e-instruct-maas"), + "meta/llama-4-maverick-17b-128e-instruct-maas" + ); + assert.equal( + normalizeVertexModelId("publishers/deepseek-ai/models/deepseek-v3.2-maas"), + "deepseek-ai/deepseek-v3.2-maas" + ); + assert.equal(normalizeVertexModelId("DeepSeek-V4-Pro"), "deepseek-ai/DeepSeek-V4-Pro"); + assert.equal(normalizeVertexModelId("Qwen3.6-35B-A3B"), "qwen/Qwen3.6-35B-A3B"); + assert.equal(normalizeVertexModelId("GLM-5.1-FP8"), "zai-org/GLM-5.1-FP8"); + + assert.equal(getVertexModelTransport("gemini-3.1-pro-preview"), "gemini"); + assert.equal(getVertexModelTransport("anthropic/claude-sonnet-5"), "anthropic"); + assert.equal(getVertexModelTransport("mistralai/mistral-medium-3"), "mistral"); + assert.equal(getVertexModelTransport("future-publisher/future-chat-maas"), "openai"); + assert.equal(getVertexModelTargetFormat("anthropic/claude-sonnet-5"), "claude"); + assert.equal(getVertexModelTargetFormat("mistralai/mistral-medium-3"), "openai"); + assert.equal(getVertexModelTargetFormat("future-publisher/future-chat-maas"), "openai"); + assert.equal(getVertexModelTargetFormat("gemini-3.1-pro-preview"), null); +}); + +test("Vertex registries contain the documented xAI MaaS models as partial live catalogs", () => { + for (const provider of ["vertex", "vertex-partner"]) { + const entry = getRegistryEntry(provider); + assert.ok(entry, provider); + assert.equal(entry.liveCatalogAuthoritative, false, provider); + + const grok = entry.models.find((model) => model.id === "xai/grok-4.6"); + assert.ok(grok, `${provider} must expose grok-4.6`); + assert.equal(grok.targetFormat, "openai"); + assert.equal(grok.supportsVision, true); + assert.equal(grok.contextLength, 524288); + } +});