Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
36 commits
Select commit Hold shift + click to select a range
0e73b3b
docs: add Gemini model metadata design spec
christopher-s Apr 2, 2026
4feea2e
docs: revise Gemini model metadata spec — add backwards compat, drop …
christopher-s Apr 2, 2026
aac99f6
docs: add Gemini model metadata implementation plan
christopher-s Apr 2, 2026
3055456
docs: fix plan — add context_length to both catalog model pushes
christopher-s Apr 2, 2026
f1805c8
feat(gemini): extract metadata from models API response
christopher-s Apr 2, 2026
325d048
feat(sync): carry Gemini metadata through model sync
christopher-s Apr 2, 2026
bd5f39e
feat(db): extend replaceCustomModels with metadata fields
christopher-s Apr 2, 2026
49ac0ca
feat(catalog): use stored inputTokenLimit for custom model context_le…
christopher-s Apr 2, 2026
faae82e
feat(v1beta): use stored token limits and metadata in Gemini models e…
christopher-s Apr 2, 2026
ea6d0f5
docs: add Gemini available models sync implementation plan
christopher-s Apr 2, 2026
668b235
docs: fix plan — add missing import for buildModelSyncInternalHeaders
christopher-s Apr 2, 2026
9e4132f
feat(db): add syncedAvailableModels namespace and CRUD functions
christopher-s Apr 2, 2026
7607cec
feat(sync): write Gemini models to syncedAvailableModels with union l…
christopher-s Apr 2, 2026
5c27e0f
feat(gemini): auto-trigger model sync when API key is saved
christopher-s Apr 2, 2026
125fb81
feat(dashboard): Gemini Available Models reads from API sync, hide Cu…
christopher-s Apr 2, 2026
5b140d2
feat(api): catalog and v1beta read from synced Gemini models
christopher-s Apr 2, 2026
464fd6d
fix(v1beta): remove Gemini duplicates — filter non-consecutive entrie…
christopher-s Apr 2, 2026
8ed0917
feat(gemini): per-connection model tracking with cleanup on key delete
christopher-s Apr 2, 2026
b4e674a
fix(gemini): no registry fallback — show 0 models when no API keys exist
christopher-s Apr 2, 2026
7f785b8
fix(gemini): refresh Available Models UI after API key add/delete
christopher-s Apr 2, 2026
2341bba
feat(gemini): progress dialog on key save, remove hardcoded registry,…
christopher-s Apr 2, 2026
3ae810a
fix(gemini): log sync errors, optimize synced models query
christopher-s Apr 2, 2026
0038fe5
feat(gemini): per-model quota lockout instead of connection-wide
christopher-s Apr 2, 2026
f8d045c
fix(gemini): per-model quota isolation — 429 on one model keeps other…
christopher-s Apr 2, 2026
03ff03e
fix(gemini): filter out locked models during credential selection
christopher-s Apr 2, 2026
35061df
refactor: consolidate per-model quota logic into shared helpers
christopher-s Apr 2, 2026
b1183c2
feat(models): add pageSize=1000 and nextPageToken pagination for Gemini
christopher-s Apr 2, 2026
a069df4
fix: address PR review — timeouts, pagination safety, standardized co…
christopher-s Apr 2, 2026
25c63c8
merge: sync with upstream origin/main (v3.4.5+v3.4.6)
christopher-s Apr 2, 2026
75daf98
fix(gemini): correct API field casing, add safety finish reasons, pag…
christopher-s Apr 2, 2026
50683e6
fix(gemini): correct misleading SAFETY comment — partial SSE content …
christopher-s Apr 2, 2026
ccab588
chore: remove superpowers plans/specs from tracking, add to .gitignore
christopher-s Apr 2, 2026
910471a
chore: remove .data/ from tracking (call_logs, db_backups, storage)
christopher-s Apr 2, 2026
80f2c79
chore: restore .agents/ tracking — do not gitignore
christopher-s Apr 2, 2026
6dcd5b7
chore: restore .agents/workflows — actively used upstream
christopher-s Apr 2, 2026
6698d33
fix(tests): update T28/T31 for gemini dynamic model sync
christopher-s Apr 2, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 5 additions & 1 deletion .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,7 @@ next-env.d.ts

# data and logs
data/
.data/
logs/*

# analysis directories (generated, not tracked)
Expand Down Expand Up @@ -153,4 +154,7 @@ vscode-extension/
typescript

# Gemini Antigravity agent data
.gemini/
.gemini/

# Superpowers plans/specs (internal tooling, not project code)
docs/superpowers/
19 changes: 3 additions & 16 deletions open-sse/config/providerRegistry.ts
Original file line number Diff line number Diff line change
Expand Up @@ -192,22 +192,9 @@ export const REGISTRY: Record<string, RegistryEntry> = {
clientSecretEnv: "GEMINI_OAUTH_CLIENT_SECRET",
clientSecretDefault: "",
},
models: [
{ id: "gemini-3.1-pro-high", name: "Gemini 3.1 Pro High" },
{ id: "gemini-3.1-pro-low", name: "Gemini 3.1 Pro Low" },
{ id: "gemini-3.1-pro", name: "Gemini 3.1 Pro" },
{ id: "gemini-3-1-pro", name: "Gemini 3.1 Pro (Alt ID)" },
{ id: "gemini-3.1-pro-preview", name: "Gemini 3.1 Pro Preview" },
{ id: "gemini-3.1-flash-lite-preview", name: "Gemini 3.1 Flash Lite Preview" },
{ id: "gemini-3-flash-preview", name: "Gemini 3 Flash Preview" },
{ id: "gemini-2.5-pro", name: "Gemini 2.5 Pro" },
{ id: "gemini-2.5-flash", name: "Gemini 2.5 Flash" },
{ id: "gemini-2.5-flash-lite", name: "Gemini 2.5 Flash Lite" },
{ id: "gemini-2.0-flash", name: "Gemini 2.0 Flash" },
{ id: "gemini-2.0-flash-exp", name: "Gemini 2.0 Flash Exp" },
{ id: "gemini-1.5-pro", name: "Gemini 1.5 Pro" },
{ id: "gemini-1.5-flash", name: "Gemini 1.5 Flash" },
],
models: [],
// Models are populated from Google's API via sync-models (per API key).
// No hardcoded fallback — show nothing until a key is added.
},

"gemini-cli": {
Expand Down
39 changes: 22 additions & 17 deletions open-sse/handlers/chatCore.ts
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,9 @@ import { refreshWithRetry } from "../services/tokenRefresh.ts";
import { createRequestLogger } from "../utils/requestLogger.ts";
import { getModelTargetFormat, PROVIDER_ID_TO_ALIAS } from "../config/providerModels.ts";
import { resolveModelAlias } from "../services/modelDeprecation.ts";
import { getUnsupportedParams, getPassthroughProviders } from "../config/providerRegistry.ts";
import { getUnsupportedParams } from "../config/providerRegistry.ts";
import { hasPerModelQuota, lockModelIfPerModelQuota } from "../services/accountFallback.ts";
import { COOLDOWN_MS } from "../config/constants.ts";
import {
buildErrorBody,
createErrorResult,
Expand Down Expand Up @@ -1375,16 +1377,12 @@ export async function handleChatCore({
`[provider] Node ${connectionId} account deactivated (${statusCode}) — disabling permanently`
);
} else if (errorType === PROVIDER_ERROR_TYPES.RATE_LIMITED) {
// For passthrough providers (e.g. Antigravity), each model has independent
// quota. A 429 on one model must NOT lock out the entire connection — other
// models may still have quota available. Use lockModel() instead.
const isPassthrough = provider && getPassthroughProviders().has(provider);
if (isPassthrough) {
const { lockModel } = await import("../services/accountFallback.ts");
const cooldown = retryAfterMs || 120_000; // 2 min default, same as COOLDOWN_MS.rateLimit
lockModel(provider, connectionId, model, "rate_limited", cooldown);
// For providers with per-model quotas (passthrough providers, Gemini),
// each model has independent quota. A 429 on one model must NOT lock out
// the entire connection — other models may still have quota available.
if (lockModelIfPerModelQuota(provider, connectionId, model, "rate_limited", retryAfterMs || COOLDOWN_MS.rateLimit)) {
console.warn(
`[provider] Node ${connectionId} model-only rate limited (${statusCode}) for ${model} - ${Math.ceil(cooldown / 1000)}s (connection stays active)`
`[provider] Node ${connectionId} model-only rate limited (${statusCode}) for ${model} - ${Math.ceil((retryAfterMs || COOLDOWN_MS.rateLimit) / 1000)}s (connection stays active)`
);
} else {
const rateLimitedUntil = new Date(Date.now() + retryAfterMs).toISOString();
Expand All @@ -1402,13 +1400,20 @@ export async function handleChatCore({
);
}
} else if (errorType === PROVIDER_ERROR_TYPES.QUOTA_EXHAUSTED) {
await updateProviderConnection(connectionId, {
testStatus: "credits_exhausted",
lastErrorType: errorType,
lastError: message,
errorCode: statusCode,
});
console.warn(`[provider] Node ${connectionId} exhausted quota (${statusCode})`);
// Providers with per-model quotas — lock the model only, not the connection
if (lockModelIfPerModelQuota(provider, connectionId, model, "quota_exhausted", retryAfterMs || COOLDOWN_MS.rateLimit)) {
console.warn(
`[provider] Node ${connectionId} model-only quota exhausted (${statusCode}) for ${model} - ${Math.ceil((retryAfterMs || COOLDOWN_MS.rateLimit) / 1000)}s (connection stays active)`
);
} else {
await updateProviderConnection(connectionId, {
testStatus: "credits_exhausted",
lastErrorType: errorType,
lastError: message,
errorCode: statusCode,
});
console.warn(`[provider] Node ${connectionId} exhausted quota (${statusCode})`);
}
} else if (errorType === PROVIDER_ERROR_TYPES.ACCOUNT_DEACTIVATED) {
await updateProviderConnection(connectionId, {
isActive: false,
Expand Down
38 changes: 37 additions & 1 deletion open-sse/services/accountFallback.ts
Original file line number Diff line number Diff line change
Expand Up @@ -100,13 +100,49 @@ export function lockModel(provider, connectionId, model, reason, cooldownMs) {
if (!model) return; // No model → skip model-level locking
ensureCleanupTimer();
const key = `${provider}:${connectionId}:${model}`;
const newUntil = Date.now() + cooldownMs;
// Preserve the longer cooldown if an existing lock has more time remaining.
// Safe without a mutex: no await between get/set, so this runs atomically
// within Node.js's single-threaded event loop.
const existing = modelLockouts.get(key);
if (existing && existing.until > newUntil) return;
modelLockouts.set(key, {
reason,
until: Date.now() + cooldownMs,
until: newUntil,
lockedAt: Date.now(),
});
}

/**
* Whether a provider should use per-model lockouts instead of connection-wide cooldowns.
* Gemini AI Studio has per-model quotas; passthrough providers have independent model limits.
*/
export function hasPerModelQuota(provider: string): boolean {
if (provider === "gemini") return true;
try {
const { getPassthroughProviders } = require("../config/providerRegistry.ts");
return getPassthroughProviders().has(provider);
} catch {
return false;
}
}

/**
* Lock a model (not connection) for a provider with per-model quotas.
* No-ops for providers that don't use per-model lockouts.
*/
export function lockModelIfPerModelQuota(
provider: string,
connectionId: string,
model: string | null,
reason: string,
cooldownMs: number
): boolean {
if (!hasPerModelQuota(provider) || !model) return false;
lockModel(provider, connectionId, model, reason, cooldownMs);
return true;
}

/**
* Check if a specific model on a specific account is locked
* @returns {boolean}
Expand Down
7 changes: 6 additions & 1 deletion open-sse/services/rateLimitManager.ts
Original file line number Diff line number Diff line change
Expand Up @@ -200,6 +200,11 @@ function getLimiterKey(provider, connectionId, model = null) {
if (provider === "codex" && model) {
return `${provider}:${getCodexRateLimitKey(connectionId, model)}`;
}
// Gemini AI Studio has per-model quotas — use model-scoped limiter keys
// so a 429 on one model doesn't pause requests for other models.
if (provider === "gemini" && model) {
return `${provider}:${connectionId}:${model}`;
}
return `${provider}:${connectionId}`;
}

Expand Down Expand Up @@ -570,7 +575,7 @@ export function updateFromResponseBody(provider, connectionId, responseBody, sta
const { retryAfterMs, reason } = parseRetryAfterFromBody(responseBody);

if (retryAfterMs && retryAfterMs > 0) {
const limiter = getLimiter(provider, connectionId, null);
const limiter = getLimiter(provider, connectionId, model);
console.log(
`🚫 [RATE-LIMIT] ${provider}:${connectionId.slice(0, 8)} — body-parsed retry: ${Math.ceil(retryAfterMs / 1000)}s (${reason})`
);
Expand Down
2 changes: 1 addition & 1 deletion open-sse/translator/helpers/geminiHelper.ts
Original file line number Diff line number Diff line change
Expand Up @@ -84,7 +84,7 @@ export function convertOpenAIContentToParts(content) {
const mimeType = mimePart.split(";")[0];

parts.push({
inlineData: { mime_type: mimeType, data: data },
inlineData: { mimeType, data },
});
}
}
Expand Down
2 changes: 1 addition & 1 deletion open-sse/translator/request/claude-to-gemini.ts
Original file line number Diff line number Diff line change
Expand Up @@ -192,7 +192,7 @@ export function claudeToGeminiRequest(model, body, stream) {
if (body.thinking?.type === "enabled" && body.thinking.budget_tokens) {
result.generationConfig.thinkingConfig = {
thinkingBudget: body.thinking.budget_tokens,
include_thoughts: true,
includeThoughts: true,
};
}

Expand Down
6 changes: 4 additions & 2 deletions open-sse/translator/request/gemini-to-openai.ts
Original file line number Diff line number Diff line change
Expand Up @@ -92,11 +92,13 @@ function convertGeminiContent(content) {
parts.push({ type: "text", text: part.text });
}

if (part.inlineData) {
if (part.inlineData || part.inline_data) {
const data = part.inlineData || part.inline_data;
const mimeType = data.mimeType || data.mime_type || "image/png";
parts.push({
type: "image_url",
image_url: {
url: `data:${part.inlineData.mimeType};base64,${part.inlineData.data}`,
url: `data:${mimeType};base64,${data.data}`,
},
});
}
Expand Down
8 changes: 4 additions & 4 deletions open-sse/translator/request/openai-to-gemini.ts
Original file line number Diff line number Diff line change
Expand Up @@ -33,7 +33,7 @@ type GeminiGenerationConfig = {
maxOutputTokens?: unknown;
thinkingConfig?: {
thinkingBudget: number;
include_thoughts: boolean;
includeThoughts: boolean;
};
responseMimeType?: string;
responseSchema?: unknown;
Expand Down Expand Up @@ -317,15 +317,15 @@ export function openaiToGeminiCLIRequest(model, body, stream) {
const budget = budgetMap[body.reasoning_effort] || getDefaultThinkingBudget(model) || 8192;
gemini.generationConfig.thinkingConfig = {
thinkingBudget: budget,
include_thoughts: true,
includeThoughts: true,
};
}

// Thinking config from Claude format
if (body.thinking?.type === "enabled" && body.thinking.budget_tokens) {
gemini.generationConfig.thinkingConfig = {
thinkingBudget: body.thinking.budget_tokens,
include_thoughts: true,
includeThoughts: true,
};
}

Expand Down Expand Up @@ -446,7 +446,7 @@ function wrapInCloudCodeEnvelopeForClaude(model, claudeRequest, credentials = nu
} else if (block.type === "image" && block.source) {
parts.push({
inlineData: {
mime_type: block.source.media_type,
mimeType: block.source.media_type,
data: block.source.data,
},
});
Expand Down
5 changes: 5 additions & 0 deletions open-sse/translator/response/gemini-to-claude.ts
Original file line number Diff line number Diff line change
Expand Up @@ -171,6 +171,11 @@ export function geminiToClaudeResponse(chunk, state) {
stopReason = "tool_use";
} else if (reason === "max_tokens" || reason === "length") {
stopReason = "max_tokens";
} else if (reason === "safety" || reason === "recitation" || reason === "blocklist") {
// Content blocked by Gemini safety. Any text streamed before this finish
// reason has already been emitted to the client — this is unavoidable in
// SSE streaming. Map to end_turn (Claude has no "content blocked" reason).
stopReason = "end_turn";
} else {
stopReason = "end_turn";
}
Expand Down
5 changes: 5 additions & 0 deletions open-sse/translator/response/gemini-to-openai.ts
Original file line number Diff line number Diff line change
Expand Up @@ -225,6 +225,11 @@ export function geminiToOpenAIResponse(chunk, state) {
if (finishReason === "stop" && state.toolCalls.size > 0) {
finishReason = "tool_calls";
}
// Content blocked by Gemini safety filters — pass through as "content_filter"
// so downstream clients can distinguish from normal completion.
if (finishReason === "safety" || finishReason === "recitation" || finishReason === "blocklist") {
finishReason = "content_filter";
}

const finalChunk: Record<string, unknown> = {
id: `chatcmpl-${state.messageId}`,
Expand Down
Loading
Loading