From 97d3fa947624443aa28906e19b7de804334f0806 Mon Sep 17 00:00:00 2001 From: Walter Riley Date: Sat, 3 Oct 2026 13:47:31 -0500 Subject: [PATCH 1/4] feat(sse): opt-in per-model concurrency caps per connection Adds optional per-model concurrency caps on a provider connection, enforced by the chatCore admission gate alongside the existing per-account cap. Squashed from the original series (incl. merge-time helper extraction): - feat(sse): add opt-in per-model concurrency caps per connection - feat(dashboard): use i18n keys for per-model concurrency editor - docs(changelog): add fragment for per-model concurrency caps - extract per-model concurrency UI/gate helpers under file-size ceilings - test(sse): align admission-gate source assertions with resolveModelSemaphore helper - fix(i18n): mirror per-model concurrency keys into all locales - chore(quality): rebaseline per-model concurrency file-size ceilings - fix(shared): move model concurrency bounds to a server-free leaf --- .../features/13700-per-model-concurrency.md | 1 + config/quality/file-size-baseline.json | 5 +- docs/architecture/RESILIENCE_GUIDE.md | 45 ++++ open-sse/executors/base/validationDispatch.ts | 6 + open-sse/handlers/chatCore.ts | 14 ++ open-sse/handlers/chatCore/executorHelpers.ts | 62 ++++- open-sse/services/accountSemaphore.ts | 17 ++ open-sse/services/rateLimitManager.ts | 7 +- .../rateLimitManager/overrideUpdates.ts | 8 +- open-sse/types.d.ts | 6 + scripts/i18n/untranslatable-keys.json | 1 + .../components/modals/EditConnectionModal.tsx | 29 ++- .../modals/ModelConcurrencyField.tsx | 29 +++ .../modals/rateLimitOverridesFromForm.ts | 46 ++++ src/domain/types.ts | 1 + src/i18n/messages/am.json | 5 +- src/i18n/messages/ar.json | 5 +- src/i18n/messages/az.json | 5 +- src/i18n/messages/bg.json | 5 +- src/i18n/messages/bn.json | 5 +- src/i18n/messages/bs.json | 5 +- src/i18n/messages/cs.json | 5 +- src/i18n/messages/da.json | 5 +- src/i18n/messages/de.json | 5 +- src/i18n/messages/el.json | 5 +- src/i18n/messages/en.json | 5 +- src/i18n/messages/es.json | 5 +- src/i18n/messages/et.json | 5 +- src/i18n/messages/fa.json | 5 +- src/i18n/messages/fi.json | 5 +- src/i18n/messages/fr.json | 5 +- src/i18n/messages/ga.json | 5 +- src/i18n/messages/gu.json | 5 +- src/i18n/messages/ha.json | 5 +- src/i18n/messages/he.json | 5 +- src/i18n/messages/hi.json | 5 +- src/i18n/messages/hr.json | 5 +- src/i18n/messages/hu.json | 5 +- src/i18n/messages/hy.json | 5 +- src/i18n/messages/id.json | 5 +- src/i18n/messages/ig.json | 5 +- src/i18n/messages/it.json | 5 +- src/i18n/messages/ja.json | 5 +- src/i18n/messages/ka.json | 5 +- src/i18n/messages/km.json | 5 +- src/i18n/messages/kn.json | 5 +- src/i18n/messages/ko.json | 5 +- src/i18n/messages/lt.json | 5 +- src/i18n/messages/lv.json | 5 +- src/i18n/messages/ml.json | 5 +- src/i18n/messages/mr.json | 5 +- src/i18n/messages/ms.json | 5 +- src/i18n/messages/mt.json | 5 +- src/i18n/messages/my.json | 5 +- src/i18n/messages/ne.json | 5 +- src/i18n/messages/nl.json | 5 +- src/i18n/messages/no.json | 5 +- src/i18n/messages/or.json | 5 +- src/i18n/messages/pa.json | 5 +- src/i18n/messages/phi.json | 5 +- src/i18n/messages/pl.json | 5 +- src/i18n/messages/pt-BR.json | 5 +- src/i18n/messages/pt.json | 5 +- src/i18n/messages/ro.json | 5 +- src/i18n/messages/ru.json | 5 +- src/i18n/messages/si.json | 5 +- src/i18n/messages/sk.json | 5 +- src/i18n/messages/sl.json | 5 +- src/i18n/messages/sr.json | 5 +- src/i18n/messages/sv.json | 5 +- src/i18n/messages/sw.json | 5 +- src/i18n/messages/ta.json | 5 +- src/i18n/messages/te.json | 5 +- src/i18n/messages/th.json | 5 +- src/i18n/messages/tr.json | 5 +- src/i18n/messages/uk-UA.json | 5 +- src/i18n/messages/ur.json | 5 +- src/i18n/messages/uz.json | 5 +- src/i18n/messages/vi.json | 5 +- src/i18n/messages/yo.json | 5 +- src/i18n/messages/zh-CN.json | 5 +- src/i18n/messages/zh-TW.json | 5 +- src/lib/db/providers/columns.ts | 114 ++++++++- src/lib/db/providers/lazyConnectionView.ts | 19 +- src/lib/providers/modelConcurrency.ts | 73 ++++++ src/shared/constants/modelConcurrency.ts | 26 ++ src/shared/validation/schemas/provider.ts | 34 +++ src/sse/services/auth.ts | 3 + src/sse/services/noAuthOptionalApiKey.ts | 6 + tests/unit/account-concurrency-cap.test.ts | 130 +++++++++- .../chatcore-hierarchical-admission.test.ts | 2 + tests/unit/model-concurrency-gate.test.ts | 232 ++++++++++++++++++ tests/unit/model-concurrency-input.test.ts | 54 ++++ .../provider-patch-model-concurrency.test.ts | 126 ++++++++++ 94 files changed, 1333 insertions(+), 98 deletions(-) create mode 100644 changelog.d/features/13700-per-model-concurrency.md create mode 100644 src/app/(dashboard)/dashboard/providers/[id]/components/modals/ModelConcurrencyField.tsx create mode 100644 src/app/(dashboard)/dashboard/providers/[id]/components/modals/rateLimitOverridesFromForm.ts create mode 100644 src/lib/providers/modelConcurrency.ts create mode 100644 src/shared/constants/modelConcurrency.ts create mode 100644 tests/unit/model-concurrency-gate.test.ts create mode 100644 tests/unit/model-concurrency-input.test.ts create mode 100644 tests/unit/provider-patch-model-concurrency.test.ts diff --git a/changelog.d/features/13700-per-model-concurrency.md b/changelog.d/features/13700-per-model-concurrency.md new file mode 100644 index 00000000000..4d281833da4 --- /dev/null +++ b/changelog.d/features/13700-per-model-concurrency.md @@ -0,0 +1 @@ +- **feat(sse):** opt-in per-model concurrency caps per connection — configure exact `modelConcurrency` ceilings inside `rateLimitOverrides` (dashboard editor or `PATCH /api/providers/[id]`), enforced as an atomic fourth gate alongside the global/provider/account semaphores with local queueing; unconfigured connections behave exactly as before ([#13700](https://github.com/diegosouzapw/OmniRoute/pull/13700)) diff --git a/config/quality/file-size-baseline.json b/config/quality/file-size-baseline.json index fbeab3f25fb..398590b5b0a 100644 --- a/config/quality/file-size-baseline.json +++ b/config/quality/file-size-baseline.json @@ -516,13 +516,14 @@ "_rebaseline_pr1043_minimax_tts": "Upstream port decolua/9router#1043 (toanalien) own growth: audioSpeech.ts 965->1061 (+96). Adds MiniMax T2A v2 TTS dispatch (handleMinimaxSpeech + hexToBytes helper) — provider entry was already in audioRegistry (format: minimax-tts) but no handler existed, falling through to the OpenAI-compatible default that fails (T2A has custom shape + hex-encoded audio + base_resp envelope). New branch sits next to the other inline provider branches (xiaomi-mimo, coqui, tortoise, aws-polly) — extracting would just create indirection. Covered by tests/unit/minimax-tts-1043.test.ts (3 tests, GREEN: success, base_resp error, invalid-hex).", "_rebaseline_pr4592_exclude_exhausted_auto": "Reconcile #4592 already-merged growth: combo.ts 2991->3036 (+45, terminal-status quota-cutoff exclusion in buildAutoCandidates + opt-in gate). Fast-gate PR->release does not run check:file-size.", "_rebaseline_2026_09_04_12737_codex_ws_premature_close": "PR #12737 own growth, re-measured after merging release/v3.8.51: open-sse/executors/codex.ts 1528->1530 (+2 over the frozen cap; the PR adds +13 lines and the tip had 11 lines of headroom), now also logging the failure and the WS close code/reason via review follow-up. The ws.onclose handler now fails the stream (failController with code upstream_websocket_closed) when the socket closes before any terminal response event, instead of finishStream(upstream_closed) silently truncating output as a completed stream. The +8 is the guard + routing through the existing failController at the existing onclose chokepoint — not extractable without hiding the close-handler semantics. Covered by the two new regression tests in tests/unit/executor-codex.test.ts (premature close emits exactly one response.failed; normal close after response.completed emits no second terminal event).", + "_rebaseline_2026_09_25_13700_per_model_concurrency": "PR #13700 (per-model concurrency caps) own growth, re-measured after rebasing onto release/v3.8.52: frozen caps open-sse/handlers/chatCore.ts 6444->6457 and src/sse/services/auth.ts 3623->3624 (actual PR growth +14 / +3 lines; the auth.ts tip had 2 lines of headroom). The per-model ceiling logic lives outside these frozen files — resolveModelSemaphore in open-sse/handlers/chatCore/executorHelpers.ts and normalizeModelConcurrencyMap in src/lib/db/providers/columns.ts (both non-frozen) — so what remains here is the irreducible call-site wiring: the model gate joins the existing composite account/provider admission array at acquireConcurrencyGates, its key/max are echoed on the pre_semaphore trace, and materializeConnection normalizes the stored map once at credential selection. Covered by tests/unit/account-concurrency-cap.test.ts, tests/unit/model-concurrency-gate.test.ts and tests/unit/chatcore-hierarchical-admission.test.ts.", "open-sse/executors/antigravity.ts": 1717, "open-sse/executors/base.ts": 1760, "open-sse/executors/chatgpt-web.ts": 5056, "open-sse/executors/codex.ts": 1584, "open-sse/executors/cursor.ts": 1868, "open-sse/executors/muse-spark-web.ts": 1405, - "open-sse/handlers/chatCore.ts": 6444, + "open-sse/handlers/chatCore.ts": 6457, "open-sse/handlers/imageGeneration.ts": 3334, "open-sse/handlers/search.ts": 1789, "open-sse/mcp-server/schemas/tools.ts": 1621, @@ -563,7 +564,7 @@ "src/shared/constants/providers/apikey/gateways.ts": 1544, "src/shared/services/cliRuntime.ts": 1296, "src/sse/handlers/chat.ts": 2586, - "src/sse/services/auth.ts": 3623, + "src/sse/services/auth.ts": 3624, "tests/unit/account-fallback-service.test.ts": 2453, "tests/unit/provider-validation-specialty.test.ts": 4656, "open-sse/services/autoCombo/virtualFactory.ts": 1258, diff --git a/docs/architecture/RESILIENCE_GUIDE.md b/docs/architecture/RESILIENCE_GUIDE.md index eff68abb158..41b61993e3a 100644 --- a/docs/architecture/RESILIENCE_GUIDE.md +++ b/docs/architecture/RESILIENCE_GUIDE.md @@ -320,6 +320,51 @@ Each provider connection can declare a `max_concurrent` ceiling Leave it empty for no limit. This is the single knob that drives the serialization layer below — set it to the account's real concurrency (e.g. GLM ~1, MiniMax ~2). +### Per-model concurrency caps (`modelConcurrency`) + +A connection can additionally declare exact per-model concurrency ceilings +inside its `rateLimitOverrides` map: + +```json +{ + "rateLimitOverrides": { + "maxConcurrent": 4, + "modelConcurrency": { "glm-5": 1, "glm-4.7": 3 } + } +} +``` + +Set it in the connection modal (**Rate limit overrides → Per-model +concurrency caps**, one `model=cap` per line) or via +`PATCH /api/providers/[id]` with the same JSON shape. Key semantics: + +- **Connection-wide vs model-specific:** `maxConcurrent` remains the shared + connection-wide ceiling. When both apply, both gates are acquired + atomically in the same composite gate (`global → provider → account → +model`); the effective behavior is the stricter applicable limit. +- **Exact model-key match:** the key is the model string passed to the + executor after routing resolution — normally the bare upstream model id + (`glm-5`), not a client-side `provider/model` alias (`zai/glm-5` does not + match `glm-5`). Values are positive-integer concurrent-request ceilings. +- **Local queueing, no discovery:** excess requests queue locally with the + existing queue/timeout semantics (typed `SEMAPHORE_TIMEOUT` / + `SEMAPHORE_QUEUE_FULL` admission errors). OmniRoute does not discover or + infer upstream policy — it enforces the exact ceilings the operator + configured. A saturated model gate never disables the provider and never + creates a permanent model lockout; upstream 429/cooldown/fallback behavior + remains the error backstop. +- **Per-connection, per-process scope:** caps are per database connection + and held in-memory, so two connections reusing the same upstream API key + do not coordinate with each other. +- **Unconfigured means unchanged:** omitting the map (or leaving the + dashboard field blank) adds no model gate. Example configuration without + asserting any universal provider limit: + +```text +glm-5=1 +glm-4.7=3 +``` + ### Quota-share request serialization When a quota-share dispatch targets a connection that declares a positive diff --git a/open-sse/executors/base/validationDispatch.ts b/open-sse/executors/base/validationDispatch.ts index e651822a412..90a749cab1c 100644 --- a/open-sse/executors/base/validationDispatch.ts +++ b/open-sse/executors/base/validationDispatch.ts @@ -7,6 +7,12 @@ export type ProviderCredentials = { expiresAt?: string; connectionId?: string; // T07: used for API key rotation index maxConcurrent?: number | null; + /** + * Optional per-model concurrency ceilings for this connection (see + * ProviderCredentials in open-sse/types.d.ts). Normalized at credential + * selection; the chat core resolves the exact-model cap fail-open. + */ + modelConcurrency?: Record | null; providerSpecificData?: Record; requestEndpointPath?: string; }; diff --git a/open-sse/handlers/chatCore.ts b/open-sse/handlers/chatCore.ts index e618526b9d2..0b756d4ffd7 100644 --- a/open-sse/handlers/chatCore.ts +++ b/open-sse/handlers/chatCore.ts @@ -87,6 +87,7 @@ export { clearCombosCache, clearUpstreamProxyConfigCache } from "./chatCore/comb import { resolveAccountSemaphoreKey, resolveAccountSemaphoreMaxConcurrency, + resolveModelSemaphore, buildClaudePromptCacheLogMeta, } from "./chatCore/executorHelpers.ts"; import { @@ -3182,6 +3183,13 @@ async function handleChatCoreInner({ connectionId: attemptConnectionId, credentials: execCreds, }); + // Opt-in per-model ceiling; joins the composite gate below. + const modelGate = resolveModelSemaphore({ + provider, + model: modelToCall, + connectionId: attemptConnectionId, + credentials: execCreds, + }); const canonicalProviderKey = resolveProviderId(String(provider).trim().toLowerCase()); const providerConcurrency = resilienceSettings.providerQuotaOverrides[canonicalProviderKey] @@ -3190,6 +3198,8 @@ async function handleChatCoreInner({ trace("pre_semaphore", { semaphoreKey: accountSemaphoreKey, max: accountSemaphoreMaxConcurrency, + modelSemaphoreKey: modelGate.key, + modelMax: modelGate.maxConcurrency, }); if (accountSemaphoreKey && accountSemaphoreMaxConcurrency != null) { updatePendingScope(pendingScope, { @@ -3216,6 +3226,10 @@ async function handleChatCoreInner({ key: accountSemaphoreKey || "", maxConcurrency: accountSemaphoreKey ? accountSemaphoreMaxConcurrency : null, }, + { + key: modelGate.key || "", + maxConcurrency: modelGate.key ? modelGate.maxConcurrency : null, + }, ], { timeoutMs: maxWaitMs, diff --git a/open-sse/handlers/chatCore/executorHelpers.ts b/open-sse/handlers/chatCore/executorHelpers.ts index f2c2e969286..828010f8f33 100644 --- a/open-sse/handlers/chatCore/executorHelpers.ts +++ b/open-sse/handlers/chatCore/executorHelpers.ts @@ -1,5 +1,8 @@ import { FORMATS } from "../../translator/formats.ts"; -import { buildAccountSemaphoreKey } from "../../services/accountSemaphore.ts"; +import { + buildAccountSemaphoreKey, + buildModelSemaphoreKey, +} from "../../services/accountSemaphore.ts"; import { getHeaderValueCaseInsensitive } from "./headers.ts"; function toFiniteNumberOrNull(value: unknown): number | null { @@ -44,6 +47,27 @@ export function resolveAccountSemaphoreMaxConcurrency( return toFiniteNumberOrNull(credentials?.maxConcurrent); } +/** + * Resolve the per-model concurrency cap for one execution attempt. + * + * Exact-match on the model string passed to the executor after routing + * resolution (normally the bare upstream model id, e.g. "glm-5" — not a + * client-side `provider/model` alias). Missing credentials, a missing or + * malformed map, or a non-positive cap all resolve to null ("no model + * gate") so an unconfigured or corrupt map never blocks a request. + */ +export function resolveModelSemaphoreMaxConcurrency( + credentials: Record | null | undefined, + model: string | null | undefined +): number | null { + if (!model || typeof model !== "string" || model.trim().length === 0) return null; + const map = credentials?.modelConcurrency; + if (!map || typeof map !== "object" || Array.isArray(map)) return null; + const cap = toFiniteNumberOrNull((map as Record)[model]); + if (cap == null || !Number.isInteger(cap) || cap < 1) return null; + return cap; +} + export function resolveAccountSemaphoreKey({ provider, model, @@ -60,6 +84,42 @@ export function resolveAccountSemaphoreKey({ return buildAccountSemaphoreKey({ provider, accountKey }); } +/** + * Build the per-connection, per-model semaphore key for one execution + * attempt. Returns null when no positive cap is configured for + * `provider + connection + model`, in which case the caller must not add a + * model requirement to the composite gate (behavior unchanged). + */ +export function resolveModelSemaphoreKey({ + provider, + model, + connectionId, + credentials, +}: { + provider: string | null | undefined; + model: string; + connectionId: string | null | undefined; + credentials: Record | null | undefined; +}): string | null { + const accountKey = resolveAccountSemaphoreAccountKey(connectionId, credentials); + if (!accountKey || !provider) return null; + if (resolveModelSemaphoreMaxConcurrency(credentials, model) == null) return null; + return buildModelSemaphoreKey({ provider, accountKey, model }); +} + +/** Per-model gate (key + cap) for the composite semaphore; key null = no gate. */ +export function resolveModelSemaphore(args: { + provider: string | null | undefined; + model: string; + connectionId: string | null | undefined; + credentials: Record | null | undefined; +}): { key: string | null; maxConcurrency: number | null } { + return { + key: resolveModelSemaphoreKey(args), + maxConcurrency: resolveModelSemaphoreMaxConcurrency(args.credentials, args.model), + }; +} + export function buildClaudePromptCacheLogMeta( targetFormat: string, finalBody: Record | null | undefined, diff --git a/open-sse/services/accountSemaphore.ts b/open-sse/services/accountSemaphore.ts index 3664f38156a..549f69123b0 100644 --- a/open-sse/services/accountSemaphore.ts +++ b/open-sse/services/accountSemaphore.ts @@ -72,6 +72,23 @@ export function buildAccountSemaphoreKey({ return `${String(provider)}:${String(accountKey)}`; } +export interface ModelSemaphoreKeyParts extends AccountSemaphoreKeyParts { + model: string; +} + +/** + * Collision-safe key for the per-connection, per-model concurrency gate. + * The `model:` infix keeps model gates disjoint from account gates even + * when a model id itself contains `:` characters. + */ +export function buildModelSemaphoreKey({ + provider, + accountKey, + model, +}: ModelSemaphoreKeyParts): string { + return `${String(provider)}:${String(accountKey)}:model:${String(model)}`; +} + function isBypassed(maxConcurrency?: number | null): boolean { return maxConcurrency == null || !Number.isFinite(maxConcurrency) || maxConcurrency <= 0; } diff --git a/open-sse/services/rateLimitManager.ts b/open-sse/services/rateLimitManager.ts index 7bd415cd8af..55c9fdd13eb 100644 --- a/open-sse/services/rateLimitManager.ts +++ b/open-sse/services/rateLimitManager.ts @@ -45,6 +45,7 @@ import { import { LimiterWedgeWatchdog, WATCHDOG_INTERVAL_MS } from "./rateLimitManager/wedgeWatchdog"; import { createCancellableJob } from "./rateLimitManager/queuedJobCancel"; import { toNumber } from "@/shared/utils/numeric"; +import type { ConnectionRateLimitOverrides } from "@/lib/db/providers/columns"; import { getExecutorTimeoutMs, resolveConnectionTimeoutMs, @@ -103,7 +104,7 @@ const enabledConnections = new Set(); // Store per-connection rate limit overrides (RPM, TPM, TPD, minTime, maxConcurrent) // Populated from provider_connections.rateLimitOverrides on startup and refresh. -const connectionRateLimitOverrides = new Map>(); +const connectionRateLimitOverrides = new Map(); // Store learned limits for persistence (debounced) // One learned entry per limiter key (provider:connection[:model]). The previous @@ -266,7 +267,7 @@ export function resolveRequestQueueMaxWaitMs( */ export function resolveExecutionMaxWaitMs(connectionId?: string): number { const override = connectionId - ? (connectionRateLimitOverrides.get(connectionId) as Record | undefined) + ? (connectionRateLimitOverrides.get(connectionId) as ConnectionRateLimitOverrides | undefined) ?.executionMaxWaitMs : undefined; return resolveOverride(override, currentRequestQueueSettings.executionMaxWaitMs); @@ -536,7 +537,7 @@ export function isRateLimitEnabled(connectionId) { * connection so the next request gets a fresh limiter with the new settings. * * @param {string} connectionId - * @param {Record | null} overrides - New overrides (null/undefined clears) + * @param {ConnectionRateLimitOverrides | null} overrides - New overrides (null/undefined clears) */ export function refreshConnectionRateLimits(connectionId, overrides) { if (overrides === null || overrides === undefined) { diff --git a/open-sse/services/rateLimitManager/overrideUpdates.ts b/open-sse/services/rateLimitManager/overrideUpdates.ts index 5a9fcd812a1..04fe751907a 100644 --- a/open-sse/services/rateLimitManager/overrideUpdates.ts +++ b/open-sse/services/rateLimitManager/overrideUpdates.ts @@ -10,8 +10,10 @@ * @module services/rateLimitManager/overrideUpdates */ +import type { ConnectionRateLimitOverrides } from "@/lib/db/providers/columns"; + export function buildOverrideUpdates( - overrides: Record + overrides: ConnectionRateLimitOverrides ): Record { const updates: Record = {}; if (typeof overrides.maxConcurrent === "number" && overrides.maxConcurrent > 0) { @@ -29,13 +31,13 @@ export function buildOverrideUpdates( } export function loadOverrideMap( - target: Map>, + target: Map, connections: Array> ): void { target.clear(); for (const conn of connections) { const overrides = conn.rateLimitOverrides; if (overrides && typeof overrides === "object" && !Array.isArray(overrides)) - target.set(String(conn.id), overrides as Record); + target.set(String(conn.id), overrides as ConnectionRateLimitOverrides); } } diff --git a/open-sse/types.d.ts b/open-sse/types.d.ts index c2f35693d49..6ee5dd72494 100644 --- a/open-sse/types.d.ts +++ b/open-sse/types.d.ts @@ -19,6 +19,12 @@ export interface ProviderCredentials { connectionId: string; /** Optional per-account concurrency cap */ maxConcurrent?: number | null; + /** + * Optional per-model concurrency ceilings for this connection, keyed by + * the exact model string passed to the executor after routing resolution + * (normally the bare upstream model id). Absent/null means no model gate. + */ + modelConcurrency?: Record | null; /** User email associated with the connection */ email?: string; /** API key (for apikey auth type) */ diff --git a/scripts/i18n/untranslatable-keys.json b/scripts/i18n/untranslatable-keys.json index 053ad00c7ba..420232209da 100644 --- a/scripts/i18n/untranslatable-keys.json +++ b/scripts/i18n/untranslatable-keys.json @@ -151,6 +151,7 @@ "providers.prefixLabel", "providers.proxy", "providers.proxyConfiguredBySource", + "providers.rateLimitOverridesModelConcurrencyPlaceholder", "providers.responses", "providers.responsesApi", "providers.responsesPath", diff --git a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx index 07e0b23fe80..a9e3aa08b20 100644 --- a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx +++ b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/EditConnectionModal.tsx @@ -20,6 +20,12 @@ import { maskEmail } from "@/shared/utils/maskEmail"; import useEmailPrivacyStore from "@/store/emailPrivacyStore"; import { useNotificationStore } from "@/store/notificationStore"; import { type CodexServiceTier } from "@/lib/providers/requestDefaults"; +import type { ConnectionRateLimitOverrides } from "@/lib/db/providers/columns"; +import ModelConcurrencyField from "./ModelConcurrencyField"; +import { + buildRateLimitOverridesFromForm, + modelConcurrencyFormValue, +} from "./rateLimitOverridesFromForm"; import { resolveDashboardProviderInfo } from "../../../providerPageUtils"; import { isBaseUrlConfigurableProvider, @@ -77,7 +83,7 @@ export interface EditConnectionModalConnection { email?: string; priority?: number; maxConcurrent?: number | null; - rateLimitOverrides?: Record | null; + rateLimitOverrides?: ConnectionRateLimitOverrides | null; authType?: string; provider?: string; apiKey?: string; @@ -123,6 +129,7 @@ export default function EditConnectionModal({ minTime: "", maxWaitMs: "", rateLimitMaxConcurrent: "", + modelConcurrency: "", apiKey: "", healthCheckInterval: "" as number | "", baseUrl: "", @@ -343,6 +350,7 @@ export default function EditConnectionModal({ connection.rateLimitOverrides?.maxConcurrent != null ? String(connection.rateLimitOverrides.maxConcurrent) : "", + modelConcurrency: modelConcurrencyFormValue(connection.rateLimitOverrides), apiKey: "", // Unset per-connection override means "follow the global default" — // surface that as an empty field (0 renders as an explicit opt-out). @@ -561,16 +569,9 @@ export default function EditConnectionModal({ healthCheckInterval: formData.healthCheckInterval === "" ? undefined : formData.healthCheckInterval, }; - const overrides: Record = {}; - if (formData.rpm.trim()) overrides.rpm = Number(formData.rpm); - if (formData.rpd.trim()) overrides.rpd = Number(formData.rpd); - if (formData.tpm.trim()) overrides.tpm = Number(formData.tpm); - if (formData.tpd.trim()) overrides.tpd = Number(formData.tpd); - if (formData.minTime.trim()) overrides.minTime = Number(formData.minTime); - if (formData.maxWaitMs.trim()) overrides.maxWaitMs = Number(formData.maxWaitMs); - if (formData.rateLimitMaxConcurrent.trim()) - overrides.maxConcurrent = Number(formData.rateLimitMaxConcurrent); - updates.rateLimitOverrides = Object.keys(overrides).length > 0 ? overrides : null; + const rateLimit = buildRateLimitOverridesFromForm(formData); + if (rateLimit.error) return setSaveError(rateLimit.error); + updates.rateLimitOverrides = rateLimit.overrides; if (isAntigravityFamily) { updates.projectId = trimmedCloudCodeProjectId || null; } @@ -1303,6 +1304,12 @@ export default function EditConnectionModal({ placeholder={t("inherit")} hint={t("rateLimitOverridesMaxConcurrentHint")} /> + + setFormData({ ...formData, modelConcurrency }) + } + /> diff --git a/src/app/(dashboard)/dashboard/providers/[id]/components/modals/ModelConcurrencyField.tsx b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/ModelConcurrencyField.tsx new file mode 100644 index 00000000000..56c9b0332a0 --- /dev/null +++ b/src/app/(dashboard)/dashboard/providers/[id]/components/modals/ModelConcurrencyField.tsx @@ -0,0 +1,29 @@ +"use client"; + +import { useTranslations } from "next-intl"; +import { Textarea } from "@/shared/components"; + +export interface ModelConcurrencyFieldProps { + value: string; + onChange: (value: string) => void; +} + +/** Per-model concurrency caps editor (`model=cap`, one per line). */ +export default function ModelConcurrencyField({ value, onChange }: ModelConcurrencyFieldProps) { + const t = useTranslations("providers"); + return ( +
+ +