Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 1 addition & 3 deletions apps/api/src/routes/admin-bulk-block.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -4,12 +4,10 @@ import { app } from "@/index.js";
import { createTestUser, deleteAll } from "@/testing.js";

import { db, tables } from "@llmgateway/db";
import { MAX_BULK_BLOCK_ORGANIZATIONS } from "@llmgateway/shared";

const originalAdminEmails = process.env.ADMIN_EMAILS;

// Mirrors MAX_BULK_BLOCK_ORGANIZATIONS in apps/api/src/routes/admin.ts.
const MAX_BULK_BLOCK_ORGANIZATIONS = 500;

interface PreviewResponse {
search: string;
matched: number;
Expand Down
10 changes: 2 additions & 8 deletions apps/api/src/routes/admin.ts
Original file line number Diff line number Diff line change
Expand Up @@ -73,6 +73,8 @@ import {
type DevPlanTier,
getDevPlanPremiumWeeklyLimit,
getIncludedResetPassesRemaining,
MAX_BULK_BLOCK_ORGANIZATIONS,
MIN_BULK_BLOCK_SEARCH_LENGTH,
} from "@llmgateway/shared";
import {
getResendClient,
Expand Down Expand Up @@ -6468,14 +6470,6 @@ admin.openapi(blockOrganizationRoute, async (c) => {

// --- Bulk block organizations (filtered list) ---

// Hard ceiling on a single bulk block. A filter matching more than this is
// treated as too broad to be an intentional selection and is rejected outright
// rather than partially applied.
const MAX_BULK_BLOCK_ORGANIZATIONS = 500;
// A one or two character filter matches far too much to be a deliberate
// selection, so require something specific enough to identify a set of orgs.
const MIN_BULK_BLOCK_SEARCH_LENGTH = 3;

const bulkBlockOrganizationSchema = z.object({
id: z.string(),
name: z.string(),
Expand Down
44 changes: 44 additions & 0 deletions apps/gateway/src/models/models.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -101,6 +101,50 @@ describe("Models API", () => {
}
});

test("GET /v1/models exposes max_output per provider mapping and as the safe model-level minimum", async () => {
const res = await app.request("/v1/models?include_deactivated=true");
expect(res.status).toBe(200);
const json = await res.json();

const haiku = json.data.find(
(m: { id: string }) => m.id === "claude-haiku-4-5",
);
expect(haiku).toBeDefined();
const anthropicMapping = haiku.providers.find(
(p: { providerId: string }) => p.providerId === "anthropic",
);
expect(anthropicMapping.max_output).toBe(64000);
expect(haiku.max_output).toBe(64000);

// The model-level value must be the minimum across still-servable
// mappings that declare a limit (requests are validated against the
// serving mapping's maxOutput, and deactivated mappings can no longer
// serve), and omitted entirely when no such mapping declares one.
const now = new Date();
const definitionById = new Map(modelsList.map((m) => [m.id, m]));
for (const model of json.data) {
const definition = definitionById.get(model.id);
expect(definition).toBeDefined();
const limits = (definition!.providers as ProviderModelMapping[])
.filter((p) => !(p.deactivatedAt && now > p.deactivatedAt))
.map((p) => p.maxOutput)
.filter((limit): limit is number => limit !== undefined);

expect(model.max_output).toBe(
limits.length > 0 ? Math.min(...limits) : undefined,
);
}

// deepseek-v3.2 has a deactivated Nebius mapping declaring 32768; the
// advertised bound must come from the mappings that can still serve
// (65536), not from limits that no longer apply.
const deepseek = json.data.find(
(m: { id: string }) => m.id === "deepseek-v3.2",
);
expect(deepseek).toBeDefined();
expect(deepseek.max_output).toBe(65536);
});

test("GET /v1/models exposes reasoning_efforts on provider mappings that define them", async () => {
const res = await app.request("/v1/models");
expect(res.status).toBe(200);
Expand Down
28 changes: 28 additions & 0 deletions apps/gateway/src/models/models.ts
Original file line number Diff line number Diff line change
Expand Up @@ -92,6 +92,10 @@ const modelSchema = z.object({
description:
"Minimum prompt length (in tokens) the provider requires before a prompt-cache write can occur. cache_control markers on shorter prompts are accepted but silently not cached by the provider.",
}),
max_output: z.number().optional().openapi({
description:
"Maximum output tokens this provider mapping accepts as max_tokens; larger requests are rejected with HTTP 400. Omitted when the mapping declares no limit (any max_tokens is accepted).",
}),
stability: z
.enum(["stable", "beta", "unstable", "experimental"])
.optional(),
Expand All @@ -115,6 +119,10 @@ const modelSchema = z.object({
input_audio_hour: z.string().optional(),
}),
context_length: z.number().optional(),
max_output: z.number().optional().openapi({
description:
"Largest max_tokens value guaranteed to be accepted regardless of which provider mapping serves the request (the minimum across still-servable mappings that declare a limit; deactivated mappings are excluded). Omitted when no such mapping declares one.",
}),
Comment on lines 121 to +125
per_request_limits: z.record(z.string()).optional(),
supported_parameters: z.array(z.string()).optional(),
json_output: z.boolean(),
Expand Down Expand Up @@ -307,6 +315,7 @@ modelsApi.openapi(listModels, async (c) => {
reasoning: provider.reasoning ?? false,
reasoning_efforts: provider.reasoningEfforts,
min_cacheable_tokens: provider.minCacheableTokens,
max_output: provider.maxOutput,
stability: provider.stability ?? model.stability,
};
}),
Expand All @@ -319,6 +328,7 @@ modelsApi.openapi(listModels, async (c) => {
context_length:
Math.max(...model.providers.map((p) => p.contextSize ?? 0)) ??
undefined,
max_output: getModelLevelMaxOutput(model.providers, currentDate),
per_request_limits: getPerRequestLimits(model),
// Get supported parameters from model definitions with fallback to defaults
supported_parameters: getSupportedParametersFromModel(model),
Expand Down Expand Up @@ -369,6 +379,24 @@ function getModelLevelDate(dates: (Date | undefined)[]): string | undefined {
.toISOString();
}

// The public max_tokens bound for a model. Requests are validated against the
// maxOutput of whichever provider mapping ends up serving them, so advertise the
// minimum across mappings that declare one — the largest value guaranteed to be
// accepted regardless of routing. Mappings without a declared limit accept any
// max_tokens and therefore do not constrain the bound. Deactivated mappings can
// no longer serve requests, so they don't constrain it either; deprecated
// mappings remain routable and keep enforcing their limit, so they stay in.
function getModelLevelMaxOutput(
mappings: ProviderModelMapping[],
currentDate: Date,
): number | undefined {
const limits = mappings
.filter((p) => !(p.deactivatedAt && currentDate > p.deactivatedAt))
.map((p) => p.maxOutput)
.filter((limit): limit is number => limit !== undefined);
return limits.length > 0 ? Math.min(...limits) : undefined;
}

// Whether a provider mapping carries any pricing information at all.
function hasPricing(p: ProviderModelMapping): boolean {
return (
Expand Down
34 changes: 5 additions & 29 deletions ee/admin/src/app/organizations/[orgId]/provider-keys-table.tsx
Original file line number Diff line number Diff line change
@@ -1,4 +1,6 @@
import { ProviderKeySpendCell } from "@/components/provider-key-spend-cell";
import { ProviderKeySpendDialog } from "@/components/provider-key-spend-dialog";
import { ProviderKeyStatusBadge } from "@/components/provider-key-status-badge";
import { Badge } from "@/components/ui/badge";
import {
Table,
Expand All @@ -8,7 +10,6 @@ import {
TableHeader,
TableRow,
} from "@/components/ui/table";
import { formatUsd, hasReachedSpendLimit } from "@/lib/provider-key-spend";

import type { paths } from "@/lib/api/v1";

Expand Down Expand Up @@ -64,36 +65,11 @@ export function ProviderKeysTable({
<TableCell className="max-w-[240px] truncate text-xs text-muted-foreground">
{key.baseUrl ?? "—"}
</TableCell>
<TableCell className="text-sm tabular-nums">
<div>{formatUsd(key.usage)}</div>
{key.usageLimit !== null ? (
<div
className="text-[11px] text-muted-foreground"
title="The key is automatically deactivated once its attributed spend reaches this cap."
>
of {formatUsd(key.usageLimit)}
</div>
) : null}
<TableCell className="text-sm">
<ProviderKeySpendCell keyRow={key} />
</TableCell>
<TableCell>
{hasReachedSpendLimit(key) ? (
<Badge
variant="destructive"
title={`Automatically deactivated: spend reached the ${formatUsd(
key.usageLimit ?? "0",
)} cap. Raise or clear the limit to re-enable.`}
>
limit reached
</Badge>
) : (
<Badge
variant={
key.status === "active" ? "secondary" : "outline"
}
>
{key.status ?? "active"}
</Badge>
)}
<ProviderKeyStatusBadge keyRow={key} />
</TableCell>
<TableCell className="text-muted-foreground">
{formatDate(key.createdAt)}
Expand Down
11 changes: 4 additions & 7 deletions ee/admin/src/app/organizations/page.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,8 @@ import { requireSession } from "@/lib/require-session";
import { createServerApiClient } from "@/lib/server-api";
import { cn } from "@/lib/utils";

import { MIN_BULK_BLOCK_SEARCH_LENGTH } from "@llmgateway/shared";

type SortBy =
| "name"
| "billingEmail"
Expand All @@ -44,11 +46,6 @@ type SortBy =
| "totalSpent";
type SortOrder = "asc" | "desc";

// Mirrors MIN_BULK_BLOCK_SEARCH_LENGTH in apps/api/src/routes/admin.ts. The
// server rejects anything shorter; this only hides the entry point so the
// action never looks available for an empty or near-empty filter.
const BULK_BLOCK_MIN_SEARCH_LENGTH = 3;

function SortableHeader({
label,
sortKey,
Expand Down Expand Up @@ -252,15 +249,15 @@ export default async function OrganizationsPage({
</form>
</header>

{search.trim().length >= BULK_BLOCK_MIN_SEARCH_LENGTH && (
{search.trim().length >= MIN_BULK_BLOCK_SEARCH_LENGTH && (
<div className="flex flex-col items-start gap-2 rounded-lg border border-destructive/30 bg-destructive/5 px-4 py-3 sm:flex-row sm:items-center sm:justify-between">
<p className="text-sm text-muted-foreground">
Bulk actions apply to every organization matching the current
filter, not just this page.
</p>
<BulkBlockOrgsButton
search={search}
minSearchLength={BULK_BLOCK_MIN_SEARCH_LENGTH}
minSearchLength={MIN_BULK_BLOCK_SEARCH_LENGTH}
onPreview={handlePreviewBulkBlock}
onBulkBlock={handleBulkBlock}
/>
Expand Down
67 changes: 48 additions & 19 deletions ee/admin/src/components/bulk-block-orgs-button.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -53,7 +53,10 @@ export function BulkBlockOrgsButton({
const [preview, setPreview] = useState<BulkBlockPreview | null>(null);
const [confirmation, setConfirmation] = useState("");
const [blocking, setBlocking] = useState(false);
const [error, setError] = useState<string | null>(null);
// Preview and block failures are tracked separately: a failed block re-resolves
// the preview, and that reload must not clear the block error it accompanies.
const [previewError, setPreviewError] = useState<string | null>(null);
const [blockError, setBlockError] = useState<string | null>(null);
const [result, setResult] = useState<BulkBlockResult | null>(null);

const trimmedSearch = search.trim();
Expand All @@ -62,47 +65,68 @@ export function BulkBlockOrgsButton({
const resetState = () => {
setPreview(null);
setConfirmation("");
setError(null);
setPreviewError(null);
setBlockError(null);
setResult(null);
setPreviewLoading(false);
};

const loadPreview = async () => {
setPreview(null);
setConfirmation("");
setPreviewError(null);
setPreviewLoading(true);
try {
const response = await onPreview(trimmedSearch);
if (response.success && response.preview) {
setPreview(response.preview);
} else {
setPreviewError(response.error ?? "Failed to preview bulk block");
}
} catch (err) {
setPreviewError(
err instanceof Error ? err.message : "Failed to preview bulk block",
);
} finally {
setPreviewLoading(false);
}
};

const handleOpen = async () => {
resetState();
setOpen(true);
setPreviewLoading(true);
const response = await onPreview(trimmedSearch);
setPreviewLoading(false);
if (response.success && response.preview) {
setPreview(response.preview);
} else {
setError(response.error ?? "Failed to preview bulk block");
}
await loadPreview();
};

const handleConfirm = async () => {
if (!preview) {
return;
}
setBlocking(true);
setError(null);
setBlockError(null);
try {
const response = await onBulkBlock(preview.search, preview.blockable);
setResult(response);
if (response.success) {
setResult(response);
router.refresh();
} else {
setError(response.error ?? "Failed to bulk block organizations");
return;
}
setBlockError(response.error ?? "Failed to bulk block organizations");
} catch (err) {
setError(
setBlockError(
err instanceof Error
? err.message
: "Failed to bulk block organizations",
);
} finally {
setBlocking(false);
}
// Only reached when the block failed — the success path returns above. The
// dialog stays on the confirmation step instead of showing a summary, and
// re-resolving the set makes the admin confirm against current numbers
// rather than resubmitting a stale one. This also covers a thrown request:
// it may still have been applied server-side before the connection failed.
await loadPreview();
};

// The admin has to retype the exact number of organizations the server
Expand Down Expand Up @@ -251,10 +275,15 @@ export function BulkBlockOrgsButton({
</div>
)}

{error && (
<p className="text-sm text-destructive" role="alert">
{error}
</p>
{(blockError || previewError) && (
<div className="space-y-1" role="alert">
{blockError && (
<p className="text-sm text-destructive">{blockError}</p>
)}
{previewError && (
<p className="text-sm text-destructive">{previewError}</p>
)}
</div>
)}

<DialogFooter>
Expand Down
Loading
Loading