Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
27 changes: 27 additions & 0 deletions .env.example
Original file line number Diff line number Diff line change
Expand Up @@ -143,13 +143,40 @@ TIMEOUT_MS=5000
# GATEWAY_RATE_LIMIT_CHATPLAN_AUDIO_SPEECH_RPM=30
# GATEWAY_RATE_LIMIT_CHATPLAN_VIDEOS_RPM=12

# Per-organization in-flight concurrency limits: one fleet-wide budget per org
# across all inference endpoints (chat, messages, responses, embeddings,
# images, etc. — POST only). Unlike the RPM limits above, a slot is held for
# the response's full lifetime (including streaming) and released on close.
# Over-limit requests get a retryable 429. Enterprise orgs are NOT exempt —
# they get the elevated ENTERPRISE ceiling instead. Regular (PAYG) org
# ceilings come from the trust tier (GATEWAY_SPEND_TIER_<N>_INFLIGHT_LIMIT,
# defaults 100/200/400/1000/2000); dev/chat plans are flat. Set a value to 0
# to disable the check for that class. Defaults shown.
# GATEWAY_ORG_INFLIGHT_LIMIT_DEV=50
# GATEWAY_ORG_INFLIGHT_LIMIT_CHATPLAN=10
# GATEWAY_ORG_INFLIGHT_LIMIT_ENTERPRISE=2000
# Seconds after which an unreleased slot (e.g. from a crashed pod) is reaped.
# Must exceed the longest legitimate request duration.
# GATEWAY_ORG_INFLIGHT_STALE_SECONDS=1800

# Per-pod cap on concurrent in-flight inference requests (all orgs combined).
# Above it, requests are shed with a retryable 529 to protect the pod instead
# of piling up unbounded connections. Tune from the gateway_inflight_requests
# gauge.
# GATEWAY_MAX_INFLIGHT_REQUESTS=1000

# Accept backlog for the gateway's listen socket (clamped by the node's
# net.core.somaxconn).
# LISTEN_BACKLOG=1024

# Unified trust tiers for regular orgs. An org gets the highest tier whose age
# threshold is met, or whose lifetime-usage-spend threshold is met while the
# account is at least MIN_AGE_DAYS old (spend alone never promotes a brand-new
# account); the tier sets the per-path RPM multiplier, the daily/monthly USD
# spend caps, AND the rolling-24h top-up cap.
# Defaults shown; every value is overridable as GATEWAY_SPEND_TIER_<N>_<FIELD>
# for N in 0..4 and FIELD in AGE_DAYS, SPEND_USD, MIN_AGE_DAYS, RPM_MULTIPLIER,
# INFLIGHT_LIMIT,
# DAILY_CAP_USD, MONTHLY_CAP_USD, TOPUP_DAILY_CAP_USD.
# GATEWAY_SPEND_TIER_1_AGE_DAYS=7
# GATEWAY_SPEND_TIER_1_SPEND_USD=10
Expand Down
31 changes: 29 additions & 2 deletions apps/api/src/routes/admin-limit-hits.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -52,6 +52,13 @@ describe("admin limit hits", () => {
hitCount: 3,
blockedUsd: "450",
},
{
organizationId: "hits-org-a",
day: utcDay(0),
limitType: "concurrency",
endpointKey: "chat_completions",
hitCount: 7,
},
{
organizationId: "hits-org-b",
day: utcDay(0),
Expand Down Expand Up @@ -103,8 +110,9 @@ describe("admin limit hits", () => {
organizationId: "hits-org-a",
organizationName: "Hits Org A",
billingEmail: "a@limit-hits.test",
totalHits: 123,
totalHits: 130,
rpmHits: 120,
concurrencyHits: 7,
topUpHits: 3,
topUpBlockedUsd: 450,
daysActive: 2,
Expand All @@ -115,6 +123,7 @@ describe("admin limit hits", () => {
totalHits: 10,
spendCapHits: 10,
rpmHits: 0,
concurrencyHits: 0,
});
});

Expand All @@ -134,6 +143,19 @@ describe("admin limit hits", () => {
organizationId: "hits-org-a",
totalHits: 120,
});

const concurrencyRes = await app.request(
"/admin/limit-hits?limitType=concurrency",
{ headers: { Cookie: cookie } },
);
expect(concurrencyRes.status).toBe(200);
const concurrencyBody = await concurrencyRes.json();
expect(concurrencyBody.total).toBe(1);
expect(concurrencyBody.organizations[0]).toMatchObject({
organizationId: "hits-org-a",
totalHits: 7,
concurrencyHits: 7,
});
});

it("returns the per-organization daily breakdown", async () => {
Expand All @@ -144,14 +166,19 @@ describe("admin limit hits", () => {
expect(res.status).toBe(200);
const body = await res.json();

expect(body.hits).toHaveLength(2);
expect(body.hits).toHaveLength(3);
expect(body.hits[0]).toMatchObject({
limitType: "rpm",
endpointKey: "chat_completions",
hitCount: 120,
blockedUsd: 0,
});
expect(body.hits[1]).toMatchObject({
limitType: "concurrency",
endpointKey: "chat_completions",
hitCount: 7,
});
expect(body.hits[2]).toMatchObject({
limitType: "topup_velocity",
hitCount: 3,
blockedUsd: 450,
Expand Down
3 changes: 3 additions & 0 deletions apps/api/src/routes/admin-limit-hits.ts
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@ const limitTypeSchema = z.enum([
"spend_cap_daily",
"spend_cap_monthly",
"topup_velocity",
"concurrency",

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Badge Surface concurrency hits in the admin dashboard

Once concurrency rejections are flushed, this API accepts and returns the new type, but the summary has no concurrencyHits bucket while totalHits includes it, and the unchanged ee/admin/src/app/limit-hits/page.tsx filter and columns only cover RPM, spend caps, and top-ups. As a result, totals no longer reconcile and admins cannot select concurrency hits from the overview; add the corresponding summary field and update the admin type lists, labels, and column.

Useful? React with 👍 / 👎.

]);

const t = tables.orgLimitHitDaily;
Expand Down Expand Up @@ -54,6 +55,7 @@ const orgLimitHitsSummarySchema = z.object({
organizationCreatedAt: z.string(),
totalHits: z.number(),
rpmHits: z.number(),
concurrencyHits: z.number(),
spendCapHits: z.number(),
topUpHits: z.number(),
topUpBlockedUsd: z.number(),
Expand Down Expand Up @@ -114,6 +116,7 @@ adminLimitHits.openapi(listLimitHits, async (c) => {
organizationCreatedAt: tables.organization.createdAt,
totalHits: totalHitsExpr,
rpmHits: sql<number>`SUM(CASE WHEN ${t.limitType} = 'rpm' THEN ${t.hitCount} ELSE 0 END)::int`,
concurrencyHits: sql<number>`SUM(CASE WHEN ${t.limitType} = 'concurrency' THEN ${t.hitCount} ELSE 0 END)::int`,
spendCapHits: sql<number>`SUM(CASE WHEN ${t.limitType} IN ('spend_cap_daily', 'spend_cap_monthly') THEN ${t.hitCount} ELSE 0 END)::int`,
topUpHits: sql<number>`SUM(CASE WHEN ${t.limitType} = 'topup_velocity' THEN ${t.hitCount} ELSE 0 END)::int`,
topUpBlockedUsd: sql<string>`COALESCE(SUM(CASE WHEN ${t.limitType} = 'topup_velocity' THEN CAST(${t.blockedUsd} AS NUMERIC) ELSE 0 END), 0)`,
Expand Down
1 change: 1 addition & 0 deletions apps/docs/content/resources/error-handling.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,7 @@ The gateway maps HTTP status codes to OpenAI error types and codes as follows:
| 429 | `rate_limit_error` | `rate_limit_exceeded` |
| 499 | `invalid_request_error` | `request_cancelled` |
| 504 | `timeout_error` | `timeout` |
| 529 | `overloaded` | `overloaded` |
| 5xx | `api_error` | _(`null`)_ |

<Callout type="info">
Expand Down
105 changes: 94 additions & 11 deletions apps/docs/content/resources/rate-limits.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -47,15 +47,15 @@ The default limits (requests per minute, per organization) are:

### Trust Tiers (account age or spend)

For regular (pay-as-you-go) organizations, limits scale with a **trust tier**. An organization qualifies for a tier when its account is old enough, **or** when its lifetime usage spend is high enough **and** the account meets the tier's minimum age. The tier raises the per-endpoint RPM limits **and** the daily/monthly USD spend caps below.
For regular (pay-as-you-go) organizations, limits scale with a **trust tier**. An organization qualifies for a tier when its account is old enough, **or** when its lifetime usage spend is high enough **and** the account meets the tier's minimum age. The tier raises the per-endpoint RPM limits, the [concurrent-request ceiling](#concurrent-request-limits), **and** the daily/monthly USD spend caps below.

| Tier | Qualifies (age, or spend + min age) | RPM multiplier | Daily cap | Monthly cap |
| ---- | ----------------------------------- | -------------- | --------- | ----------- |
| 0 | new / $0 | 1× | $25 | $250 |
| 1 | 7 days **or** $10 (account ≥ 1 day) | 2× | $100 | $1,000 |
| 2 | 30 days **or** $100 (≥ 3 days) | 4× | $500 | $5,000 |
| 3 | 60 days **or** $1,000 (≥ 7 days) | 10× | $5,000 | $50,000 |
| 4 | 90 days **or** $5,000 (≥ 14 days) | 20× | $15,000 | $200,000 |
| Tier | Qualifies (age, or spend + min age) | RPM multiplier | Concurrent | Daily cap | Monthly cap |
| ---- | ----------------------------------- | -------------- | ---------- | --------- | ----------- |
| 0 | new / $0 | 1× | 100 | $25 | $250 |
| 1 | 7 days **or** $10 (account ≥ 1 day) | 2× | 200 | $100 | $1,000 |
| 2 | 30 days **or** $100 (≥ 3 days) | 4× | 400 | $500 | $5,000 |
| 3 | 60 days **or** $1,000 (≥ 7 days) | 10× | 1,000 | $5,000 | $50,000 |
| 4 | 90 days **or** $5,000 (≥ 14 days) | 20× | 2,000 | $15,000 | $200,000 |

Spend alone never promotes a brand-new account: each spend-qualified tier also requires the minimum account age shown, so the fastest possible path to Tier 4 is 14 days — no amount of day-one usage unlocks higher limits.

Expand Down Expand Up @@ -102,13 +102,51 @@ Dev plans are inference-only and only cover **chat completions, messages, respon

### Enterprise

Organizations on the **[Enterprise plan](https://llmgateway.io/enterprise) have no per-organization rate limits at all** — no requests-per-minute caps, no spend caps, and no top-up limits. Your throughput is limited only by your credit balance and any upstream provider limits.
Organizations on the **[Enterprise plan](https://llmgateway.io/enterprise) have no per-organization requests-per-minute, spend, or top-up limits**. Your request rate is limited only by your credit balance and any upstream provider limits. The only gateway limit that still applies is a greatly elevated [concurrent-request ceiling](#concurrent-request-limits).

<Callout type="info">
Need unlimited gateway throughput? [Contact us](mailto:contact@llmgateway.io)
about an enterprise plan.
</Callout>

## Concurrent Request Limits

Separately from the per-minute request limits above, each organization has one fleet-wide budget of **concurrent in-flight requests** across all inference endpoints (chat completions, messages, responses, embeddings, moderations, rerank, OCR, images, speech, transcriptions, videos, and the AI SDK surface). A slot is held for a request's full lifetime — including the entire duration of a streamed response — and freed when the response finishes or the connection closes.

This bounds what a per-minute limit cannot: long-running requests. Six hundred requests per minute that each stream for two minutes hold 1,200 connections open; the concurrency budget is what keeps that pile-up from exhausting shared gateway capacity.

For regular (pay-as-you-go) organizations the ceiling scales with the same [trust tier](#trust-tiers-account-age-or-spend) that raises the per-minute limits; Dev and Chat plans have a flat limit.

| Plan | Concurrent requests |
| ----------------------- | ------------------- |
| Regular (PAYG) — Tier 0 | 100 |
| Regular (PAYG) — Tier 1 | 200 |
| Regular (PAYG) — Tier 2 | 400 |
| Regular (PAYG) — Tier 3 | 1,000 |
| Regular (PAYG) — Tier 4 | 2,000 |
| Dev plan | 50 |
| Chat plan | 10 |
| Enterprise | 2,000 |

Unlike the per-minute limits, **Enterprise organizations are not exempt** — they get the elevated ceiling instead. Requests over the limit receive a retryable `429`:

```http
HTTP/1.1 429 Too Many Requests
Retry-After: 1
```

```json
{
"error": {
"message": "Too many concurrent requests for this organization (limit: 100). Retry shortly, or reduce request concurrency.",
"type": "rate_limit_error",
"code": "rate_limit_exceeded"
}
}
```

Because slots free up continuously as in-flight requests complete, retrying after a short backoff typically succeeds — there is no fixed window to wait out. If you consistently hit the concurrency limit, reduce your client-side parallelism or [contact us](mailto:contact@llmgateway.io) about raising your ceiling.

## Free Models

Free models (models with zero input and output pricing) have additional rate limits that depend on your account's credit status:
Expand Down Expand Up @@ -171,12 +209,57 @@ When you exceed a rate limit, you'll receive a `429 Too Many Requests` response:

This uses the standard OpenAI-compatible error envelope. Requests to the Anthropic-compatible `/v1/messages` endpoint receive the Anthropic error shape instead. See [Error Handling](/resources/error-handling) for the full format and status-code reference.

## Gateway Overload (529)

Separately from per-account rate limits, the gateway protects itself from
transient overload. When a single gateway instance is holding too many
concurrent in-flight inference requests at once — across all organizations
combined (for example during a traffic spike, or when an upstream provider is
slow and connections pile up) — it sheds excess inference requests with an
`HTTP 529` response instead of letting them queue indefinitely. Non-inference
endpoints such as the models list are unaffected:

```http
HTTP/1.1 529
Retry-After: 1
```

```json
{
"error": {
"message": "Gateway overloaded, please retry",
"type": "overloaded",
"code": "overloaded"
}
}
```

Requests to the Anthropic-compatible `/v1/messages` endpoint receive the
equivalent Anthropic envelope (`{ "type": "error", "error": { "type": "overloaded_error" } }`),
matching Anthropic's own `529` behavior.

<Callout type="info">
A `529` is **transient and retryable** — it reflects momentary capacity, not a
quota on your account. Unlike a `429`, it is not tied to your credits or model
tier, and retrying after a short delay (honoring the `Retry-After` header)
will typically succeed.
</Callout>

How `529` differs from `429`:

| | `429 Too Many Requests` | `529` Overloaded |
| --------- | ---------------------------------------------------------------------------------------------- | -------------------------------------- |
| Cause | Your organization exceeded its request rate or [concurrency](#concurrent-request-limits) limit | The gateway is momentarily at capacity |
| Scope | Per organization / API key | Transient, server-side |
| Fix | Slow down or reduce concurrency; add credits for [elevated limits](#elevated-rate-limits) | Retry after a short delay |
| Retryable | After the window resets (rate) or as soon as an in-flight request finishes (concurrency) | Yes, immediately with backoff |

## Best Practices

- **Respect `Retry-After`.** Implement exponential backoff when you receive `429` responses, starting from the `Retry-After` value.
- **Respect `Retry-After`.** Implement exponential backoff when you receive `429` or `529` responses, starting from the `Retry-After` value.
- **Watch the headers.** Monitor `X-RateLimit-Remaining` to back off before you hit the limit.
- **Spread traffic across endpoints.** Limits are per endpoint, so unrelated workloads don't compete for the same budget.
- **Scale with usage.** Regular organizations unlock higher limits automatically as lifetime spend grows; [contact us](mailto:contact@llmgateway.io) about an [Enterprise plan](https://llmgateway.io/enterprise) to remove per-organization limits entirely.
- **Scale with usage.** Regular organizations unlock higher limits automatically as lifetime spend grows; [contact us](mailto:contact@llmgateway.io) about an [Enterprise plan](https://llmgateway.io/enterprise) to remove the per-minute limits and get an elevated concurrency ceiling.

<Callout type="success">
Adding even a small amount of credits to your account (e.g., $10) will
Expand Down
47 changes: 18 additions & 29 deletions apps/gateway/src/app.ts
Original file line number Diff line number Diff line change
Expand Up @@ -30,10 +30,8 @@ import { isUpstreamTermination } from "./chat/tools/normalize-streaming-error.js
import { embeddingsRoute } from "./embeddings/route.js";
import { imagesRoute } from "./images/route.js";
import { keyRoute } from "./key/route.js";
import {
buildAnthropicErrorBody,
buildOpenAIErrorBody,
} from "./lib/error-response.js";
import { backpressureMiddleware } from "./lib/backpressure.js";
import { renderGatewayError } from "./lib/error-response.js";
import { mcpHandler, registerMcpOAuthRoutes } from "./mcp/mcp.js";
import { corsMiddleware } from "./middleware/cors.js";
import { orgRateLimitMiddleware } from "./middleware/org-rate-limit.js";
Expand All @@ -49,8 +47,6 @@ import { transcriptionsRoute } from "./transcriptions/route.js";
import { videosRoute } from "./videos/route.js";

import type { ServerTypes } from "./vars.js";
import type { Context } from "hono";
import type { ContentfulStatusCode } from "hono/utils/http-status";

export const config = {
servers: [
Expand Down Expand Up @@ -95,13 +91,22 @@ app.use("*", requestLifecycleMiddleware);
app.use("*", honoRequestLogger);
app.use("*", corsMiddleware);

// Per-organization, per-path rate limiting. Registered before the other
// request gates (content-type validation) and ahead of every downstream
// DB check and rate limiter in the route handlers (credit checks, free-model
// and provider rate limits), so an over-limit org is rejected as early as
// possible. Enterprise orgs are exempt and limits scale with the
// organization's lifetime spend tier. Only configured `/v1/*` paths are
// throttled; everything else passes through.
// Shed excess inference load early so each pod fast-fails with a retryable
// 529 instead of piling up unbounded connections. Only inference endpoints
// are counted — everything else completes near-instantly and keeps working
// under overload. Registered after CORS so shed responses still carry the
// Access-Control-* headers browser clients need to surface the 529, and
// before the org limiter so pod protection costs no Redis/DB lookups.
app.use("*", backpressureMiddleware);

// Per-organization, per-path rate limiting plus the per-org in-flight
// concurrency cap. Registered before the other request gates (content-type
// validation) and ahead of every downstream DB check and rate limiter in the
// route handlers (credit checks, free-model and provider rate limits), so an
// over-limit org is rejected as early as possible. Enterprise orgs skip the
// RPM limits but get an elevated concurrency ceiling; regular org RPM limits
// scale with the organization's lifetime spend tier. Only configured `/v1/*`
// paths are throttled; everything else passes through.
app.use("*", orgRateLimitMiddleware);

// Middleware to check for application/json content type on POST requests
Expand All @@ -128,22 +133,6 @@ app.use("*", async (c, next) => {
return await next();
});

// Renders a gateway-level error in a provider-compatible shape. The Anthropic
// `/v1/messages` endpoint expects Anthropic's `{ type: "error", error: {...} }`
// envelope; every other (OpenAI-compatible) endpoint expects OpenAI's
// `{ error: { message, type, param, code } }` envelope.
function renderGatewayError(
c: Context<ServerTypes>,
status: number,
message: string,
) {
const jsonStatus = status as ContentfulStatusCode;
if (c.req.path.startsWith("/v1/messages")) {
return c.json(buildAnthropicErrorBody({ message, status }), jsonStatus);
}
return c.json(buildOpenAIErrorBody({ message, status }), jsonStatus);
}

app.onError((error, c) => {
if (error instanceof UnsupportedAudioFormatError) {
logger.warn("Unsupported audio format", {
Expand Down
Loading
Loading