diff --git a/apps/gateway/src/lib/org-rate-limit.spec.ts b/apps/gateway/src/lib/org-rate-limit.spec.ts index 4edbc5f617..b28004f0a7 100644 --- a/apps/gateway/src/lib/org-rate-limit.spec.ts +++ b/apps/gateway/src/lib/org-rate-limit.spec.ts @@ -319,16 +319,15 @@ describe("getBaseLimit", () => { expect(getBaseLimit(chatConfig, "regular")).toBe(chatConfig.defaultRpm); }); - it("uses the tighter per-plan defaults for dev and chat plans", () => { - expect(getBaseLimit(chatConfig, "dev")).toBe(chatConfig.devDefaultRpm); + it("doubles every regular endpoint default for DevPass", () => { + for (const config of PATH_RATE_LIMITS) { + expect(getBaseLimit(config, "dev")).toBe(config.defaultRpm * 2); + } + }); + + it("uses the tighter per-plan default for chat plans", () => { expect(getBaseLimit(chatConfig, "chat")).toBe(chatConfig.chatDefaultRpm); - expect(chatConfig.devDefaultRpm).toBeLessThan(chatConfig.defaultRpm); expect(chatConfig.chatDefaultRpm).toBeLessThan(chatConfig.defaultRpm); - // Dev (devpass) is relaxed relative to the chat plan: it's an anti-abuse - // backstop, not a product cap. - expect(chatConfig.devDefaultRpm).toBeGreaterThanOrEqual( - chatConfig.chatDefaultRpm, - ); }); it("honors a per-plan-class env override", () => { @@ -386,10 +385,9 @@ describe("checkOrgRateLimit", () => { expect(redis.zadd).not.toHaveBeenCalled(); }); - it("applies the tighter base limit for dev (devpass) plans", async () => { - // No env override: dev plan falls back to its default, which is below the - // regular default. - vi.mocked(redis.zcard).mockResolvedValue(chatConfig.devDefaultRpm); + it("applies the doubled base limit for DevPass plans", async () => { + const devLimit = getBaseLimit(chatConfig, "dev"); + vi.mocked(redis.zcard).mockResolvedValue(devLimit); const future = Date.now() + 30_000; vi.mocked(redis.zrange).mockResolvedValue(["m", future.toString()]); @@ -402,7 +400,7 @@ describe("checkOrgRateLimit", () => { ); expect(result.allowed).toBe(false); - expect(result.limit).toBe(chatConfig.devDefaultRpm); + expect(result.limit).toBe(devLimit); }); it("resolves the spend-tier multiplier only once the base limit is hit", async () => { diff --git a/apps/gateway/src/lib/org-rate-limit.ts b/apps/gateway/src/lib/org-rate-limit.ts index 8b03e5701a..c76c59e4a2 100644 --- a/apps/gateway/src/lib/org-rate-limit.ts +++ b/apps/gateway/src/lib/org-rate-limit.ts @@ -198,10 +198,11 @@ export interface RateLimitResult { * using a Redis sliding window. Mirrors the free-model limiter: on any Redis * error the request is allowed so that limiter outages never block traffic. * - * The base limit depends on the org's `planClass` (dev/chat plans are much - * tighter). `getMultiplier` is resolved lazily: the spend tier only matters - * once the org is already at or above its base limit, so the (cached but still - * extra) spend lookup is skipped entirely for the common under-limit case. + * The base limit depends on the org's `planClass` (DevPass gets twice the + * regular base; chat plans use a tighter base). `getMultiplier` is resolved + * lazily: the spend tier only matters once the org is already at or above its + * base limit, so the (cached but still extra) spend lookup is skipped entirely + * for the common under-limit case. * Higher tiers can only raise the limit, so a request under the base limit is * always allowed regardless of tier. */ diff --git a/apps/gateway/src/middleware/org-rate-limit.spec.ts b/apps/gateway/src/middleware/org-rate-limit.spec.ts index 73c1f7a7da..ac75e5b451 100644 --- a/apps/gateway/src/middleware/org-rate-limit.spec.ts +++ b/apps/gateway/src/middleware/org-rate-limit.spec.ts @@ -32,7 +32,6 @@ const chatConfig: PathRateLimitConfig = { key: "chat_completions", prefix: "/v1/chat/completions", defaultRpm: 600, - devDefaultRpm: 120, chatDefaultRpm: 60, }; diff --git a/apps/gateway/src/middleware/org-rate-limit.ts b/apps/gateway/src/middleware/org-rate-limit.ts index 01389834e0..ef1474830d 100644 --- a/apps/gateway/src/middleware/org-rate-limit.ts +++ b/apps/gateway/src/middleware/org-rate-limit.ts @@ -24,8 +24,8 @@ import type { Context, Next } from "hono"; * - Only `/v1/*` endpoints with a configured limit are throttled; other paths * (health, metrics, docs, mcp/oauth) pass through. * - Enterprise organizations are exempt. - * - Dev ("devpass") and chat plan orgs get their own, much tighter per-path - * limits and are not eligible for the spend-tier multiplier. + * - DevPass orgs get twice the regular per-path base limit; chat plan orgs use + * their own tighter limits. Neither gets the spend-tier multiplier. * - Regular (pay-as-you-go) org limits scale with their lifetime spend tier. * - Requests without a resolvable API token are passed through so the * downstream handler can return the appropriate auth error. @@ -76,10 +76,11 @@ export async function orgRateLimitMiddleware( const planClass = getPlanClass(organization); - // Only regular (pay-as-you-go) orgs get a spend-tier boost; dev/chat plans - // stay on their flat, tight limit. The multiplier is resolved lazily inside - // checkOrgRateLimit and only once the org has reached its base limit, so the - // common under-limit path skips the spend lookup entirely. + // Only regular (pay-as-you-go) orgs get a spend-tier boost; DevPass stays at + // twice the regular base and chat plans stay on their flat limit. The + // multiplier is resolved lazily inside checkOrgRateLimit and only once the org + // has reached its base limit, so the common under-limit path skips the spend + // lookup entirely. const result = await checkOrgRateLimit( organizationId, config, diff --git a/packages/shared/src/spend-tier.ts b/packages/shared/src/spend-tier.ts index 026b2b1f9b..b45a30ae23 100644 --- a/packages/shared/src/spend-tier.ts +++ b/packages/shared/src/spend-tier.ts @@ -12,6 +12,8 @@ const DAY_MS = 86_400_000; export type PlanClass = "regular" | "dev" | "chat"; +const DEVPASS_RATE_LIMIT_MULTIPLIER = 2; + export interface PathRateLimitConfig { /** Stable identifier used in the Redis key and env var names. */ key: string; @@ -19,8 +21,6 @@ export interface PathRateLimitConfig { prefix: string; /** Default requests per minute for regular (pay-as-you-go) orgs. */ defaultRpm: number; - /** Default requests per minute for dev ("devpass") plan orgs. */ - devDefaultRpm: number; /** Default requests per minute for chat plan orgs. */ chatDefaultRpm: number; } @@ -35,84 +35,72 @@ export const PATH_RATE_LIMITS: readonly PathRateLimitConfig[] = [ key: "chat_completions", prefix: "/v1/chat/completions", defaultRpm: 600, - devDefaultRpm: 120, chatDefaultRpm: 60, }, { key: "messages", prefix: "/v1/messages", defaultRpm: 600, - devDefaultRpm: 120, chatDefaultRpm: 60, }, { key: "responses", prefix: "/v1/responses", defaultRpm: 600, - devDefaultRpm: 120, chatDefaultRpm: 60, }, { key: "embeddings", prefix: "/v1/embeddings", defaultRpm: 1200, - devDefaultRpm: 120, chatDefaultRpm: 120, }, { key: "moderations", prefix: "/v1/moderations", defaultRpm: 1200, - devDefaultRpm: 120, chatDefaultRpm: 120, }, { key: "rerank", prefix: "/v1/rerank", defaultRpm: 1200, - devDefaultRpm: 120, chatDefaultRpm: 120, }, { key: "models", prefix: "/v1/models", defaultRpm: 1200, - devDefaultRpm: 120, chatDefaultRpm: 120, }, { key: "ocr", prefix: "/v1/ocr", defaultRpm: 300, - devDefaultRpm: 120, chatDefaultRpm: 30, }, { key: "images", prefix: "/v1/images", defaultRpm: 300, - devDefaultRpm: 120, chatDefaultRpm: 30, }, { key: "audio_speech", prefix: "/v1/audio/speech", defaultRpm: 300, - devDefaultRpm: 120, chatDefaultRpm: 30, }, { key: "audio_transcriptions", prefix: "/v1/audio/transcriptions", defaultRpm: 300, - devDefaultRpm: 120, chatDefaultRpm: 30, }, { key: "videos", prefix: "/v1/videos", defaultRpm: 120, - devDefaultRpm: 120, chatDefaultRpm: 12, }, // Realtime session-secret minting (`POST /v1/realtime/client_secrets`). @@ -123,21 +111,18 @@ export const PATH_RATE_LIMITS: readonly PathRateLimitConfig[] = [ key: "realtime", prefix: "/v1/realtime", defaultRpm: 120, - devDefaultRpm: 120, chatDefaultRpm: 12, }, { key: "key", prefix: "/v1/key", defaultRpm: 1200, - devDefaultRpm: 120, chatDefaultRpm: 120, }, { key: "credits", prefix: "/v1/credits", defaultRpm: 300, - devDefaultRpm: 120, chatDefaultRpm: 30, }, // AI SDK Gateway protocol surface. All four spec-version prefixes forward @@ -148,28 +133,24 @@ export const PATH_RATE_LIMITS: readonly PathRateLimitConfig[] = [ key: "ai_sdk", prefix: "/v1/ai", defaultRpm: 600, - devDefaultRpm: 120, chatDefaultRpm: 60, }, { key: "ai_sdk", prefix: "/v2/ai", defaultRpm: 600, - devDefaultRpm: 120, chatDefaultRpm: 60, }, { key: "ai_sdk", prefix: "/v3/ai", defaultRpm: 600, - devDefaultRpm: 120, chatDefaultRpm: 60, }, { key: "ai_sdk", prefix: "/v4/ai", defaultRpm: 600, - devDefaultRpm: 120, chatDefaultRpm: 60, }, ]; @@ -492,7 +473,7 @@ export function getBaseLimit( ): number { const fallback = planClass === "dev" - ? config.devDefaultRpm + ? config.defaultRpm * DEVPASS_RATE_LIMIT_MULTIPLIER : planClass === "chat" ? config.chatDefaultRpm : config.defaultRpm;