Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 21 additions & 0 deletions apps/gateway/src/lib/timeout-config.ts
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,20 @@ export function getTimeoutMs(): number {
return 180000;
}

/**
* Gets the upstream video creation timeout.
* Video job creation should fail fast enough to return a JSON error before
* callers hit their own route/runtime timeout limits.
* Default: 45 seconds or 5 seconds below the gateway timeout, whichever is lower.
*/
export function getVideoCreateTimeoutMs(): number {
const envValue = Number(process.env.AI_VIDEO_CREATE_TIMEOUT_MS);
if (envValue > 0) {
return envValue;
}
return Math.min(45000, Math.max(1000, getGatewayTimeoutMs() - 5000));
}

// Legacy exports for backwards compatibility (read at module load time)
// These should be avoided in new code - use the getter functions instead
export const GATEWAY_TIMEOUT_MS = getGatewayTimeoutMs();
Expand All @@ -66,6 +80,13 @@ export function createTimeoutSignal(): AbortSignal {
return AbortSignal.timeout(getTimeoutMs());
}

/**
* Creates an AbortSignal that will abort after the video creation timeout.
*/
export function createVideoCreateTimeoutSignal(): AbortSignal {
return AbortSignal.timeout(getVideoCreateTimeoutMs());
}

/**
* Combines a streaming timeout signal with an optional cancellation signal.
* If the cancellation signal is provided, the request will abort on either timeout or cancellation.
Expand Down
35 changes: 35 additions & 0 deletions apps/gateway/src/test-utils/mock-openai-server.ts
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,26 @@ function extractTimeoutDelay(content: string): number | null {
return null;
}

function extractProviderSpecificTimeoutDelay(
content: string,
provider: "obsidian" | "avalanche" | "vertex",
): number | null {
const token =
provider === "vertex"
? "TRIGGER_VERTEX_ONLY_TIMEOUT"
: provider === "avalanche"
? "TRIGGER_AVALANCHE_ONLY_TIMEOUT"
: "TRIGGER_OBSIDIAN_ONLY_TIMEOUT";
const match = content.match(new RegExp(`${token}_(\\d+)`));
if (match) {
return parseInt(match[1], 10);
}
if (content.includes(token)) {
return 5000;
}
return null;
}

// Helper to extract a specific HTTP status code from message content
// e.g., "TRIGGER_STATUS_429" -> { statusCode: 429, errorResponse: {...} }
function extractStatusCodeTrigger(
Expand Down Expand Up @@ -511,6 +531,10 @@ mockOpenAIServer.post("/v1/videos", async (c) => {
? await c.req.parseBody({ all: true })
: await c.req.json();
const prompt = typeof body.prompt === "string" ? body.prompt : "";
const timeoutDelay = extractProviderSpecificTimeoutDelay(prompt, "obsidian");
if (timeoutDelay) {

Copilot AI Mar 23, 2026

Copy link

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

extractProviderSpecificTimeoutDelay can legitimately return 0 (e.g., ..._0), but the current if (timeoutDelay) check treats 0 as falsy and skips the delay. Use an explicit null check (e.g., timeoutDelay !== null) so the behavior matches the trigger value exactly.

Suggested change
if (timeoutDelay) {
if (timeoutDelay !== null) {

Copilot uses AI. Check for mistakes.
await delay(timeoutDelay);
}
const statusTrigger = extractStatusCodeTrigger(prompt);
if (statusTrigger) {
c.status(statusTrigger.statusCode as any);
Expand Down Expand Up @@ -560,6 +584,10 @@ mockOpenAIServer.post("/v1/videos", async (c) => {
mockOpenAIServer.post("/api/v1/veo/generate", async (c) => {
const body = await c.req.json();
const prompt = typeof body.prompt === "string" ? body.prompt : "";
const timeoutDelay = extractProviderSpecificTimeoutDelay(prompt, "avalanche");
if (timeoutDelay) {

Copilot AI Mar 23, 2026

Copy link

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

extractProviderSpecificTimeoutDelay can return 0, but if (timeoutDelay) treats it as falsy and won’t apply the delay. Prefer checking timeoutDelay !== null (or typeof timeoutDelay === 'number') before calling delay(...).

Suggested change
if (timeoutDelay) {
if (typeof timeoutDelay === "number") {

Copilot uses AI. Check for mistakes.
await delay(timeoutDelay);
}
if (
prompt.includes("TRIGGER_STATUS_500_AVALANCHE_ONLY") ||
prompt.includes("TRIGGER_AVALANCHE_ONLY_500")
Expand Down Expand Up @@ -690,6 +718,13 @@ mockOpenAIServer.post(
"string"
? ((body.instances[0] as Record<string, unknown>).prompt as string)
: "";
const timeoutDelay = extractProviderSpecificTimeoutDelay(
prompt,
"vertex",
);
if (timeoutDelay) {

Copilot AI Mar 23, 2026

Copy link

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

extractProviderSpecificTimeoutDelay returns number | null; using if (timeoutDelay) skips the delay for 0. Switch to an explicit null check so ..._0 behaves as intended and to avoid relying on truthiness.

Suggested change
if (timeoutDelay) {
if (timeoutDelay !== null) {

Copilot uses AI. Check for mistakes.
await delay(timeoutDelay);
}
if (
prompt.includes("TRIGGER_STATUS_500_VERTEX_ONLY") ||
prompt.includes("TRIGGER_VERTEX_ONLY_500")
Expand Down
159 changes: 159 additions & 0 deletions apps/gateway/src/videos/videos.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -510,6 +510,165 @@ describe("videos", () => {
}
});

test("/v1/videos returns 504 when upstream create times out before job creation", async () => {
const originalVideoCreateTimeout = process.env.AI_VIDEO_CREATE_TIMEOUT_MS;
process.env.AI_VIDEO_CREATE_TIMEOUT_MS = "50";

try {
await db.insert(tables.apiKey).values({
id: "token-id",
token: "real-token",
projectId: "project-id",
description: "Test API Key",
createdBy: "user-id",
});

await db.insert(tables.providerKey).values({
id: "provider-key-id",
token: "sk-test-key",
provider: "obsidian",
organizationId: "org-id",
baseUrl: mockServerUrl,
});

const res = await app.request("/v1/videos", {
method: "POST",
headers: {
"Content-Type": "application/json",
Authorization: "Bearer real-token",
},
body: JSON.stringify({
model: "obsidian/veo-3.1-generate-preview",
prompt:
"TRIGGER_OBSIDIAN_ONLY_TIMEOUT_200 A robot dancing in the rain",
seconds: 8,
}),
});

expect(res.status).toBe(504);
const json = await res.json();
expect(JSON.stringify(json)).toContain(
"Video creation timed out while waiting for the upstream provider",
);

const videoJobs = await db.query.videoJob.findMany();
expect(videoJobs).toHaveLength(0);
} finally {
if (originalVideoCreateTimeout !== undefined) {
process.env.AI_VIDEO_CREATE_TIMEOUT_MS = originalVideoCreateTimeout;
} else {
delete process.env.AI_VIDEO_CREATE_TIMEOUT_MS;
}
}
});

test("/v1/videos falls back after an upstream create timeout", async () => {
const originalGoogleCloudProject = process.env.LLM_GOOGLE_CLOUD_PROJECT;
const originalGoogleVertexRegion = process.env.LLM_GOOGLE_VERTEX_REGION;
const originalVideoCreateTimeout = process.env.AI_VIDEO_CREATE_TIMEOUT_MS;
process.env.LLM_GOOGLE_CLOUD_PROJECT = "test-project";
process.env.LLM_GOOGLE_VERTEX_REGION = "us-central1";
process.env.AI_VIDEO_CREATE_TIMEOUT_MS = "50";

try {
await db.insert(tables.apiKey).values({
id: "token-id",
token: "real-token",
projectId: "project-id",
description: "Test API Key",
createdBy: "user-id",
});

await db.insert(tables.providerKey).values([
{
id: "provider-key-avalanche",
token: "sk-avalanche-key",
provider: "avalanche",
organizationId: "org-id",
baseUrl: mockServerUrl,
},
{
id: "provider-key-vertex",
token: "vertex-test-token",
provider: "google-vertex",
organizationId: "org-id",
baseUrl: mockServerUrl,
},
]);

await setRoutingMetrics("veo-3.1-generate-preview", "avalanche", {
uptime: 70,
latency: 300,
throughput: 50,
});
await setRoutingMetrics("veo-3.1-generate-preview", "google-vertex", {
uptime: 99.9,
latency: 80,
throughput: 180,
});

const res = await app.request("/v1/videos", {
method: "POST",
headers: {
"Content-Type": "application/json",
Authorization: "Bearer real-token",
},
body: JSON.stringify({
model: "veo-3.1-generate-preview",
prompt:
"TRIGGER_VERTEX_ONLY_TIMEOUT_200 A cinematic city skyline at dusk",
size: "1920x1080",
seconds: 8,
}),
});

expect(res.status).toBe(200);

const json = await res.json();
const videoJob = await db.query.videoJob.findFirst({
where: { id: { eq: json.id } },
});
expect(videoJob?.usedProvider).toBe("avalanche");
expect(
videoJob?.routingMetadata?.routing?.map((attempt) => ({
provider: attempt.provider,
succeeded: attempt.succeeded,
status_code: attempt.status_code,
error_type: attempt.error_type,
})),
).toEqual([
{
provider: "google-vertex",
succeeded: false,
status_code: 504,
error_type: "upstream_error",
},
{
provider: "avalanche",
succeeded: true,
status_code: 200,
error_type: "none",
},
]);
} finally {
if (originalGoogleCloudProject !== undefined) {
process.env.LLM_GOOGLE_CLOUD_PROJECT = originalGoogleCloudProject;
} else {
delete process.env.LLM_GOOGLE_CLOUD_PROJECT;
}
if (originalGoogleVertexRegion !== undefined) {
process.env.LLM_GOOGLE_VERTEX_REGION = originalGoogleVertexRegion;
} else {
delete process.env.LLM_GOOGLE_VERTEX_REGION;
}
if (originalVideoCreateTimeout !== undefined) {
process.env.AI_VIDEO_CREATE_TIMEOUT_MS = originalVideoCreateTimeout;
} else {
delete process.env.AI_VIDEO_CREATE_TIMEOUT_MS;
}
}
});

test("/v1/videos supports retrieve and content for completed jobs", async () => {
await db.insert(tables.apiKey).values({
id: "token-id",
Expand Down
30 changes: 29 additions & 1 deletion apps/gateway/src/videos/videos.ts
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,11 @@ import {
findProviderKey,
} from "@/lib/cached-queries.js";
import { validateModelAccess } from "@/lib/iam.js";
import {
createVideoCreateTimeoutSignal,
getVideoCreateTimeoutMs,
isTimeoutError,
} from "@/lib/timeout-config.js";

import {
getCheapestFromAvailableProviders,
Expand Down Expand Up @@ -1958,7 +1963,30 @@ async function fetchUpstreamJson(
url: string,
init: RequestInit,
): Promise<Record<string, unknown>> {
const response = await fetch(url, init);
const timeoutSignal = createVideoCreateTimeoutSignal();
const signal = init.signal
? AbortSignal.any([init.signal, timeoutSignal])
: timeoutSignal;
let response: Response;
try {
response = await fetch(url, {
...init,
signal,
});
} catch (error) {
if (isTimeoutError(error)) {
logger.warn("Upstream video create request timed out", {
url,
timeoutMs: getVideoCreateTimeoutMs(),
});
throw new HTTPException(504, {
Comment on lines +1978 to +1982

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P1 Badge Avoid falling back after a timed-out create POST

The new 504 here is treated as retryable by the outer loop in apps/gateway/src/videos/videos.ts:2811-2844, so a timeout on provider A now immediately sends the same create request to provider B. That is unsafe for these side-effecting POSTs: createObsidianVideoJob and createAvalancheVideoJob build plain create payloads with no idempotency key or cancellation handle, so a slow provider can still enqueue the first job after our client gives up. In that case one user request can create two upstream videos (and two billable jobs), while we only persist the fallback attempt.

Useful? React with 👍 / 👎.

message:
"Video creation timed out while waiting for the upstream provider. Please try again.",
});
}

throw error;
}
const text = await response.text();
let body: Record<string, unknown> = {};

Expand Down
49 changes: 39 additions & 10 deletions apps/playground/src/app/api/video/route.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,19 @@

export const maxDuration = 60;

function getVideoCreateProxyTimeoutMs(): number {
const envValue = Number(process.env.PLAYGROUND_VIDEO_CREATE_TIMEOUT_MS);
if (envValue > 0) {
return envValue;
}

return Math.max(1000, maxDuration * 1000 - 5000);

Check failure on line 16 in apps/playground/src/app/api/video/route.ts

View workflow job for this annotation

GitHub Actions / autofix

Unexpected mix of '*' and '-'. Use parentheses to clarify the intended order of operations

Check failure on line 16 in apps/playground/src/app/api/video/route.ts

View workflow job for this annotation

GitHub Actions / autofix

Unexpected mix of '*' and '-'. Use parentheses to clarify the intended order of operations

Check failure on line 16 in apps/playground/src/app/api/video/route.ts

View workflow job for this annotation

GitHub Actions / lint / run

Unexpected mix of '*' and '-'. Use parentheses to clarify the intended order of operations

Check failure on line 16 in apps/playground/src/app/api/video/route.ts

View workflow job for this annotation

GitHub Actions / lint / run

Unexpected mix of '*' and '-'. Use parentheses to clarify the intended order of operations
}

function isTimeoutError(error: unknown): boolean {
return error instanceof Error && error.name === "TimeoutError";
}

export async function POST(req: Request) {
const user = await getUser();
if (!user) {
Expand All @@ -31,16 +44,32 @@
const requestBody = await req.json();
const noFallback = req.headers.get("x-no-fallback");

const response = await fetch(`${gatewayBaseUrl}/v1/videos`, {
method: "POST",
headers: {
"Content-Type": "application/json",
Authorization: `Bearer ${apiKey}`,
"x-source": "chat.llmgateway.io",
...(noFallback ? { "x-no-fallback": noFallback } : {}),
},
body: JSON.stringify(requestBody),
});
let response: Response;
try {
response = await fetch(`${gatewayBaseUrl}/v1/videos`, {
method: "POST",
headers: {
"Content-Type": "application/json",
Authorization: `Bearer ${apiKey}`,
"x-source": "chat.llmgateway.io",
...(noFallback ? { "x-no-fallback": noFallback } : {}),
},
body: JSON.stringify(requestBody),
signal: AbortSignal.timeout(getVideoCreateProxyTimeoutMs()),
Comment on lines +57 to +58

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Badge Propagate the proxy timeout to the gateway request

This new timeout makes the playground return a 504 after ~55s, but the gateway video-create path never threads the incoming request abort signal into its upstream work (I checked apps/gateway/src/videos/videos.ts and the create helpers only use their own timeout signal). For requests that spend a long time preprocessing/uploading images or timing out on one provider before falling back, the UI can now report failure while /v1/videos keeps running and eventually inserts a videoJob; if the user retries after that JSON error, they can duplicate the generation and billing.

Useful? React with 👍 / 👎.

});
} catch (error) {
if (isTimeoutError(error)) {
return NextResponse.json(
{
error:
"Video creation timed out before the gateway responded. Please try again.",
},
{ status: 504 },
);
}

throw error;
}

const responseBody = await readGatewayResponseBody(response);

Expand Down
Loading