Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
295 changes: 4 additions & 291 deletions apps/gateway/src/chat/chat.ts
Original file line number Diff line number Diff line change
Expand Up @@ -47,7 +47,10 @@ import {
type WebSearchTool,
} from "@llmgateway/models";

import { completionsRequestSchema } from "./schemas/completions.js";
import { convertImagesToBase64 } from "./tools/convert-images-to-base64.js";
import { createLogEntry } from "./tools/create-log-entry.js";
import { estimateTokensFromContent } from "./tools/estimate-tokens-from-content.js";
import { estimateTokens } from "./tools/estimate-tokens.js";
import { extractContent } from "./tools/extract-content.js";
import { extractCustomHeaders } from "./tools/extract-custom-headers.js";
Expand All @@ -57,80 +60,16 @@ import { extractToolCalls } from "./tools/extract-tool-calls.js";
import { getFinishReasonFromError } from "./tools/get-finish-reason-from-error.js";
import { getProviderEnv } from "./tools/get-provider-env.js";
import { healJsonResponse } from "./tools/heal-json-response.js";
import { isModelTrulyFree } from "./tools/is-model-truly-free.js";
import { convertAwsEventStreamToSSE } from "./tools/parse-aws-eventstream.js";
import { parseProviderResponse } from "./tools/parse-provider-response.js";
import { transformResponseToOpenai } from "./tools/transform-response-to-openai.js";
import { transformStreamingToOpenai } from "./tools/transform-streaming-to-openai.js";
import { type ChatMessage, DEFAULT_TOKENIZER_MODEL } from "./tools/types.js";
import { validateFreeModelUsage } from "./tools/validate-free-model-usage.js";

import type { ImageObject } from "./tools/types.js";
import type { ServerTypes } from "@/vars.js";

/**
* Checks if a model is truly free (has free flag AND no per-request pricing)
*/
function isModelTrulyFree(modelInfo: ModelDefinition): boolean {
if (!modelInfo.free) {
return false;
}
// Check if any provider has a per-request cost
return !modelInfo.providers.some((p) => p.requestPrice && p.requestPrice > 0);
}

/**
* Estimates tokens from content length using simple division
*/
export function estimateTokensFromContent(content: string): number {
return Math.max(1, Math.round(content.length / 4));
}

/**
* Converts external image URLs to base64 data URLs
* Used for providers like Alibaba that return external URLs instead of base64
*/
async function convertImagesToBase64(
images: ImageObject[],
): Promise<ImageObject[]> {
return await Promise.all(
images.map(async (image): Promise<ImageObject> => {
const url = image.image_url.url;
// Skip if already a data URL
if (url.startsWith("data:")) {
return image;
}

try {
const response = await fetch(url);
if (!response.ok) {
logger.warn("Failed to fetch image for base64 conversion", {
url,
status: response.status,
});
return image;
}

const contentType = response.headers.get("content-type") || "image/png";
const arrayBuffer = await response.arrayBuffer();
const base64 = Buffer.from(arrayBuffer).toString("base64");

return {
type: "image_url",
image_url: {
url: `data:${contentType};base64,${base64}`,
},
};
} catch (error) {
logger.warn("Error converting image to base64", {
url,
error: error instanceof Error ? error.message : String(error),
});
return image;
}
}),
);
}

/**
* Checks if any messages contain images (image_url or image type content)
* Used to filter providers that don't support vision
Expand All @@ -155,232 +94,6 @@ function messagesContainImages(messages: BaseMessage[]): boolean {

export const chat = new OpenAPIHono<ServerTypes>();

const completionsRequestSchema = z.object({
model: z.string().openapi({
example: "gpt-5",
}),
messages: z.array(
z.object({
role: z.string().openapi({
example: "user",
}),
content: z.union([
z.string().openapi({
example: "Hello!",
}),
z.array(
z.union([
z.object({
type: z.literal("text"),
text: z.string(),
}),
z.object({
type: z.literal("image_url"),
image_url: z.object({
url: z.string(),
detail: z.enum(["low", "high", "auto"]).optional(),
}),
}),
]),
),
]),
name: z.string().optional(),
tool_call_id: z.string().optional(),
tool_calls: z
.array(
z.object({
id: z.string(),
type: z.literal("function"),
function: z.object({
name: z.string(),
arguments: z.string(),
}),
}),
)
.optional()
.openapi({
description:
"A list of tool calls generated by the model in this message.",
example: [
{
id: "call_abc123",
type: "function",
function: {
name: "get_current_weather",
arguments: '{"location": "Boston, MA"}',
},
},
],
}),
}),
),
temperature: z
.number()
.nullable()
.optional()
.transform((val) => (val === null ? undefined : val))
.openapi({
example: 0.7,
}),
max_tokens: z
.number()
.nullable()
.optional()
.transform((val) => (val === null ? undefined : val))
.openapi({
example: 1000,
}),
top_p: z
.number()
.nullable()
.optional()
.transform((val) => (val === null ? undefined : val))
.openapi({
example: 0.9,
}),
frequency_penalty: z
.number()
.nullable()
.optional()
.transform((val) => (val === null ? undefined : val))
.openapi({
example: 0.0,
}),
presence_penalty: z
.number()
.nullable()
.optional()
.transform((val) => (val === null ? undefined : val))
.openapi({
example: 0.0,
}),
response_format: z
.union([
z.object({
type: z.enum(["text", "json_object"]).openapi({
example: "json_object",
}),
}),
z.object({
type: z.literal("json_schema"),
json_schema: z.object({
name: z.string(),
description: z.string().optional(),
schema: z.record(z.any()),
strict: z.boolean().optional(),
}),
}),
])
.optional(),
stream: z.boolean().optional().default(false),
tools: z
.array(
z.union([
z.object({
type: z.literal("function"),
function: z.object({
name: z.string(),
description: z.string().optional(),
parameters: z.record(z.any()).optional(),
}),
}),
z.object({
type: z.literal("web_search"),
user_location: z
.object({
city: z.string().optional(),
region: z.string().optional(),
country: z.string().optional(),
timezone: z.string().optional(),
})
.optional(),
search_context_size: z.enum(["low", "medium", "high"]).optional(),
max_uses: z.number().optional(),
}),
]),
)
.optional(),
tool_choice: z
.union([
z.literal("auto"),
z.literal("none"),
z.literal("required"),
z.object({
type: z.literal("function"),
function: z.object({
name: z.string(),
}),
}),
])
.optional(),
reasoning_effort: z
.enum(["minimal", "low", "medium", "high"])
.nullable()
.optional()
.transform((val) => (val === null ? undefined : val))
.openapi({
description: "Controls the reasoning effort for reasoning-capable models",
example: "medium",
}),
effort: z
.enum(["low", "medium", "high"])
.nullable()
.optional()
.transform((val) => (val === null ? undefined : val))
.openapi({
description:
"Controls the computational effort for supported models (currently only claude-opus-4-5-20251101)",
example: "medium",
}),
free_models_only: z.boolean().optional().default(false).openapi({
description:
"When used with auto routing, only route to free models (models with zero input and output pricing)",
example: false,
}),
no_reasoning: z.boolean().optional().default(false).openapi({
description:
"When used with auto routing, exclude reasoning models from selection",
example: false,
}),
// Z.ai specific parameter - not documented in OpenAPI
sensitive_word_check: z
.object({
status: z.enum(["DISABLE", "ENABLE"]),
})
.optional(),
// Image generation config (Google and Alibaba)
image_config: z
.object({
aspect_ratio: z.string().optional(),
image_size: z.string().optional(),
n: z.number().optional(),
seed: z.number().optional(),
})
.optional(),
// Web search enablement
web_search: z.boolean().optional().default(false).openapi({
description:
"Enable native web search for models that support it. When enabled, the model can search the web for real-time information.",
example: true,
}),
// Plugins configuration
plugins: z
.array(
z.object({
id: z.enum(["response-healing"]).openapi({
description: "Plugin identifier",
example: "response-healing",
}),
}),
)
.optional()
.openapi({
description:
"Plugins to enable for this request. Currently supported: response-healing (automatically repairs malformed JSON responses when using response_format)",
example: [{ id: "response-healing" }],
}),
});

const completions = createRoute({
operationId: "v1_chat_completions",
summary: "Chat Completions",
Expand Down
Loading
Loading