Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
31 changes: 31 additions & 0 deletions litellm/model_prices_and_context_window_backup.json
Original file line number Diff line number Diff line change
Expand Up @@ -22176,14 +22176,17 @@
},
"gpt-4.1-2025-04-14": {
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_priority": 8.75e-07,
"input_cost_per_token": 2e-06,
"input_cost_per_token_priority": 3.5e-06,
"input_cost_per_token_batches": 1e-06,
"litellm_provider": "openai",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 8e-06,
"output_cost_per_token_priority": 1.4e-05,
"output_cost_per_token_batches": 4e-06,
"supported_endpoints": [
"/v1/chat/completions",
Expand Down Expand Up @@ -22247,14 +22250,17 @@
},
"gpt-4.1-mini-2025-04-14": {
"cache_read_input_token_cost": 1e-07,
"cache_read_input_token_cost_priority": 1.75e-07,
"input_cost_per_token": 4e-07,
"input_cost_per_token_priority": 7e-07,
"input_cost_per_token_batches": 2e-07,
"litellm_provider": "openai",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 1.6e-06,
"output_cost_per_token_priority": 2.8e-06,
"output_cost_per_token_batches": 8e-07,
"supported_endpoints": [
"/v1/chat/completions",
Expand Down Expand Up @@ -22317,14 +22323,17 @@
},
"gpt-4.1-nano-2025-04-14": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_priority": 5e-08,
"input_cost_per_token": 1e-07,
"input_cost_per_token_priority": 2e-07,
"input_cost_per_token_batches": 5e-08,
"litellm_provider": "openai",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 4e-07,
"output_cost_per_token_priority": 8e-07,
"output_cost_per_token_batches": 2e-07,
"supported_endpoints": [
"/v1/chat/completions",
Expand Down Expand Up @@ -22393,14 +22402,17 @@
},
"gpt-4o-2024-08-06": {
"cache_read_input_token_cost": 1.25e-06,
"cache_read_input_token_cost_priority": 2.125e-06,
"input_cost_per_token": 2.5e-06,
"input_cost_per_token_priority": 4.25e-06,
"input_cost_per_token_batches": 1.25e-06,
"litellm_provider": "openai",
"max_input_tokens": 128000,
"max_output_tokens": 16384,
"max_tokens": 16384,
"mode": "chat",
"output_cost_per_token": 1e-05,
"output_cost_per_token_priority": 1.7e-05,
"output_cost_per_token_batches": 5e-06,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
Expand All @@ -22413,14 +22425,17 @@
},
"gpt-4o-2024-11-20": {
"cache_read_input_token_cost": 1.25e-06,
"cache_read_input_token_cost_priority": 2.125e-06,
"input_cost_per_token": 2.5e-06,
"input_cost_per_token_priority": 4.25e-06,
"input_cost_per_token_batches": 1.25e-06,
"litellm_provider": "openai",
"max_input_tokens": 128000,
"max_output_tokens": 16384,
"max_tokens": 16384,
"mode": "chat",
"output_cost_per_token": 1e-05,
"output_cost_per_token_priority": 1.7e-05,
"output_cost_per_token_batches": 5e-06,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
Expand Down Expand Up @@ -22720,14 +22735,17 @@
},
"gpt-4o-mini-2024-07-18": {
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_priority": 1.25e-07,
"input_cost_per_token": 1.5e-07,
"input_cost_per_token_priority": 2.5e-07,
"input_cost_per_token_batches": 7.5e-08,
"litellm_provider": "openai",
"max_input_tokens": 128000,
"max_output_tokens": 16384,
"max_tokens": 16384,
"mode": "chat",
"output_cost_per_token": 6e-07,
"output_cost_per_token_priority": 1e-06,
"output_cost_per_token_batches": 3e-07,
"search_context_cost_per_query": {
"search_context_size_high": 0.03,
Expand Down Expand Up @@ -25077,6 +25095,7 @@
"cache_read_input_token_cost": 5e-09,
"cache_read_input_token_cost_flex": 2.5e-09,
"input_cost_per_token": 5e-08,
"input_cost_per_token_priority": 2.5e-06,
"input_cost_per_token_flex": 2.5e-08,
"litellm_provider": "openai",
"max_input_tokens": 272000,
Expand Down Expand Up @@ -29304,13 +29323,19 @@
},
"o3-2025-04-16": {
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_flex": 2.5e-07,
"cache_read_input_token_cost_priority": 8.75e-07,
"input_cost_per_token": 2e-06,
"input_cost_per_token_flex": 1e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "openai",
"max_input_tokens": 200000,
"max_output_tokens": 100000,
"max_tokens": 100000,
"mode": "chat",
"output_cost_per_token": 8e-06,
"output_cost_per_token_flex": 4e-06,
"output_cost_per_token_priority": 1.4e-05,
"supported_endpoints": [
"/v1/responses",
"/v1/chat/completions",
Expand Down Expand Up @@ -29525,13 +29550,19 @@
},
"o4-mini-2025-04-16": {
"cache_read_input_token_cost": 2.75e-07,
"cache_read_input_token_cost_flex": 1.375e-07,
"cache_read_input_token_cost_priority": 5e-07,
"input_cost_per_token": 1.1e-06,
"input_cost_per_token_flex": 5.5e-07,
"input_cost_per_token_priority": 2e-06,
"litellm_provider": "openai",
"max_input_tokens": 200000,
"max_output_tokens": 100000,
"max_tokens": 100000,
"mode": "chat",
"output_cost_per_token": 4.4e-06,
"output_cost_per_token_flex": 2.2e-06,
"output_cost_per_token_priority": 8e-06,
"supports_function_calling": true,
"supports_parallel_function_calling": false,
"supports_pdf_input": true,
Expand Down
31 changes: 31 additions & 0 deletions model_prices_and_context_window.json
Original file line number Diff line number Diff line change
Expand Up @@ -22251,14 +22251,17 @@
},
"gpt-4.1-2025-04-14": {
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_priority": 8.75e-07,
"input_cost_per_token": 2e-06,
"input_cost_per_token_priority": 3.5e-06,

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Medium: Priority pricing bypasses budget reservation

These _priority rates are used during final cost calculation, while budget_reservation.py estimates requests exclusively from input_cost_per_token and output_cost_per_token. A user can send priority requests against these models that pass admission using the lower standard estimate but consume up to roughly twice the reserved key or team budget. Update the reservation estimator to select the service-tier-specific input, output, cache, and reasoning rates from request_body["service_tier"], with the same fallback behavior as final cost calculation.

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Reservation already ignored service_tier for the base aliases' existing priority keys; this sync doesn't widen that pre-existing gap: tier-aware reservation deserves its own PR

"input_cost_per_token_batches": 1e-06,
"litellm_provider": "openai",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 8e-06,
"output_cost_per_token_priority": 1.4e-05,
"output_cost_per_token_batches": 4e-06,
"supported_endpoints": [
"/v1/chat/completions",
Expand Down Expand Up @@ -22322,14 +22325,17 @@
},
"gpt-4.1-mini-2025-04-14": {
"cache_read_input_token_cost": 1e-07,
"cache_read_input_token_cost_priority": 1.75e-07,
"input_cost_per_token": 4e-07,
"input_cost_per_token_priority": 7e-07,
"input_cost_per_token_batches": 2e-07,
"litellm_provider": "openai",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 1.6e-06,
"output_cost_per_token_priority": 2.8e-06,
"output_cost_per_token_batches": 8e-07,
"supported_endpoints": [
"/v1/chat/completions",
Expand Down Expand Up @@ -22392,14 +22398,17 @@
},
"gpt-4.1-nano-2025-04-14": {
"cache_read_input_token_cost": 2.5e-08,
"cache_read_input_token_cost_priority": 5e-08,
"input_cost_per_token": 1e-07,
"input_cost_per_token_priority": 2e-07,
"input_cost_per_token_batches": 5e-08,
"litellm_provider": "openai",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 4e-07,
"output_cost_per_token_priority": 8e-07,
"output_cost_per_token_batches": 2e-07,
"supported_endpoints": [
"/v1/chat/completions",
Expand Down Expand Up @@ -22468,14 +22477,17 @@
},
"gpt-4o-2024-08-06": {
"cache_read_input_token_cost": 1.25e-06,
"cache_read_input_token_cost_priority": 2.125e-06,
"input_cost_per_token": 2.5e-06,
"input_cost_per_token_priority": 4.25e-06,
"input_cost_per_token_batches": 1.25e-06,
"litellm_provider": "openai",
"max_input_tokens": 128000,
"max_output_tokens": 16384,
"max_tokens": 16384,
"mode": "chat",
"output_cost_per_token": 1e-05,
"output_cost_per_token_priority": 1.7e-05,
"output_cost_per_token_batches": 5e-06,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
Expand All @@ -22488,14 +22500,17 @@
},
"gpt-4o-2024-11-20": {
"cache_read_input_token_cost": 1.25e-06,
"cache_read_input_token_cost_priority": 2.125e-06,
"input_cost_per_token": 2.5e-06,
"input_cost_per_token_priority": 4.25e-06,
"input_cost_per_token_batches": 1.25e-06,
"litellm_provider": "openai",
"max_input_tokens": 128000,
"max_output_tokens": 16384,
"max_tokens": 16384,
"mode": "chat",
"output_cost_per_token": 1e-05,
"output_cost_per_token_priority": 1.7e-05,
"output_cost_per_token_batches": 5e-06,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
Expand Down Expand Up @@ -22795,14 +22810,17 @@
},
"gpt-4o-mini-2024-07-18": {
"cache_read_input_token_cost": 7.5e-08,
"cache_read_input_token_cost_priority": 1.25e-07,
"input_cost_per_token": 1.5e-07,
"input_cost_per_token_priority": 2.5e-07,
"input_cost_per_token_batches": 7.5e-08,
"litellm_provider": "openai",
"max_input_tokens": 128000,
"max_output_tokens": 16384,
"max_tokens": 16384,
"mode": "chat",
"output_cost_per_token": 6e-07,
"output_cost_per_token_priority": 1e-06,
"output_cost_per_token_batches": 3e-07,
"search_context_cost_per_query": {
"search_context_size_high": 0.03,
Expand Down Expand Up @@ -25152,6 +25170,7 @@
"cache_read_input_token_cost": 5e-09,
"cache_read_input_token_cost_flex": 2.5e-09,
"input_cost_per_token": 5e-08,
"input_cost_per_token_priority": 2.5e-06,
"input_cost_per_token_flex": 2.5e-08,
"litellm_provider": "openai",
"max_input_tokens": 272000,
Expand Down Expand Up @@ -29379,13 +29398,19 @@
},
"o3-2025-04-16": {
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_flex": 2.5e-07,
"cache_read_input_token_cost_priority": 8.75e-07,
"input_cost_per_token": 2e-06,
"input_cost_per_token_flex": 1e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "openai",
"max_input_tokens": 200000,
"max_output_tokens": 100000,
"max_tokens": 100000,
"mode": "chat",
"output_cost_per_token": 8e-06,
"output_cost_per_token_flex": 4e-06,
"output_cost_per_token_priority": 1.4e-05,
"supported_endpoints": [
"/v1/responses",
"/v1/chat/completions",
Expand Down Expand Up @@ -29600,13 +29625,19 @@
},
"o4-mini-2025-04-16": {
"cache_read_input_token_cost": 2.75e-07,
"cache_read_input_token_cost_flex": 1.375e-07,
"cache_read_input_token_cost_priority": 5e-07,
"input_cost_per_token": 1.1e-06,
"input_cost_per_token_flex": 5.5e-07,
"input_cost_per_token_priority": 2e-06,
"litellm_provider": "openai",
"max_input_tokens": 200000,
"max_output_tokens": 100000,
"max_tokens": 100000,
"mode": "chat",
"output_cost_per_token": 4.4e-06,
"output_cost_per_token_flex": 2.2e-06,
"output_cost_per_token_priority": 8e-06,
"supports_function_calling": true,
"supports_parallel_function_calling": false,
"supports_pdf_input": true,
Expand Down
32 changes: 32 additions & 0 deletions tests/test_litellm/test_model_prices_schema.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@

import importlib.util
import json
import re
from pathlib import Path

import jsonschema
Expand Down Expand Up @@ -97,3 +98,34 @@ def test_schema_accepts_minimal_and_unknown_optional_fields(committed_schema: di
validator = build_validator(committed_schema)
assert validator.is_valid({"some-model": {"litellm_provider": "openai"}})
assert validator.is_valid({"some-model": {"litellm_provider": "openai", "brand_new_field": {"nested": True}}})


DATED_VARIANT = re.compile(r"^(.*?)-(\d{4}-\d{2}-\d{2})$")
SERVICE_TIER_SUFFIXES = ("_flex", "_priority")


def tier_anchor(tier_key: str) -> str:
matched = next(suffix for suffix in SERVICE_TIER_SUFFIXES if tier_key.endswith(suffix))
return tier_key[: -len(matched)]


def test_dated_variants_carry_base_alias_service_tier_pricing(prices: dict):
Comment thread
greptile-apps[bot] marked this conversation as resolved.
drifted = [
f"{name}: missing {tier_key}={base[tier_key]} (base alias {match.group(1)})"
for name, entry in prices.items()
if isinstance(entry, dict)
for match in [DATED_VARIANT.match(name)]
if match is not None
for base in [prices.get(match.group(1))]
if isinstance(base, dict)
for tier_key in base
if tier_key.endswith(SERVICE_TIER_SUFFIXES)
and tier_anchor(tier_key) in base
and entry.get(tier_anchor(tier_key)) == base[tier_anchor(tier_key)]
and entry.get(tier_key) != base[tier_key]
]
assert drifted == [], (
"dated model variants are missing flex/priority pricing their base alias has; "
"sync the tier keys so service-tier requests against pinned snapshots are not "
"billed at standard rates:\n" + "\n".join(drifted)
)
Loading