Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
14 commits
Select commit Hold shift + click to select a range
c8a2d8c
feat(proxy): add LiteLLM_DailyGlobalSpend key-free rollup for the usa…
yassin-berriai Sep 15, 2026
ad8de0e
fix(proxy): roll up only closed days into LiteLLM_DailyGlobalSpend an…
yassin-berriai Sep 15, 2026
84c098d
fix(proxy): fold late-arriving per-key spend into already rolled-up g…
yassin-berriai Sep 16, 2026
225fc53
Merge branch 'litellm_usage_key_free_aggregate_split' into litellm_da…
yassin-berriai Sep 16, 2026
0601d2b
feat(proxy): carry response time metrics through LiteLLM_DailyGlobalS…
yassin-berriai Sep 16, 2026
7c5fa53
Merge remote-tracking branch 'origin/litellm_usage_key_free_aggregate…
yassin-berriai Sep 16, 2026
834313a
test(proxy): assert the global rollup split and scheduler through beh…
yassin-berriai Sep 16, 2026
f25d659
fix(proxy): take the closed-day cutoff for the global spend rollup fr…
yassin-berriai Sep 16, 2026
3b2b9bf
Merge branch 'litellm_usage_key_free_aggregate_split' into litellm_da…
yassin-berriai Sep 17, 2026
f631301
Merge branch 'litellm_usage_key_free_aggregate_split' into litellm_da…
yassin-berriai Sep 18, 2026
b92820d
Merge branch 'litellm_usage_key_free_aggregate_split' into litellm_da…
yassin-berriai Sep 18, 2026
6a6ae2d
Merge remote-tracking branch 'origin/main' into litellm_daily_global_…
yassin-berriai Sep 18, 2026
abf530f
fix(proxy): never rewind the daily global spend marker from an overla…
yassin-berriai Sep 18, 2026
3449ae9
fix(proxy): advance the daily global spend marker in one conditional …
yassin-berriai Sep 18, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,35 @@
-- CreateTable
CREATE TABLE IF NOT EXISTS "LiteLLM_DailyGlobalSpend" (
"id" TEXT NOT NULL,
"date" TEXT NOT NULL,
"model" TEXT,
"model_group" TEXT,
"custom_llm_provider" TEXT,
"mcp_namespaced_tool_name" TEXT,
"endpoint" TEXT,
"prompt_tokens" BIGINT NOT NULL DEFAULT 0,
"completion_tokens" BIGINT NOT NULL DEFAULT 0,
"cache_read_input_tokens" BIGINT NOT NULL DEFAULT 0,
"cache_creation_input_tokens" BIGINT NOT NULL DEFAULT 0,
"compression_saved_tokens" BIGINT NOT NULL DEFAULT 0,
"compression_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0,
"prompt_caching_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0,
"gateway_injected_caching_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0,
"autorouter_savings_spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0,
"spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0,
"api_requests" BIGINT NOT NULL DEFAULT 0,
"successful_requests" BIGINT NOT NULL DEFAULT 0,
"failed_requests" BIGINT NOT NULL DEFAULT 0,
"total_response_time_ms" BIGINT NOT NULL DEFAULT 0,
"timed_requests" BIGINT NOT NULL DEFAULT 0,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"updated_at" TIMESTAMP(3) NOT NULL,

CONSTRAINT "LiteLLM_DailyGlobalSpend_pkey" PRIMARY KEY ("id")
);

-- CreateIndex
CREATE INDEX IF NOT EXISTS "LiteLLM_DailyGlobalSpend_date_idx" ON "LiteLLM_DailyGlobalSpend"("date");

-- CreateIndex
CREATE UNIQUE INDEX IF NOT EXISTS "LiteLLM_DailyGlobalSpend_date_model_model_group_custom_llm__key" ON "LiteLLM_DailyGlobalSpend"("date", "model", "model_group", "custom_llm_provider", "mcp_namespaced_tool_name", "endpoint");
31 changes: 31 additions & 0 deletions litellm-proxy-extras/litellm_proxy_extras/schema.prisma
Original file line number Diff line number Diff line change
Expand Up @@ -820,6 +820,37 @@ model LiteLLM_DailyUserSpend {
@@index([endpoint])
}

// Key-free daily rollup of LiteLLM_DailyUserSpend, read by the global usage view
model LiteLLM_DailyGlobalSpend {
id String @id @default(uuid())
date String
model String?
model_group String?
custom_llm_provider String?
mcp_namespaced_tool_name String?
endpoint String?
prompt_tokens BigInt @default(0)
completion_tokens BigInt @default(0)
cache_read_input_tokens BigInt @default(0)
cache_creation_input_tokens BigInt @default(0)
compression_saved_tokens BigInt @default(0)
compression_savings_spend Float @default(0.0)
prompt_caching_savings_spend Float @default(0.0)
gateway_injected_caching_savings_spend Float @default(0.0)
autorouter_savings_spend Float @default(0.0)
spend Float @default(0.0)
api_requests BigInt @default(0)
successful_requests BigInt @default(0)
failed_requests BigInt @default(0)
total_response_time_ms BigInt @default(0)
timed_requests BigInt @default(0)
created_at DateTime @default(now())
updated_at DateTime @updatedAt

@@unique([date, model, model_group, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
}

// Track daily organization spend metrics per model and key
model LiteLLM_DailyOrganizationSpend {
id String @id @default(uuid())
Expand Down
3 changes: 3 additions & 0 deletions litellm/constants.py
Original file line number Diff line number Diff line change
Expand Up @@ -2061,6 +2061,9 @@
# Deployments named in the lapsed-window alert before it is truncated, so a fleet-wide
# expiry cannot produce an alert too large for the channel delivering it.
PTU_LAPSED_ALERT_LIMIT: Final[int] = 10
DAILY_GLOBAL_SPEND_RECONCILE_JOB_ID: Final[str] = "daily_global_spend_reconcile_job"
DAILY_GLOBAL_SPEND_RECONCILE_LOCK_TTL_SECONDS: Final[int] = 3600
DAILY_GLOBAL_SPEND_RECONCILED_THROUGH_PARAM: Final[str] = "daily_global_spend_reconciled_through"
# Slack allowed when deciding a sentinel row is stale. The row's updated_at and the
# run's cutoff are stamped by different hosts, so clock skew between them must not let
# one run delete a charge another just wrote. A stale row is hours old and a concurrent
Expand Down
74 changes: 70 additions & 4 deletions litellm/proxy/management_endpoints/common_daily_activity.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
from litellm._logging import verbose_proxy_logger
from litellm.constants import PTU_SENTINEL_API_KEY, USAGE_TOP_API_KEYS_LIMIT
from litellm.proxy._types import CommonProxyErrors
from litellm.proxy.spend_tracking.daily_global_spend_rollup import GLOBAL_SPEND_TABLE_NAME, reconciled_through
from litellm.proxy.spend_tracking.key_metadata_recovery import (
attach_user_emails,
recover_double_hashed_key_metadata,
Expand Down Expand Up @@ -750,6 +751,66 @@ def _rollup_metric_select(table_name: str) -> str:
_MODEL_GROUP_EXPR: Final = "COALESCE(NULLIF(model_group, ''), model)"


_KEY_FREE_SOURCE_COLUMNS: Final = (
"date",
"model",
"model_group",
"custom_llm_provider",
"mcp_namespaced_tool_name",
"endpoint",
"spend",
"prompt_tokens",
"completion_tokens",
"cache_read_input_tokens",
"cache_creation_input_tokens",
"compression_saved_tokens",
"compression_savings_spend",
"prompt_caching_savings_spend",
"gateway_injected_caching_savings_spend",
"autorouter_savings_spend",
"api_requests",
"successful_requests",
"failed_requests",
"total_response_time_ms",
"timed_requests",
)


async def global_rollup_reconciled_through(prisma_client: PrismaClient, query: _AggregatedQueryKwargs) -> str | None:
"""The last day ``LiteLLM_DailyGlobalSpend`` can answer the key-free arm for, or None to
read it all from the per-key table.

Only an unfiltered read of the user table sums to the same rows as the global table. The
marker read is served from the config cache, so this is not a database round trip per request.
"""
if query["table_name"] != "litellm_dailyuserspend":
return None
if query["entity_id"] is not None or query["api_key"] is not None or query["exclude_entity_ids"]:
return None
Comment thread
greptile-apps[bot] marked this conversation as resolved.
try:
return await reconciled_through(prisma_client)
except Exception as exc: # noqa: BLE001 # the per-key table is always a correct answer, so never fail the read
verbose_proxy_logger.warning("Could not read the daily global spend marker, using the per-key table: %s", exc)
return None


def _key_free_source(pg_table: str, where_clause: str, marker_param: str | None) -> str:
"""The relation the key-free arm aggregates: the per-key table alone, or the global rollup
for days through the marker plus the per-key table for the days still open after it."""
if marker_param is None:
return f'"{pg_table}"\n WHERE {where_clause}'
columns: Final = ", ".join(_KEY_FREE_SOURCE_COLUMNS)
return f"""(
SELECT {columns}
FROM "{GLOBAL_SPEND_TABLE_NAME}"
WHERE {where_clause} AND date <= {marker_param}
UNION ALL
SELECT {columns}
FROM "{pg_table}"
WHERE {where_clause} AND date > {marker_param}
) AS key_free_source"""


def _build_aggregated_sql_query(
*,
table_name: str,
Expand All @@ -762,6 +823,7 @@ def _build_aggregated_sql_query(
exclude_entity_ids: list[str] | None = None, # mutable-ok: filter union shared with the paginated path
timezone_offset_minutes: int | None = None,
include_current_utc_day: bool = False,
global_rollup_through: str | None = None,
) -> tuple[str, list[str]]: # mutable-ok: SQL text plus its ordered $N params
"""Build the GROUPING SETS query for aggregated daily activity.

Expand All @@ -786,6 +848,7 @@ def _build_aggregated_sql_query(
exclude_entity_ids=exclude_entity_ids,
)
sentinel_param: Final = f"${len(where_params) + 1}"
marker_param: Final = None if global_rollup_through is None else f"${len(where_params) + 2}"
metric_select: Final = _rollup_metric_select(table_name)

# TODO: drop the successful_requests/failed_requests aggregates (and the
Expand All @@ -806,8 +869,7 @@ def _build_aggregated_sql_query(
custom_llm_provider, mcp_namespaced_tool_name,
endpoint) AS group_level,
NULL::bigint AS distinct_api_keys,{metric_select}
FROM "{pg_table}"
WHERE {where_clause}
FROM {_key_free_source(pg_table, where_clause, marker_param)}
GROUP BY GROUPING SETS (
(date),
(date, model),
Expand Down Expand Up @@ -850,7 +912,8 @@ def _build_aggregated_sql_query(
))
"""

return sql_query, [*where_params, PTU_SENTINEL_API_KEY]
marker_params: Final = () if global_rollup_through is None else (global_rollup_through,)
return sql_query, [*where_params, PTU_SENTINEL_API_KEY, *marker_params]


def _build_entity_rollup_sql_query(
Expand Down Expand Up @@ -1395,7 +1458,10 @@ async def get_daily_activity_aggregated(
timezone_offset_minutes=timezone_offset_minutes,
include_current_utc_day=include_current_utc_day,
)
sql_query, sql_params = _build_aggregated_sql_query(**query_kwargs)
sql_query, sql_params = _build_aggregated_sql_query(
**query_kwargs,
global_rollup_through=await global_rollup_reconciled_through(prisma_client, query_kwargs),
)
entity_query: Final = _build_entity_rollup_sql_query(**query_kwargs) if include_entity_breakdown else None

raw_rows, raw_entity_rows = await asyncio.gather(
Expand Down
43 changes: 43 additions & 0 deletions litellm/proxy/proxy_server.py
Original file line number Diff line number Diff line change
Expand Up @@ -262,6 +262,7 @@ def generate_feedback_box():
APSCHEDULER_MISFIRE_GRACE_TIME,
APSCHEDULER_REPLACE_EXISTING,
CLI_SSO_SESSION_TTL_SECONDS,
DAILY_GLOBAL_SPEND_RECONCILE_JOB_ID,
DAYS_IN_A_MONTH,
DEFAULT_HEALTH_CHECK_INTERVAL,
DEFAULT_MODEL_CREATED_AT_TIME,
Expand Down Expand Up @@ -687,6 +688,9 @@ def generate_feedback_box():
from litellm.proxy.search_endpoints.endpoints import router as search_router
from litellm.proxy.shutdown.graceful_shutdown_manager import GracefulShutdownManager
from litellm.proxy.spend_tracking.budget_reservation import get_budget_window_start
from litellm.proxy.spend_tracking.daily_global_spend_rollup import (
run_scheduled_daily_global_spend_reconcile,
)
from litellm.proxy.spend_tracking.spend_counter_batch import (
PendingSpendIncrement,
active_spend_counter_batch,
Expand Down Expand Up @@ -10016,6 +10020,12 @@ async def initialize_scheduled_background_jobs(

await cls._initialize_spend_tracking_background_jobs(scheduler=scheduler)

cls._initialize_daily_global_spend_reconcile_job(
scheduler=scheduler,
proxy_logging_obj=proxy_logging_obj,
prisma_client=prisma_client,
)

### PTU DAILY ROLLUP ###
from litellm.proxy.spend_tracking.ptu_feature_flag import (
is_ptu_cost_attribution_enabled,
Expand Down Expand Up @@ -10357,6 +10367,39 @@ async def _initialize_expired_ui_session_key_cleanup_background_job(cls, schedul
"LITELLM_EXPIRED_UI_SESSION_KEY_CLEANUP_ENABLED=true to enable)"
)

@classmethod
def _initialize_daily_global_spend_reconcile_job(
cls,
scheduler: AsyncIOScheduler,
proxy_logging_obj: ProxyLogging,
prisma_client: PrismaClient,
) -> None:
async def alert(message: str) -> None:
await proxy_logging_obj.alerting_handler(
message=message,
level="High",
alert_type=AlertType.failed_tracking_spend,
)

async def reconcile() -> None:
await run_scheduled_daily_global_spend_reconcile(
prisma_client,
pod_lock_manager=proxy_logging_obj.db_spend_update_writer.pod_lock_manager,
alert=alert,
)

scheduler.add_job(
reconcile,
"cron",
hour=0,
minute=30,
timezone="UTC",
id=DAILY_GLOBAL_SPEND_RECONCILE_JOB_ID,
replace_existing=True,
misfire_grace_time=APSCHEDULER_MISFIRE_GRACE_TIME,
next_run_time=datetime.now(timezone.utc) + timedelta(minutes=2),
)

@classmethod
async def _initialize_slack_alerting_jobs(
cls,
Expand Down
31 changes: 31 additions & 0 deletions litellm/proxy/schema.prisma
Original file line number Diff line number Diff line change
Expand Up @@ -820,6 +820,37 @@ model LiteLLM_DailyUserSpend {
@@index([endpoint])
}

// Key-free daily rollup of LiteLLM_DailyUserSpend, read by the global usage view
model LiteLLM_DailyGlobalSpend {
id String @id @default(uuid())
date String
model String?
model_group String?
custom_llm_provider String?
mcp_namespaced_tool_name String?
endpoint String?
prompt_tokens BigInt @default(0)
completion_tokens BigInt @default(0)
cache_read_input_tokens BigInt @default(0)
cache_creation_input_tokens BigInt @default(0)
compression_saved_tokens BigInt @default(0)
compression_savings_spend Float @default(0.0)
prompt_caching_savings_spend Float @default(0.0)
gateway_injected_caching_savings_spend Float @default(0.0)
autorouter_savings_spend Float @default(0.0)
spend Float @default(0.0)
api_requests BigInt @default(0)
successful_requests BigInt @default(0)
failed_requests BigInt @default(0)
total_response_time_ms BigInt @default(0)
timed_requests BigInt @default(0)
created_at DateTime @default(now())
updated_at DateTime @updatedAt

@@unique([date, model, model_group, custom_llm_provider, mcp_namespaced_tool_name, endpoint])
@@index([date])
}

// Track daily organization spend metrics per model and key
model LiteLLM_DailyOrganizationSpend {
id String @id @default(uuid())
Expand Down
Loading
Loading