Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -98,9 +98,9 @@ def _sanitize_user_api_key_auth(auth: Any) -> Any:
return auth


def _classifier_call_metadata(metadata: dict[str, Any] | None) -> dict[str, Any] | None:
def _classifier_call_metadata(metadata: dict[str, Any] | None) -> dict[str, Any]:
if not metadata:
return metadata
return {}
return {
k: _sanitize_user_api_key_auth(v) if k == "user_api_key_auth" else v
for k, v in metadata.items()
Expand Down Expand Up @@ -763,8 +763,8 @@ async def _semantic_tier_override(self, user_message: str, request_kwargs: dict)
# embedding call. Forwarding it would let the embedding's cost callback finalize the
# reservation, so the routed completion's own callback then skips incrementing the
# key/team budget. Key/team attribution fields are preserved for spend logging.
metadata = _classifier_call_metadata(request_kwargs.get("metadata")) or {}
litellm_metadata = _classifier_call_metadata(request_kwargs.get("litellm_metadata")) or {}
metadata = _classifier_call_metadata(request_kwargs.get("metadata"))
litellm_metadata = _classifier_call_metadata(request_kwargs.get("litellm_metadata"))
query_vector = (
await encoder.aencode_queries([user_message], metadata=metadata, litellm_metadata=litellm_metadata)
)[0]
Expand Down
1 change: 1 addition & 0 deletions tests/e2e/coverage_registry/reliability.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@
- {id: reliability.routing.cost_based.picks_lowest_cost, module: reliability, tier: P1, behavior: routing, variant: cost_based, assertions: [picks_lowest_cost], exercised_on: [chat_completions, messages], source: "router_strategy/lowest_cost.py", rationale: "Spend-aware routing"}
- {id: reliability.routing.usage_based.picks_under_tpm, module: reliability, tier: P0, behavior: routing, variant: usage_based, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "router_strategy/lowest_tpm_rpm_v2.py", rationale: "Routes to lowest-TPM deployment; prevents over-allocation"}
- {id: reliability.routing.least_busy.picks_lowest_traffic, module: reliability, tier: P1, behavior: routing, variant: least_busy, assertions: [picks_lowest_traffic], exercised_on: [chat_completions, messages], source: "router_strategy/least_busy.py", rationale: "Fewest in-flight requests"}
- {id: reliability.routing.complexity_llm_classifier.routes_by_llm_tier, module: reliability, tier: P1, behavior: routing, variant: complexity_llm_classifier, assertions: [routes_by_llm_tier], exercised_on: [chat_completions], source: "router_strategy/complexity_router/complexity_router.py", fail_before_fix: proven, rationale: "v2 auto-router LLM complexity classifier runs over the proxy and routes by semantic tier instead of silently crashing on absent litellm_metadata and falling back to heuristic scoring"}
- {id: reliability.cache.exact.returns_cached, module: reliability, tier: P1, behavior: cache, variant: exact, assertions: [returns_cached], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/caching.py", rationale: "Response cache returns cached on exact match"}
- {id: reliability.cache.prompt_caching_model_select.returns_cached, module: reliability, tier: P1, behavior: cache, variant: prompt_caching_model_select, assertions: [returns_cached], exercised_on: [chat_completions], source: "router_utils/prompt_caching_cache.py", rationale: "Selects model supporting prompt caching for cacheable prefix"}
- {id: reliability.circuit_breaker.redis.trips_then_recovers, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/redis_cache.py:99", rationale: "Redis breaker CLOSED->OPEN->HALF_OPEN; guards all cache/rate-limit ops"}
Expand Down
17 changes: 17 additions & 0 deletions tests/e2e/docker-compose.yml
Original file line number Diff line number Diff line change
Expand Up @@ -66,6 +66,23 @@ configs:
model: openai/text-embedding-3-small
api_key: os.environ/OPENAI_API_KEY

# v2 auto-router with the LLM complexity classifier. SIMPLE stays on the
# openai backend; every higher tier routes to the anthropic backend, so the
# served deployment (read back from the spend log's model) reveals whether
# the LLM classifier actually ran or silently fell back to heuristic scoring.
- model_name: complexity-smart-router
litellm_params:
model: auto_router/complexity_router
complexity_router_config:
classifier_type: llm
classifier_llm_config:
model: gpt-5.5
tiers:
SIMPLE: gpt-5.5
MEDIUM: claude-haiku-4-5
COMPLEX: claude-haiku-4-5
REASONING: claude-haiku-4-5

services:
litellm:
image: ghcr.io/berriai/litellm:main-latest
Expand Down
20 changes: 20 additions & 0 deletions tests/e2e/router/complexity_router_client.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
"""Client for the complexity auto-router e2e tests.

The suite drives the shared /chat/completions and spend-log reads on the Gateway,
so this client only carries the Gateway the shared lifecycle needs for cleanup.
"""

from __future__ import annotations

from dataclasses import dataclass

from e2e_gateway import Gateway, build_gateway


@dataclass(frozen=True, slots=True)
class ComplexityRouterClient:
gateway: Gateway


def build_client() -> ComplexityRouterClient:
return ComplexityRouterClient(gateway=build_gateway())
15 changes: 15 additions & 0 deletions tests/e2e/router/conftest.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
"""Router suite's `client` fixture.

The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker
live in the parent tests/e2e/conftest.py. ComplexityRouterClient holds the shared
Gateway, so the `resources` fixture cleans up keys this suite creates.
"""

import pytest

from complexity_router_client import ComplexityRouterClient, build_client


@pytest.fixture(scope="session")
def client() -> ComplexityRouterClient:
return build_client()
62 changes: 62 additions & 0 deletions tests/e2e/router/test_complexity_router_e2e.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,62 @@
"""Live e2e: the v2 auto-router's LLM complexity classifier actually runs over the
proxy and drives routing, instead of silently crashing and falling back to the
local heuristic scorer.

The regression this guards (complexity_router.py `_classifier_call_metadata`
returning None when the request carries no `litellm_metadata`, which the classifier
sub-call then fed into a `.update`, raising `'NoneType' object has no attribute
'update'`) was invisible from the outside: the router caught the error and answered
from heuristic scoring, so every request still returned 200. The only tell is which
tier, and therefore which backend, served the request.

`complexity-smart-router` (see the inline config in docker-compose.yml) pins SIMPLE
to the openai backend and every higher tier to the anthropic backend. "Is P equal
to NP?" is lexically trivial, so the heuristic scorer lands it in SIMPLE (openai),
but any competent LLM classifier reads it as a hard reasoning question and lands it
above SIMPLE (anthropic). The served deployment is read back from the spend log's
`model`, so anthropic proves the classifier ran and openai proves it silently fell
back - the exact failure before the fix.
"""

import pytest

from complexity_router_client import ComplexityRouterClient
from e2e_http import unwrap
from models import ChatBody, ChatMessage

pytestmark = pytest.mark.e2e

ROUTER_MODEL = "complexity-smart-router"
# Lexically simple (heuristic -> SIMPLE) but a hard reasoning question (LLM -> above SIMPLE).
LEXICALLY_SIMPLE_HARD_PROMPT = "Is P equal to NP?"
# SIMPLE tier backend; served only when the classifier silently falls back to heuristic.
HEURISTIC_TIER_MODEL = "openai/gpt-5.5"
# MEDIUM/COMPLEX/REASONING tier backend; served only when the LLM classifier runs.
LLM_TIER_MODEL = "anthropic/claude-haiku-4-5"


class TestComplexityRouterLlmClassifier:
@pytest.mark.covers("reliability.routing.complexity_llm_classifier.routes_by_llm_tier")
def test_llm_classifier_runs_and_routes_by_semantic_tier(
self, client: ComplexityRouterClient, scoped_key: str
) -> None:
chat = unwrap(
client.gateway.chat(
scoped_key,
ChatBody(
model=ROUTER_MODEL,
messages=[ChatMessage(role="user", content=LEXICALLY_SIMPLE_HARD_PROMPT)],
max_tokens=16,
),
)
)
assert chat.choices, f"router returned no choices: {chat}"

rows = client.gateway.poll_logs_for_key(scoped_key, min_rows=1)
served = [row.model for row in rows]
assert served == [LLM_TIER_MODEL], (
f"expected the request to be served by {LLM_TIER_MODEL!r} (the higher-tier "
f"backend the LLM classifier picks for a hard prompt), but the spend log shows "
f"{served!r}. {HEURISTIC_TIER_MODEL!r} means the LLM classifier silently failed "
f"and the router fell back to heuristic scoring (SIMPLE) - the pre-fix regression"
)
10 changes: 10 additions & 0 deletions tests/test_litellm/router_strategy/test_complexity_router.py
Original file line number Diff line number Diff line change
Expand Up @@ -2387,6 +2387,16 @@ def test_cost_callback_cannot_recover_reservation_from_sanitized_metadata(self):
assert sanitized["user_api_key_auth"] is not None
assert _get_budget_reservation_from_metadata(sanitized) is None

def test_returns_empty_dict_for_missing_metadata(self):
from litellm.router_strategy.complexity_router.complexity_router import (
_classifier_call_metadata,
)

for absent in (None, {}):
result = _classifier_call_metadata(absent)
assert result == {}
assert isinstance(result, dict)

def test_sanitized_auth_keeps_access_group_fields_and_leaves_original_untouched(self):
from litellm.proxy._types import UserAPIKeyAuth
from litellm.router_strategy.complexity_router.complexity_router import (
Expand Down
Loading