Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
19 commits
Select commit Hold shift + click to select a range
f2479cc
Merge pull request #34200 from BerriAI/litellm_internal_staging
yuneng-berri Jul 22, 2026
7495b1f
Merge pull request #34450 from BerriAI/litellm_internal_staging
yuneng-berri Jul 24, 2026
0cd588a
Merge pull request #34519 from BerriAI/litellm_internal_staging
yuneng-berri Jul 24, 2026
9ead580
Merge pull request #34864 from BerriAI/litellm_internal_staging
yuneng-berri Jul 28, 2026
e1afe2e
test(e2e): bound the post-/model/new servable wait at 40s
mubashir1osmani Jul 28, 2026
5aa66ea
fix(e2e): wait one default DB reload interval of continuous listing
mubashir1osmani Jul 29, 2026
5953a66
test(e2e): drop proxy_client model-servable unit tests
mubashir1osmani Jul 29, 2026
38d03fd
fix(e2e): never skip the final deadline-clamped model-servable poll
mubashir1osmani Jul 29, 2026
87be33f
fix(e2e): reject first listing that returns after the 40s deadline
mubashir1osmani Jul 29, 2026
2cd62cf
Merge pull request #35020 from BerriAI/litellm_hotfix_e2e_model_serva…
yuneng-berri Jul 29, 2026
82fa669
test(e2e): poll MCP tools across multi-worker lag (#35047)
mubashir1osmani Jul 29, 2026
cad32fd
Merge pull request #35049 from BerriAI/litellm_hotfix_35047_mcp_e2e_poll
yuneng-berri Jul 29, 2026
122f935
Merge pull request #35285 from BerriAI/litellm_internal_staging
yuneng-berri Jul 30, 2026
de706a3
Merge pull request #35328 from BerriAI/litellm_internal_staging
mateo-berri Jul 31, 2026
a79f598
Merge pull request #35501 from BerriAI/litellm_internal_staging
yuneng-berri Aug 3, 2026
cfe8552
Merge pull request #35836 from BerriAI/litellm_internal_staging
yuneng-berri Aug 4, 2026
ead6252
Merge pull request #35876 from BerriAI/litellm_internal_staging
yuneng-berri Aug 5, 2026
714fff6
Merge pull request #36057 from BerriAI/litellm_internal_staging
yuneng-berri Aug 7, 2026
09323fc
chore(ci): sync main into internal staging
yuneng-berri Aug 8, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
161 changes: 139 additions & 22 deletions tests/e2e/proxy_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -76,12 +76,127 @@

RowsPredicate = Callable[[list[SpendLogRow]], bool]

# After /model/new, poll data-plane /v1/models until the model is listed (or fail).
# Bound by MODEL_SERVABLE_TIMEOUT so a stuck reload does not burn the spend
# poll_timeout (120s). Return on first listing; settle_propagation owns the separate
# wait that lets every worker and replica reload before the caller uses the model.
MODEL_SERVABLE_TIMEOUT = 40.0
MODEL_SERVABLE_DB_SYNC_SECONDS = 0.0
MODEL_SERVABLE_INTERVAL = 2.0
# Cap each /v1/models poll so one slow request cannot outlast the remaining budget.
MODEL_SERVABLE_REQUEST_TIMEOUT = 5.0


@dataclass(frozen=True, slots=True)
class Servable:
"""The data plane listed the model within the deadline."""


@dataclass(frozen=True, slots=True)
class NotServable:
"""The deadline passed without the data plane listing the model.

`last_result` is the final /v1/models read, so the caller can tell "the proxy
answered but omitted the model" (propagation) from "the read itself failed"
(network/auth) when reporting."""

last_result: Result[ModelsListResponse] | None


ServableOutcome = Servable | NotServable


def await_servable(
list_models: Callable[[float], Result[ModelsListResponse]],
*,
model_name: str,
timeout: float,
interval: float,
request_timeout: float,
db_sync_seconds: float,
now: Callable[[], float],
sleep: Callable[[float], None],
) -> ServableOutcome:
"""Poll until `model_name` is listed long enough for every worker to DB-sync.

First listing must happen within `timeout`. After that, the model must stay
listed continuously for `db_sync_seconds` (any miss resets the continuous
window). `db_sync_seconds=0` returns on the first listing. Each poll's request
timeout is clamped to the remaining budget. Sleeps only min(interval, time left)
so a final deadline-clamped poll is never skipped just because a full interval
does not fit. Clock and sleep are injected."""
started = now()
first_seen_at: float | None = None
last_result: Result[ModelsListResponse] | None = None
while True:
t = now()
phase_deadline = (
started + timeout if first_seen_at is None else first_seen_at + db_sync_seconds
)
remaining = phase_deadline - t
if remaining <= 0:
if (
last_result is not None
and first_seen_at is not None
and (db_sync_seconds <= 0 or t - first_seen_at >= db_sync_seconds)
):
return Servable()
return NotServable(last_result=last_result)

poll_timeout = min(request_timeout, remaining)
last_result = list_models(poll_timeout)
listed = isinstance(last_result, Success) and any(
entry.id == model_name for entry in last_result.data.data
)
t = now()
if not listed:
first_seen_at = None
elif first_seen_at is None:
if t > started + timeout:
return NotServable(last_result=last_result)
first_seen_at = t
if db_sync_seconds <= 0:
return Servable()
elif t - first_seen_at >= db_sync_seconds:
return Servable()

phase_deadline = (
started + timeout if first_seen_at is None else first_seen_at + db_sync_seconds
)
wait = min(interval, phase_deadline - now())
if wait > 0:
sleep(wait)


def servable_timeout_message(
*,
model_name: str,
timeout: float,
db_sync_seconds: float,
last_result: Result[ModelsListResponse] | None,
) -> str:
last_error = (
f"; last /v1/models poll did not succeed: {last_result}"
if last_result is not None and not isinstance(last_result, Success)
else ""
)
return (
f"model {model_name!r} was created but never became servable on the data "
f"plane within {timeout}s of first listing (plus {db_sync_seconds}s continuous "
f"DB sync) after /model/new (control/data-plane propagation or "
f"STORE_MODEL_IN_DB reload issue){last_error}"
)


@dataclass(frozen=True, slots=True)
class ProxyClient:
transport: Transport
poll_timeout: float = 120.0
poll_interval: float = 5.0
model_servable_timeout: float = MODEL_SERVABLE_TIMEOUT
model_servable_db_sync_seconds: float = MODEL_SERVABLE_DB_SYNC_SECONDS
model_servable_interval: float = MODEL_SERVABLE_INTERVAL
model_servable_request_timeout: float = MODEL_SERVABLE_REQUEST_TIMEOUT

# ---- keys / customers (satisfies lifecycle.ResourceClient) ----------

Expand Down Expand Up @@ -192,33 +307,35 @@ def create_model(
return model_id

def _await_model_servable(self, model_name: str) -> None:
"""Block until the data plane lists `model_name`, or fail loudly if it does
not within poll_timeout (a real propagation/config problem, surfaced here
instead of as a downstream "Invalid model name passed")."""
deadline = time.monotonic() + self.poll_timeout
last_result: Result[ModelsListResponse] | None = None
while time.monotonic() < deadline:
last_result = self.transport.get(
"""Block until the data plane lists `model_name`, or fail at model_servable_timeout."""
outcome = await_servable(
lambda poll_timeout: self.transport.get(
"/v1/models",
headers=self.transport.master,
params=NoBody(),
response_type=ModelsListResponse,
)
if isinstance(last_result, Success) and any(
entry.id == model_name for entry in last_result.data.data
):
return
time.sleep(self.poll_interval)
last_error = (
f"; last /v1/models poll did not succeed: {last_result}"
if last_result is not None and not isinstance(last_result, Success)
else ""
)
raise AssertionError(
f"model {model_name!r} was created but never became servable on the data "
f"plane within {self.poll_timeout}s of /model/new (control/data-plane "
f"propagation or STORE_MODEL_IN_DB reload issue){last_error}"
timeout=poll_timeout,
),
model_name=model_name,
timeout=self.model_servable_timeout,
interval=self.model_servable_interval,
request_timeout=self.model_servable_request_timeout,
db_sync_seconds=self.model_servable_db_sync_seconds,
now=time.monotonic,
sleep=time.sleep,
)
match outcome:
case Servable():
return
case NotServable(last_result=last_result):
raise AssertionError(
servable_timeout_message(
model_name=model_name,
timeout=self.model_servable_timeout,
db_sync_seconds=self.model_servable_db_sync_seconds,
last_result=last_result,
)
)

def update_model(self, model_id: str, litellm_params: LiteLLMParamsBody) -> None:
"""Merge `litellm_params` over the deployment `model_id`'s stored params via
Expand Down
13 changes: 11 additions & 2 deletions tests/e2e/transport.py
Original file line number Diff line number Diff line change
Expand Up @@ -58,6 +58,7 @@ def get[R: BaseModel](
headers: BaseModel,
params: BaseModel,
response_type: type[R],
timeout: float | None = None,
) -> Result[R]: ...

def delete[R: BaseModel](
Expand Down Expand Up @@ -136,13 +137,16 @@ def get[R: BaseModel](
headers: BaseModel,
params: BaseModel,
response_type: type[R],
timeout: float | None = None,
) -> Result[R]:
"""`timeout` overrides the transport-wide request_timeout for this call, for
pollers whose own deadline is shorter than it."""
return e2e_http.get(
self._url(path),
headers=headers,
params=params,
response_type=response_type,
timeout=self.request_timeout,
timeout=self.request_timeout if timeout is None else timeout,
)

def delete[R: BaseModel](
Expand Down Expand Up @@ -336,9 +340,14 @@ def get[R: BaseModel](
headers: BaseModel,
params: BaseModel,
response_type: type[R],
timeout: float | None = None,
) -> Result[R]:
return self._route(path).get(
path, headers=headers, params=params, response_type=response_type
path,
headers=headers,
params=params,
response_type=response_type,
timeout=timeout,
)

def delete[R: BaseModel](
Expand Down
Loading