Skip to content
Merged
22 changes: 2 additions & 20 deletions litellm/llms/azure/chat/gpt_5_transformation.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
from litellm.llms.openai.chat.gpt_5_transformation import (
OpenAIGPT5Config,
_get_effort_level,
is_gpt_reasoning_series_name,
)
from litellm.types.llms.openai import AllMessageValues

Expand Down Expand Up @@ -35,26 +36,7 @@ def _supports_reasoning_effort_level(cls, model: str, level: str) -> bool:

@classmethod
def is_model_gpt_5_model(cls, model: str) -> bool:
"""Check if the Azure model string refers to a gpt-5 variant.

Accepts both explicit gpt-5 model names and the ``gpt5_series/`` prefix
used for manual routing.
"""
# The gpt-5-chat* family (gpt-5-chat, gpt-5-chat-latest, gpt-5-chat-2025-08-07,
# …) are regular chat models: they support temperature and tool_choice but NOT
# reasoning_effort. They must NOT be routed through the GPT-5 reasoning path.
#
# Versioned chat models such as gpt-5.3-chat and gpt-5.1-chat ARE reasoning
# models and must stay on the GPT-5 path. The distinguishing feature is that
# the gpt-5-chat family has a literal "-chat" immediately after "gpt-5"
# (i.e. "gpt-5-chat…"), while versioned chat models interpose a minor version
# number (i.e. "gpt-5.<digit>-chat").
#
# Using a startswith("gpt-5-chat") prefix check on the normalized name (rather
# than a substring check) makes this boundary explicit and avoids any ambiguity
# if future model names coincidentally contain "gpt-5-chat" as an interior run.
_normalized: Final = model.split("/")[-1] # strip provider prefix, e.g. "azure/"
return ("gpt-5" in model and not _normalized.startswith("gpt-5-chat")) or "gpt5_series" in model
return is_gpt_reasoning_series_name(model) or "gpt5_series" in model

def get_supported_openai_params(self, model: str) -> list[str]:
"""Get supported parameters for Azure OpenAI GPT-5 models.
Expand Down
26 changes: 11 additions & 15 deletions litellm/llms/openai/chat/gpt_5_transformation.py
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,14 @@ def _get_effort_level(value: str | dict | None) -> str | None:
return None


GPT_REASONING_SERIES_MARKERS: Final = ("gpt-5", "gpt-6")


def is_gpt_reasoning_series_name(model: str) -> bool:
normalized: Final = model.split("/")[-1]
return any(marker in model for marker in GPT_REASONING_SERIES_MARKERS) and not normalized.startswith("gpt-5-chat")
Comment on lines +47 to +52

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 GPT-6 capabilities are hardcoded The new marker check and GPT-6 prefix check classify model behavior in code. Repository rules require model-specific flags in model metadata, read through get_model_info. This requirement must be met before merging

Rule Used: What: Do not hardcode model-specific flags in the codebase. Instead, put them in model_prices_and_context_window.json and then read them in via get_model_info Why: Prevents need for users to upgrade litellm each time a new model supports this featu... (source)

Knowledge Base Used: Provider adapters and capabilities



class OpenAIGPT5Config(OpenAIGPTConfig):
"""Configuration for gpt-5 models including GPT-5-Codex variants.

Expand All @@ -56,21 +64,7 @@ class OpenAIGPT5Config(OpenAIGPTConfig):

@classmethod
def is_model_gpt_5_model(cls, model: str) -> bool:
# The gpt-5-chat* family (gpt-5-chat, gpt-5-chat-latest, gpt-5-chat-2025-08-07,
# …) are regular chat models: they support temperature and tool_choice but NOT
# reasoning_effort. They must NOT be routed through the GPT-5 reasoning path.
#
# Versioned chat models such as gpt-5.3-chat and gpt-5.1-chat ARE reasoning
# models and must stay on the GPT-5 path. The distinguishing feature is that
# the gpt-5-chat family has a literal "-chat" immediately after "gpt-5"
# (i.e. "gpt-5-chat…"), while versioned chat models interpose a minor version
# number (i.e. "gpt-5.<digit>-chat").
#
# Using a startswith("gpt-5-chat") prefix check on the normalized name (rather
# than a substring check) makes this boundary explicit and avoids any ambiguity
# if future model names coincidentally contain "gpt-5-chat" as an interior run.
_normalized: Final = model.split("/")[-1] # strip provider prefix, e.g. "openai/"
return "gpt-5" in model and not _normalized.startswith("gpt-5-chat")
return is_gpt_reasoning_series_name(model)

@classmethod
def is_model_gpt_5_search_model(cls, model: str) -> bool:
Expand Down Expand Up @@ -105,6 +99,8 @@ def is_model_gpt_5_4_model(cls, model: str) -> bool:
def is_model_gpt_5_4_plus_model(cls, model: str) -> bool:
"""Check if the model is gpt-5.4 or newer (5.4, 5.5, 5.6, etc., including pro)."""
model_name: Final = model.split("/")[-1]
if model_name.startswith("gpt-6"):
return True
if not model_name.startswith("gpt-5."):
return False
try:
Expand Down
3 changes: 2 additions & 1 deletion litellm/llms/openai/responses/transformation.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@
)
from litellm.litellm_core_utils.url_utils import encode_url_path_segment
from litellm.llms.base_llm.responses.transformation import BaseResponsesAPIConfig
from litellm.llms.openai.chat.gpt_5_transformation import is_gpt_reasoning_series_name
from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import *
from litellm.types.responses.main import *
Expand Down Expand Up @@ -48,7 +49,7 @@ def _is_gpt_5_model(model: str) -> bool:
parts: Final = model.split("/")
if len(parts) > 1 and parts[0] not in ("openai",):
return False
return "gpt-5" in model and "gpt-5-chat" not in model
return is_gpt_reasoning_series_name(model)

@staticmethod
def _supports_reasoning_effort_none(model: str) -> bool:
Expand Down
19 changes: 15 additions & 4 deletions litellm/proxy/common_utils/reset_budget_job.py
Original file line number Diff line number Diff line change
Expand Up @@ -115,6 +115,14 @@ def _tag_cache_keys(row: _TagRow) -> tuple[str, ...]:
return (f"tag:{row.tag_name}",)


def _enduser_counter_key(row: _EndUserRow) -> str:
return f"spend:end_user:{row.user_id}"


def _enduser_cache_keys(row: _EndUserRow) -> tuple[str, ...]:
return (f"end_user_id:{row.user_id}",)


def _budget_link_where(
budget_ids: Sequence[str],
extra: Mapping[str, object] = MappingProxyType({}),
Expand Down Expand Up @@ -346,6 +354,7 @@ async def _collect_budget_cascade(self, budgets_to_reset: Sequence[LiteLLM_Budge
where=_budget_link_where(budget_ids, _SPENT_ROWS_WHERE),
log_subject="tags",
)
endusers: Final[tuple[_EndUserRow, ...]] = await self._collect_endusers_to_reset(budget_ids)
return _BudgetCascade(
budgets=tuple(budgets_to_reset),
budget_ids=budget_ids,
Expand All @@ -357,18 +366,20 @@ async def _collect_budget_cascade(self, budgets_to_reset: Sequence[LiteLLM_Budge
for b in budgets_to_reset
if b.budget_id is not None and b.budget_duration is not None
),
endusers=await self._collect_endusers_to_reset(budget_ids),
endusers=endusers,
counter_keys=(
*(_team_membership_counter_key(row) for row in team_memberships),
*(_key_counter_key(row) for row in keys),
*(_org_counter_key(row) for row in orgs),
*(_tag_counter_key(row) for row in tags),
*(_enduser_counter_key(row) for row in endusers),

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P1 New spend can be erased If a request is charged while a large tier’s counters are being cleared after commit, the later reset can overwrite that charge with zero. Subsequent budget checks then undercount the new window’s spend

Knowledge Base Used: Spend, budgets, and rate limits

),
cache_keys=(
*(key for row in team_memberships for key in _team_membership_cache_keys(row)),
*(key for row in keys for key in _key_cache_keys(row)),
*(key for row in orgs for key in _org_cache_keys(row)),
*(key for row in tags for key in _tag_cache_keys(row)),
*(key for row in endusers for key in _enduser_cache_keys(row)),
Comment on lines +375 to +382

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Large tiers delay resets The job now writes a counter and deletes a cache entry sequentially for every end user. On a tier with tens of thousands of users, this sweep can delay later budget resets; please batch or bound it

Note: If this suggestion doesn't match your team's coding style, reply to this and let me know. I'll remember it for next time!

),
)

Expand All @@ -383,14 +394,14 @@ async def _commit_budget_cascade(self, cascade: _BudgetCascade) -> None:
if not cascade.budget_ids:
return

enduser_ids: Final = tuple(row.user_id for row in cascade.endusers)
async with budget_cascade_unit_of_work(self.prisma_client.db.batch_) as uow:
uow.team_memberships.queue_spend_zero(where=_budget_link_where(cascade.budget_ids))
uow.keys.queue_spend_zero(where=_budget_link_where(cascade.budget_ids, _LINKED_KEYS_WHERE))
uow.organizations.queue_spend_zero(where=_budget_link_where(cascade.budget_ids, _SPENT_ROWS_WHERE))
uow.tags.queue_spend_zero(where=_budget_link_where(cascade.budget_ids, _SPENT_ROWS_WHERE))
if enduser_ids:
uow.endusers.queue_spend_zero(where={"user_id": {"in": list(enduser_ids)}})
uow.endusers.queue_spend_zero(where=_budget_link_where(cascade.budget_ids, _SPENT_ROWS_WHERE))

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P1 Newly linked users stay blocked If an end user joins a due tier after users are collected, this update clears their database spend but misses their counter. Auth still reads that counter, so the user can remain blocked after the reset

Knowledge Base Used: Spend, budgets, and rate limits

if litellm.max_end_user_budget_id in cascade.budget_ids:
uow.endusers.queue_spend_zero(where={"budget_id": None, **_SPENT_ROWS_WHERE})
for budget_id, budget_reset_at in cascade.budget_resets:
uow.budgets.queue_window_advance(budget_id=budget_id, budget_reset_at=budget_reset_at)

Expand Down
6 changes: 3 additions & 3 deletions tests/litellm_utils_tests/test_proxy_budget_reset.py
Original file line number Diff line number Diff line change
Expand Up @@ -407,7 +407,7 @@ async def get_data_mock(table_name, *args, **kwargs):

enduser_writes = [c for c in batch_calls if c["table"] == "enduser"]
assert len(enduser_writes) == 1
assert enduser_writes[0]["where"]["user_id"]["in"] == [f"user{i}" for i in range(1, 7)]
assert enduser_writes[0]["where"] == {"budget_id": {"in": ["budget1"]}, "spend": {"gt": 0}}
assert enduser_writes[0]["data"] == {"spend": 0}

budget_writes = [c for c in batch_calls if c["table"] == "budget"]
Expand Down Expand Up @@ -608,7 +608,7 @@ async def fake_reset_team(team, current_time, reset_settings=None):
assert len([c for c in batch_calls if c["table"] == "team_membership"]) == 1
enduser_writes = [c for c in batch_calls if c["table"] == "enduser"]
assert len(enduser_writes) == 1
assert enduser_writes[0]["where"] == {"user_id": {"in": ["user1"]}}
assert enduser_writes[0]["where"] == {"budget_id": {"in": ["budget1"]}, "spend": {"gt": 0}}
assert enduser_writes[0]["data"] == {"spend": 0}

# Check the new batch write path: 2 keys + 1 user (user1 failed) + 2 teams.
Expand Down Expand Up @@ -1038,7 +1038,7 @@ async def fake_get_data(*, table_name, query_type, **kwargs):

enduser_writes = [c for c in batch_calls if c["table"] == "enduser"]
assert len(enduser_writes) == 1
assert enduser_writes[0]["where"] == {"user_id": {"in": ["user1", "user2"]}}
assert enduser_writes[0]["where"] == {"budget_id": {"in": ["budget1"]}, "spend": {"gt": 0}}

proxy_logging_obj.service_logging_obj.async_service_success_hook.assert_called_once()
(
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -299,3 +299,15 @@ def test_azure_gpt5_1_does_not_support_logprobs(config: AzureOpenAIGPT5Config):
supported_params = config.get_supported_openai_params(model="gpt-5.1")
assert "logprobs" not in supported_params
assert "top_logprobs" not in supported_params


def test_azure_gpt_6_astra_takes_the_reasoning_series_request_shape():
params = litellm.get_optional_params(
model="gpt-6-astra",
custom_llm_provider="azure",
max_tokens=100,
reasoning_effort="max",
)
assert params["max_completion_tokens"] == 100
assert "max_tokens" not in params
assert params["reasoning_effort"] == "max"
Original file line number Diff line number Diff line change
Expand Up @@ -1540,3 +1540,13 @@ def test_phase_roundtrip_output_to_input(self):
assert validated[0]["phase"] == "commentary"
assert validated[1]["phase"] == "final_answer"
assert "phase" not in validated[2]


@pytest.mark.parametrize("effort", [None, "low"])
def test_gpt_6_astra_drops_temperature_on_the_responses_path(effort):
mapped = OpenAIResponsesAPIConfig().map_openai_params(
response_api_optional_params={"temperature": 0, **({"reasoning": {"effort": effort}} if effort else {})},
model="gpt-6-astra",
drop_params=True,
)
assert "temperature" not in mapped
14 changes: 14 additions & 0 deletions tests/test_litellm/llms/openai/test_gpt5_transformation.py
Original file line number Diff line number Diff line change
Expand Up @@ -1309,3 +1309,17 @@ def test_responses_gpt54_allow_temperature_effort_none(
drop_params=False,
)
assert params["temperature"] == 0.7


def test_gpt_6_astra_takes_the_reasoning_series_request_shape():
params = litellm.get_optional_params(
model="gpt-6-astra",
custom_llm_provider="openai",
max_tokens=100,
reasoning_effort="max",
verbosity="low",
)
assert params["max_completion_tokens"] == 100
assert "max_tokens" not in params
assert params["reasoning_effort"] == "max"
assert params["verbosity"] == "low"
4 changes: 4 additions & 0 deletions tests/test_litellm/llms/openai/test_is_model_gpt_5_model.py
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,8 @@

# Models that MUST be classified as GPT-5 (routed through GPT-5 reasoning path)
GPT5_MODELS = [
"gpt-6-astra",
"openai/gpt-6-astra",
"gpt-5",
"gpt-5.1",
"gpt-5.2",
Expand Down Expand Up @@ -120,6 +122,8 @@ def test_gpt5_chat_family_is_excluded(self):
# /v1/responses bridge (when reasoning_effort is set and tools are passed) on
# is_model_gpt_5_4_plus_model, so the gpt-5.6 family must land on the True side.
GPT5_4_PLUS_MODELS = [
"gpt-6-astra",
"openai/gpt-6-astra",
"gpt-5.4",
"gpt-5.5",
"gpt-5.5-pro",
Expand Down
Loading
Loading