Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
45 changes: 32 additions & 13 deletions litellm/model_prices_and_context_window_backup.json
Original file line number Diff line number Diff line change
Expand Up @@ -1652,7 +1652,7 @@
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_assistant_prefill": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
Expand Down Expand Up @@ -1681,7 +1681,7 @@
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_assistant_prefill": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
Expand Down Expand Up @@ -1710,7 +1710,7 @@
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_assistant_prefill": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
Expand Down Expand Up @@ -1739,7 +1739,7 @@
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_assistant_prefill": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
Expand Down Expand Up @@ -1768,7 +1768,7 @@
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_assistant_prefill": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
Expand Down Expand Up @@ -1797,7 +1797,7 @@
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_assistant_prefill": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
Expand Down Expand Up @@ -2470,7 +2470,7 @@
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 1.5e-05,
"supports_assistant_prefill": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
Expand Down Expand Up @@ -10240,7 +10240,7 @@
"search_context_size_medium": 0.01
},
"supports_adaptive_thinking": true,
"supports_assistant_prefill": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
Expand Down Expand Up @@ -18980,6 +18980,7 @@
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_vision": true
Expand Down Expand Up @@ -28199,7 +28200,7 @@
"output_cost_per_token": 1.5e-05,
"output_cost_per_token_above_200k_tokens": 2.25e-05,
"source": "https://openrouter.ai/anthropic/claude-sonnet-4.6",
"supports_assistant_prefill": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
Expand Down Expand Up @@ -28237,7 +28238,7 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.5e-05,
"supports_assistant_prefill": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
Expand Down Expand Up @@ -30272,6 +30273,7 @@
"mode": "responses",
"supports_web_search": true,
"supports_reasoning": false,
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_output_config": true
},
Expand All @@ -30280,6 +30282,7 @@
"mode": "responses",
"supports_web_search": true,
"supports_reasoning": false,
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_output_config": true
},
Expand Down Expand Up @@ -31303,6 +31306,22 @@
"mode": "chat",
"supports_computer_use": true
},
"snowflake/claude-sonnet-4-6": {
"cache_read_input_token_cost": 0.0000003,
"input_cost_per_token": 0.000003,
"litellm_provider": "snowflake",
"max_input_tokens": 200000,
"max_output_tokens": 16384,
"max_tokens": 16384,
"mode": "chat",
"output_cost_per_token": 0.000015,
"supports_assistant_prefill": false,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_vision": true
},
"snowflake/deepseek-r1": {
"litellm_provider": "snowflake",
"max_input_tokens": 32768,
Expand Down Expand Up @@ -33453,7 +33472,7 @@
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 2.5e-05,
"supports_assistant_prefill": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_prompt_caching": true,
Expand Down Expand Up @@ -34942,7 +34961,7 @@
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 1.5e-05,
"supports_assistant_prefill": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
Expand Down Expand Up @@ -42350,7 +42369,7 @@
"max_tokens": 64000,
"mode": "chat",
"output_cost_per_token": 1.5e-05,
"supports_assistant_prefill": true,
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
Expand Down
56 changes: 34 additions & 22 deletions litellm/router.py
Original file line number Diff line number Diff line change
Expand Up @@ -108,6 +108,7 @@
)
from litellm.router_utils.fallback_event_handlers import (
_check_non_standard_fallback_format,
build_mid_stream_continuation_messages,
get_fallback_model_group,
run_async_fallback,
)
Expand Down Expand Up @@ -2233,17 +2234,18 @@ async def stream_with_fallbacks():
# would waste tokens and confuse the model.
initial_kwargs["messages"] = messages
else:
initial_kwargs["messages"] = messages + [
{
"role": "system",
"content": "You are a helpful assistant. You are given a message and you need to respond to it. You are also given a generated content. You need to respond to the message in continuation of the generated content. Do not repeat the same content. Your response should be in continuation of this text: ",
},
{
"role": "assistant",
"content": e.generated_content,
"prefix": True,
},
]
# Prefill-resume only where the model accepts assistant
# prefill (Claude Sonnet 4.6+/Opus 4.6+ return a 400 for
# it); otherwise the partial text rides a user message.
initial_kwargs["messages"] = (
build_mid_stream_continuation_messages(
messages=messages,
generated_content=e.generated_content,
model_group=model_group,
fallbacks=fallbacks,
resolve_underlying_model=self._underlying_model_for_group,
)
)
self._update_kwargs_before_fallbacks(
model=model_group, kwargs=initial_kwargs
)
Expand Down Expand Up @@ -2793,17 +2795,18 @@ def stream_with_fallbacks():
if e.is_pre_first_chunk or not e.generated_content:
initial_kwargs["messages"] = messages
else:
initial_kwargs["messages"] = messages + [
{
"role": "system",
"content": "You are a helpful assistant. You are given a message and you need to respond to it. You are also given a generated content. You need to respond to the message in continuation of the generated content. Do not repeat the same content. Your response should be in continuation of this text: ",
},
{
"role": "assistant",
"content": e.generated_content,
"prefix": True,
},
]
# Prefill-resume only where the model accepts assistant
# prefill (Claude Sonnet 4.6+/Opus 4.6+ return a 400 for
# it); otherwise the partial text rides a user message.
initial_kwargs["messages"] = (
build_mid_stream_continuation_messages(
messages=messages,
generated_content=e.generated_content,
model_group=model_group,
fallbacks=fallbacks,
resolve_underlying_model=router_self._underlying_model_for_group,
)
)
router_self._update_kwargs_before_fallbacks(
model=model_group, kwargs=initial_kwargs
)
Expand Down Expand Up @@ -9049,6 +9052,15 @@ def get_deployment_by_model_group_name(
raise Exception("Model Name invalid - {}".format(type(model)))
return None

def _underlying_model_for_group(self, model_group_name: str) -> str | None:
"""Registry model name the group resolves to (e.g. the alias
``production-claude`` -> ``anthropic/claude-sonnet-4-6``), or None when
the group has no deployment."""
deployment = self.get_deployment_by_model_group_name(model_group_name)
if deployment is None:
return None
return deployment.litellm_params.model

def get_deployment_credentials_with_provider(
self, model_id: str
) -> Optional[Dict[str, Any]]:
Expand Down
Loading
Loading