From 031cc1b8870691fd23f555a2f1fc2c26419967b2 Mon Sep 17 00:00:00 2001 From: Filippo Mattia Menghi Date: Wed, 10 Jun 2026 10:09:45 +0200 Subject: [PATCH 1/2] fix(router): route aspeech through async_function_with_fallbacks Router.aspeech selected a deployment and awaited litellm.aspeech directly, so TTS requests got no retry on failure and no failover to backup deployments; the except block only fired an exception alert and re-raised. Every other router endpoint (acompletion, aembedding, atranscription, arerank) already delegates to async_function_with_fallbacks Mirror the atranscription pattern: move deployment selection and the litellm.aspeech call into a private _aspeech method, then have the public aspeech set kwargs["original_function"] = self._aspeech and await self.async_function_with_fallbacks(**kwargs). _aspeech also picks up the shared _get_async_openai_model_client helper and the same total/success/fail call accounting the sibling endpoints use Fixes #27778. --- litellm/router.py | 58 +++++++++++------ .../test_router_endpoints.py | 64 +++++++++++++++++++ 2 files changed, 101 insertions(+), 21 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index d0f4e5ff44d8..47f76d00f9d4 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -4042,18 +4042,38 @@ async def aspeech(self, model: str, input: str, voice: str, **kwargs): ``` """ try: + kwargs["model"] = model kwargs["input"] = input kwargs["voice"] = voice + kwargs["original_function"] = self._aspeech + self._update_kwargs_before_fallbacks(model=model, kwargs=kwargs) + response = await self.async_function_with_fallbacks(**kwargs) + return response + except Exception as e: + asyncio.create_task( + send_llm_exception_alert( + litellm_router_instance=self, + request_kwargs=kwargs, + error_traceback_str=traceback.format_exc(), + original_exception=e, + ) + ) + raise e + + async def _aspeech(self, model: str, input: str, voice: str, **kwargs): + model_name = model + try: + verbose_router_logger.debug( + f"Inside _aspeech()- model: {model}; kwargs: {kwargs}" + ) deployment = await self.async_get_available_deployment( model=model, messages=[{"role": "user", "content": "prompt"}], specific_deployment=kwargs.pop("specific_deployment", None), request_kwargs=kwargs, ) - self._update_kwargs_before_fallbacks(model=model, kwargs=kwargs) data = deployment["litellm_params"].copy() - data["model"] for k, v in self.default_litellm_params.items(): if ( k not in kwargs @@ -4062,37 +4082,33 @@ async def aspeech(self, model: str, input: str, voice: str, **kwargs): elif k == "metadata": kwargs[k].update(v) - potential_model_client = self._get_client( - deployment=deployment, kwargs=kwargs, client_type="async" + model_client = self._get_async_openai_model_client( + deployment=deployment, + kwargs=kwargs, ) - # check if provided keys == client keys # - dynamic_api_key = kwargs.get("api_key", None) - if ( - dynamic_api_key is not None - and potential_model_client is not None - and dynamic_api_key != potential_model_client.api_key - ): - model_client = None - else: - model_client = potential_model_client + self.total_calls[model_name] += 1 response = await litellm.aspeech( **{ **data, + "input": input, + "voice": voice, "client": model_client, **kwargs, } ) + + self.success_calls[model_name] += 1 + verbose_router_logger.info( + f"litellm.aspeech(model={model_name})\033[32m 200 OK\033[0m" + ) return response except Exception as e: - asyncio.create_task( - send_llm_exception_alert( - litellm_router_instance=self, - request_kwargs=kwargs, - error_traceback_str=traceback.format_exc(), - original_exception=e, - ) + verbose_router_logger.info( + f"litellm.aspeech(model={model_name})\033[31m Exception {str(e)}\033[0m" ) + if model_name is not None: + self.fail_calls[model_name] += 1 raise e async def arerank(self, model: str, **kwargs): diff --git a/tests/router_unit_tests/test_router_endpoints.py b/tests/router_unit_tests/test_router_endpoints.py index 3f0afe2a5a69..03a9baa9864f 100644 --- a/tests/router_unit_tests/test_router_endpoints.py +++ b/tests/router_unit_tests/test_router_endpoints.py @@ -198,6 +198,70 @@ async def test_audio_speech_router(mode): assert test_logger.standard_logging_object["model_group"] == "tts" +@pytest.mark.asyncio +async def test_aspeech_fallbacks_on_deployment_failure(): + router = Router( + model_list=[ + { + "model_name": "tts-main", + "litellm_params": {"model": "openai/tts-1", "api_key": "fake-key"}, + }, + { + "model_name": "tts-backup", + "litellm_params": {"model": "openai/tts-1-hd", "api_key": "fake-key"}, + }, + ], + fallbacks=[{"tts-main": ["tts-backup"]}], + num_retries=0, + ) + + called_models = [] + + async def mock_aspeech(*args, **kwargs): + called_models.append(kwargs["model"]) + if kwargs["model"] == "openai/tts-1": + raise litellm.InternalServerError( + message="deployment down", + llm_provider="openai", + model="tts-1", + ) + return MagicMock() + + with patch("litellm.aspeech", side_effect=mock_aspeech): + response = await router.aspeech( + model="tts-main", + input="the quick brown fox jumped over the lazy dogs", + voice="alloy", + ) + + assert response is not None + assert called_models == ["openai/tts-1", "openai/tts-1-hd"] + + +@pytest.mark.asyncio +async def test_aspeech_success_returns_response(): + router = Router( + model_list=[ + { + "model_name": "tts", + "litellm_params": {"model": "openai/tts-1", "api_key": "fake-key"}, + }, + ] + ) + + mock_response = MagicMock() + with patch("litellm.aspeech", return_value=mock_response) as mock_aspeech: + response = await router.aspeech( + model="tts", + input="the quick brown fox jumped over the lazy dogs", + voice="alloy", + ) + + assert response is mock_response + mock_aspeech.assert_called_once() + assert mock_aspeech.call_args.kwargs["model"] == "openai/tts-1" + + @pytest.mark.asyncio() async def test_rerank_endpoint(model_list): from litellm.types.utils import RerankResponse From 467c03392cd76972c574550e9a53ade9433f20fc Mon Sep 17 00:00:00 2001 From: Filippo Mattia Menghi Date: Wed, 10 Jun 2026 11:00:47 +0200 Subject: [PATCH 2/2] fix(router): apply deployment kwargs and rpm semaphore in _aspeech Bring _aspeech fully in line with _atranscription: call _update_kwargs_with_deployment so deployment metadata, model_info, timeout, and default litellm params flow into the request, and wrap the litellm.aspeech call with the max_parallel_requests semaphore plus async_routing_strategy_pre_call_checks so TTS respects rpm limits the same way the other router endpoints do Also add a unit test that exercises _aspeech directly and asserts the deployment metadata reaches the underlying call --- litellm/router.py | 38 ++++++++++++++----- .../test_router_endpoints.py | 26 +++++++++++++ 2 files changed, 55 insertions(+), 9 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index 47f76d00f9d4..ba63813b7475 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -4067,28 +4067,23 @@ async def _aspeech(self, model: str, input: str, voice: str, **kwargs): verbose_router_logger.debug( f"Inside _aspeech()- model: {model}; kwargs: {kwargs}" ) + parent_otel_span = _get_parent_otel_span_from_kwargs(kwargs) deployment = await self.async_get_available_deployment( model=model, messages=[{"role": "user", "content": "prompt"}], specific_deployment=kwargs.pop("specific_deployment", None), request_kwargs=kwargs, ) - data = deployment["litellm_params"].copy() - for k, v in self.default_litellm_params.items(): - if ( - k not in kwargs - ): # prioritize model-specific params > default router params - kwargs[k] = v - elif k == "metadata": - kwargs[k].update(v) + self._update_kwargs_with_deployment(deployment=deployment, kwargs=kwargs) + data = deployment["litellm_params"].copy() model_client = self._get_async_openai_model_client( deployment=deployment, kwargs=kwargs, ) self.total_calls[model_name] += 1 - response = await litellm.aspeech( + response = litellm.aspeech( **{ **data, "input": input, @@ -4098,6 +4093,31 @@ async def _aspeech(self, model: str, input: str, voice: str, **kwargs): } ) + ### CONCURRENCY-SAFE RPM CHECKS ### + rpm_semaphore = self._get_client( + deployment=deployment, + kwargs=kwargs, + client_type="max_parallel_requests", + ) + + if rpm_semaphore is not None and isinstance( + rpm_semaphore, asyncio.Semaphore + ): + async with rpm_semaphore: + """ + - Check rpm limits before making the call + - If allowed, increment the rpm limit (allows global value to be updated, concurrency-safe) + """ + await self.async_routing_strategy_pre_call_checks( + deployment=deployment, parent_otel_span=parent_otel_span + ) + response = await response + else: + await self.async_routing_strategy_pre_call_checks( + deployment=deployment, parent_otel_span=parent_otel_span + ) + response = await response + self.success_calls[model_name] += 1 verbose_router_logger.info( f"litellm.aspeech(model={model_name})\033[32m 200 OK\033[0m" diff --git a/tests/router_unit_tests/test_router_endpoints.py b/tests/router_unit_tests/test_router_endpoints.py index 03a9baa9864f..f4c6fbefe05e 100644 --- a/tests/router_unit_tests/test_router_endpoints.py +++ b/tests/router_unit_tests/test_router_endpoints.py @@ -262,6 +262,32 @@ async def test_aspeech_success_returns_response(): assert mock_aspeech.call_args.kwargs["model"] == "openai/tts-1" +@pytest.mark.asyncio +async def test_aspeech_sets_deployment_metadata(): + router = Router( + model_list=[ + { + "model_name": "tts", + "litellm_params": {"model": "openai/tts-1", "api_key": "fake-key"}, + }, + ] + ) + + mock_response = MagicMock() + with patch("litellm.aspeech", return_value=mock_response) as mock_aspeech: + response = await router._aspeech( + model="tts", + input="the quick brown fox jumped over the lazy dogs", + voice="alloy", + ) + + assert response is mock_response + metadata = mock_aspeech.call_args.kwargs["metadata"] + assert metadata["deployment"] == "openai/tts-1" + assert metadata["deployment_model_name"] == "tts" + assert metadata["model_info"]["id"] is not None + + @pytest.mark.asyncio() async def test_rerank_endpoint(model_list): from litellm.types.utils import RerankResponse