From 382c627b44bba004f2c23d4120cae8889aa906f5 Mon Sep 17 00:00:00 2001 From: Sina Azizi Date: Mon, 17 Aug 2026 09:53:00 -0700 Subject: [PATCH 1/3] fix(responses): accept per-request timeout on aresponses aresponses() was the one call path left out of #1263: it has no timeout parameter, and since ResponsesParams is extra="forbid", timeout= raises a ValidationError. Declare it on the signature and route it through _validate_and_forward_timeout like acompletion/amessages. --- src/any_llm/any_llm.py | 6 ++++++ tests/unit/test_responses.py | 16 ++++++++++++++++ 2 files changed, 22 insertions(+) diff --git a/src/any_llm/any_llm.py b/src/any_llm/any_llm.py index 15602de00..babfb4165 100644 --- a/src/any_llm/any_llm.py +++ b/src/any_llm/any_llm.py @@ -1222,6 +1222,7 @@ async def aresponses( prompt_cache_key: str | None = None, prompt_cache_retention: str | None = None, conversation: str | dict[str, Any] | None = None, + timeout: float | None = None, # noqa: ASYNC109 # forwarded to the provider SDK, which owns the timeout extra_body: dict[str, Any] | None = None, **kwargs: Any, ) -> ResponseResource | Response | AsyncIterator[ResponseStreamEvent] | ParsedResponse[Any]: @@ -1269,6 +1270,10 @@ async def aresponses( prompt_cache_key: A key to use when reading from or writing to the prompt cache. prompt_cache_retention: How long to retain a prompt cache entry created by this request. conversation: The conversation to associate this response with (ID string or ConversationParam object). + timeout: Per-request timeout in seconds, passed through to the provider's client/SDK call. + It is a transport option, not request-body JSON. Providers with no per-request + timeout raise `UnsupportedParameterError`; set a timeout on their client via + client_args instead. extra_body: Additional fields to merge into an OpenAI-compatible Responses request body. **kwargs: Additional provider-specific arguments that will be passed to the provider's API call. @@ -1327,6 +1332,7 @@ async def aresponses( ) provider_kwargs: dict[str, Any] = {} + self._validate_and_forward_timeout(timeout, provider_kwargs) if extra_body is not None: provider_kwargs["extra_body"] = extra_body result = await self._aresponses(params, **provider_kwargs) diff --git a/tests/unit/test_responses.py b/tests/unit/test_responses.py index e43889271..b195f8fea 100644 --- a/tests/unit/test_responses.py +++ b/tests/unit/test_responses.py @@ -115,3 +115,19 @@ def add(a: int, b: int) -> int: assert tools[1]["parameters"] == {} assert tools[1]["strict"] is True assert tools[3] == {"type": "web_search"} + + +@pytest.mark.asyncio +async def test_timeout_forwarded_to_provider_not_params() -> None: + """timeout is an SDK request option: it must reach the provider call, never ResponsesParams.""" + from unittest.mock import AsyncMock, patch + + from any_llm import AnyLLM + + llm = AnyLLM.create("openai", api_key="test-key") + with patch.object(type(llm), "_aresponses", new=AsyncMock(return_value=object())) as mock_aresponses: + await llm.aresponses("gpt-4.1-mini", "hello", timeout=60) + + assert mock_aresponses.call_args.kwargs["timeout"] == 60 + params = mock_aresponses.call_args.args[0] + assert "timeout" not in params.model_dump(exclude_none=True) From 447a1cc122043ae8764f0c8543c96f8cab95e2a9 Mon Sep 17 00:00:00 2001 From: Sina Azizi Date: Mon, 17 Aug 2026 10:04:24 -0700 Subject: [PATCH 2/3] chore(responses): shorten timeout docstring, move test imports to module scope --- src/any_llm/any_llm.py | 5 +---- tests/unit/test_responses.py | 4 ---- 2 files changed, 1 insertion(+), 8 deletions(-) diff --git a/src/any_llm/any_llm.py b/src/any_llm/any_llm.py index babfb4165..a53ffd055 100644 --- a/src/any_llm/any_llm.py +++ b/src/any_llm/any_llm.py @@ -1270,10 +1270,7 @@ async def aresponses( prompt_cache_key: A key to use when reading from or writing to the prompt cache. prompt_cache_retention: How long to retain a prompt cache entry created by this request. conversation: The conversation to associate this response with (ID string or ConversationParam object). - timeout: Per-request timeout in seconds, passed through to the provider's client/SDK call. - It is a transport option, not request-body JSON. Providers with no per-request - timeout raise `UnsupportedParameterError`; set a timeout on their client via - client_args instead. + timeout: Per-request timeout in seconds, forwarded to the provider's SDK call. extra_body: Additional fields to merge into an OpenAI-compatible Responses request body. **kwargs: Additional provider-specific arguments that will be passed to the provider's API call. diff --git a/tests/unit/test_responses.py b/tests/unit/test_responses.py index b195f8fea..8a5caa7e6 100644 --- a/tests/unit/test_responses.py +++ b/tests/unit/test_responses.py @@ -120,10 +120,6 @@ def add(a: int, b: int) -> int: @pytest.mark.asyncio async def test_timeout_forwarded_to_provider_not_params() -> None: """timeout is an SDK request option: it must reach the provider call, never ResponsesParams.""" - from unittest.mock import AsyncMock, patch - - from any_llm import AnyLLM - llm = AnyLLM.create("openai", api_key="test-key") with patch.object(type(llm), "_aresponses", new=AsyncMock(return_value=object())) as mock_aresponses: await llm.aresponses("gpt-4.1-mini", "hello", timeout=60) From c83e27ef027e99710eafbaf40ff93dc5679d77d4 Mon Sep 17 00:00:00 2001 From: njbrake Date: Wed, 19 Aug 2026 17:50:54 +0000 Subject: [PATCH 3/3] fix(responses): expose timeout on the public responses helpers AnyLLM.aresponses accepts the per-request timeout, but any_llm.responses and any_llm.aresponses never declared it, so callers of the functional API could only pass it as an untyped **kwargs passthrough that type checkers and the docstrings did not mention. #1263 declared it on completion, acompletion, messages and amessages; do the same for the two Responses helpers. Also restore the docstring wording the sibling methods use. huggingface supports the Responses API while declaring TIMEOUT_SUPPORT "unsupported", so the UnsupportedParameterError path is reachable here and worth naming. Tests cover signature exposure, an explicit None staying unset, both the sync and async helpers, and the unsupported-provider rejection. Co-Authored-By: Claude Opus 5 (1M context) --- src/any_llm/any_llm.py | 6 +++- src/any_llm/api.py | 14 +++++++++ tests/unit/test_responses.py | 59 +++++++++++++++++++++++++++++++++--- 3 files changed, 74 insertions(+), 5 deletions(-) diff --git a/src/any_llm/any_llm.py b/src/any_llm/any_llm.py index a53ffd055..c9c1e84da 100644 --- a/src/any_llm/any_llm.py +++ b/src/any_llm/any_llm.py @@ -1270,7 +1270,11 @@ async def aresponses( prompt_cache_key: A key to use when reading from or writing to the prompt cache. prompt_cache_retention: How long to retain a prompt cache entry created by this request. conversation: The conversation to associate this response with (ID string or ConversationParam object). - timeout: Per-request timeout in seconds, forwarded to the provider's SDK call. + timeout: Per-request timeout in seconds, passed through to the provider's client/SDK. + An explicit ``None`` is treated the same as omitting it (the provider's default + applies), so it cannot request an unbounded timeout. Providers that have no + per-request timeout raise `UnsupportedParameterError`; set a timeout on their + client via `client_args` instead. extra_body: Additional fields to merge into an OpenAI-compatible Responses request body. **kwargs: Additional provider-specific arguments that will be passed to the provider's API call. diff --git a/src/any_llm/api.py b/src/any_llm/api.py index 1afd8a1f8..0346f12fb 100644 --- a/src/any_llm/api.py +++ b/src/any_llm/api.py @@ -300,6 +300,7 @@ def responses( prompt_cache_key: str | None = None, prompt_cache_retention: str | None = None, conversation: str | dict[str, Any] | None = None, + timeout: float | None = None, extra_body: dict[str, Any] | None = None, client_args: dict[str, Any] | None = None, **kwargs: Any, @@ -354,6 +355,11 @@ def responses( prompt_cache_key: A key to use when reading from or writing to the prompt cache. prompt_cache_retention: How long to retain a prompt cache entry created by this request. conversation: The conversation to associate this response with (ID string or ConversationParam object). + timeout: Per-request timeout in seconds, passed through to the provider's client/SDK. + An explicit ``None`` is treated the same as omitting it (the provider's default + applies), so it cannot request an unbounded timeout. Providers that have no + per-request timeout raise `UnsupportedParameterError`; set a timeout on their + client via `client_args` instead. extra_body: Additional fields to merge into an OpenAI-compatible Responses request body. client_args: Additional provider-specific arguments that will be passed to the provider's client instantiation. **kwargs: Additional provider-specific arguments that will be passed to the provider's API call. @@ -410,6 +416,7 @@ def responses( prompt_cache_key=prompt_cache_key, prompt_cache_retention=prompt_cache_retention, conversation=conversation, + timeout=timeout, extra_body=extra_body, **kwargs, ) @@ -449,6 +456,7 @@ async def aresponses( prompt_cache_key: str | None = None, prompt_cache_retention: str | None = None, conversation: str | dict[str, Any] | None = None, + timeout: float | None = None, # noqa: ASYNC109 # forwarded to the provider SDK, which owns the timeout extra_body: dict[str, Any] | None = None, client_args: dict[str, Any] | None = None, **kwargs: Any, @@ -503,6 +511,11 @@ async def aresponses( prompt_cache_key: A key to use when reading from or writing to the prompt cache. prompt_cache_retention: How long to retain a prompt cache entry created by this request. conversation: The conversation to associate this response with (ID string or ConversationParam object). + timeout: Per-request timeout in seconds, passed through to the provider's client/SDK. + An explicit ``None`` is treated the same as omitting it (the provider's default + applies), so it cannot request an unbounded timeout. Providers that have no + per-request timeout raise `UnsupportedParameterError`; set a timeout on their + client via `client_args` instead. extra_body: Additional fields to merge into an OpenAI-compatible Responses request body. client_args: Additional provider-specific arguments that will be passed to the provider's client instantiation. **kwargs: Additional provider-specific arguments that will be passed to the provider's API call. @@ -559,6 +572,7 @@ async def aresponses( prompt_cache_key=prompt_cache_key, prompt_cache_retention=prompt_cache_retention, conversation=conversation, + timeout=timeout, extra_body=extra_body, **kwargs, ) diff --git a/tests/unit/test_responses.py b/tests/unit/test_responses.py index 8a5caa7e6..374e0f039 100644 --- a/tests/unit/test_responses.py +++ b/tests/unit/test_responses.py @@ -1,3 +1,4 @@ +from inspect import signature from typing import Any, cast from unittest.mock import AsyncMock, patch @@ -5,7 +6,8 @@ from pydantic import ValidationError from any_llm import AnyLLM -from any_llm.api import aresponses +from any_llm.api import aresponses, responses +from any_llm.exceptions import UnsupportedParameterError from any_llm.types.responses import ResponsesParams @@ -117,13 +119,62 @@ def add(a: int, b: int) -> int: assert tools[3] == {"type": "web_search"} +def test_responses_exposes_timeout_parameter() -> None: + """The Responses entry points advertise timeout the same way the completion ones do.""" + assert "timeout" in signature(responses).parameters + assert "timeout" in signature(aresponses).parameters + assert "timeout" in signature(AnyLLM.aresponses).parameters + + @pytest.mark.asyncio -async def test_timeout_forwarded_to_provider_not_params() -> None: +@pytest.mark.parametrize(("requested_timeout", "expected"), [(60, 60), (None, None)]) +async def test_timeout_forwarded_to_provider_not_params( + requested_timeout: float | None, expected: float | None +) -> None: """timeout is an SDK request option: it must reach the provider call, never ResponsesParams.""" llm = AnyLLM.create("openai", api_key="test-key") with patch.object(type(llm), "_aresponses", new=AsyncMock(return_value=object())) as mock_aresponses: - await llm.aresponses("gpt-4.1-mini", "hello", timeout=60) + await llm.aresponses("gpt-4.1-mini", "hello", timeout=requested_timeout) - assert mock_aresponses.call_args.kwargs["timeout"] == 60 + assert mock_aresponses.call_args.kwargs.get("timeout") == expected params = mock_aresponses.call_args.args[0] assert "timeout" not in params.model_dump(exclude_none=True) + + +@pytest.mark.asyncio +async def test_aresponses_helper_forwards_timeout_to_provider() -> None: + """The public aresponses() helper routes timeout through to the provider call.""" + provider = AnyLLM.create("openai", api_key="test-key") + with ( + patch("any_llm.any_llm.AnyLLM.create", return_value=provider), + patch.object(provider, "_aresponses", new=AsyncMock(return_value=object())) as mock_aresponses, + ): + await aresponses("gpt-4.1-mini", "hello", provider="openai", api_key="test-key", timeout=60) + + assert mock_aresponses.call_args.kwargs["timeout"] == 60 + + +def test_responses_helper_forwards_timeout_to_provider_sync() -> None: + """The synchronous responses() helper routes timeout through the same path.""" + provider = AnyLLM.create("openai", api_key="test-key") + with ( + patch("any_llm.any_llm.AnyLLM.create", return_value=provider), + patch.object(provider, "_aresponses", new=AsyncMock(return_value=object())) as mock_aresponses, + ): + responses("gpt-4.1-mini", "hello", provider="openai", api_key="test-key", timeout=60) + + assert mock_aresponses.call_args.kwargs["timeout"] == 60 + + +@pytest.mark.asyncio +async def test_aresponses_rejects_timeout_for_unsupported_provider() -> None: + """A provider declaring no per-request timeout support is rejected before the request is built.""" + llm = AnyLLM.create("openai", api_key="test-key") + llm.TIMEOUT_SUPPORT = "unsupported" + with ( + patch.object(llm, "_aresponses", new=AsyncMock()) as mock_aresponses, + pytest.raises(UnsupportedParameterError, match="timeout"), + ): + await llm.aresponses("gpt-4.1-mini", "hello", timeout=60) + + mock_aresponses.assert_not_called()