Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -39,12 +39,12 @@ mistral = [
anthropic = []

gemini = [
"google-genai>=1.51.0",
"google-genai>=1.70.0",
"google-cloud-storage",
]

vertexai = [
"google-genai>=1.51.0",
"google-genai>=1.70.0",
"google-cloud-storage",
]

Expand Down
17 changes: 14 additions & 3 deletions src/any_llm/any_llm.py
Original file line number Diff line number Diff line change
Expand Up @@ -54,6 +54,7 @@
from any_llm.utils.structured_output import (
build_parsed_message,
is_structured_output_type,
normalize_output_config,
parse_json_content,
parse_responses_output,
)
Expand Down Expand Up @@ -132,6 +133,9 @@ class AnyLLM(ABC):
SUPPORTS_MESSAGES: bool = True
"""Anthropic Messages API (all providers support it via conversion)"""

SUPPORTS_MESSAGES_STRUCTURED_OUTPUT_STREAMING: bool = False
"""Whether Messages structured output can be streamed by this provider."""

PROMPT_CACHE_KEY_SUPPORT: Literal["unsupported", "supported", "passthrough"] = "unsupported"
"""Whether prompt_cache_key is supported, forwarded to a router, or rejected."""

Expand Down Expand Up @@ -960,7 +964,8 @@ async def amessages(
``output_config``. Either a Pydantic ``BaseModel``/dataclass **type** (typed
``parsed_output``) or a raw Anthropic ``output_config`` **dict** for non-Pydantic
JSON schemas (``parsed_output`` holds the parsed JSON). The call returns
Anthropic's ``ParsedMessage``. Not supported with ``stream=True``.
Anthropic's ``ParsedMessage`` for non-streaming requests. Providers with native
support can stream schema-constrained Messages events instead.
timeout: Per-request timeout in seconds, passed through to the provider's client/SDK.
An explicit ``None`` is treated the same as omitting it (the provider's default
applies), so it cannot request an unbounded timeout. Providers that have no
Expand All @@ -973,12 +978,13 @@ async def amessages(
iterator of MessageStreamEvent (if streaming).

Raises:
ValueError: If `output_format` is combined with `stream=True`.
ValueError: If `output_format` is combined with `stream=True` for a provider that
does not support streaming structured output.
NotImplementedError: If `container`, `context_management`, or `betas` is used
with a provider that has no native Anthropic Messages API.

"""
if output_format is not None and stream:
if output_format is not None and stream and not self.SUPPORTS_MESSAGES_STRUCTURED_OUTPUT_STREAMING:
msg = "stream is not supported for output_format"
raise ValueError(msg)

Expand Down Expand Up @@ -1014,6 +1020,11 @@ async def amessages(
# case); for the raw-dict case and for all bridged providers it returns a MessageResponse,
# so build the same ParsedMessage shape from the response's JSON text here.
if output_format is not None and isinstance(result, MessageResponse):
if isinstance(output_format, dict):
format_config = normalize_output_config(output_format).get("format")
schema = format_config.get("schema") if isinstance(format_config, dict) else None
if not isinstance(schema, dict) or not schema:
return result
return build_parsed_message(result, output_format)

return result
Expand Down
8 changes: 4 additions & 4 deletions src/any_llm/api.py
Original file line number Diff line number Diff line change
Expand Up @@ -636,8 +636,8 @@ def messages(
output_format: Structured output, mirroring Anthropic's ``messages.parse``/``output_config``.
Either a Pydantic ``BaseModel``/dataclass **type** (typed ``parsed_output``) or a raw
Anthropic ``output_config`` **dict** for non-Pydantic JSON schemas (``parsed_output``
holds the parsed JSON). The call returns Anthropic's ``ParsedMessage``. Not supported
with streaming.
holds the parsed JSON). Non-streaming calls return Anthropic's ``ParsedMessage``;
providers with native support can stream schema-constrained Messages events.
timeout: Per-request timeout in seconds, passed through to the provider's client/SDK.
An explicit ``None`` is treated the same as omitting it (the provider's default
applies), so it cannot request an unbounded timeout. Providers that have no
Expand Down Expand Up @@ -744,8 +744,8 @@ async def amessages(
output_format: Structured output, mirroring Anthropic's ``messages.parse``/``output_config``.
Either a Pydantic ``BaseModel``/dataclass **type** (typed ``parsed_output``) or a raw
Anthropic ``output_config`` **dict** for non-Pydantic JSON schemas (``parsed_output``
holds the parsed JSON). The call returns Anthropic's ``ParsedMessage``. Not supported
with streaming.
holds the parsed JSON). Non-streaming calls return Anthropic's ``ParsedMessage``;
providers with native support can stream schema-constrained Messages events.
timeout: Per-request timeout in seconds, passed through to the provider's client/SDK.
An explicit ``None`` is treated the same as omitting it (the provider's default
applies), so it cannot request an unbounded timeout. Providers that have no
Expand Down
12 changes: 11 additions & 1 deletion src/any_llm/providers/anthropic/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -236,6 +236,7 @@ class BaseAnthropicProvider(AnyLLM, ABC):
SUPPORTS_LIST_MODELS = False
SUPPORTS_BATCH = True
SUPPORTS_RERANK = False
SUPPORTS_MESSAGES_STRUCTURED_OUTPUT_STREAMING = True

# The Anthropic SDK accepts a per-request `timeout` on messages.create, so it forwards unchanged.
TIMEOUT_SUPPORT = "native"
Expand Down Expand Up @@ -322,7 +323,8 @@ async def _amessages(
(which drives the GA ``output_config`` primitive) and returns the SDK's ``ParsedMessage``
unchanged. When it is a raw ``output_config`` dict, passes it straight to native
``messages.create(output_config=...)`` and returns a ``MessageResponse`` (the base layer
then builds the matching ``ParsedMessage`` from its JSON text).
then builds the matching ``ParsedMessage`` from its JSON text). Streaming requests use
``messages.stream`` with the matching typed or raw output configuration.
"""
header_betas = _pop_anthropic_beta_header(kwargs)
betas = _messages_betas(params, header_betas)
Expand All @@ -335,6 +337,14 @@ async def _amessages(
if betas:
native_kwargs["betas"] = betas
native_kwargs.update(kwargs)
if params.stream:
if is_structured_output_type(params.output_format):
native_kwargs["output_format"] = params.output_format
else:
native_kwargs["output_config"] = normalize_output_config(
cast("dict[str, Any]", params.output_format)
)
return self._stream_messages_async(use_beta=use_beta, **native_kwargs)
if is_structured_output_type(params.output_format):
with _translating_nonstreaming_guard(self, params.max_tokens):
parsed = await messages_resource.parse(output_format=params.output_format, **native_kwargs)
Expand Down
4 changes: 3 additions & 1 deletion src/any_llm/providers/anthropic/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -307,7 +307,6 @@ def _create_openai_chunk_from_anthropic_chunk(chunk: Any, model_id: str) -> Chat
delta["extra_content"] = {"anthropic": {"stop_details": stop_details}}

elif isinstance(chunk, MessageStopEvent):
finish_reason = None
if hasattr(chunk, "message") and chunk.message.usage:
anthropic_usage = chunk.message.usage
cache_read = anthropic_usage.cache_read_input_tokens or 0
Expand All @@ -319,6 +318,9 @@ def _create_openai_chunk_from_anthropic_chunk(chunk: Any, model_id: str) -> Chat
"total_tokens": total_prompt_tokens + anthropic_usage.output_tokens,
"prompt_tokens_details": PromptTokensDetails(cached_tokens=cache_read) if cache_read else None,
}
# The stop event carries no delta or finish_reason, only usage. Leave choices
# empty so it matches the trailing usage-only chunk OpenAI-compatible providers emit.
return ChatCompletionChunk.model_validate(chunk_dict)

choice = {
"index": 0,
Expand Down
41 changes: 28 additions & 13 deletions src/any_llm/providers/bedrock/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,23 @@
# Titan, Llama, ...) have no equivalent verified mechanism, so they keep raising.
_STRUCTURED_OUTPUT_TOOL_NAME = "any_llm_structured_output"

_FinishReason = Literal["stop", "length", "tool_calls", "content_filter", "function_call"]

# The Converse API reports nine stop reasons (botocore's bedrock-runtime service model, shape
# "StopReason"). OpenAI has no counterpart for "stop_sequence", "malformed_model_output" and
# "malformed_tool_use", so those fall through to the "stop" default along with any reason a
# future service model adds. The rest do have one, and without it a guardrail block or a
# context overflow looks like a normal completion to callers, including the structured-output
# guard in any_llm.py that is supposed to raise ContentFilterFinishReasonError.
BEDROCK_STOP_REASON_TO_FINISH_REASON: dict[str, _FinishReason] = {
"end_turn": "stop",
"max_tokens": "length",
"model_context_window_exceeded": "length",
"tool_use": "tool_calls",
"content_filtered": "content_filter",
"guardrail_intervened": "content_filter",
}

REASONING_EFFORT_TO_THINKING_BUDGETS = {
"minimal": 1024,
"low": 2048,
Expand All @@ -46,6 +63,13 @@
}


def _map_stop_reason(stop_reason: Any) -> _FinishReason:
"""Map a Converse API stopReason onto the OpenAI finish_reason vocabulary."""
if not isinstance(stop_reason, str):
return "stop"
return BEDROCK_STOP_REASON_TO_FINISH_REASON.get(stop_reason, "stop")


def _is_anthropic_model(model_id: str) -> bool:
"""Return True if the Bedrock model id refers to an Anthropic Claude model.

Expand Down Expand Up @@ -521,8 +545,7 @@ def _convert_response(response: dict[str, Any]) -> ChatCompletion:
)

content = "".join(content_parts)
stop_reason = response.get("stopReason")
finish_reason: Literal["stop", "length"] = "length" if stop_reason == "max_tokens" else "stop"
finish_reason = _map_stop_reason(response.get("stopReason"))

message = ChatCompletionMessage(
role="assistant",
Expand All @@ -534,9 +557,7 @@ def _convert_response(response: dict[str, Any]) -> ChatCompletion:
choices_out.append(
Choice(
index=0,
finish_reason=cast(
"Literal['stop', 'length', 'tool_calls', 'content_filter', 'function_call']", finish_reason
),
finish_reason=finish_reason,
message=message,
)
)
Expand Down Expand Up @@ -570,7 +591,7 @@ def _create_openai_chunk_from_aws_chunk(

content: str | None = None
reasoning_content: str | None = None
finish_reason: Literal["stop", "length", "tool_calls"] | None = None
finish_reason: _FinishReason | None = None
tool_call: ChoiceDeltaToolCall | None = None
usage: CompletionUsage | None = None

Expand Down Expand Up @@ -616,13 +637,7 @@ def _create_openai_chunk_from_aws_chunk(
),
)
elif "messageStop" in chunk:
stop_reason = chunk["messageStop"]["stopReason"]
if stop_reason == "max_tokens":
finish_reason = "length"
elif stop_reason == "tool_use":
finish_reason = "tool_calls"
else:
finish_reason = "stop"
finish_reason = _map_stop_reason(chunk["messageStop"]["stopReason"])
elif "messageStart" in chunk:
content = ""
elif "metadata" in chunk:
Expand Down
85 changes: 65 additions & 20 deletions src/any_llm/providers/deepseek/deepseek.py
Original file line number Diff line number Diff line change
@@ -1,22 +1,49 @@
import re
from collections.abc import AsyncIterator
from typing import Any

from typing_extensions import override

from any_llm.exceptions import InvalidRequestError
from any_llm.providers.deepseek.utils import (
_inject_cached_tokens,
_inject_cached_tokens_chunk,
_inject_reasoning_extra_content,
_preprocess_messages,
)
from any_llm.providers.openai.base import BaseOpenAIProvider
from any_llm.types.completion import ChatCompletion, ChatCompletionChunk, CompletionParams
from any_llm.types.completion import ChatCompletion, ChatCompletionChunk, CompletionParams, ReasoningEffort

# The two legacy API model names being discontinued 2026-07-24 in favor of deepseek-v4-flash /
# deepseek-v4-pro. See https://api-docs.deepseek.com/updates/#date-2026-04-24. They hard-code
# their own thinking behavior (non-thinking / thinking respectively) and don't need, and may not
# accept, the `thinking` request toggle added below for the new model family.
_LEGACY_MODEL_IDS = frozenset({"deepseek-chat", "deepseek-reasoner"})
# Each entry maps an any-llm effort to DeepSeek's top-level effort and thinking toggle.
# DeepSeek Chat accepts low, high, and max, maps medium and xhigh to high, and does not
# accept OpenAI's minimal value.
# https://api-docs.deepseek.com/guides/thinking_mode/
_REASONING_CONTROLS: dict[ReasoningEffort | None, tuple[str | None, str | None]] = {
None: (None, None),
"auto": (None, None),
"none": (None, "disabled"),
"low": ("low", "enabled"),
"medium": ("high", "enabled"),
"high": ("high", "enabled"),
"xhigh": ("high", "enabled"),
"max": ("max", "enabled"),
}

# These normalized fields are absent from the current DeepSeek Chat schema, except for the two
# penalty fields, which the API marks deprecated and ineffective. They remain accepted by
# any_llm's shared interface for compatibility but are not sent to DeepSeek.
# https://api-docs.deepseek.com/api/create-chat-completion
_UNSUPPORTED_DEEPSEEK_FIELDS = frozenset(
{
"frequency_penalty",
"logit_bias",
"n",
"parallel_tool_calls",
"presence_penalty",
"seed",
"service_tier",
}
)


class DeepseekProvider(BaseOpenAIProvider):
Expand All @@ -36,25 +63,43 @@ class DeepseekProvider(BaseOpenAIProvider):
def _convert_completion_params(params: CompletionParams, **kwargs: Any) -> dict[str, Any]:
"""DeepSeek only accepts ``max_tokens``, not ``max_completion_tokens``.

Also maps ``reasoning_effort`` to DeepSeek's ``thinking`` toggle for the V4 model
family. DeepSeek's V4 models default to thinking mode ENABLED when the toggle is
omitted from the request (see https://api-docs.deepseek.com/guides/thinking_mode), so
any_llm explicitly defaults it to disabled here -- matching the legacy ``deepseek-chat``
behavior -- unless the caller opts in via ``reasoning_effort``. A caller-supplied
``extra_body`` override is respected and not clobbered.
DeepSeek's V4 models default to enabled thinking with high effort, so ``None`` and the
normalized ``auto`` sentinel leave both controls absent. An explicit ``none`` uses the
provider's thinking toggle. Caller-supplied ``extra_body`` values take precedence.
"""
converted_params = BaseOpenAIProvider._convert_completion_params(params, **kwargs)
if "max_completion_tokens" in converted_params:
converted_params["max_tokens"] = converted_params.pop("max_completion_tokens")

if params.model_id not in _LEGACY_MODEL_IDS:
# ``"auto"`` means "no explicit reasoning requested" (BaseOpenAIProvider._acompletion
# normalizes it to this provider's default before we run), so it is treated the same
# as ``None``/``"none"`` here -- matching every other provider's converter and keeping
# this self-contained even if called directly with ``"auto"``.
thinking_disabled = params.reasoning_effort in (None, "none", "auto")
extra_body = converted_params.setdefault("extra_body", {})
extra_body.setdefault("thinking", {"type": "disabled" if thinking_disabled else "enabled"})
user_id = converted_params.pop("user", None)
for field in _UNSUPPORTED_DEEPSEEK_FIELDS:
converted_params.pop(field, None)

converted_params.pop("reasoning_effort", None)
controls = _REASONING_CONTROLS.get(params.reasoning_effort)
if controls is None:
msg = f"reasoning_effort {params.reasoning_effort!r} is not supported by DeepSeek Chat"
raise InvalidRequestError(msg, provider_name=DeepseekProvider.PROVIDER_NAME)
reasoning_effort, thinking_type = controls
if reasoning_effort is not None:
converted_params["reasoning_effort"] = reasoning_effort
thinking = {"type": thinking_type} if thinking_type is not None else None

if user_id is not None or thinking is not None:
extra_body = dict(converted_params.get("extra_body") or {})
converted_params["extra_body"] = extra_body
if user_id is not None and "user_id" not in extra_body:
# DeepSeek's user_id contract is stricter than any-llm's shared user field.
# https://api-docs.deepseek.com/quick_start/rate_limit/#setting-user_id
if re.fullmatch(r"[a-zA-Z0-9_-]{1,512}", user_id) is None:
msg = (
"DeepSeek user_id must contain only ASCII letters, digits, underscores, or hyphens "
"and be between 1 and 512 characters"
)
raise InvalidRequestError(msg, provider_name=DeepseekProvider.PROVIDER_NAME)
extra_body["user_id"] = user_id
if thinking is not None:
extra_body.setdefault("thinking", thinking)
return converted_params

@staticmethod
Expand Down
18 changes: 8 additions & 10 deletions src/any_llm/providers/deepseek/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -44,15 +44,13 @@ def _convert_structured_type_to_deepseek_json(
return modified_messages


def _reinject_reasoning_content(messages: list[dict[str, Any]]) -> list[dict[str, Any]]:
"""Restore ``reasoning_content`` on replayed assistant tool-call turns.
def _reinject_reasoning_content(messages: list[dict[str, Any]], *, replay_reasoning: bool) -> list[dict[str, Any]]:
"""Restore ``reasoning_content`` when the current request carries tools.

DeepSeek's thinking mode requires ``reasoning_content`` to be passed back verbatim on any
assistant turn that performed a tool call, or the API returns a 400. any_llm's shared
message serialization (``AnyLLM.acompletion``) strips the normalized ``reasoning`` field
before replaying a ``ChatCompletionMessage`` back as a request message, so we restore it
here from the ``extra_content["deepseek"]`` side-channel populated by
``_inject_reasoning_extra_content`` when the response was first received.
DeepSeek requires all previous assistant reasoning to be replayed when ``tools`` is present,
including turns that did not call a tool. any_llm's shared message serialization strips the
normalized ``reasoning`` field, so this restores it from the provider side-channel populated
by ``_inject_reasoning_extra_content``.

Reference: https://api-docs.deepseek.com/guides/thinking_mode#tool-calls

Expand All @@ -64,7 +62,7 @@ def _reinject_reasoning_content(messages: list[dict[str, Any]]) -> list[dict[str
for message in messages:
extra_content = message.get("extra_content")
cleaned = {k: v for k, v in message.items() if k != "extra_content"} if extra_content is not None else message
if message.get("role") == "assistant" and message.get("tool_calls") is not None:
if replay_reasoning and message.get("role") == "assistant":
deepseek_extra = extra_content.get("deepseek") if isinstance(extra_content, dict) else None
if isinstance(deepseek_extra, dict) and isinstance(deepseek_extra.get("reasoning_content"), str):
result.append({**cleaned, "reasoning_content": deepseek_extra["reasoning_content"]})
Expand All @@ -81,7 +79,7 @@ def _preprocess_messages(params: CompletionParams) -> CompletionParams:
params.response_format = {"type": "json_object"}
params.messages = modified_messages

params.messages = _reinject_reasoning_content(params.messages)
params.messages = _reinject_reasoning_content(params.messages, replay_reasoning=params.tools is not None)

return params

Expand Down
Loading