From db0fb28b5885e928a1dc91335ec1cecfe834aa59 Mon Sep 17 00:00:00 2001 From: Jeremy Schoemaker Date: Sun, 23 Aug 2026 17:58:32 -0500 Subject: [PATCH] =?UTF-8?q?fix(mistral):=20trim=20reasoning=20at=20the=20o?= =?UTF-8?q?pening=20response=20tag=20in=20the=20streaming=20path=20too=20?= =?UTF-8?q?=F0=9F=94=A7?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #1302 fixed this for the non-streaming converter (_create_mistral_completion_from_response) but the streaming converter (_create_openai_chunk_from_mistral_chunk) never got the same treatment. With stream=True, a Magistral response that wraps its answer in ... inside the thinking delta comes out with delta.content=None on every chunk - the whole answer, tags included, sits only in delta.reasoning.content. A caller following the standard OpenAI streaming contract (accumulate delta.content) gets an empty response with no error. Factored the split/trim logic out of the non-streaming function into a shared _split_response_tag_from_reasoning() helper and call it from both converters, so the two paths can't drift apart a third time. Streaming semantics note: this only catches the case where both and land in the same chunk's reasoning text; a marker split across two chunks is not reassembled, and that limitation is called out in the docstring. --- src/any_llm/providers/mistral/utils.py | 42 ++++++++++--- tests/unit/providers/test_mistral_provider.py | 63 +++++++++++++++++++ 2 files changed, 97 insertions(+), 8 deletions(-) diff --git a/src/any_llm/providers/mistral/utils.py b/src/any_llm/providers/mistral/utils.py index 09402c482..1203ec565 100644 --- a/src/any_llm/providers/mistral/utils.py +++ b/src/any_llm/providers/mistral/utils.py @@ -158,6 +158,34 @@ def _extract_mistral_content_and_reasoning( return content, reasoning_content, thinking_signature +def _split_response_tag_from_reasoning( + content: str | None, reasoning_content: str | None +) -> tuple[str | None, str | None]: + """Recover an answer that Mistral wrapped in `` tags inside the thinking trace. + + Some Mistral reasoning models (e.g. Magistral) sometimes return an empty/None content + and instead wrap the real answer in `...` inside the reasoning + content. When that happens, pull the answer out of the reasoning and trim the reasoning + down to what came before the opening tag. + + Used by both the non-streaming and streaming converters so the two paths can't drift + apart again the way they did when this was only fixed in one of them (see #1302). + + Note: this only handles the tags landing in a single string. In streaming, if the + `` or `` marker itself is split across two chunks, this will not + catch it - each chunk's reasoning text is checked independently. + """ + if ( + content is None + and reasoning_content + and "" in reasoning_content + and "" in reasoning_content + ): + content = reasoning_content.split("")[1].split("")[0] + reasoning_content = reasoning_content.split("")[0] + return content, reasoning_content + + def _create_mistral_completion_from_response( response_data: MistralChatCompletionResponse, model: str ) -> ChatCompletion: @@ -208,14 +236,7 @@ def _create_mistral_completion_from_response( # if the content is none, see if it accidentally ended up in the reasoning content (aka ). # This is a bug in the mistral provider/model return - if ( - content is None - and reasoning_content - and "" in reasoning_content - and "" in reasoning_content - ): - content = reasoning_content.split("")[1].split("")[0] - reasoning_content = reasoning_content.split("")[0] + content, reasoning_content = _split_response_tag_from_reasoning(content, reasoning_content) message = ChatCompletionMessage( role="assistant", @@ -290,6 +311,11 @@ def _create_openai_chunk_from_mistral_chunk(event: CompletionEvent) -> ChatCompl else: content = str(choice.delta.content) + # Mirrors the non-streaming converter's recovery of an answer that Mistral wrapped + # in tags inside the thinking trace (see #1302). This only catches the + # case where both tags land in the same chunk; see _split_response_tag_from_reasoning. + content, reasoning_content = _split_response_tag_from_reasoning(content, reasoning_content) + role = None if choice.delta.role: role = cast("Literal['developer', 'system', 'user', 'assistant', 'tool']", choice.delta.role) diff --git a/tests/unit/providers/test_mistral_provider.py b/tests/unit/providers/test_mistral_provider.py index 7dab1ea7d..7aef8fc87 100644 --- a/tests/unit/providers/test_mistral_provider.py +++ b/tests/unit/providers/test_mistral_provider.py @@ -1351,6 +1351,69 @@ def test_create_openai_chunk_captures_thinking_signature() -> None: assert chunk.choices[0].delta.extra_content == {"mistral": {"signature": "sig-abc"}} +def test_create_openai_chunk_strips_response_block_from_reasoning() -> None: + """Streaming must recover a -wrapped answer the same way non-streaming does. + + #1302 fixed this for `_create_mistral_completion_from_response` but left the streaming + converter untouched, so `stream=True` silently returned an empty `delta.content` for the + same payload. This is the streaming half of that fix. + """ + pytest.importorskip("mistralai") + from mistralai.client.models import TextChunk, ThinkChunk + + from any_llm.providers.mistral.utils import _create_openai_chunk_from_mistral_chunk + + choice = Mock() + choice.index = 0 + choice.delta.content = [ + ThinkChunk(thinking=[TextChunk(text="Let me work it out.The answer is 42.")]) + ] + choice.delta.role = "assistant" + choice.delta.tool_calls = None + choice.finish_reason = None + + event = Mock() + event.data.id = "chatcmpl-abc" + event.data.created = 1_700_000_000 + event.data.model = "magistral-medium-latest" + event.data.choices = [choice] + event.data.usage = None + + chunk = _create_openai_chunk_from_mistral_chunk(event) + + assert chunk.choices[0].delta.content == "The answer is 42." + assert chunk.choices[0].delta.reasoning is not None + assert chunk.choices[0].delta.reasoning.content == "Let me work it out." + + +def test_create_openai_chunk_leaves_plain_reasoning_unchanged() -> None: + """Regression: a normal streaming chunk with plain reasoning and no tags is unaffected.""" + pytest.importorskip("mistralai") + from mistralai.client.models import TextChunk, ThinkChunk + + from any_llm.providers.mistral.utils import _create_openai_chunk_from_mistral_chunk + + choice = Mock() + choice.index = 0 + choice.delta.content = [ThinkChunk(thinking=[TextChunk(text="just thinking, nothing special")])] + choice.delta.role = "assistant" + choice.delta.tool_calls = None + choice.finish_reason = None + + event = Mock() + event.data.id = "chatcmpl-abc" + event.data.created = 1_700_000_000 + event.data.model = "mistral-medium-3-5" + event.data.choices = [choice] + event.data.usage = None + + chunk = _create_openai_chunk_from_mistral_chunk(event) + + assert chunk.choices[0].delta.content is None + assert chunk.choices[0].delta.reasoning is not None + assert chunk.choices[0].delta.reasoning.content == "just thinking, nothing special" + + @pytest.mark.asyncio async def test_timeout_is_translated_to_timeout_ms() -> None: """The seconds-based any-llm ``timeout`` must become the SDK's ``timeout_ms``.