Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions plugins/model-providers/anthropic/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,7 @@ def fetch_models(
base_url="https://api.anthropic.com",
auth_type="api_key",
default_aux_model="claude-haiku-4-5-20251001",
supports_multimodal_tool_content=True,
)

register_provider(anthropic)
1 change: 1 addition & 0 deletions plugins/model-providers/openai-codex/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@
env_vars=(), # OAuth external — no API key
base_url="https://chatgpt.com/backend-api/codex",
auth_type="oauth_external",
supports_multimodal_tool_content=True,
)

register_provider(openai_codex)
1 change: 1 addition & 0 deletions plugins/model-providers/openrouter/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -103,6 +103,7 @@ def build_api_kwargs_extras(
signup_url="https://openrouter.ai/keys",
base_url="https://openrouter.ai/api/v1",
models_url="https://openrouter.ai/api/v1/models",
supports_multimodal_tool_content=True,
fallback_models=(
"anthropic/claude-sonnet-4.6",
"openai/gpt-5.4",
Expand Down
1 change: 1 addition & 0 deletions plugins/model-providers/xai/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@
env_vars=("XAI_API_KEY",),
base_url="https://api.x.ai/v1",
auth_type="api_key",
supports_multimodal_tool_content=True,
)

register_provider(xai)
7 changes: 7 additions & 0 deletions providers/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -68,6 +68,13 @@ class ProviderProfile:
# ── Client-level quirks (set once at client construction) ─
default_headers: dict[str, str] = field(default_factory=dict)

# Whether the provider accepts multimodal content (list of text + image_url
# parts) inside ``role: "tool"`` messages. Most OpenAI-compatible providers
# only accept plain-string content for tool messages even when they support
# images in user messages. Default False — only providers known to support
# it should set True (e.g. Anthropic, OpenAI GPT-4o).
supports_multimodal_tool_content: bool = False

# ── Request-level quirks ─────────────────────────────────
# Temperature: None = use caller's default, OMIT_TEMPERATURE = don't send
fixed_temperature: Any = None
Expand Down
25 changes: 24 additions & 1 deletion run_agent.py
Original file line number Diff line number Diff line change
Expand Up @@ -3182,6 +3182,24 @@ def _model_supports_vision(self) -> bool:
except Exception:
return False

def _provider_supports_multimodal_tool_content(self) -> bool:
"""Return True if the provider accepts multimodal content in tool messages.

Many providers support images in user messages but require string-only
content in ``role: "tool"`` messages. This checks the provider profile
flag; defaults to True for known multimodal-friendly providers
(OpenAI, Anthropic) and False otherwise. See #27344.
"""
try:
from providers import get_provider_profile
profile = get_provider_profile(self.provider)
if profile is not None:
return bool(profile.supports_multimodal_tool_content)
except Exception:
pass
# Conservative default: assume no multimodal tool support unless declared.
return False

def _preprocess_anthropic_content(self, content: Any, role: str) -> Any:
if not self._content_has_image_parts(content):
return content
Expand Down Expand Up @@ -3321,7 +3339,12 @@ def _tool_result_content_for_active_model(self, tool_name: str, result: Any) ->
return content

if self._model_supports_vision():
return content
# Even if the model supports images in user messages, the provider
# may not accept multimodal content (list of text+image parts) inside
# tool-role messages. Check the provider profile flag; fall back to
# text summary when not supported. (#27344)
if self._provider_supports_multimodal_tool_content():
return content

summary = _multimodal_text_summary(result)
if tool_name == "computer_use":
Expand Down
34 changes: 33 additions & 1 deletion tests/tools/test_computer_use.py
Original file line number Diff line number Diff line change
Expand Up @@ -618,6 +618,7 @@ def test_computer_use_image_result_preserved_for_vision_model(self):
from run_agent import AIAgent

agent = object.__new__(AIAgent)
agent.provider = "openrouter"
result = {
"_multimodal": True,
"content": [
Expand All @@ -626,12 +627,43 @@ def test_computer_use_image_result_preserved_for_vision_model(self):
],
}

with patch.object(agent, "_model_supports_vision", return_value=True):
with patch.object(agent, "_model_supports_vision", return_value=True), \
patch.object(agent, "_provider_supports_multimodal_tool_content", return_value=True):
content = agent._tool_result_content_for_active_model("computer_use", result)

assert content is result["content"]
assert any(part.get("type") == "image_url" for part in content)

def test_computer_use_image_result_downgraded_for_vision_model_without_tool_support(self):
"""Vision-capable model whose provider doesn't support multimodal tool content.

Regression test for #27344: MiMo supports images in user messages but
requires string content in tool messages. The multimodal image should
be downgraded to text_summary, not passed through as a list.
"""
from run_agent import AIAgent

agent = object.__new__(AIAgent)
agent.provider = "xiaomi"
agent.model = "mimo-v2.5"
result = {
"_multimodal": True,
"content": [
{"type": "text", "text": "screen captured"},
{"type": "image_url", "image_url": {"url": "data:image/png;base64,x"}},
],
"text_summary": "screen captured summary",
}

with patch.object(agent, "_model_supports_vision", return_value=True), \
patch.object(agent, "_provider_supports_multimodal_tool_content", return_value=False):
content = agent._tool_result_content_for_active_model("computer_use", result)

# Should NOT contain image_url — must be downgraded to error + text_summary
parsed = json.loads(content)
assert "computer_use returned screenshot/image content" in parsed["error"]
assert "image_url" not in str(content)

def test_other_multimodal_tool_uses_text_summary_for_text_only_model(self):
from run_agent import AIAgent

Expand Down
Loading