Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
55 changes: 44 additions & 11 deletions agent/title_generator.py
Original file line number Diff line number Diff line change
Expand Up @@ -315,6 +315,24 @@ def _extract_title_text(content: str) -> str:
return raw.strip("\"'").strip()


def _response_format_rejected(exc: Exception) -> bool:
"""Return whether *exc* is an HTTP 400 rejecting the request shape.

Some OpenAI-compatible backends (vLLM translating ``json_schema`` into a
``guided_grammar`` it can't compile, DeepSeek/Kimi rejecting the type
outright) 400 on strict structured output but accept the looser
``json_object`` mode. A 400 for an unrelated reason (bad auth, invalid
model) will just 400 again on retry and fall through to the outer
handler, so this check only needs to be cheap, not exhaustive.
"""
status = getattr(exc, "status_code", None) or getattr(
getattr(exc, "response", None), "status_code", None
)
if status == 400:
return True
return "error code: 400" in str(exc).lower()


def _clean_title(text: str) -> Optional[str]:
"""Normalize a model-produced title, or None when nothing usable remains."""
title = " ".join((text or "").split())
Expand Down Expand Up @@ -391,18 +409,33 @@ def generate_title(
{"role": "user", "content": user_snippet},
]

call_kwargs = dict(
task="title_generation",
messages=messages,
# A title is a handful of tokens. The old 500-token ceiling let a
# chatty model burn seconds generating prose we then threw away.
max_tokens=64,
temperature=0.3,
timeout=timeout,
main_runtime=main_runtime,
)
try:
response = call_llm(
task="title_generation",
messages=messages,
# A title is a handful of tokens. The old 500-token ceiling let a
# chatty model burn seconds generating prose we then threw away.
max_tokens=64,
temperature=0.3,
timeout=timeout,
main_runtime=main_runtime,
extra_body={"response_format": _TITLE_RESPONSE_FORMAT},
)
try:
response = call_llm(
**call_kwargs,
extra_body={"response_format": _TITLE_RESPONSE_FORMAT},
)
except Exception as schema_exc:
if not _response_format_rejected(schema_exc):
raise
logger.info(
"Title provider rejected json_schema response_format (HTTP "
"400); retrying with json_object"
)
response = call_llm(
**call_kwargs,
extra_body={"response_format": {"type": "json_object"}},
)
content = response.choices[0].message.content or ""
return _clean_title(_extract_title_text(content))
except Exception as e:
Expand Down
63 changes: 63 additions & 0 deletions tests/agent/test_title_generator.py
Original file line number Diff line number Diff line change
Expand Up @@ -92,6 +92,69 @@ def test_truncates_long_titles(self):



def test_retries_with_json_object_on_http_400_status_code(self):
"""A provider that 400s on strict json_schema (status_code attr set,
as the openai SDK does) gets one retry with the looser json_object
mode instead of losing the title outright."""
calls = []

def mock_call_llm(**kwargs):
calls.append(kwargs)
if len(calls) == 1:
exc = RuntimeError("bad request")
exc.status_code = 400
raise exc
resp = MagicMock()
resp.choices = [MagicMock()]
resp.choices[0].message.content = '{"title": "Fix login button"}'
return resp

with patch("agent.title_generator.call_llm", side_effect=mock_call_llm):
assert generate_title("fix the login button") == "Fix login button"

assert len(calls) == 2
assert calls[0]["extra_body"]["response_format"]["type"] == "json_schema"
assert calls[1]["extra_body"] == {"response_format": {"type": "json_object"}}

def test_retries_on_vllm_guided_grammar_rejection(self):
"""vLLM backends that translate json_schema into an uncompilable
guided_grammar reject with a plain HTTP 400 whose message never
mentions "response_format" — only the status must gate the retry."""
calls = []

def mock_call_llm(**kwargs):
calls.append(kwargs)
if len(calls) == 1:
raise RuntimeError(
"Error code: 400 - {'error': {'message': \"guided_grammar "
"has compile_grammar_error: No module named 'xgrammar'\", "
"'type': 'invalid_request_error', 'code': '400'}}"
)
resp = MagicMock()
resp.choices = [MagicMock()]
resp.choices[0].message.content = '{"title": "Debug xgrammar error"}'
return resp

with patch("agent.title_generator.call_llm", side_effect=mock_call_llm):
assert generate_title("why does xgrammar fail") == "Debug xgrammar error"

assert len(calls) == 2

def test_does_not_retry_on_unrelated_error(self):
"""A non-400 failure (e.g. payment/rate-limit) must not retry — it
would just burn a second call for a request shape that was never
the problem."""
calls = []

def mock_call_llm(**kwargs):
calls.append(kwargs)
raise RuntimeError("openrouter 402: credits exhausted")

with patch("agent.title_generator.call_llm", side_effect=mock_call_llm):
assert generate_title("question") is None

assert len(calls) == 1

def test_invokes_failure_callback_on_exception(self):
"""failure_callback must fire so the user sees a warning (issue #15775)."""
captured = []
Expand Down
Loading