Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
40 changes: 40 additions & 0 deletions cron/jobs.py
Original file line number Diff line number Diff line change
Expand Up @@ -3021,6 +3021,33 @@ def list_jobs(include_disabled: bool = False) -> List[Dict[str, Any]]:
return jobs


def _pin_fallback_reasoning_effort(chain: Any, effort: str) -> Any:
"""Copy of a job ``fallback`` chain with every EXPLICIT per-entry
``reasoning_effort`` set to ``effort``; ``None`` when nothing changes.

Accepts the list form and the single-dict form the scheduler accepts
(``cron/scheduler.py`` job-chain resolution). Entries with no effort are
left as-is: they already inherit the job pin at fallback time.
"""
single = isinstance(chain, dict)
entries = [chain] if single else chain
if not isinstance(entries, list):
return None
out, changed = [], False
for entry in entries:
if (
isinstance(entry, dict)
and str(entry.get("reasoning_effort") or "").strip()
and entry.get("reasoning_effort") != effort
):
entry = {**entry, "reasoning_effort": effort}
changed = True
out.append(entry)
if not changed:
return None
return out[0] if single else out


def update_job(job_id: str, updates: Dict[str, Any]) -> Optional[Dict[str, Any]]:
"""Update a job by ID, refreshing derived schedule fields when needed."""
# Block mutation of immutable fields. ``id`` in particular is a filesystem
Expand Down Expand Up @@ -3063,6 +3090,19 @@ def update_job(job_id: str, updates: Dict[str, Any]) -> Optional[Dict[str, Any]]
updates["reasoning_effort"] = _normalize_reasoning_effort(
updates["reasoning_effort"]
)
# The pin must govern the job's OWN fallback chain too: an
# entry with an explicit ``reasoning_effort`` overrides the
# pin on the fallback turn (chat_completion_helpers #21256),
# so `cron edit --reasoning-effort low` used to leave xhigh
# fallbacks running xhigh (t_ef1ba08b). Setting a pin rewrites
# explicit entry values; entries without one already inherit.
# Clearing (None) leaves entries alone; the CLI reports them.
if updates["reasoning_effort"] is not None and "fallback" not in updates:
_fb = _pin_fallback_reasoning_effort(
job.get("fallback"), updates["reasoning_effort"]
)
if _fb is not None:
updates["fallback"] = _fb

# Normalize repeat the same way create_job does. Callers pass
# either the stored dict shape ({"times": N, "completed": M}) or
Expand Down
37 changes: 37 additions & 0 deletions hermes_cli/cron.py
Original file line number Diff line number Diff line change
Expand Up @@ -724,9 +724,46 @@ def cron_edit(args):
print(" Continuity: on (each run sees the previous run's output)")
if updated.get("workdir"):
print(f" Workdir: {updated['workdir']}")
if getattr(args, "reasoning_effort", None) is not None:
from cron.jobs import get_job

after = get_job(job["id"]) or {}
print(f" Reasoning effort: {after.get('reasoning_effort') or 'config default (pin cleared)'}")
for line in _fallback_effort_lines(job, after):
print(line)
return 0


def _fallback_effort_lines(before: dict, after: dict) -> list:
"""One line per job-level fallback entry saying what the effort edit did.

An entry with its own ``reasoning_effort`` overrides the job pin on the
fallback turn, so an edit that leaves one behind must say so (t_ef1ba08b).
"""
def _chain(job):
fb = job.get("fallback")
return [fb] if isinstance(fb, dict) else (fb if isinstance(fb, list) else [])

prev_chain, lines = _chain(before), []
for i, entry in enumerate(_chain(after)):
if not isinstance(entry, dict):
continue
label = f"{entry.get('provider') or '?'}/{entry.get('model') or '?'}"
prev_entry = prev_chain[i] if i < len(prev_chain) and isinstance(prev_chain[i], dict) else {}
prev, cur = prev_entry.get("reasoning_effort"), entry.get("reasoning_effort")
if not str(cur or "").strip():
lines.append(f" Fallback[{i}] {label}: inherits the job effort")
elif prev != cur:
lines.append(f" Fallback[{i}] {label}: reasoning_effort {prev} -> {cur}")
elif cur == after.get("reasoning_effort"):
lines.append(f" Fallback[{i}] {label}: reasoning_effort {cur} (unchanged)")
else:
lines.append(
f" Fallback[{i}] {label}: NOT touched, keeps its own reasoning_effort {cur!r}"
)
return lines


def _job_action(action: str, job_id: str, success_verb: str) -> int:
_stateless_reset = None
if action == "run":
Expand Down
7 changes: 5 additions & 2 deletions hermes_cli/kanban.py
Original file line number Diff line number Diff line change
Expand Up @@ -4339,8 +4339,11 @@ def _cmd_complete(args: argparse.Namespace) -> int:
outcome = _completion_outcome(conn, tid, last_event)
if not done:
failed.append(tid)
print(f"cannot complete {tid}: {outcome or '(unknown id or terminal state)'}",
file=sys.stderr)
print(
f"cannot complete {tid}: "
f"{outcome or kb.explain_complete_refusal(conn, tid, expected_run_id=_worker_run_id_for(tid))}",
file=sys.stderr,
)
else:
after = kb.get_task(conn, tid)
if getattr(after, "status", None) == "review":
Expand Down
42 changes: 42 additions & 0 deletions hermes_cli/kanban_db.py
Original file line number Diff line number Diff line change
Expand Up @@ -23421,6 +23421,48 @@ def latest_run(conn: sqlite3.Connection, task_id: str) -> Optional[Run]:
return Run.from_row(row) if row else None


def explain_complete_refusal(
conn: sqlite3.Connection,
task_id: str,
*,
expected_run_id: Optional[int] = None,
) -> str:
"""Why ``complete_task`` returned False, in one clause, read after the fact.

Replaces the generic "unknown id or terminal state": a worker retrying a
timed-out complete needs to see "already done by <who> at <when>", not
a guess (t_ef1ba08b).
"""
task = get_task(conn, task_id)
if task is None:
return "unknown id"
if task.status in ("done", "archived"):
row = conn.execute(
"SELECT profile, outcome FROM task_runs WHERE task_id = ? AND ended_at IS NOT NULL "
"ORDER BY ended_at DESC, id DESC LIMIT 1",
(task_id,),
).fetchone()
who = (row["profile"] if row else None) or task.assignee or "unknown"
outcome = f", outcome {row['outcome']}" if row and row["outcome"] else ""
when = (
time.strftime("%Y-%m-%d %H:%M:%S %Z", time.localtime(task.completed_at))
if task.completed_at
else "unknown time"
)
state = "already done" if task.status == "done" else "archived (was done)" if task.completed_at else "archived"
return f"{state} by {who} at {when}{outcome}"
if task.status not in ("running", "ready", "blocked", "review"):
return f"status is {task.status!r}; complete needs running/ready/blocked/review"
if expected_run_id is not None and task.current_run_id != expected_run_id:
return (
f"run {expected_run_id} is no longer the current run "
f"(current: {task.current_run_id}); another run owns this card"
)
if not _parents_satisfied(conn, task_id):
return "a parent task is not done"
return f"refused while status is {task.status!r} (state changed concurrently?)"


def latest_summary(conn: sqlite3.Connection, task_id: str) -> Optional[str]:
"""Return the latest non-null ``task_runs.summary`` for ``task_id``.

Expand Down
55 changes: 55 additions & 0 deletions tests/cron/test_per_job_reasoning_effort.py
Original file line number Diff line number Diff line change
Expand Up @@ -97,6 +97,61 @@ def test_update_clears_with_empty_string(self):
update_job(job["id"], {"reasoning_effort": None})
assert get_job(job["id"]).get("reasoning_effort") in (None, "")

# t_ef1ba08b: an explicit per-entry fallback effort overrides the pin on
# the fallback turn, so setting the pin must rewrite those entries.
def test_update_pin_rewrites_explicit_fallback_efforts(self):
job = create_job(prompt="brief", schedule="every 1h", reasoning_effort="xhigh")
chain = [
{"provider": "openai-codex", "model": "gpt-a", "reasoning_effort": "xhigh"},
{"provider": "claude-bpr", "model": "opus-b"},
{"provider": "claude-bpr", "model": "opus-c", "reasoning_effort": "minimal"},
]
update_job(job["id"], {"fallback": chain})
update_job(job["id"], {"reasoning_effort": "low"})
stored = get_job(job["id"])
assert stored["reasoning_effort"] == "low"
assert [e.get("reasoning_effort") for e in stored["fallback"]] == ["low", None, "low"]
assert [e["model"] for e in stored["fallback"]] == ["gpt-a", "opus-b", "opus-c"]

def test_update_pin_rewrites_single_dict_fallback(self):
job = create_job(prompt="brief", schedule="every 1h")
update_job(job["id"], {"fallback": {"provider": "p", "model": "m", "reasoning_effort": "high"}})
update_job(job["id"], {"reasoning_effort": "medium"})
assert get_job(job["id"])["fallback"] == {"provider": "p", "model": "m", "reasoning_effort": "medium"}

def test_clearing_pin_leaves_fallback_entries_alone(self):
job = create_job(prompt="brief", schedule="every 1h", reasoning_effort="xhigh")
chain = [{"provider": "p", "model": "m", "reasoning_effort": "xhigh"}]
update_job(job["id"], {"fallback": chain})
update_job(job["id"], {"reasoning_effort": None})
assert get_job(job["id"])["fallback"] == chain

def test_explicit_fallback_in_same_update_wins(self):
job = create_job(prompt="brief", schedule="every 1h")
chain = [{"provider": "p", "model": "m", "reasoning_effort": "xhigh"}]
update_job(job["id"], {"reasoning_effort": "low", "fallback": chain})
assert get_job(job["id"])["fallback"] == chain

def test_cli_reports_fallback_entries(self):
import hermes_cli.cron as mod
before = {"reasoning_effort": "xhigh", "fallback": [
{"provider": "a", "model": "x", "reasoning_effort": "xhigh"},
{"provider": "b", "model": "y"},
{"provider": "c", "model": "z", "reasoning_effort": "high"},
]}
after = {"reasoning_effort": None, "fallback": before["fallback"]}
lines = mod._fallback_effort_lines(before, after)
assert "Fallback[0] a/x: NOT touched, keeps its own reasoning_effort 'xhigh'" in lines[0]
assert "Fallback[1] b/y: inherits the job effort" in lines[1]
after = {"reasoning_effort": "low", "fallback": [
{"provider": "a", "model": "x", "reasoning_effort": "low"},
{"provider": "b", "model": "y"},
{"provider": "c", "model": "z", "reasoning_effort": "low"},
]}
lines = mod._fallback_effort_lines(before, after)
assert "Fallback[0] a/x: reasoning_effort xhigh -> low" in lines[0]
assert "Fallback[2] c/z: reasoning_effort high -> low" in lines[2]


# ---------------------------------------------------------------------------
# The cronjob tool: validation + threading
Expand Down
51 changes: 51 additions & 0 deletions tests/hermes_cli/test_kanban_complete_refusal.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
"""complete_task refusals name the real reason (t_ef1ba08b).

A worker retrying a timed-out complete used to read "unknown id or terminal
state" for both an unknown id and a card that was already closed.
"""
from __future__ import annotations

from pathlib import Path

import pytest

from hermes_cli import kanban_db as kb


@pytest.fixture
def kanban_home(tmp_path, monkeypatch):
"""Isolated HERMES_HOME with an empty kanban DB."""
home = tmp_path / ".hermes"
home.mkdir()
monkeypatch.setenv("HERMES_HOME", str(home))
monkeypatch.setattr(Path, "home", lambda: tmp_path)
kb.init_db()
return home


def test_unknown_id(kanban_home):
with kb.connect() as conn:
assert kb.explain_complete_refusal(conn, "t_doesnotexist") == "unknown id"


def test_already_done_names_who_and_when(kanban_home):
with kb.connect() as conn:
t = kb.create_task(conn, title="x", assignee="daedalus")
claimed = kb.claim_task(conn, t)
assert claimed is not None
assert kb.complete_task(conn, t, summary="done", expected_run_id=claimed.current_run_id)
assert not kb.complete_task(conn, t, summary="again")
msg = kb.explain_complete_refusal(conn, t)
assert msg.startswith("already done by daedalus at 20"), msg
assert "outcome completed" in msg
assert "unknown id" not in msg


def test_stale_run_is_named(kanban_home):
with kb.connect() as conn:
t = kb.create_task(conn, title="x", assignee="daedalus")
run_id = kb.claim_task(conn, t).current_run_id
stale = run_id + 1000
assert not kb.complete_task(conn, t, summary="s", expected_run_id=stale)
msg = kb.explain_complete_refusal(conn, t, expected_run_id=stale)
assert f"run {stale} is no longer the current run (current: {run_id})" in msg
3 changes: 2 additions & 1 deletion tools/kanban_tools.py
Original file line number Diff line number Diff line change
Expand Up @@ -952,7 +952,8 @@ def _handle_complete(args: dict, **kw) -> str:
)
if not ok:
return tool_error(
f"could not complete {tid} (unknown id or already terminal)"
f"could not complete {tid}: "
f"{kb.explain_complete_refusal(conn, tid, expected_run_id=_worker_run_id(tid))}"
)
run = kb.latest_run(conn, tid)
after = kb.get_task(conn, tid)
Expand Down
Loading