Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -1122,6 +1122,10 @@ Isolation model:
- After `kanban.failure_limit` consecutive non-success attempts on the
same task (default: 2), the dispatcher auto-blocks it to prevent spin
loops.
- Opt-in exception: with `kanban.retriage_on_timeout: true`, a breaker
trip caused by consecutive **timeouts** sends the task back to Triage
for decomposition (once per task) instead of blocking it — timeouts
are deterministic, so splitting beats blind retries.

Full user-facing docs: `website/docs/user-guide/features/kanban.md`.

Expand Down
24 changes: 23 additions & 1 deletion gateway/kanban_watchers.py
Original file line number Diff line number Diff line change
Expand Up @@ -164,7 +164,11 @@ async def _kanban_notifier_watcher(self, interval: float = 5.0) -> None:

# "status" covers dashboard drag-drop and `_set_status_direct()`
# writes — surface those transitions to subscribers too.
TERMINAL_KINDS = ("completed", "blocked", "gave_up", "crashed", "timed_out", "status", "archived", "unblocked")
# "retriaged" (kanban.retriage_on_timeout) is delivered so a
# subscriber who just saw the task's `timed_out` event also sees
# that the dispatcher recovered it into Triage for decomposition
# rather than giving up.
TERMINAL_KINDS = ("completed", "blocked", "gave_up", "crashed", "timed_out", "retriaged", "status", "archived", "unblocked")
# Subscriptions are removed only when the task reaches a truly final
# status (done / archived). We used to also unsub on any terminal
# event kind (gave_up / crashed / timed_out / blocked), but that
Expand Down Expand Up @@ -874,6 +878,23 @@ async def _kanban_dispatcher_watcher(self) -> None:
)
failure_limit = _kb.DEFAULT_FAILURE_LIMIT

# Retriage-on-timeout: opt-in — a task whose breaker trips on
# consecutive timeouts goes back to Triage for decomposition
# instead of blocking (see kanban_db._record_task_failure).
retriage_on_timeout = bool(kanban_cfg.get("retriage_on_timeout", False))
if retriage_on_timeout:
logger.info("kanban dispatcher: retriage_on_timeout enabled")
_ad_enabled, _ = _resolve_auto_decompose_settings(_load_config)
if not _ad_enabled:
# Not fatal — manual `hermes kanban decompose` still works —
# but without auto-decompose a retriaged task sits in Triage
# until someone acts, which is easy to miss.
logger.warning(
"kanban dispatcher: retriage_on_timeout is enabled but "
"auto_decompose is disabled — retriaged tasks will wait "
"in Triage for a manual decompose"
)

# Read stale_timeout_seconds — 0 disables stale detection.
raw_stale = kanban_cfg.get("dispatch_stale_timeout_seconds", 0)
try:
Expand Down Expand Up @@ -1022,6 +1043,7 @@ def _tick_once_for_board(slug: str) -> "Optional[object]":
stale_timeout_seconds=stale_timeout_seconds,
default_assignee=default_assignee,
max_in_progress_per_profile=max_in_progress_per_profile,
retriage_on_timeout=retriage_on_timeout,
)
except sqlite3.DatabaseError as exc:
if _is_corrupt_board_db_error(exc):
Expand Down
11 changes: 11 additions & 0 deletions hermes_cli/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -2786,6 +2786,17 @@ def _ensure_hermes_home_managed(home: Path):
# same task/profile (spawn_failed, timed_out, or crashed). Reassignment
# resets the streak for the new profile.
"failure_limit": 2,
# When true, a task whose circuit breaker trips on consecutive
# TIMEOUTS is sent back to Triage for decomposition (with a
# failure-context block appended to its body) instead of being
# auto-blocked. A timeout is deterministic — blind retries of a
# task that needs more than max_runtime_seconds fail identically
# — so subdivision is the productive recovery. At most one
# retriage per task; the second trip blocks normally. Works best
# with auto_decompose: true (otherwise the task waits in Triage
# for a manual `hermes kanban decompose`). Crashes and spawn
# failures never retriage. Default off — opt-in behavior change.
"retriage_on_timeout": False,
# Worker stdout/stderr logs rotate at spawn time. Defaults preserve
# the historical 2 MiB + one-backup behavior; long-running workers can
# raise these to keep more early failure evidence.
Expand Down
15 changes: 14 additions & 1 deletion hermes_cli/kanban.py
Original file line number Diff line number Diff line change
Expand Up @@ -320,7 +320,13 @@ def build_parser(parent_subparsers: argparse._SubParsersAction) -> argparse.Argu
"worktree under the project's primary repo with a "
"deterministic branch. See `hermes project list`.")
p_create.add_argument("--tenant", default=None, help="Tenant namespace")
p_create.add_argument("--priority", type=int, default=0, help="Priority tiebreaker")
p_create.add_argument(
"--priority", type=int, default=None,
help=(
"Priority tiebreaker (higher = picked sooner). Omitted: inherit "
"the highest parent priority when --parent is given, else 0."
),
)
p_create.add_argument("--triage", action="store_true",
help="Park in triage — a specifier will flesh out the spec and promote to todo")
p_create.add_argument("--idempotency-key", default=None,
Expand Down Expand Up @@ -2246,11 +2252,13 @@ def _coerce_positive_int(value):
max_spawn = cli_max if cli_max is not None else _coerce_positive_int(
_kanban_cfg.get("max_spawn")
)
retriage_on_timeout = bool(_kanban_cfg.get("retriage_on_timeout", False))
except Exception:
default_assignee = None
max_in_progress_per_profile = None
max_in_progress = None
max_spawn = getattr(args, "max", None)
retriage_on_timeout = False
with kb.connect_closing() as conn:
res = kb.dispatch_once(
conn,
Expand All @@ -2260,12 +2268,14 @@ def _coerce_positive_int(value):
failure_limit=getattr(args, "failure_limit", kb.DEFAULT_SPAWN_FAILURE_LIMIT),
default_assignee=default_assignee,
max_in_progress_per_profile=max_in_progress_per_profile,
retriage_on_timeout=retriage_on_timeout,
)
if getattr(args, "json", False):
print(json.dumps({
"reclaimed": res.reclaimed,
"crashed": res.crashed,
"timed_out": res.timed_out,
"retriaged": res.retriaged,
"stale": res.stale,
"auto_blocked": res.auto_blocked,
"promoted": res.promoted,
Expand All @@ -2289,6 +2299,9 @@ def _coerce_positive_int(value):
print(f"Timed out: {len(res.timed_out)}")
if res.timed_out:
print(f" {', '.join(res.timed_out)}")
if res.retriaged:
print(f"Retriaged: {len(res.retriaged)}")
print(f" {', '.join(res.retriaged)}")
print(f"Stale: {len(res.stale)}")
if res.stale:
print(f" {', '.join(res.stale)}")
Expand Down
Loading