diff --git a/tests/tools/test_terminal_tool.py b/tests/tools/test_terminal_tool.py index 8dbce065ce19..bbb29c04d0f2 100644 --- a/tests/tools/test_terminal_tool.py +++ b/tests/tools/test_terminal_tool.py @@ -27,7 +27,7 @@ def test_terminal_schema_advertises_persistent_env_state(): assert "exported environment variables persist between calls" in description assert "activate a virtualenv" in description - assert "do not re-source the same environment before every command" in description + assert "once per session" in description def test_printf_literal_sudo_does_not_trigger_rewrite(monkeypatch): diff --git a/tools/code_execution_tool.py b/tools/code_execution_tool.py index 12c7c285ae0a..3df99f48388f 100644 --- a/tools/code_execution_tool.py +++ b/tools/code_execution_tool.py @@ -2027,25 +2027,22 @@ def build_execute_code_schema(enabled_sandbox_tools: set = None, ) description = ( - "Run a Python script that can call Hermes tools programmatically. " - "Use this when you need 3+ tool calls with processing logic between them, " - "need to filter/reduce large tool outputs before they enter your context, " - "need conditional branching (if X then Y else Z), or need to loop " - "(fetch N pages, process N files, retry on failure).\n\n" - "Use normal tool calls instead when: single tool call with no processing, " - "you need to see the full result and apply complex reasoning, " - "or the task requires interactive user input.\n\n" + "Run a Python script that calls Hermes tools programmatically. " + "Use when you need 3+ tool calls with logic between them: " + "filtering/reducing large outputs before they enter context, " + "conditional branching, or loops (N pages/files, retry on failure). " + "Use normal tool calls for single calls, results you must reason " + "over in full, or anything needing user interaction.\n\n" f"Available via `from hermes_tools import ...`:\n\n" f"{tool_lines}\n\n" "Limits: 5-minute timeout, 50KB stdout cap, max 50 tool calls per script. " "terminal() is foreground-only (no background or pty).\n\n" f"{cwd_note}\n\n" - "Print your final result to stdout. Use Python stdlib (json, re, math, csv, " - "datetime, collections, etc.) for processing between tool calls.\n\n" - "Also available (no import needed — built into hermes_tools):\n" - " json_parse(text: str) — json.loads with strict=False; use for terminal() output with control chars\n" - " shell_quote(s: str) — shlex.quote(); use when interpolating dynamic strings into shell commands\n" - " retry(fn, max_attempts=3, delay=2) — retry with exponential backoff for transient failures" + "Print your final result to stdout; stdlib (json, re, csv, datetime, ...) " + "is available for processing.\n\n" + "Built-in helpers (no import): json_parse(text) — tolerant json.loads for " + "terminal() output; shell_quote(s) — shlex.quote for dynamic shell args; " + "retry(fn, max_attempts=3, delay=2) — exponential backoff for transient failures." ) return { diff --git a/tools/terminal_tool.py b/tools/terminal_tool.py index 8b89393b6786..4191c48ac2ae 100644 --- a/tools/terminal_tool.py +++ b/tools/terminal_tool.py @@ -1057,25 +1057,13 @@ def _transform_sudo_command(command: str | None) -> tuple[str | None, str | None # Tool description for LLM TERMINAL_TOOL_DESCRIPTION = """Execute shell commands on a Linux environment. Filesystem, current working directory, and exported environment variables persist between calls. -Do NOT use cat/head/tail to read files — use read_file instead. -Do NOT use grep/rg/find to search — use search_files instead. -Do NOT use ls to list directories — use search_files(target='files') instead. -Do NOT use sed/awk to edit files — use patch instead. -Do NOT use echo/cat heredoc to create files — use write_file instead. -Reserve terminal for: builds, installs, git, processes, scripts, network, package managers, and anything that needs a shell. -Because exported environment state persists, activate a virtualenv or export setup variables once per session; do not re-source the same environment before every command unless a command proves the shell state was reset. - -Foreground (default): Commands return INSTANTLY when done, even if the timeout is high. Set timeout=300 for long builds/scripts — you'll still get the result in seconds if it's fast. Prefer foreground for short commands. -Background: Set background=true to get a session_id. Almost always pair with notify_on_complete=true — bg without notify runs SILENTLY and you have no way to learn it finished short of calling process(action='poll') yourself. Two legitimate uses: - (1) Long-lived processes that never exit (servers, watchers, daemons) — silent is correct, there's no exit to notify on. - (2) Long-running bounded tasks (tests, builds, deploys, CI pollers, batch jobs) — MUST set notify_on_complete=true. Without it you'll either forget to poll or sit blocked waiting for the user to surface the result. -For servers/watchers, do NOT use shell-level background wrappers (nohup/disown/setsid/trailing '&') in foreground mode. Use background=true so Hermes can track lifecycle and output. -After starting a server, verify readiness with a health check or log signal, then run tests in a separate terminal() call. Avoid blind sleep loops. -Use process(action="poll") for progress checks, process(action="wait") to block until done. -Working directory: Use 'workdir' for per-command cwd. When a command changes the session cwd (cd, pushd), the result includes a "cwd" field with the directory you ended in — trust it instead of prefixing every command with 'cd'. -PTY mode: Set pty=true for interactive CLI tools (Codex, Claude Code, Python REPL). - -Do NOT use vim/nano/interactive tools without pty=true — they hang without a pseudo-terminal. Pipe git output to cat if it might page. +Do NOT use cat/head/tail (use read_file), grep/rg/find/ls (use search_files), sed/awk (use patch), or echo/heredoc file creation (use write_file). Reserve terminal for: builds, installs, git, processes, scripts, network, package managers, and anything that needs a shell. +Environment state persists: activate a virtualenv or export variables once per session, not before every command. + +Foreground (default): returns INSTANTLY when the command finishes, even with a high timeout — set timeout generously for long builds. +Background: set background=true (returns a session_id). Pair with notify_on_complete=true for bounded tasks; leave silent only for servers/daemons that never exit. Never use nohup/setsid/trailing '&' — use background=true so Hermes tracks the process. After starting a server, verify readiness with a health check, then act in a separate call; no blind sleep loops. Manage with process(action="poll"/"wait"). +Working directory: use 'workdir' for per-command cwd. When a command changes the session cwd (cd, pushd), the result includes a "cwd" field — trust it instead of prefixing every command with 'cd'. +PTY: set pty=true for interactive CLIs (they hang without it). Pipe git output to cat if it might page. """ # Global state for environment lifecycle management @@ -3349,7 +3337,7 @@ def check_terminal_requirements() -> bool: }, "background": { "type": "boolean", - "description": "Run the command in the background. Almost always pair with notify_on_complete=true — without it, the process runs silently and you'll have no way to learn it finished short of calling process(action='poll') yourself (easy to forget, leading to silent blindness on long jobs). Two legitimate patterns: (1) Long-lived processes that never exit (servers, watchers, daemons) — these stay silent because there's no exit to notify on. (2) Long-running bounded tasks (tests, builds, deploys, CI pollers, batch jobs) — these MUST set notify_on_complete=true. For short commands, prefer foreground with a generous timeout instead.", + "description": "Run in the background, returning a session_id. Pair with notify_on_complete=true for anything with a defined end (tests, builds, deploys) — without it the process runs silently. Only servers/watchers/daemons that never exit should stay silent. Short commands: prefer foreground with a generous timeout.", "default": False }, "timeout": { @@ -3368,13 +3356,13 @@ def check_terminal_requirements() -> bool: }, "notify_on_complete": { "type": "boolean", - "description": "When true (and background=true), you'll be automatically notified exactly once when the process finishes. **This is the right choice for almost every long-running task** — tests, builds, deployments, multi-item batch jobs, anything that takes over a minute and has a defined end. Use this and keep working on other things; the system notifies you on exit. MUTUALLY EXCLUSIVE with watch_patterns — when both are set, watch_patterns is dropped.", + "description": "With background=true: get exactly one notification when the process exits. The right choice for nearly every bounded long task — set it and keep working. MUTUALLY EXCLUSIVE with watch_patterns (watch_patterns is dropped when both are set).", "default": False }, "watch_patterns": { "type": "array", "items": {"type": "string"}, - "description": "Strings to watch for in background process output. HARD RATE LIMIT: at most 1 notification per 15 seconds per process — matches arriving inside the cooldown are dropped. After 3 consecutive 15-second windows with dropped matches, watch_patterns is automatically disabled for that process and promoted to notify_on_complete behavior (one notification on exit, no more mid-process spam). USE ONLY for truly rare, one-shot mid-process signals on LONG-LIVED processes that will never exit on their own — e.g. ['Application startup complete'] on a server so you know when to hit its endpoint, or ['migration done'] on a daemon. DO NOT use for: (1) end-of-run markers like 'DONE'/'PASS' — use notify_on_complete instead; (2) error patterns like 'ERROR'/'Traceback' in loops or multi-item batch jobs — they fire on every iteration and you'll hit the strike limit fast; (3) anything you'd ever combine with notify_on_complete. When in doubt, choose notify_on_complete. MUTUALLY EXCLUSIVE with notify_on_complete — set one, not both." + "description": "Strings to watch for in background output. ONLY for rare one-shot mid-process signals on processes that never exit (e.g. ['Application startup complete'] on a server). NOT for end-of-run markers (use notify_on_complete) and NOT for per-iteration patterns like 'ERROR' in loops — rate-limited to 1 notification/15s; repeated over-firing auto-disables it and falls back to notify-on-exit. When in doubt, use notify_on_complete. MUTUALLY EXCLUSIVE with notify_on_complete." } }, "required": ["command"]