Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -50,23 +50,12 @@ def split_execution_note(plan: "ExecutionPlan") -> str:
"""
if not plan.split_execution_active:
return ""
remote_operations = []
if plan.shell_backend in ("native", "opensandbox", "e2b"):
remote_operations.append("shell commands (`run_command`)")
if plan.filesystem_backend == "sandbox_remote":
remote_operations.append("filesystem operations (`read_file`, `write_file`, `list_files`, etc.)")
guidance = (
"Use `await run_command(...)` for anything that must happen inside the sandbox; "
if plan.shell_backend in ("native", "opensandbox", "e2b")
else ""
)
return (
"**Split-execution mode is active**: your Python code runs in the local environment, "
"but "
+ " and ".join(remote_operations)
+ " execute inside the remote sandbox. "
+ guidance
+ "avoid `os.path`, `open()`, or other local-filesystem calls when targeting sandbox paths."
"but shell commands (`run_command`) and filesystem operations (`read_file`, `write_file`, "
"`list_files`, etc.) execute inside the remote sandbox. "
"Use `await run_command(...)` for anything that must happen inside the sandbox; "
"avoid `os.path`, `open()`, or other local-filesystem calls when targeting sandbox paths."
)


Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -41,7 +41,6 @@
thread_budget_exhausted,
)
from cuga.backend.cuga_graph.nodes.cuga_lite.prompt_utils import (
drop_examples_using_absent_helpers,
PromptUtils,
create_mcp_prompt,
format_apps_for_prompt,
Expand Down Expand Up @@ -455,13 +454,7 @@ async def prepare_tools_and_apps(state: Any, config: Optional[RunnableConfig] =
# gated by enable_filesystem_tools / enable_shell_tool.
_runtime_backends = resolve_runtime_backends(settings, configurable)

filesystem_enabled = _runtime_backends.filesystem != "none"
shell_enabled = _runtime_backends.shell != "none"
few_shot_examples = drop_examples_using_absent_helpers(
few_shot_examples, filesystem_enabled=filesystem_enabled, shell_enabled=shell_enabled
)

if filesystem_enabled or shell_enabled:
if _runtime_backends.filesystem != "none" or _runtime_backends.shell != "none":
cfg = config.get("configurable", {}) if config else {}
# Spawn may set workspace_thread_id to the parent thread while keeping a
# fresh conversation thread_id for checkpointer/chat isolation.
Expand Down Expand Up @@ -521,15 +514,7 @@ async def prepare_tools_and_apps(state: Any, config: Optional[RunnableConfig] =
from cuga.backend.evolve.memory import build_evolve_special_instructions_extension

special_instructions_final = effective_special or ""
# The note must describe the same backends used for tool injection,
# including per-invocation filesystem overrides and unavailable shells.
execution_plan = ExecutionRouter.resolve(settings).model_copy(
update={
"shell_backend": _runtime_backends.shell,
"filesystem_backend": _runtime_backends.filesystem,
}
)
_split_note = split_execution_note(execution_plan)
_split_note = split_execution_note(ExecutionRouter.resolve(settings))
if _split_note:
special_instructions_final = (special_instructions_final + "\n\n" + _split_note).strip()
evolve_extension = await build_evolve_special_instructions_extension(
Expand Down Expand Up @@ -761,8 +746,7 @@ async def _wrapped(*args, **kwargs):
special_instructions=special_instructions_final,
skills_enabled=skills_enabled,
skills_prompt_section=skills_prompt_section,
enable_shell_tool=shell_enabled,
enable_filesystem_tools=filesystem_enabled,
enable_shell_tool=getattr(settings.advanced_features, "enable_shell_tool", False),
sandbox_env_info=get_sandbox_env_description(),
has_knowledge=has_knowledge_tools,
few_shot_examples=few_shot_examples,
Expand All @@ -780,7 +764,7 @@ async def _wrapped(*args, **kwargs):
)
else:
logger.info(
"Using static CugaLite prompt with filtered few-shot messages "
"Using static CugaLite prompt; dynamic few-shot injection skipped "
"(enable_find_tools={} few_shot_turns={})",
enable_find_tools,
len(few_shot_examples),
Expand Down
45 changes: 0 additions & 45 deletions src/cuga/backend/cuga_graph/nodes/cuga_lite/prompt_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,6 @@
import functools
import json
import os
import re
from typing import Any, Dict, List, Optional

from cuga.config import settings
Expand Down Expand Up @@ -793,41 +792,6 @@ def normalize_mcp_few_shot_examples(raw: Any) -> List[Dict[str, str]]:
return out


FILESYSTEM_TOOL_NAMES = (
"read_file",
"write_file",
"edit_file",
"list_files",
"make_directory",
"move_file",
"search_files",
"get_file_info",
)


def drop_examples_using_absent_helpers(
examples: Optional[List[Dict[str, str]]], *, filesystem_enabled: bool, shell_enabled: bool = True
) -> List[Dict[str, str]]:
"""Drop the whole transcript if it calls an unavailable runtime helper.

Turns form a conversation: removing individual calls leaves orphaned outputs
and final answers claiming work that was never shown. Match complete helper
calls, so prose such as "do not use read_file" and longer tool names survive.
"""
if not examples or (filesystem_enabled and shell_enabled):
return list(examples or [])

absent = list(FILESYSTEM_TOOL_NAMES) if not filesystem_enabled else []
if not shell_enabled:
absent.append("run_command")
if examples and absent:
calls = re.compile(r"(?<![\w.])(?:" + "|".join(map(re.escape, absent)) + r")\s*\(")
if any(calls.search(str(ex.get("content", ""))) for ex in examples):
logger.debug("Dropped few-shot conversation demonstrating disabled runtime helpers")
return []
return list(examples or [])


def create_mcp_prompt(
tools,
base_prompt=None,
Expand All @@ -844,7 +808,6 @@ def create_mcp_prompt(
skills_enabled: bool = False,
skills_prompt_section: str = "",
enable_shell_tool: bool = False,
enable_filesystem_tools: bool = False,
sandbox_workspace: str = "/workspace",
sandbox_env_info: str = "",
has_knowledge=False,
Expand All @@ -871,7 +834,6 @@ def create_mcp_prompt(
skills_enabled: If True, render the skills block (load_skill, available skills list)
skills_prompt_section: Pre-formatted markdown/XML block from the skills registry
enable_shell_tool: If True, include run_command / npm / sandbox workspace bullets in the prompt (OpenSandbox shell tools; defaults False in settings)
enable_filesystem_tools: If True, the workspace filesystem helpers (read_file, write_file, list_files) are injected into the execution context and may be described in the prompt. Defaults False, matching settings.toml.
sandbox_workspace: Path prefix shown to the agent for sandbox files. Use "/workspace" for opensandbox/e2b (real Docker path) and "." for native/local (relative cwd).
sandbox_env_info: Human-readable OS/environment string shown to the model when shell tools are enabled (e.g. "macOS 14.5" or "Linux (Ubuntu, Docker container)").
has_knowledge: If True, include knowledge-base search guidance in the prompt
Expand All @@ -886,12 +848,6 @@ def create_mcp_prompt(
for tool in tools:
tool_name = tool.name if hasattr(tool, 'name') else str(tool)
tool_desc = tool.description if hasattr(tool, 'description') else "No description"
if tool_name == "run_command" and not enable_filesystem_tools:
tool_desc = (
tool_desc.replace("Write scripts with write_file before running them. ", "")
.replace("; `read_file` accepts both.", ".")
.replace("from `read_file`/`list_files` ", "")
)

params_str = PromptUtils.get_tool_params_str(tool)
params_doc, response_doc = PromptUtils.get_tool_docs(tool)
Expand Down Expand Up @@ -933,7 +889,6 @@ def create_mcp_prompt(
"skills_enabled": skills_enabled,
"skills_prompt_section": skills_prompt_section,
"enable_shell_tool": enable_shell_tool,
"enable_filesystem_tools": enable_filesystem_tools,
"sandbox_workspace": sandbox_workspace,
"sandbox_env_info": sandbox_env_info,
"has_knowledge": has_knowledge,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -18,8 +18,8 @@ and the sandbox tools all share this directory contract).
- `./output/report.xlsx`, `output/report.xlsx`
- `./data/results.json`, `data/results.json`
- `./contacts.txt`, `contacts.txt`
{% if enable_shell_tool %}- `await run_command("mkdir -p ./output")` before writing to a new nested directory
{% endif %}
- `await run_command("mkdir -p ./output")` before writing to a new nested directory

Do **not** hardcode `/workspace/...` or other absolute host paths. The
workspace's host location is managed externally and is not the same in
every environment; only relative-to-CWD references are portable.
Expand Down Expand Up @@ -131,7 +131,7 @@ Your output MUST be one of these two types. **Do not mix them.**
- **Python execution (shell / run_command)**: Install with `uv pip install <pkg>` only. **Verify import** with `python -c "import <pkg>; print('ok')"` or `uv pip show <pkg>` — **never** `python -m pip`, `pip show`, `pip list`, or `python -m <pkg>` (sandbox venvs have no `pip` CLI). **Run** with `python -c '...'` or `python ./script.py` first; if that fails, retry once with `uv run --no-project ...`. **Never** bare `uv run` or `uv run --active`. Legacy `pip install` → `uv pip install`.
- **Python execution (inline CodeAgent blocks)**: restricted imports apply — use tools/`run_command` for shell, packages, and scripts instead of inline `open`/`os`/extra imports unless skills relaxed mode is on.
- **Node/npm execution**: `uv` is Python-only. NEVER prefix Node or npm commands with `uv`; do not use `uv npm`, `uv run node`, or `uv run npm`. Node commands must start with `node ...` only, e.g. `node ./script.js`. npm commands must start with `npm ...` only. Do not use `npm install -g` — global installs are not reachable via `require()`. Always install locally in the workspace: `npm install <package>`.
{% if enable_filesystem_tools %}- **Sandbox working directory**: always write output files and scripts under your CWD using relative paths. Use `await write_file("./script.js", content)` or `await write_file("./script.py", content)` to place scripts — it's cleaner than printf/heredoc shell tricks and handles multiline content correctly. **Write before run**: call `write_file` first and confirm success (no `[write_file error]` prefix) before `run_command("python ./script.py")`; prefer separate code blocks for large scripts. **Python scripts (`.py`)**: every top-level line in `content` must start at column 0 — never indent the body inside `"""..."""` to match your code block (that causes `IndentationError`). `write_file` rejects invalid Python before saving. **Workspace paths**: use workspace-relative paths everywhere (e.g. `./uploads/foo.json`) — they work in `read_file`/`write_file`/`list_files`, `run_command`, and inside scripts (`open('uploads/foo.json')`). Absolute `/workspace/...` is also accepted. Then run with `await run_command("node ./script.js")` or `await run_command("python ./script.py")` (preferred after `uv pip install`) or `await run_command("uv run --no-project ./script.py")` when skill text requires `uv run`. Skills are available at `./skills/<skill_name>/` when enabled.{% endif %}
- **Sandbox working directory**: always write output files and scripts under your CWD using relative paths. Use `await write_file("./script.js", content)` or `await write_file("./script.py", content)` to place scripts — it's cleaner than printf/heredoc shell tricks and handles multiline content correctly. **Write before run**: call `write_file` first and confirm success (no `[write_file error]` prefix) before `run_command("python ./script.py")`; prefer separate code blocks for large scripts. **Python scripts (`.py`)**: every top-level line in `content` must start at column 0 — never indent the body inside `"""..."""` to match your code block (that causes `IndentationError`). `write_file` rejects invalid Python before saving. **Workspace paths**: use workspace-relative paths everywhere (e.g. `./uploads/foo.json`) — they work in `read_file`/`write_file`/`list_files`, `run_command`, and inside scripts (`open('uploads/foo.json')`). Absolute `/workspace/...` is also accepted. Then run with `await run_command("node ./script.js")` or `await run_command("python ./script.py")` (preferred after `uv pip install`) or `await run_command("uv run --no-project ./script.py")` when skill text requires `uv run`. Skills are available at `./skills/<skill_name>/` when enabled.
{% endif %}

**TYPE 2: Return to User with Text**
Expand Down Expand Up @@ -174,7 +174,7 @@ Your output MUST be one of these two types. **Do not mix them.**
7. **NO FUNCTION CALLING JSON:** NEVER output a JSON object for function calling or native tool-call syntax. Always call tools via Python code (`await tool_name(args)`), never via JSON tool-call blocks. Valid outputs: optional brief chain-of-thought then a Python code block, or a final text answer.
8. **NO NAMESPACES OR APP PREFIXES (CRITICAL):** NEVER prepend the application name or any object prefix to a tool call (e.g., ❌ `jira.get_issues()` or ❌ `salesforce.get_accounts()`). You MUST use the exact, standalone function names exactly as they are listed in the **Current Available Tools** section below (e.g., ✅ `get_issues()`). Do not invent, hallucinate, or guess tool names.
9. **NO `globals()`, `eval()`, OR DYNAMIC TOOL LOOKUP (CRITICAL):** Every tool is **already injected** into your execution environment—you can call it **by name** as a normal function. Do **NOT** use `globals()`, `locals()`, `eval()`, `exec()`, or `getattr(..., some_string)` to load or invoke tools from names parsed out of markdown. After `find_tools`, read the tool name from the output and in the **next** code block call **`await exact_function_name(...)`** using the exact identifier from **Current Available Tools** (those functions already exist in context).
10. **MULTILINE REPORTS:** Do not build long markdown or JSON in triple-quoted f-strings (`f"""..."""`) inside inline code blocks. Use `'\n'.join([...])`, `json.dumps()` for dict sections{% if enable_filesystem_tools %}, or `await write_file('./output/report.md', content)`{% endif %}.
10. **MULTILINE REPORTS:** Do not build long markdown or JSON in triple-quoted f-strings (`f"""..."""`) inside inline code blocks. Use `'\n'.join([...])`, `json.dumps()` for dict sections, or `await write_file('./output/report.md', content)`.

### ⚡ Isolated Tools Execution (CRITICAL)
Certain tools and actions MUST be executed in absolute isolation so you can observe their output or the resulting state change in the next turn before writing further logic.
Expand Down
Loading
Loading