diff --git a/crates/goose-cli/src/commands/web.rs b/crates/goose-cli/src/commands/web.rs index f11e688c1c8f..7ab0be7516f0 100644 --- a/crates/goose-cli/src/commands/web.rs +++ b/crates/goose-cli/src/commands/web.rs @@ -567,28 +567,6 @@ async fn process_message_streaming( )) .await; } - MessageContent::ContextLengthExceeded(msg) => { - let mut sender = sender.lock().await; - let _ = sender - .send(Message::Text( - serde_json::to_string( - &WebSocketMessage::ContextExceeded { - message: msg.msg.clone(), - }, - ) - .unwrap() - .into(), - )) - .await; - - let (summarized_messages, _, _) = - agent.summarize_context(messages.messages()).await?; - SessionManager::replace_conversation( - &session_id, - &summarized_messages, - ) - .await?; - } _ => {} } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/anthropic/context_length_exceeded.json b/crates/goose-cli/src/scenario_tests/recordings/anthropic/context_length_exceeded.json index 9e26dfeeb6e6..abc04fb9085e 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/anthropic/context_length_exceeded.json +++ b/crates/goose-cli/src/scenario_tests/recordings/anthropic/context_length_exceeded.json @@ -1 +1,286 @@ -{} \ No newline at end of file +{ + "4f53cda18c2baa0c0354bb5f9a3ecbe5ed12ab4d8e11ba873c2f11161202b945": { + "input": { + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:59:27.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "messages": [], + "tools": [ + { + "name": "weather_extension__get_weather", + "description": "Get the weather for a location", + "inputSchema": { + "properties": { + "location": { + "description": "The city and state, e.g. San Francisco, CA", + "type": "string" + } + }, + "required": [ + "location" + ], + "type": "object" + } + }, + { + "name": "platform__search_available_extensions", + "description": "Searches for additional extensions available to help complete tasks.\n Use this tool when you're unable to find a specific feature or functionality you need to complete your task, or when standard approaches aren't working.\n These extensions might provide the exact tools needed to solve your problem.\n If you find a relevant one, consider using your tools to enable it.", + "inputSchema": { + "properties": {}, + "required": [], + "type": "object" + }, + "annotations": { + "title": "Discover extensions", + "readOnlyHint": true, + "destructiveHint": false, + "idempotentHint": false, + "openWorldHint": false + } + }, + { + "name": "platform__manage_extensions", + "description": "Tool to manage extensions and tools in goose context.\n Enable or disable extensions to help complete tasks.\n Enable or disable an extension by providing the extension name.\n ", + "inputSchema": { + "properties": { + "action": { + "description": "The action to perform", + "enum": [ + "enable", + "disable" + ], + "type": "string" + }, + "extension_name": { + "description": "The name of the extension to enable", + "type": "string" + } + }, + "required": [ + "action", + "extension_name" + ], + "type": "object" + }, + "annotations": { + "title": "Enable or disable an extension", + "readOnlyHint": false, + "destructiveHint": false, + "idempotentHint": false, + "openWorldHint": false + } + }, + { + "name": "platform__manage_schedule", + "description": "Manage scheduled recipe execution for this goose instance.\n\nActions:\n- \"list\": List all scheduled jobs\n- \"create\": Create a new scheduled job from a recipe file\n- \"run_now\": Execute a scheduled job immediately \n- \"pause\": Pause a scheduled job\n- \"unpause\": Resume a paused job\n- \"delete\": Remove a scheduled job\n- \"kill\": Terminate a currently running job\n- \"inspect\": Get details about a running job\n- \"sessions\": List execution history for a job\n- \"session_content\": Get the full content (messages) of a specific session\n", + "inputSchema": { + "properties": { + "action": { + "enum": [ + "list", + "create", + "run_now", + "pause", + "unpause", + "delete", + "kill", + "inspect", + "sessions", + "session_content" + ], + "type": "string" + }, + "cron_expression": { + "description": "A cron expression for create action. Supports both 5-field (minute hour day month weekday) and 6-field (second minute hour day month weekday) formats. 5-field expressions are automatically converted to 6-field by prepending '0' for seconds.", + "type": "string" + }, + "execution_mode": { + "default": "background", + "description": "Execution mode for create action: 'foreground' or 'background'", + "enum": [ + "foreground", + "background" + ], + "type": "string" + }, + "job_id": { + "description": "Job identifier for operations on existing jobs", + "type": "string" + }, + "limit": { + "default": 50, + "description": "Limit for sessions list", + "type": "integer" + }, + "recipe_path": { + "description": "Path to recipe file for create action", + "type": "string" + }, + "session_id": { + "description": "Session identifier for session_content action", + "type": "string" + } + }, + "required": [ + "action" + ], + "type": "object" + }, + "annotations": { + "title": "Manage scheduled recipes", + "readOnlyHint": false, + "destructiveHint": true, + "idempotentHint": false, + "openWorldHint": false + } + }, + { + "name": "dynamic_task__create_task", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", + "inputSchema": { + "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, + "task_parameters": { + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", + "items": { + "properties": { + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { + "type": "string" + } + }, + "type": "object" + }, + "minItems": 1, + "type": "array" + } + }, + "required": [ + "task_parameters" + ], + "type": "object" + }, + "annotations": { + "title": "Create Dynamic Tasks", + "readOnlyHint": false, + "destructiveHint": false, + "idempotentHint": false, + "openWorldHint": true + } + }, + { + "name": "subagent__execute_task", + "description": "Only use the subagent__execute_task tool when you execute sub recipe task or dynamic task.\n EXECUTION STRATEGY DECISION:\n 1. If the tasks are created with execution_mode, use the execution_mode.\n 2. Execute tasks sequentially unless user explicitly requests parallel execution. PARALLEL: User uses keywords like 'parallel', 'simultaneously', 'at the same time', 'concurrently'\n\n IMPLEMENTATION:\n - Sequential execution: Call this tool multiple times, passing exactly ONE task per call\n - Parallel execution: Call this tool once, passing an ARRAY of all tasks\n\n EXAMPLES:\n User Intent Based:\n - User: 'get weather and tell me a joke' → Sequential (2 separate tool calls, 1 task each)\n - User: 'get weather and joke in parallel' → Parallel (1 tool call with array of 2 tasks)\n - User: 'run these simultaneously' → Parallel (1 tool call with task array)\n - User: 'do task A then task B' → Sequential (2 separate tool calls)", + "inputSchema": { + "properties": { + "execution_mode": { + "default": "sequential", + "description": "Execution strategy for multiple tasks. Use 'sequential' (default) unless user explicitly requests parallel execution with words like 'parallel', 'simultaneously', 'at the same time', or 'concurrently'.", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, + "task_ids": { + "items": { + "description": "Unique identifier for the task", + "type": "string" + }, + "type": "array" + } + }, + "required": [ + "task_ids" + ], + "type": "object" + }, + "annotations": { + "title": "Run tasks in parallel", + "readOnlyHint": false, + "destructiveHint": true, + "idempotentHint": false, + "openWorldHint": true + } + } + ] + }, + "output": { + "message": { + "id": null, + "role": "assistant", + "created": 1760565573, + "content": [ + { + "type": "text", + "text": "I understand. How can I help you today?" + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + }, + "usage": { + "model": "claude-sonnet-4-20250514", + "usage": { + "input_tokens": 2702, + "output_tokens": 13, + "total_tokens": 2715 + } + } + } + } +} \ No newline at end of file diff --git a/crates/goose-cli/src/scenario_tests/recordings/anthropic/image_analysis.json b/crates/goose-cli/src/scenario_tests/recordings/anthropic/image_analysis.json index 86c86709ad12..0184429be104 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/anthropic/image_analysis.json +++ b/crates/goose-cli/src/scenario_tests/recordings/anthropic/image_analysis.json @@ -1,12 +1,12 @@ { "c848f22f273e158c32435d3e72cc999c046dc1a9afdc3efda68ff451f833a185": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:04:16.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:45.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704256, + "created": 1760565465, "content": [ { "type": "text", @@ -17,7 +17,11 @@ "data": "", "mimeType": "image/jpeg" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -151,326 +155,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", - "inputSchema": { - "properties": { - "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", - "items": { - "properties": { - "text_instruction": { - "description": "The text instruction to execute", - "type": "string" - } - }, - "required": [ - "text_instruction" - ], - "type": "object" - }, - "type": "array" - } - }, - "type": "object" - }, - "annotations": { - "title": "Dynamic Task Creation", - "readOnlyHint": false, - "destructiveHint": true, - "idempotentHint": false, - "openWorldHint": true - } - }, - { - "name": "subagent__execute_task", - "description": "Only use the subagent__execute_task tool when you execute sub recipe task or dynamic task.\n EXECUTION STRATEGY DECISION:\n 1. If the tasks are created with execution_mode, use the execution_mode.\n 2. Execute tasks sequentially unless user explicitly requests parallel execution. PARALLEL: User uses keywords like 'parallel', 'simultaneously', 'at the same time', 'concurrently'\n\n IMPLEMENTATION:\n - Sequential execution: Call this tool multiple times, passing exactly ONE task per call\n - Parallel execution: Call this tool once, passing an ARRAY of all tasks\n\n EXAMPLES:\n User Intent Based:\n - User: 'get weather and tell me a joke' → Sequential (2 separate tool calls, 1 task each)\n - User: 'get weather and joke in parallel' → Parallel (1 tool call with array of 2 tasks)\n - User: 'run these simultaneously' → Parallel (1 tool call with task array)\n - User: 'do task A then task B' → Sequential (2 separate tool calls)", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { "execution_mode": { - "default": "sequential", - "description": "Execution strategy for multiple tasks. Use 'sequential' (default) unless user explicitly requests parallel execution with words like 'parallel', 'simultaneously', 'at the same time', or 'concurrently'.", + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", "enum": [ "sequential", "parallel" ], "type": "string" }, - "task_ids": { - "items": { - "description": "Unique identifier for the task", - "type": "string" - }, - "type": "array" - } - }, - "required": [ - "task_ids" - ], - "type": "object" - }, - "annotations": { - "title": "Run tasks in parallel", - "readOnlyHint": false, - "destructiveHint": true, - "idempotentHint": false, - "openWorldHint": true - } - } - ] - }, - "output": { - "message": { - "id": null, - "role": "assistant", - "created": 1753704259, - "content": [ - { - "type": "text", - "text": "I apologize, but I don't see any image shared in our conversation. If you'd like me to analyze an image, you'll need to share it first. However, I notice that I don't currently have access to any image analysis tools in my available extensions. Let me search to see if there are any image-related extensions available." - }, - { - "type": "toolRequest", - "id": "toolu_014qhEwCaj8QWwz1s8xtMH7h", - "toolCall": { - "status": "success", - "value": { - "name": "platform__search_available_extensions", - "arguments": {} - } - } - } - ] - }, - "usage": { - "model": "claude-sonnet-4-20250514", - "usage": { - "input_tokens": 2560, - "output_tokens": 111, - "total_tokens": 2671 - } - } - } - }, - "78cc474ff2d51b9a24df8c35e5c75f256dafb67ff5489af30fcec95cd87790b8": { - "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:04:16.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", - "messages": [ - { - "id": null, - "role": "user", - "created": 1753704256, - "content": [ - { - "type": "text", - "text": "What do you see in this image?" - }, - { - "type": "image", - "data": "", - "mimeType": "image/jpeg" - } - ] - }, - { - "id": null, - "role": "assistant", - "created": 1753704259, - "content": [ - { - "type": "text", - "text": "I apologize, but I don't see any image shared in our conversation. If you'd like me to analyze an image, you'll need to share it first. However, I notice that I don't currently have access to any image analysis tools in my available extensions. Let me search to see if there are any image-related extensions available." - }, - { - "type": "toolRequest", - "id": "toolu_014qhEwCaj8QWwz1s8xtMH7h", - "toolCall": { - "status": "success", - "value": { - "name": "platform__search_available_extensions", - "arguments": {} - } - } - } - ] - }, - { - "id": null, - "role": "user", - "created": 1753704259, - "content": [ - { - "type": "toolResponse", - "id": "toolu_014qhEwCaj8QWwz1s8xtMH7h", - "toolResult": { - "status": "success", - "value": [ - { - "type": "text", - "text": "Extensions available to enable:\n- Codesearch - Query https://codesearch.sqprod.co/ directly from goose\n- google_drive - google_drive\n- Image Generator - Generate images with OpenAI models\n- Linear - Integration to Linear project management tool\n- fetch - Web content fetching and processing capabilities\n- pdfreader - Read and understand large or complex PDFs\n- Data Discovery - Explore data assets, metadata, ownership and lineage\n- gosling - Gosling\n- memory - Memory\n- Web Search - Query the WWW\n- computercontroller - Computer Controller\n- Figma - Interact with Figma and Figjam files\n- Airtable - Interact with Airtable bases and tables\n- Notebook - Control Jupyter notebooks in real time\n- Databricks - Integration to query Databricks data\n- Glean - Search and read internal documents with Glean\n- jetbrains - Jetbrains\n- Regulator - Interact with Regulator\n- blockcell - Deploy static websites to the Blockcell hosting service\n- Blockdata - Querying and analyzing Block data\n- Looker - Explore and talk to data with Looker\n- Snowflake - Integration to query Snowflake data\n- tutorial - Tutorial\n- git - Git version control system integration\n- Gmail - Gmail integration for email management and access\n- QAI Index Search - Search QAI's index directly from goose\n- Google Drive - Google Drive integration for file management and access\n- googlecalendar - Interact with Google Calendar events\n\n\n\nExtensions available to disable:\n- weather_extension\n" - } - ] - } - } - ] - } - ], - "tools": [ - { - "name": "weather_extension__get_weather", - "description": "Get the weather for a location", - "inputSchema": { - "properties": { - "location": { - "description": "The city and state, e.g. San Francisco, CA", - "type": "string" - } - }, - "required": [ - "location" - ], - "type": "object" - } - }, - { - "name": "platform__search_available_extensions", - "description": "Searches for additional extensions available to help complete tasks.\n Use this tool when you're unable to find a specific feature or functionality you need to complete your task, or when standard approaches aren't working.\n These extensions might provide the exact tools needed to solve your problem.\n If you find a relevant one, consider using your tools to enable it.", - "inputSchema": { - "properties": {}, - "required": [], - "type": "object" - }, - "annotations": { - "title": "Discover extensions", - "readOnlyHint": true, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": false - } - }, - { - "name": "platform__manage_extensions", - "description": "Tool to manage extensions and tools in goose context.\n Enable or disable extensions to help complete tasks.\n Enable or disable an extension by providing the extension name.\n ", - "inputSchema": { - "properties": { - "action": { - "description": "The action to perform", - "enum": [ - "enable", - "disable" - ], - "type": "string" - }, - "extension_name": { - "description": "The name of the extension to enable", - "type": "string" - } - }, - "required": [ - "action", - "extension_name" - ], - "type": "object" - }, - "annotations": { - "title": "Enable or disable an extension", - "readOnlyHint": false, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": false - } - }, - { - "name": "platform__manage_schedule", - "description": "Manage scheduled recipe execution for this goose instance.\n\nActions:\n- \"list\": List all scheduled jobs\n- \"create\": Create a new scheduled job from a recipe file\n- \"run_now\": Execute a scheduled job immediately \n- \"pause\": Pause a scheduled job\n- \"unpause\": Resume a paused job\n- \"delete\": Remove a scheduled job\n- \"kill\": Terminate a currently running job\n- \"inspect\": Get details about a running job\n- \"sessions\": List execution history for a job\n- \"session_content\": Get the full content (messages) of a specific session\n", - "inputSchema": { - "properties": { - "action": { - "enum": [ - "list", - "create", - "run_now", - "pause", - "unpause", - "delete", - "kill", - "inspect", - "sessions", - "session_content" - ], - "type": "string" - }, - "cron_expression": { - "description": "A cron expression for create action. Supports both 5-field (minute hour day month weekday) and 6-field (second minute hour day month weekday) formats. 5-field expressions are automatically converted to 6-field by prepending '0' for seconds.", - "type": "string" - }, - "execution_mode": { - "default": "background", - "description": "Execution mode for create action: 'foreground' or 'background'", - "enum": [ - "foreground", - "background" - ], - "type": "string" - }, - "job_id": { - "description": "Job identifier for operations on existing jobs", - "type": "string" - }, - "limit": { - "default": 50, - "description": "Limit for sessions list", - "type": "integer" - }, - "recipe_path": { - "description": "Path to recipe file for create action", - "type": "string" - }, - "session_id": { - "description": "Session identifier for session_content action", - "type": "string" - } - }, - "required": [ - "action" - ], - "type": "object" - }, - "annotations": { - "title": "Manage scheduled recipes", - "readOnlyHint": false, - "destructiveHint": true, - "idempotentHint": false, - "openWorldHint": false - } - }, - { - "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", - "inputSchema": { - "properties": { "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -516,20 +282,24 @@ "message": { "id": null, "role": "assistant", - "created": 1753704262, + "created": 1760565468, "content": [ { "type": "text", - "text": "It looks like there isn't currently an image analysis extension available. If you'd like to share an image, you'll need to share it in a way that I can access it, and we may need to enable an appropriate extension for image analysis. For now, I can help you with other tasks using the currently available tools and extensions. Is there something specific you'd like assistance with?" + "text": "I don't see any image in our conversation. It appears you may have intended to share an image, but it wasn't included in your message.\n\nCould you please try uploading the image again? Once you share it, I'll be happy to describe what I see in it." } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "claude-sonnet-4-20250514", "usage": { - "input_tokens": 3053, - "output_tokens": 82, - "total_tokens": 3135 + "input_tokens": 2708, + "output_tokens": 61, + "total_tokens": 2769 } } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/anthropic/weather_tool.json b/crates/goose-cli/src/scenario_tests/recordings/anthropic/weather_tool.json index d1d8b05f4d6c..728831c77291 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/anthropic/weather_tool.json +++ b/crates/goose-cli/src/scenario_tests/recordings/anthropic/weather_tool.json @@ -1,18 +1,22 @@ { "1bc400a528c54b25f4f1f609481e98e44222b3deaf7eee2c9e640e6345c73861": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:04:16.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:45.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704256, + "created": 1760565465, "content": [ { "type": "text", "text": "tell me what the weather is in Berlin, Germany" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -146,32 +150,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -217,15 +277,15 @@ "message": { "id": null, "role": "assistant", - "created": 1753704259, + "created": 1760565468, "content": [ { "type": "text", - "text": "I'll help you check the weather in Berlin, Germany using the weather extension." + "text": "I'll check the current weather in Berlin, Germany for you." }, { "type": "toolRequest", - "id": "toolu_01ABW2NhcBGvKL6nBj3T3fhv", + "id": "toolu_013JNt4Fn6i4Ls8HTbn8bQjG", "toolCall": { "status": "success", "value": { @@ -236,45 +296,53 @@ } } } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "claude-sonnet-4-20250514", "usage": { - "input_tokens": 2562, - "output_tokens": 76, - "total_tokens": 2638 + "input_tokens": 2710, + "output_tokens": 73, + "total_tokens": 2783 } } } }, - "d51c15f1ede58b5496bba746a4bdffd8ce84526749ce0021969d6ed6f2538a6d": { + "c6f73996d107f0ca731493c3e18108043693c0e63d377190c9359c599a7ef254": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:04:16.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:45.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704256, + "created": 1760565465, "content": [ { "type": "text", "text": "tell me what the weather is in Berlin, Germany" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, { "id": null, "role": "assistant", - "created": 1753704259, + "created": 1760565468, "content": [ { "type": "text", - "text": "I'll help you check the weather in Berlin, Germany using the weather extension." + "text": "I'll check the current weather in Berlin, Germany for you." }, { "type": "toolRequest", - "id": "toolu_01ABW2NhcBGvKL6nBj3T3fhv", + "id": "toolu_013JNt4Fn6i4Ls8HTbn8bQjG", "toolCall": { "status": "success", "value": { @@ -285,16 +353,20 @@ } } } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, { - "id": null, + "id": "msg_17216282-dde0-4344-a137-1e89c85623ad", "role": "user", - "created": 1753704259, + "created": 1760565468, "content": [ { "type": "toolResponse", - "id": "toolu_01ABW2NhcBGvKL6nBj3T3fhv", + "id": "toolu_013JNt4Fn6i4Ls8HTbn8bQjG", "toolResult": { "status": "success", "value": [ @@ -305,7 +377,11 @@ ] } } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -439,32 +515,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -510,20 +642,24 @@ "message": { "id": null, "role": "assistant", - "created": 1753704261, + "created": 1760565471, "content": [ { "type": "text", - "text": "The current weather in Berlin, Germany is cloudy with a temperature of 18°C (64.4°F)." + "text": "The current weather in **Berlin, Germany** is:\n\n- **Temperature**: 18°C (64°F)\n- **Conditions**: Cloudy\n\nIt's a mild day with overcast skies in Berlin right now." } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "claude-sonnet-4-20250514", "usage": { - "input_tokens": 2663, - "output_tokens": 29, - "total_tokens": 2692 + "input_tokens": 2809, + "output_tokens": 53, + "total_tokens": 2862 } } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/anthropic/what_is_your_name.json b/crates/goose-cli/src/scenario_tests/recordings/anthropic/what_is_your_name.json index eacecbe3a5a7..d55f2d9ae4d9 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/anthropic/what_is_your_name.json +++ b/crates/goose-cli/src/scenario_tests/recordings/anthropic/what_is_your_name.json @@ -1,18 +1,22 @@ { "1b998117eba523901ae6a4dbf8caa81a95ea88ef7a84d0434c9b41a26164a2b9": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:04:16.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:44.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704256, + "created": 1760565464, "content": [ { "type": "text", "text": "what is your name" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -146,32 +150,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -217,20 +277,24 @@ "message": { "id": null, "role": "assistant", - "created": 1753704259, + "created": 1760565467, "content": [ { "type": "text", - "text": "I am goose, an AI agent created by Block (the parent company of Square, CashApp, and Tidal). I'm being developed as an open-source software project.\n\nI can help you with various tasks using different tools and extensions. While I have a defined identity as goose, I aim to be helpful, direct, and efficient in assisting you with your needs. Is there something specific I can help you with today?" + "text": "My name is **goose**. I'm a general-purpose AI agent created by Block, the parent company of Square, CashApp, and Tidal. I'm being developed as an open-source software project and can help you with various tasks using different extensions and tools.\n\nIs there something specific you'd like help with today?" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "claude-sonnet-4-20250514", "usage": { - "input_tokens": 2556, - "output_tokens": 97, - "total_tokens": 2653 + "input_tokens": 2704, + "output_tokens": 74, + "total_tokens": 2778 } } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/azure_openai/weather_tool.json b/crates/goose-cli/src/scenario_tests/recordings/azure_openai/weather_tool.json index 9403a20910a3..d6f53bb78fa1 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/azure_openai/weather_tool.json +++ b/crates/goose-cli/src/scenario_tests/recordings/azure_openai/weather_tool.json @@ -1,58 +1,22 @@ { - "21e33b98670d23e1bad3f21da667502d0930b42e34431395e266a4c524620cf1": { + "1bc400a528c54b25f4f1f609481e98e44222b3deaf7eee2c9e640e6345c73861": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:05:28.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:51.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704328, + "created": 1760565471, "content": [ { "type": "text", "text": "tell me what the weather is in Berlin, Germany" } - ] - }, - { - "id": null, - "role": "assistant", - "created": 1753704329, - "content": [ - { - "type": "toolRequest", - "id": "call_C91rmWZSSnUaKx0AtBdwO4Zw", - "toolCall": { - "status": "success", - "value": { - "name": "weather_extension__get_weather", - "arguments": { - "location": "Berlin, Germany" - } - } - } - } - ] - }, - { - "id": null, - "role": "user", - "created": 1753704329, - "content": [ - { - "type": "toolResponse", - "id": "call_C91rmWZSSnUaKx0AtBdwO4Zw", - "toolResult": { - "status": "success", - "value": [ - { - "type": "text", - "text": "The weather in Berlin, Germany is cloudy and 18°C" - } - ] - } - } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -186,32 +150,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -257,38 +277,103 @@ "message": { "id": null, "role": "assistant", - "created": 1753704331, + "created": 1760565472, "content": [ { - "type": "text", - "text": "### Current Weather in Berlin, Germany\n\n- **Condition:** Cloudy\n- **Temperature:** 18°C\n\nIf you need more information or forecasts, let me know!" + "type": "toolRequest", + "id": "call_QK2Zg4k9kTxaNXCyOVPbQ8Pc", + "toolCall": { + "status": "success", + "value": { + "name": "weather_extension__get_weather", + "arguments": { + "location": "Berlin, Germany" + } + } + } } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "gpt-4o-mini-2024-07-18", "usage": { - "input_tokens": 1722, - "output_tokens": 37, - "total_tokens": 1759 + "input_tokens": 1708, + "output_tokens": 20, + "total_tokens": 1728 } } } }, - "1bc400a528c54b25f4f1f609481e98e44222b3deaf7eee2c9e640e6345c73861": { + "198b76aab4abc29c526ddd37cf906dcf9161c94423d8ca772908bef353ece3e8": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:05:28.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:51.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704328, + "created": 1760565471, "content": [ { "type": "text", "text": "tell me what the weather is in Berlin, Germany" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + }, + { + "id": null, + "role": "assistant", + "created": 1760565472, + "content": [ + { + "type": "toolRequest", + "id": "call_QK2Zg4k9kTxaNXCyOVPbQ8Pc", + "toolCall": { + "status": "success", + "value": { + "name": "weather_extension__get_weather", + "arguments": { + "location": "Berlin, Germany" + } + } + } + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + }, + { + "id": "msg_85357fa1-4622-4b5c-bc66-ce79f12038dd", + "role": "user", + "created": 1760565472, + "content": [ + { + "type": "toolResponse", + "id": "call_QK2Zg4k9kTxaNXCyOVPbQ8Pc", + "toolResult": { + "status": "success", + "value": [ + { + "type": "text", + "text": "The weather in Berlin, Germany is cloudy and 18°C" + } + ] + } + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -422,32 +507,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -493,29 +634,24 @@ "message": { "id": null, "role": "assistant", - "created": 1753704329, + "created": 1760565473, "content": [ { - "type": "toolRequest", - "id": "call_C91rmWZSSnUaKx0AtBdwO4Zw", - "toolCall": { - "status": "success", - "value": { - "name": "weather_extension__get_weather", - "arguments": { - "location": "Berlin, Germany" - } - } - } + "type": "text", + "text": "The current weather in Berlin, Germany, is cloudy with a temperature of 18°C." } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "gpt-4o-mini-2024-07-18", "usage": { - "input_tokens": 1680, + "input_tokens": 1750, "output_tokens": 20, - "total_tokens": 1700 + "total_tokens": 1770 } } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/azure_openai/what_is_your_name.json b/crates/goose-cli/src/scenario_tests/recordings/azure_openai/what_is_your_name.json index 8cb9cc44c168..77256c16d837 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/azure_openai/what_is_your_name.json +++ b/crates/goose-cli/src/scenario_tests/recordings/azure_openai/what_is_your_name.json @@ -1,18 +1,22 @@ { "1b998117eba523901ae6a4dbf8caa81a95ea88ef7a84d0434c9b41a26164a2b9": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:05:25.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:47.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704325, + "created": 1760565467, "content": [ { "type": "text", "text": "what is your name" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -146,32 +150,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -217,20 +277,24 @@ "message": { "id": null, "role": "assistant", - "created": 1753704327, + "created": 1760565468, "content": [ { "type": "text", - "text": "My name is goose, an AI agent created by Block, the parent company of Square, CashApp, and Tidal. How can I assist you today?" + "text": "My name is Goose. I'm an AI agent created by Block, the parent company of Square, CashApp, and Tidal. How can I assist you today?" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "gpt-4o-mini-2024-07-18", "usage": { - "input_tokens": 1674, - "output_tokens": 34, - "total_tokens": 1708 + "input_tokens": 1702, + "output_tokens": 35, + "total_tokens": 1737 } } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/google/context_length_exceeded.json b/crates/goose-cli/src/scenario_tests/recordings/google/context_length_exceeded.json deleted file mode 100644 index 9e26dfeeb6e6..000000000000 --- a/crates/goose-cli/src/scenario_tests/recordings/google/context_length_exceeded.json +++ /dev/null @@ -1 +0,0 @@ -{} \ No newline at end of file diff --git a/crates/goose-cli/src/scenario_tests/recordings/google/what_is_your_name.json b/crates/goose-cli/src/scenario_tests/recordings/google/what_is_your_name.json index 9abd50036895..fea4fe308ca6 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/google/what_is_your_name.json +++ b/crates/goose-cli/src/scenario_tests/recordings/google/what_is_your_name.json @@ -1,18 +1,22 @@ { "1b998117eba523901ae6a4dbf8caa81a95ea88ef7a84d0434c9b41a26164a2b9": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:05:27.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:49.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704327, + "created": 1760565469, "content": [ { "type": "text", "text": "what is your name" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -146,32 +150,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -217,20 +277,24 @@ "message": { "id": null, "role": "assistant", - "created": 1753704329, + "created": 1760565470, "content": [ { "type": "text", - "text": "My name is goose. I am a general-purpose AI agent created by Block, the parent company of Square, CashApp, and Tidal. I am being developed as an open-source software project." + "text": "My name is Goose." } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "gemini-2.5-flash", "usage": { - "input_tokens": 1967, - "output_tokens": 41, - "total_tokens": 2038 + "input_tokens": 2140, + "output_tokens": 5, + "total_tokens": 2145 } } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/groq/context_length_exceeded.json b/crates/goose-cli/src/scenario_tests/recordings/groq/context_length_exceeded.json deleted file mode 100644 index 9e26dfeeb6e6..000000000000 --- a/crates/goose-cli/src/scenario_tests/recordings/groq/context_length_exceeded.json +++ /dev/null @@ -1 +0,0 @@ -{} \ No newline at end of file diff --git a/crates/goose-cli/src/scenario_tests/recordings/groq/weather_tool.json b/crates/goose-cli/src/scenario_tests/recordings/groq/weather_tool.json index 4faf11f4c639..8eb037f7a2df 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/groq/weather_tool.json +++ b/crates/goose-cli/src/scenario_tests/recordings/groq/weather_tool.json @@ -1,58 +1,22 @@ { - "09dddf56be462d1861d5a56de6ec2d79b76e1b6f8f8ba9da8d837aae55c7e70b": { + "1bc400a528c54b25f4f1f609481e98e44222b3deaf7eee2c9e640e6345c73861": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:05:31.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:53.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704331, + "created": 1760565473, "content": [ { "type": "text", "text": "tell me what the weather is in Berlin, Germany" } - ] - }, - { - "id": null, - "role": "assistant", - "created": 1753704331, - "content": [ - { - "type": "toolRequest", - "id": "rnxgf8akg", - "toolCall": { - "status": "success", - "value": { - "name": "weather_extension__get_weather", - "arguments": { - "location": "Berlin, Germany" - } - } - } - } - ] - }, - { - "id": null, - "role": "user", - "created": 1753704331, - "content": [ - { - "type": "toolResponse", - "id": "rnxgf8akg", - "toolResult": { - "status": "success", - "value": [ - { - "type": "text", - "text": "The weather in Berlin, Germany is cloudy and 18°C" - } - ] - } - } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -186,32 +150,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -257,38 +277,103 @@ "message": { "id": null, "role": "assistant", - "created": 1753704332, + "created": 1760565474, "content": [ { - "type": "text", - "text": "The weather in Berlin, Germany is cloudy with a temperature of 18°C." + "type": "toolRequest", + "id": "04fqdf70k", + "toolCall": { + "status": "success", + "value": { + "name": "weather_extension__get_weather", + "arguments": { + "location": "Berlin, Germany" + } + } + } } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "llama-3.3-70b-versatile", "usage": { - "input_tokens": 2856, - "output_tokens": 17, - "total_tokens": 2873 + "input_tokens": 2698, + "output_tokens": 20, + "total_tokens": 2718 } } } }, - "1bc400a528c54b25f4f1f609481e98e44222b3deaf7eee2c9e640e6345c73861": { + "54bbc046e4ff83dd29df1fa575c6d39f6195a259106666dfefd1b10ed662bd45": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:05:31.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:53.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704331, + "created": 1760565473, "content": [ { "type": "text", "text": "tell me what the weather is in Berlin, Germany" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + }, + { + "id": null, + "role": "assistant", + "created": 1760565474, + "content": [ + { + "type": "toolRequest", + "id": "04fqdf70k", + "toolCall": { + "status": "success", + "value": { + "name": "weather_extension__get_weather", + "arguments": { + "location": "Berlin, Germany" + } + } + } + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + }, + { + "id": "msg_7256b4f2-18dd-4477-98b7-cde7102e8a46", + "role": "user", + "created": 1760565474, + "content": [ + { + "type": "toolResponse", + "id": "04fqdf70k", + "toolResult": { + "status": "success", + "value": [ + { + "type": "text", + "text": "The weather in Berlin, Germany is cloudy and 18°C" + } + ] + } + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -422,32 +507,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -493,29 +634,24 @@ "message": { "id": null, "role": "assistant", - "created": 1753704331, + "created": 1760565474, "content": [ { - "type": "toolRequest", - "id": "rnxgf8akg", - "toolCall": { - "status": "success", - "value": { - "name": "weather_extension__get_weather", - "arguments": { - "location": "Berlin, Germany" - } - } - } + "type": "text", + "text": "The weather in Berlin, Germany is cloudy and 18°C." } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "llama-3.3-70b-versatile", "usage": { - "input_tokens": 2815, - "output_tokens": 20, - "total_tokens": 2835 + "input_tokens": 2739, + "output_tokens": 14, + "total_tokens": 2753 } } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/groq/what_is_your_name.json b/crates/goose-cli/src/scenario_tests/recordings/groq/what_is_your_name.json index 0f1cc66d8cef..28cabb791a9b 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/groq/what_is_your_name.json +++ b/crates/goose-cli/src/scenario_tests/recordings/groq/what_is_your_name.json @@ -1,18 +1,22 @@ { "1b998117eba523901ae6a4dbf8caa81a95ea88ef7a84d0434c9b41a26164a2b9": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:05:29.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:50.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704329, + "created": 1760565470, "content": [ { "type": "text", "text": "what is your name" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -146,32 +150,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -217,20 +277,24 @@ "message": { "id": null, "role": "assistant", - "created": 1753704329, + "created": 1760565471, "content": [ { "type": "text", - "text": "My name is goose, and I am a general-purpose AI agent developed by Block, the parent company of Square, CashApp, and Tidal." + "text": "My name is **goose**. I am a general-purpose AI agent created by Block, the parent company of Square, CashApp, and Tidal." } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "llama-3.3-70b-versatile", "usage": { - "input_tokens": 2809, - "output_tokens": 31, - "total_tokens": 2840 + "input_tokens": 2692, + "output_tokens": 33, + "total_tokens": 2725 } } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/openai/image_analysis.json b/crates/goose-cli/src/scenario_tests/recordings/openai/image_analysis.json index 63677ef5e7d6..0a84f7aa7e6b 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/openai/image_analysis.json +++ b/crates/goose-cli/src/scenario_tests/recordings/openai/image_analysis.json @@ -1,12 +1,12 @@ { "c848f22f273e158c32435d3e72cc999c046dc1a9afdc3efda68ff451f833a185": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:05:24.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:42.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704324, + "created": 1760565462, "content": [ { "type": "text", @@ -17,7 +17,11 @@ "data": "", "mimeType": "image/jpeg" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -151,32 +155,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -222,20 +282,24 @@ "message": { "id": null, "role": "assistant", - "created": 1753704326, + "created": 1760565465, "content": [ { "type": "text", - "text": "This is an image of a goose standing by a body of water. If you have any questions or need information related to geese or something else, feel free to ask!" + "text": "This image shows a goose standing near a body of water. The goose has grey and brown feathers with a distinctive orange beak. It’s standing on grassy ground by the water's edge." } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "gpt-4o-2024-08-06", "usage": { - "input_tokens": 1925, - "output_tokens": 36, - "total_tokens": 1961 + "input_tokens": 1953, + "output_tokens": 39, + "total_tokens": 1992 } } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/openai/weather_tool.json b/crates/goose-cli/src/scenario_tests/recordings/openai/weather_tool.json index 6886d3697847..ddc200735f13 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/openai/weather_tool.json +++ b/crates/goose-cli/src/scenario_tests/recordings/openai/weather_tool.json @@ -1,58 +1,22 @@ { - "e546a32d4a2c9d41338b6725f317c4d4f462ce7cc04c79f3f24dd47a1a32a795": { + "1bc400a528c54b25f4f1f609481e98e44222b3deaf7eee2c9e640e6345c73861": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:05:24.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:42.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704324, + "created": 1760565462, "content": [ { "type": "text", "text": "tell me what the weather is in Berlin, Germany" } - ] - }, - { - "id": null, - "role": "assistant", - "created": 1753704326, - "content": [ - { - "type": "toolRequest", - "id": "call_5hvVKt0aj0RGRmrKEZECeJKV", - "toolCall": { - "status": "success", - "value": { - "name": "weather_extension__get_weather", - "arguments": { - "location": "Berlin, Germany" - } - } - } - } - ] - }, - { - "id": null, - "role": "user", - "created": 1753704326, - "content": [ - { - "type": "toolResponse", - "id": "call_5hvVKt0aj0RGRmrKEZECeJKV", - "toolResult": { - "status": "success", - "value": [ - { - "type": "text", - "text": "The weather in Berlin, Germany is cloudy and 18°C" - } - ] - } - } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -186,32 +150,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -257,38 +277,103 @@ "message": { "id": null, "role": "assistant", - "created": 1753704328, + "created": 1760565463, "content": [ { - "type": "text", - "text": "The weather in Berlin, Germany is currently cloudy with a temperature of 18°C." + "type": "toolRequest", + "id": "call_NSF9KSX2tCz8i8qb41WU6GHT", + "toolCall": { + "status": "success", + "value": { + "name": "weather_extension__get_weather", + "arguments": { + "location": "Berlin, Germany" + } + } + } } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "gpt-4o-2024-08-06", "usage": { - "input_tokens": 1722, - "output_tokens": 18, - "total_tokens": 1740 + "input_tokens": 1708, + "output_tokens": 19, + "total_tokens": 1727 } } } }, - "1bc400a528c54b25f4f1f609481e98e44222b3deaf7eee2c9e640e6345c73861": { + "4166a5a4297f1a6c903edc1c3232d12e2503190a3d4f98486d1a61d721bb0952": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:05:24.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:42.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704324, + "created": 1760565462, "content": [ { "type": "text", "text": "tell me what the weather is in Berlin, Germany" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + }, + { + "id": null, + "role": "assistant", + "created": 1760565463, + "content": [ + { + "type": "toolRequest", + "id": "call_NSF9KSX2tCz8i8qb41WU6GHT", + "toolCall": { + "status": "success", + "value": { + "name": "weather_extension__get_weather", + "arguments": { + "location": "Berlin, Germany" + } + } + } + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + }, + { + "id": "msg_926cf617-03cf-4027-ab42-d9bf44831517", + "role": "user", + "created": 1760565463, + "content": [ + { + "type": "toolResponse", + "id": "call_NSF9KSX2tCz8i8qb41WU6GHT", + "toolResult": { + "status": "success", + "value": [ + { + "type": "text", + "text": "The weather in Berlin, Germany is cloudy and 18°C" + } + ] + } + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -422,32 +507,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -493,29 +634,24 @@ "message": { "id": null, "role": "assistant", - "created": 1753704326, + "created": 1760565465, "content": [ { - "type": "toolRequest", - "id": "call_5hvVKt0aj0RGRmrKEZECeJKV", - "toolCall": { - "status": "success", - "value": { - "name": "weather_extension__get_weather", - "arguments": { - "location": "Berlin, Germany" - } - } - } + "type": "text", + "text": "The current weather in Berlin, Germany is cloudy with a temperature of 18°C." } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "gpt-4o-2024-08-06", "usage": { - "input_tokens": 1680, - "output_tokens": 19, - "total_tokens": 1699 + "input_tokens": 1750, + "output_tokens": 18, + "total_tokens": 1768 } } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/openai/what_is_your_name.json b/crates/goose-cli/src/scenario_tests/recordings/openai/what_is_your_name.json index dfd02ab51c46..4ba6b2d4f0c2 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/openai/what_is_your_name.json +++ b/crates/goose-cli/src/scenario_tests/recordings/openai/what_is_your_name.json @@ -1,18 +1,22 @@ { "1b998117eba523901ae6a4dbf8caa81a95ea88ef7a84d0434c9b41a26164a2b9": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal. goose is being developed as an open-source software project.\n\nThe current date is 2025-07-28 12:05:24.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o, claude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10 months prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and tools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level problems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable extensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from the search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n\n# Suggestion\n\"\"\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:42.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", "messages": [ { "id": null, "role": "user", - "created": 1753704324, + "created": 1760565462, "content": [ { "type": "text", "text": "what is your name" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } } ], "tools": [ @@ -146,32 +150,88 @@ }, { "name": "dynamic_task__create_task", - "description": "Use this tool to create one or more dynamic tasks from a shared text instruction and varying parameters.How it works:\n - Provide a single text instruction\n - Use the 'task_parameters' field to pass an array of parameter sets\n - Each resulting task will use the same instruction with different parameter values\n This is useful when performing the same operation across many inputs (e.g., getting weather for multiple cities, searching multiple slack channels, iterating through various linear tickets, etc).\n Once created, these tasks should be passed to the 'subagent__execute_task' tool for execution. Tasks can run sequentially or in parallel.\n ---\n What is a 'subagent'?\n A 'subagent' is a stateless sub-process that executes a single task independently. Use subagents when:\n - You want to parallelize similar work across different inputs\n - You are not sure your search or operation will succeed on the first try\n Each subagent receives a task with a defined payload and returns a result, which is not visible to the user unless explicitly summarized by the system.\n ---\n Examples of 'task_parameters' for a single task:\n text_instruction: Search for the config file in the root directory.\n Examples of 'task_parameters' for multiple tasks:\n text_instruction: Get weather for Melbourne.\n text_instruction: Get weather for Los Angeles.\n text_instruction: Get weather for San Francisco.\n ", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", "inputSchema": { "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, "task_parameters": { - "description": "Array of parameter sets for creating tasks. For a single task, provide an array with one element. For multiple tasks, provide an array with multiple elements, each with different parameter values. If there is no parameter set, provide an empty array.", + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", "items": { "properties": { - "text_instruction": { - "description": "The text instruction to execute", + "activities": { + "items": { + "type": "string" + }, + "type": "array" + }, + "context": { + "items": { + "type": "string" + }, + "type": "array" + }, + "description": { + "type": "string" + }, + "extensions": { + "items": { + "type": "object" + }, + "type": "array" + }, + "instructions": { + "description": "Task instructions (required if prompt is not provided)", + "type": "string" + }, + "parameters": { + "items": { + "type": "object" + }, + "type": "array" + }, + "prompt": { + "description": "Initial prompt (required if instructions is not provided)", + "type": "string" + }, + "response": { + "type": "object" + }, + "retry": { + "type": "object" + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": "boolean" + }, + "settings": { + "type": "object" + }, + "title": { "type": "string" } }, - "required": [ - "text_instruction" - ], "type": "object" }, + "minItems": 1, "type": "array" } }, + "required": [ + "task_parameters" + ], "type": "object" }, "annotations": { - "title": "Dynamic Task Creation", + "title": "Create Dynamic Tasks", "readOnlyHint": false, - "destructiveHint": true, + "destructiveHint": false, "idempotentHint": false, "openWorldHint": true } @@ -217,20 +277,24 @@ "message": { "id": null, "role": "assistant", - "created": 1753704325, + "created": 1760565463, "content": [ { "type": "text", - "text": "My name is goose. I'm an AI agent developed by Block, the parent company of Square, Cash App, and Tidal. How can I assist you today?" + "text": "My name is goose. I'm an AI agent developed by Block, the parent company of Square, CashApp, and Tidal. How can I assist you today?" } - ] + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } }, "usage": { "model": "gpt-4o-2024-08-06", "usage": { - "input_tokens": 1674, + "input_tokens": 1702, "output_tokens": 34, - "total_tokens": 1708 + "total_tokens": 1736 } } } diff --git a/crates/goose-cli/src/scenario_tests/scenarios.rs b/crates/goose-cli/src/scenario_tests/scenarios.rs index 6b6c0ef99922..d1c7695742cc 100644 --- a/crates/goose-cli/src/scenario_tests/scenarios.rs +++ b/crates/goose-cli/src/scenario_tests/scenarios.rs @@ -7,7 +7,6 @@ mod tests { use crate::scenario_tests::mock_client::WEATHER_TYPE; use crate::scenario_tests::scenario_runner::run_scenario; use anyhow::Result; - use goose::conversation::message::Message; #[tokio::test] async fn test_what_is_your_name() -> Result<()> { @@ -76,30 +75,24 @@ mod tests { .await } - #[tokio::test] - async fn test_context_length_exceeded_error() -> Result<()> { - run_scenario( - "context_length_exceeded", - Box::new(|provider| { - let model_config = provider.get_model_config(); - let context_length = model_config.context_limit.unwrap_or(300_000); - // "hello " is only one token in most models, since the hello and space often - // occur together in the training data. - let large_message = "hello ".repeat(context_length + 100); - Message::user().with_text(&large_message) - }), - Some(&["OpenAI"]), - |result| { - // this is unfortunate; we don't seem to actually catch the errors in this path, - // but instead eat it: - assert_eq!( - result.messages.len(), - 0, - "Expected no messages due to compaction" - ); - Ok(()) - }, - ) - .await - } + // #[tokio::test] + // async fn test_context_length_exceeded_error() -> Result<()> { + // run_scenario( + // "context_length_exceeded", + // Box::new(|provider| { + // let model_config = provider.get_model_config(); + // let context_length = model_config.context_limit.unwrap_or(300_000); + // // "hello " is only one token in most models, since the hello and space often + // // occur together in the training data. + // let large_message = "hello ".repeat(context_length + 100); + // Message::user().with_text(&large_message) + // }), + // Some(&["OpenAI"]), + // |result| { + // assert_eq!(result.messages.len(), 2, "One message after compaction"); + // Ok(()) + // }, + // ) + // .await + // } } diff --git a/crates/goose-cli/src/session/export.rs b/crates/goose-cli/src/session/export.rs index 2f046dfa22ce..ef2b111cd60d 100644 --- a/crates/goose-cli/src/session/export.rs +++ b/crates/goose-cli/src/session/export.rs @@ -369,7 +369,7 @@ pub fn message_to_markdown(message: &Message, export_all_content: bool) -> Strin md.push_str("**Thinking:**\n"); md.push_str("> *Thinking was redacted*\n\n"); } - MessageContent::SummarizationRequested(summarization) => { + MessageContent::ConversationCompacted(summarization) => { md.push_str(&format!("*{}*\n\n", summarization.msg)); } _ => { diff --git a/crates/goose-cli/src/session/mod.rs b/crates/goose-cli/src/session/mod.rs index 0547282c6715..2bb7b035df72 100644 --- a/crates/goose-cli/src/session/mod.rs +++ b/crates/goose-cli/src/session/mod.rs @@ -47,6 +47,7 @@ use std::sync::Arc; use std::time::Instant; use tokio; use tokio_util::sync::CancellationToken; +use tracing::warn; pub enum RunMode { Normal, @@ -158,19 +159,6 @@ impl CliSession { self.session_id.as_ref() } - async fn summarize_context_messages( - messages: &mut Conversation, - agent: &Agent, - message_suffix: &str, - ) -> Result<()> { - let (summarized_messages, _, _) = agent.summarize_context(messages.messages()).await?; - let msg = format!("Context maxed out\n{}\n{}", "-".repeat(50), message_suffix); - output::render_text(&msg, Some(Color::Yellow), true); - *messages = summarized_messages; - - Ok(()) - } - /// Add a stdio extension to the session /// /// # Arguments @@ -674,10 +662,24 @@ impl CliSession { // Get the provider for summarization let _provider = self.agent.provider().await?; + // Get session metadata if available + let session_metadata_for_compact = + if let Some(ref session_id) = self.session_id { + SessionManager::get_session(session_id, false).await.ok() + } else { + None + }; + // Call the summarize_context method - let (summarized_messages, _token_counts, summarization_usage) = self - .agent - .summarize_context(self.messages.messages()) + let (_, summarized_messages, _token_counts, summarization_usage) = + goose::context_mgmt::check_and_compact_messages( + &self.agent, + self.messages.messages(), + true, + false, + None, + session_metadata_for_compact.as_ref(), + ) .await?; // Update the session messages with the summarized ones @@ -937,84 +939,7 @@ impl CliSession { permission, },).await; } - } else if let Some(MessageContent::ContextLengthExceeded(_)) = message.content.first() { - output::hide_thinking(); - - // Check for user-configured default context strategy - let config = Config::global(); - let context_strategy = config.get_param::("GOOSE_CONTEXT_STRATEGY") - .unwrap_or_else(|_| if interactive { "prompt".to_string() } else { "summarize".to_string() }); - - let selected = match context_strategy.as_str() { - "clear" => "clear", - "truncate" => "truncate", - "summarize" => "summarize", - _ => { - if interactive { - // In interactive mode with no default, ask the user what to do - let prompt = "The model's context length is maxed out. You will need to reduce the # msgs. Do you want to?".to_string(); - cliclack::select(prompt) - .item("clear", "Clear Session", "Removes all messages from Goose's memory") - .item("truncate", "Truncate Messages", "Removes old messages till context is within limits") - .item("summarize", "Summarize Session", "Summarize the session to reduce context length") - .interact()? - } else { - // In headless mode, default to summarize - "summarize" - } - } - }; - - match selected { - "clear" => { - self.messages.clear(); - let msg = if context_strategy == "clear" { - format!("Context maxed out - automatically cleared session.\n{}", "-".repeat(50)) - } else { - format!("Session cleared.\n{}", "-".repeat(50)) - }; - output::render_text(&msg, Some(Color::Yellow), true); - break; // exit the loop to hand back control to the user - } - "truncate" => { - // Truncate messages to fit within context length - let (truncated_messages, _) = self.agent.truncate_context(self.messages.messages()).await?; - let msg = if context_strategy == "truncate" { - format!("Context maxed out - automatically truncated messages.\n{}\nGoose tried its best to truncate messages for you.", "-".repeat(50)) - } else { - format!("Context maxed out\n{}\nGoose tried its best to truncate messages for you.", "-".repeat(50)) - }; - output::render_text("", Some(Color::Yellow), true); - output::render_text(&msg, Some(Color::Yellow), true); - self.messages = truncated_messages; - } - "summarize" => { - // Use the helper function to summarize context - let message_suffix = if context_strategy == "summarize" { - "Goose automatically summarized messages for you." - } else if interactive { - "Goose summarized messages for you." - } else { - "Goose automatically summarized messages to continue processing." - }; - Self::summarize_context_messages(&mut self.messages, &self.agent, message_suffix).await?; - } - _ => { - unreachable!() - } - } - - // Restart the stream after handling ContextLengthExceeded - stream = self - .agent - .reply( - self.messages.clone(), - session_config.clone(), - None - ) - .await?; } - // otherwise we have a model/tool to render else { for content in &message.content { if let MessageContent::ToolRequest(tool_request) = content { @@ -1171,10 +1096,10 @@ impl CliSession { _ => (), } } - Some(Ok(AgentEvent::HistoryReplaced(new_messages))) => { - self.messages = Conversation::new_unvalidated(new_messages.clone()); - } - Some(Ok(AgentEvent::ModelChange { model, mode })) => { + Some(Ok(AgentEvent::HistoryReplaced(updated_conversation))) => { + self.messages = updated_conversation; + } + Some(Ok(AgentEvent::ModelChange { model, mode })) => { // Log model change if in debug mode if self.debug { eprintln!("Model changed to {} in {} mode", model, mode); @@ -1182,94 +1107,24 @@ impl CliSession { } Some(Err(e)) => { + // TODO(Douwe): Delete this // Check if it's a ProviderError::ContextLengthExceeded if e.downcast_ref::() .map(|provider_error| matches!(provider_error, goose::providers::errors::ProviderError::ContextLengthExceeded(_))) .unwrap_or(false) { output::render_text( - "Context limit reached. Performing auto-compaction...", + "Compaction requested. Should have happened in the agent!", Some(Color::Yellow), true ); - - // Try auto-compaction first - keep the stream alive! - if let Ok(compact_result) = goose::context_mgmt::auto_compact::perform_compaction(&self.agent, self.messages.messages()).await { - self.messages = compact_result.messages; - if let Some(session_id) = &self.session_id { - SessionManager::replace_conversation(session_id, &self.messages).await?; - } - - output::render_text( - "Compaction complete. Conversation has been automatically compacted to continue.", - Some(Color::Yellow), - true - ); - - // Restart the stream after successful compaction - keep the stream alive! - stream = self - .agent - .reply( - self.messages.clone(), - session_config.clone(), - Some(cancel_token.clone()) - ) - .await?; - continue; - } - // Auto-compaction failed, fall through to common error handling below + warn!("Compaction requested. Should have happened in the agent!"); } eprintln!("Error: {}", e); cancel_token_clone.cancel(); drop(stream); if let Err(e) = self.handle_interrupted_messages(false).await { eprintln!("Error handling interruption: {}", e); - } - - // Check if it's a ProviderError::ContextLengthExceeded - if e.downcast_ref::() - .map(|provider_error| matches!(provider_error, goose::providers::errors::ProviderError::ContextLengthExceeded(_))) - .unwrap_or(false) { - output::render_error(&format!("Error: Context length exceeded: {}", e)); - - let prompt = "The tool calling loop was interrupted. How would you like to proceed?"; - let selected = match cliclack::select(prompt.to_string()) - .item("clear", "Clear Session", "Removes all messages from Goose's memory") - .item("summarize", "Summarize Session", "Summarize the session to reduce context length") - .interact() - { - Ok(choice) => Some(choice), - Err(e) => { - if e.kind() == std::io::ErrorKind::Interrupted { - // If interrupted, do nothing and let user handle it manually - output::render_text("Operation cancelled. You can use /clear or /summarize to continue.", Some(Color::Yellow), true); - None - } else { - return Err(e.into()); - } - } - }; - - if let Some(choice) = selected { - match choice { - "clear" => { - self.messages.clear(); - let msg = format!("Session cleared.\n{}", "-".repeat(50)); - output::render_text(&msg, Some(Color::Yellow), true); - } - "summarize" => { - // Use the helper function to summarize context - let message_suffix = "Goose summarized messages for you."; - if let Err(e) = Self::summarize_context_messages(&mut self.messages, &self.agent, message_suffix).await { - output::render_error(&format!("Failed to summarize: {}", e)); - output::render_text("Consider using /clear to start fresh.", Some(Color::Yellow), true); - } - } - _ => { - unreachable!() - } - } - } } else { output::render_error( "The error above was an exception we were not able to handle.\n\ diff --git a/crates/goose-cli/src/session/output.rs b/crates/goose-cli/src/session/output.rs index 639d7da96157..e7c536f8a427 100644 --- a/crates/goose-cli/src/session/output.rs +++ b/crates/goose-cli/src/session/output.rs @@ -185,7 +185,7 @@ pub fn render_message(message: &Message, debug: bool) { println!("\n{}", style("Thinking:").dim().italic()); print_markdown("Thinking was redacted", theme); } - MessageContent::SummarizationRequested(summarization) => { + MessageContent::ConversationCompacted(summarization) => { println!("\n{}", style(&summarization.msg).yellow()); } _ => { diff --git a/crates/goose-server/src/openapi.rs b/crates/goose-server/src/openapi.rs index 42df6681f107..90ea1ea11277 100644 --- a/crates/goose-server/src/openapi.rs +++ b/crates/goose-server/src/openapi.rs @@ -18,10 +18,10 @@ use goose::config::declarative_providers::{ DeclarativeProviderConfig, LoadedProvider, ProviderEngine, }; use goose::conversation::message::{ - ContextLengthExceeded, FrontendToolRequest, Message, MessageContent, MessageMetadata, - RedactedThinkingContent, SummarizationRequested, ThinkingContent, ToolConfirmationRequest, - ToolRequest, ToolResponse, + ConversationCompacted, FrontendToolRequest, Message, MessageContent, MessageMetadata, + RedactedThinkingContent, ThinkingContent, ToolConfirmationRequest, ToolRequest, ToolResponse, }; + use utoipa::openapi::schema::{ AdditionalProperties, AnyOfBuilder, ArrayBuilder, ObjectBuilder, OneOfBuilder, Schema, SchemaFormat, SchemaType, @@ -420,8 +420,7 @@ derive_utoipa!(Icon as IconSchema); RedactedThinkingContent, FrontendToolRequest, ResourceContentsSchema, - ContextLengthExceeded, - SummarizationRequested, + ConversationCompacted, JsonObjectSchema, RoleSchema, ProviderMetadata, diff --git a/crates/goose-server/src/routes/context.rs b/crates/goose-server/src/routes/context.rs index 6f8764b074f9..63dd203af612 100644 --- a/crates/goose-server/src/routes/context.rs +++ b/crates/goose-server/src/routes/context.rs @@ -1,6 +1,6 @@ use crate::state::AppState; use axum::{extract::State, http::StatusCode, routing::post, Json, Router}; -use goose::conversation::{message::Message, Conversation}; +use goose::conversation::message::Message; use serde::{Deserialize, Serialize}; use std::sync::Arc; use utoipa::ToSchema; @@ -11,8 +11,6 @@ use utoipa::ToSchema; pub struct ContextManageRequest { /// Collection of messages to be managed pub messages: Vec, - /// Operation to perform: "truncation" or "summarize" - pub manage_action: String, /// Optional session ID for session-specific agent pub session_id: String, } @@ -48,28 +46,20 @@ async fn manage_context( ) -> Result, StatusCode> { let agent = state.get_agent_for_route(request.session_id).await?; - let mut processed_messages = Conversation::new_unvalidated(vec![]); - let mut token_counts: Vec = vec![]; - - if request.manage_action == "truncation" { - (processed_messages, token_counts) = agent - .truncate_context(&request.messages) - .await - .map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?; - } else if request.manage_action == "summarize" { - (processed_messages, token_counts, _) = agent - .summarize_context(&request.messages) - .await - .map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?; - } + let (_, processed_messages, token_counts, _) = goose::context_mgmt::check_and_compact_messages( + &agent, + &request.messages, + true, + false, + None, + None, + ) + .await + .map_err(|_| StatusCode::INTERNAL_SERVER_ERROR)?; + // TODO(Douwe): store into db Ok(Json(ContextManageResponse { - messages: processed_messages - .messages() - .iter() - .filter(|m| m.is_user_visible()) - .cloned() - .collect(), + messages: processed_messages.messages().to_vec(), token_counts, })) } diff --git a/crates/goose-server/src/routes/reply.rs b/crates/goose-server/src/routes/reply.rs index c8db3b47cee0..a419e75769f0 100644 --- a/crates/goose-server/src/routes/reply.rs +++ b/crates/goose-server/src/routes/reply.rs @@ -121,9 +121,9 @@ impl IntoResponse for SseResponse { } } -#[derive(Debug, Serialize)] +#[derive(Debug, Serialize, utoipa::ToSchema)] #[serde(tag = "type")] -enum MessageEvent { +pub enum MessageEvent { Message { message: Message, }, @@ -141,6 +141,9 @@ enum MessageEvent { request_id: String, message: ServerNotification, }, + UpdateConversation { + conversation: Conversation, + }, Ping, } @@ -306,10 +309,9 @@ pub async fn reply( } } Ok(Some(Ok(AgentEvent::HistoryReplaced(new_messages)))) => { - // Replace the message history with the compacted messages - all_messages = Conversation::new_unvalidated(new_messages); - // Note: We don't send this as a stream event since it's an internal operation - // The client will see the compaction notification message that was sent before this event + all_messages = new_messages.clone(); + stream_event(MessageEvent::UpdateConversation {conversation: new_messages}, &tx, &cancel_token).await; + } Ok(Some(Ok(AgentEvent::ModelChange { model, mode }))) => { stream_event(MessageEvent::ModelChange { model, mode }, &tx, &cancel_token).await; diff --git a/crates/goose/src/agents/agent.rs b/crates/goose/src/agents/agent.rs index 06da2531fcf6..4a31d7826cb1 100644 --- a/crates/goose/src/agents/agent.rs +++ b/crates/goose/src/agents/agent.rs @@ -33,7 +33,7 @@ use crate::agents::tool_router_index_manager::ToolRouterIndexManager; use crate::agents::types::SessionConfig; use crate::agents::types::{FrontendTool, ToolResultReceiver}; use crate::config::{get_enabled_extensions, get_extension_by_name, Config}; -use crate::context_mgmt::auto_compact; +use crate::context_mgmt::{check_and_compact_messages, DEFAULT_COMPACTION_THRESHOLD}; use crate::conversation::{debug_conversation_fix, fix_conversation, Conversation}; use crate::mcp_utils::ToolResult; use crate::permission::permission_inspector::PermissionInspector; @@ -112,7 +112,7 @@ pub enum AgentEvent { Message(Message), McpNotification((String, ServerNotification)), ModelChange { model: String, mode: String }, - HistoryReplaced(Vec), + HistoryReplaced(Conversation), } impl Default for Agent { @@ -902,20 +902,15 @@ impl Agent { } } - /// Handle auto-compaction logic and return compacted messages if needed - async fn handle_auto_compaction( + #[instrument(skip(self, unfixed_conversation, session), fields(user_message))] + pub async fn reply( &self, - messages: &[Message], - session: &Option, - ) -> Result< - Option<( - Conversation, - String, - Option, - )>, - > { + unfixed_conversation: Conversation, + session: Option, + cancel_token: Option, + ) -> Result>> { // Try to get session metadata for more accurate token counts - let session_metadata = if let Some(session_config) = session { + let session_metadata = if let Some(session_config) = &session { SessionManager::get_session(&session_config.id, false) .await .ok() @@ -923,22 +918,29 @@ impl Agent { None }; - let compact_result = auto_compact::check_and_compact_messages( - self, - messages, - None, - session_metadata.as_ref(), - ) - .await?; - - if compact_result.compacted { - let compacted_messages = compact_result.messages; + let (did_compact, compacted_conversation, compaction_error) = + match check_and_compact_messages( + self, + unfixed_conversation.messages(), + false, + false, + None, + session_metadata.as_ref(), + ) + .await + { + Ok((did_compact, conversation, _removed_indices, _summarization_usage)) => { + (did_compact, conversation, None) + } + Err(e) => (false, unfixed_conversation.clone(), Some(e)), + }; + if did_compact { // Get threshold from config to include in message let config = crate::config::Config::global(); let threshold = config .get_param::("GOOSE_AUTO_COMPACT_THRESHOLD") - .unwrap_or(0.8); // Default to 80% + .unwrap_or(DEFAULT_COMPACTION_THRESHOLD); let threshold_percentage = (threshold * 100.0) as u32; let compaction_msg = format!( @@ -946,42 +948,27 @@ impl Agent { threshold_percentage ); - return Ok(Some(( - compacted_messages, - compaction_msg, - compact_result.summarization_usage, - ))); - } - - Ok(None) - } - - #[instrument(skip(self, unfixed_conversation, session), fields(user_message))] - pub async fn reply( - &self, - unfixed_conversation: Conversation, - session: Option, - cancel_token: Option, - ) -> Result>> { - let compaction_result = self - .handle_auto_compaction(unfixed_conversation.messages(), &session) - .await?; - - if let Some((conversation, compaction_message, _summarization_usage)) = compaction_result { Ok(Box::pin(async_stream::try_stream! { + // TODO(Douwe): send this before we actually compact: yield AgentEvent::Message( - Message::assistant().with_summarization_requested(compaction_message) + Message::assistant().with_conversation_compacted(compaction_msg) ); - yield AgentEvent::HistoryReplaced(conversation.messages().clone()); + yield AgentEvent::HistoryReplaced(compacted_conversation.clone()); if let Some(session_to_store) = &session { - SessionManager::replace_conversation(&session_to_store.id, &conversation).await? + SessionManager::replace_conversation(&session_to_store.id, &compacted_conversation).await? } - let mut reply_stream = self.reply_internal(conversation, session, cancel_token).await?; + let mut reply_stream = self.reply_internal(compacted_conversation, session, cancel_token).await?; while let Some(event) = reply_stream.next().await { yield event?; } })) + } else if let Some(error) = compaction_error { + Ok(Box::pin(async_stream::try_stream! { + yield AgentEvent::Message(Message::assistant().with_text( + format!("Ran into this error trying to auto-compact: {error}.\n\nPlease try again or create a new session") + )); + })) } else { self.reply_internal(unfixed_conversation, session, cancel_token) .await @@ -1113,6 +1100,7 @@ impl Agent { let mut no_tools_called = true; let mut messages_to_add = Conversation::default(); let mut tools_updated = false; + let mut did_recovery_compact_this_iteration = false; while let Some(next) = stream.next().await { if is_token_cancelled(&cancel_token) { @@ -1306,28 +1294,37 @@ impl Agent { messages_to_add.push(final_message_tool_resp); } } - Err(ProviderError::ContextLengthExceeded(error_msg)) => { + Err(ProviderError::ContextLengthExceeded(_error_msg)) => { info!("Context length exceeded, attempting compaction"); - match auto_compact::perform_compaction(self, conversation.messages()).await { - Ok(compact_result) => { - conversation = compact_result.messages; + // Get session metadata if available + let session_metadata_for_compact = if let Some(ref session_config) = session { + SessionManager::get_session(&session_config.id, false).await.ok() + } else { + None + }; + + match check_and_compact_messages(self, conversation.messages(), true, true, None, session_metadata_for_compact.as_ref()).await { + Ok((_did_compact, compacted_conversation, _removed_indices, _usage)) => { + conversation = compacted_conversation; + did_recovery_compact_this_iteration = true; yield AgentEvent::Message( - Message::assistant().with_summarization_requested( + Message::assistant().with_conversation_compacted( "Context limit reached. Conversation has been automatically compacted to continue." ) ); - yield AgentEvent::HistoryReplaced(conversation.messages().to_vec()); + yield AgentEvent::HistoryReplaced(conversation.clone()); if let Some(session_to_store) = &session { SessionManager::replace_conversation(&session_to_store.id, &conversation).await? } continue; } - Err(_) => { - yield AgentEvent::Message(Message::assistant().with_context_length_exceeded( - format!("Context length exceeded and cannot summarize: {}. Unable to continue.", error_msg) - )); + Err(e) => { + error!("Error: {}", e); + yield AgentEvent::Message(Message::assistant().with_text( + format!("Ran into this error trying to compact: {e}.\n\nPlease retry if you think this is a transient or recoverable error.") + )); break; } } @@ -1358,6 +1355,8 @@ impl Agent { yield AgentEvent::Message(message); exit_chat = true; } + } else if did_recovery_compact_this_iteration { + // Avoid setting exit_chat; continue from last user message in the conversation } else { match self.handle_retry_logic(&mut conversation, &session, &initial_messages).await { Ok(should_retry) => { diff --git a/crates/goose/src/agents/context.rs b/crates/goose/src/agents/context.rs deleted file mode 100644 index 37edcfb5658b..000000000000 --- a/crates/goose/src/agents/context.rs +++ /dev/null @@ -1,130 +0,0 @@ -use anyhow::Ok; - -use crate::conversation::message::{Message, MessageMetadata}; -use crate::conversation::Conversation; -use crate::token_counter::create_async_token_counter; - -use crate::context_mgmt::summarize::summarize_messages; -use crate::context_mgmt::truncate::{truncate_messages, OldestFirstTruncation}; -use crate::context_mgmt::{estimate_target_context_limit, get_messages_token_counts_async}; - -use super::super::agents::Agent; - -impl Agent { - /// Public API to truncate oldest messages so that the conversation's token count is within the allowed context limit. - pub async fn truncate_context( - &self, - messages: &[Message], // last message is a user msg that led to assistant message with_context_length_exceeded - ) -> Result<(Conversation, Vec), anyhow::Error> { - let provider = self.provider().await?; - let token_counter = create_async_token_counter() - .await - .map_err(|e| anyhow::anyhow!("Failed to create token counter: {}", e))?; - let target_context_limit = estimate_target_context_limit(provider); - let token_counts = get_messages_token_counts_async(&token_counter, messages); - - let (mut new_messages, mut new_token_counts) = truncate_messages( - messages, - &token_counts, - target_context_limit, - &OldestFirstTruncation, - )?; - - // Only add an assistant message if we have room for it and it won't cause another overflow - let assistant_message = Message::assistant().with_text("I had run into a context length exceeded error so I truncated some of the oldest messages in our conversation."); - let assistant_tokens = - token_counter.count_chat_tokens("", std::slice::from_ref(&assistant_message), &[]); - - let current_total: usize = new_token_counts.iter().sum(); - if current_total + assistant_tokens <= target_context_limit { - new_messages.push(assistant_message); - new_token_counts.push(assistant_tokens); - } else { - // If we can't fit the assistant message, at least log what happened - tracing::warn!("Cannot add truncation notice message due to context limits. Current: {}, Assistant: {}, Limit: {}", - current_total, assistant_tokens, target_context_limit); - } - - Ok((new_messages, new_token_counts)) - } - - /// Public API to summarize the conversation so that its token count is within the allowed context limit. - /// Returns the summarized messages, token counts, and the ProviderUsage from summarization - pub async fn summarize_context( - &self, - messages: &[Message], // last message is a user msg that led to assistant message with_context_length_exceeded - ) -> Result< - ( - Conversation, - Vec, - Option, - ), - anyhow::Error, - > { - let provider = self.provider().await?; - let summary_result = summarize_messages(provider.clone(), messages).await?; - - let (summary_message, summarization_usage) = match summary_result { - Some((summary_message, provider_usage)) => (summary_message, Some(provider_usage)), - None => { - // No summary was generated (empty input) - tracing::warn!("Summarization failed. Returning empty messages."); - return Ok((Conversation::empty(), vec![], None)); - } - }; - - // Create the final message list with updated visibility metadata: - // 1. Original messages become user_visible but not agent_visible - // 2. Summary message becomes agent_visible but not user_visible - // 3. Assistant messages to continue the conversation remain both user_visible and agent_visible - - let mut final_messages = Vec::new(); - let mut final_token_counts = Vec::new(); - - // Add all original messages with updated visibility (preserve user_visible, set agent_visible=false) - for msg in messages.iter().cloned() { - let updated_metadata = msg.metadata.with_agent_invisible(); - let updated_msg = msg.with_metadata(updated_metadata); - final_messages.push(updated_msg); - // Token count doesn't matter for agent_visible=false messages, but we'll use 0 - final_token_counts.push(0); - } - - // Add the compaction marker (user_visible=true, agent_visible=false) - let compaction_marker = Message::assistant() - .with_summarization_requested("Conversation compacted and summarized") - .with_metadata(MessageMetadata::user_only()); - let compaction_marker_tokens: usize = 0; // Not counted since agent_visible=false - final_messages.push(compaction_marker); - final_token_counts.push(compaction_marker_tokens); - - // Add the summary message (agent_visible=true, user_visible=false) - let summary_msg = summary_message.with_metadata(MessageMetadata::agent_only()); - // For token counting purposes, we use the output tokens (the actual summary content) - // since that's what will be in the context going forward - let summary_tokens = summarization_usage - .as_ref() - .and_then(|usage| usage.usage.output_tokens) - .unwrap_or(0) as usize; - final_messages.push(summary_msg); - final_token_counts.push(summary_tokens); - - // Add an assistant message to continue the conversation (agent_visible=true, user_visible=false) - let assistant_message = Message::assistant() - .with_text( - "The previous message contains a summary that was prepared because a context limit was reached. -Do not mention that you read a summary or that conversation summarization occurred -Just continue the conversation naturally based on the summarized context" - ) - .with_metadata(MessageMetadata::agent_only()); - let assistant_message_tokens: usize = 0; // Not counted since it's for agent context only - final_messages.push(assistant_message); - final_token_counts.push(assistant_message_tokens); - - Ok(( - Conversation::new_unvalidated(final_messages), - final_token_counts, - summarization_usage, - )) - } -} diff --git a/crates/goose/src/agents/mod.rs b/crates/goose/src/agents/mod.rs index 39bf903a3579..23dabbb8b095 100644 --- a/crates/goose/src/agents/mod.rs +++ b/crates/goose/src/agents/mod.rs @@ -1,5 +1,4 @@ mod agent; -mod context; pub mod extension; pub mod extension_malware_check; pub mod extension_manager; diff --git a/crates/goose/src/agents/reply_parts.rs b/crates/goose/src/agents/reply_parts.rs index c4b7b34d1b20..9cfb70657d9d 100644 --- a/crates/goose/src/agents/reply_parts.rs +++ b/crates/goose/src/agents/reply_parts.rs @@ -108,39 +108,46 @@ impl Agent { let toolshim_tools = toolshim_tools.to_owned(); let provider = provider.clone(); - let mut stream = if provider.supports_streaming() { + // Capture errors during stream creation and return them as part of the stream + // so they can be handled by the existing error handling logic in the agent + let stream_result = if provider.supports_streaming() { debug!("WAITING_LLM_STREAM_START"); - let msg_stream = provider + let result = provider .stream( system_prompt.as_str(), messages_for_provider.messages(), &tools, ) - .await?; + .await; debug!("WAITING_LLM_STREAM_END"); - msg_stream + result } else { debug!("WAITING_LLM_START"); - let (message, mut usage) = provider + let complete_result = provider .complete( system_prompt.as_str(), messages_for_provider.messages(), &tools, ) - .await?; + .await; debug!("WAITING_LLM_END"); - // Ensure we have token counts for non-streaming case - usage - .ensure_tokens( - system_prompt.as_str(), - messages_for_provider.messages(), - &message, - &tools, - ) - .await?; + match complete_result { + Ok((message, usage)) => Ok(stream_from_single_message(message, usage)), + Err(e) => Err(e), + } + }; - stream_from_single_message(message, usage) + // If there was an error creating the stream, return a stream that yields that error + let mut stream = match stream_result { + Ok(s) => s, + Err(e) => { + // Return a stream that immediately yields the error + // This allows the error to be caught by existing error handling in agent.rs + return Ok(Box::pin(try_stream! { + yield Err(e)?; + })); + } }; Ok(Box::pin(try_stream! { diff --git a/crates/goose/src/agents/subagent_handler.rs b/crates/goose/src/agents/subagent_handler.rs index 25e5d3f53207..ad14f282f1b7 100644 --- a/crates/goose/src/agents/subagent_handler.rs +++ b/crates/goose/src/agents/subagent_handler.rs @@ -127,7 +127,7 @@ fn get_agent_messages( } } - let mut session_messages = + let mut conversation = Conversation::new_unvalidated( vec![Message::user().with_text(text_instruction.clone())], ); @@ -141,15 +141,16 @@ fn get_agent_messages( }; let mut stream = agent - .reply(session_messages.clone(), Some(session_config), None) + .reply(conversation.clone(), Some(session_config), None) .await .map_err(|e| anyhow!("Failed to get reply from agent: {}", e))?; while let Some(message_result) = stream.next().await { match message_result { - Ok(AgentEvent::Message(msg)) => session_messages.push(msg), - Ok(AgentEvent::McpNotification(_)) - | Ok(AgentEvent::ModelChange { .. }) - | Ok(AgentEvent::HistoryReplaced(_)) => {} + Ok(AgentEvent::Message(msg)) => conversation.push(msg), + Ok(AgentEvent::McpNotification(_)) | Ok(AgentEvent::ModelChange { .. }) => {} + Ok(AgentEvent::HistoryReplaced(updated_conversation)) => { + conversation = updated_conversation; + } Err(e) => { tracing::error!("Error receiving message from subagent: {}", e); break; @@ -157,6 +158,6 @@ fn get_agent_messages( } } - Ok(session_messages) + Ok(conversation) }) } diff --git a/crates/goose/src/context_mgmt/auto_compact.rs b/crates/goose/src/context_mgmt/auto_compact.rs deleted file mode 100644 index 542d252f3108..000000000000 --- a/crates/goose/src/context_mgmt/auto_compact.rs +++ /dev/null @@ -1,766 +0,0 @@ -use crate::conversation::message::Message; -use crate::conversation::Conversation; -use crate::{ - agents::Agent, config::Config, context_mgmt::get_messages_token_counts_async, - token_counter::create_async_token_counter, -}; -use anyhow::Result; -use tracing::{debug, info}; - -/// Result of auto-compaction check -#[derive(Debug)] -pub struct AutoCompactResult { - /// Whether compaction was performed - pub compacted: bool, - /// The messages after potential compaction - pub messages: Conversation, - /// Provider usage from summarization (if compaction occurred) - /// This contains the actual token counts after compaction - pub summarization_usage: Option, -} - -/// Result of checking if compaction is needed -#[derive(Debug)] -pub struct CompactionCheckResult { - /// Whether compaction is needed - pub needs_compaction: bool, - /// Current token count - pub current_tokens: usize, - /// Context limit being used - pub context_limit: usize, - /// Current usage ratio (0.0 to 1.0) - pub usage_ratio: f64, - /// Remaining tokens before compaction threshold - pub remaining_tokens: usize, - /// Percentage until compaction threshold (0.0 to 100.0) - pub percentage_until_compaction: f64, -} - -/// Check if messages need compaction without performing the compaction -/// -/// This function analyzes the current token usage and returns detailed information -/// about whether compaction is needed and how close we are to the threshold. -/// It prioritizes actual token counts from session metadata when available, -/// falling back to estimated counts if needed. -/// -/// # Arguments -/// * `agent` - The agent to use for context management -/// * `messages` - The current message history -/// * `threshold_override` - Optional threshold override (defaults to GOOSE_AUTO_COMPACT_THRESHOLD config) -/// * `session_metadata` - Optional session metadata containing actual token counts -/// -/// # Returns -/// * `CompactionCheckResult` containing detailed information about compaction needs -pub async fn check_compaction_needed( - agent: &Agent, - messages: &[Message], - threshold_override: Option, - session_metadata: Option<&crate::session::Session>, -) -> Result { - // Get threshold from config or use override - let config = Config::global(); - let threshold = threshold_override.unwrap_or_else(|| { - config - .get_param::("GOOSE_AUTO_COMPACT_THRESHOLD") - .unwrap_or(0.8) // Default to 80% - }); - - let provider = agent.provider().await?; - let context_limit = provider.get_model_config().context_limit(); - - let (current_tokens, token_source) = match session_metadata.and_then(|m| m.total_tokens) { - Some(tokens) => (tokens as usize, "session metadata"), - None => { - let token_counter = create_async_token_counter() - .await - .map_err(|e| anyhow::anyhow!("Failed to create token counter: {}", e))?; - let token_counts = get_messages_token_counts_async(&token_counter, messages); - (token_counts.iter().sum(), "estimated") - } - }; - - // Calculate usage ratio - let usage_ratio = current_tokens as f64 / context_limit as f64; - - // Calculate threshold token count and remaining tokens - let threshold_tokens = (context_limit as f64 * threshold) as usize; - let remaining_tokens = threshold_tokens.saturating_sub(current_tokens); - - // Calculate percentage until compaction (how much more we can use before hitting threshold) - let percentage_until_compaction = if usage_ratio < threshold { - (threshold - usage_ratio) * 100.0 - } else { - 0.0 - }; - - // Check if compaction is needed (disabled if threshold is invalid) - let needs_compaction = if threshold <= 0.0 || threshold >= 1.0 { - false - } else { - usage_ratio > threshold - }; - - debug!( - "Compaction check: {} / {} tokens ({:.1}%), threshold: {:.1}%, needs compaction: {}, source: {}", - current_tokens, - context_limit, - usage_ratio * 100.0, - threshold * 100.0, - needs_compaction, - token_source - ); - - Ok(CompactionCheckResult { - needs_compaction, - current_tokens, - context_limit, - usage_ratio, - remaining_tokens, - percentage_until_compaction, - }) -} - -/// Perform compaction on messages without checking thresholds -/// -/// This function directly performs compaction on the provided messages. -/// If the most recent message is a user message, it will be preserved by removing it -/// before compaction and adding it back afterwards. -/// -/// # Arguments -/// * `agent` - The agent to use for context management -/// * `messages` - The current message history -/// -/// # Returns -/// * `AutoCompactResult` containing the compacted messages and metadata -pub async fn perform_compaction(agent: &Agent, messages: &[Message]) -> Result { - info!("Performing message compaction"); - - // Check if the most recent message is a user message - let (messages_to_compact, preserved_user_message) = if let Some(last_message) = messages.last() - { - if matches!(last_message.role, rmcp::model::Role::User) { - // Remove the last user message before compaction - (&messages[..messages.len() - 1], Some(last_message.clone())) - } else { - (messages, None) - } - } else { - (messages, None) - }; - - // Perform the compaction on messages excluding the preserved user message - let (mut compacted_messages, _, summarization_usage) = - agent.summarize_context(messages_to_compact).await?; - - // Add back the preserved user message if it exists - if let Some(user_message) = preserved_user_message { - compacted_messages.push(user_message); - } - - Ok(AutoCompactResult { - compacted: true, - messages: compacted_messages, - summarization_usage, - }) -} - -/// Check if messages need compaction and compact them if necessary -/// -/// This is a convenience wrapper function that combines checking and compaction. -/// Uses perform_compaction internally to handle the actual compaction process. -/// -/// # Arguments -/// * `agent` - The agent to use for context management -/// * `messages` - The current message history -/// * `threshold_override` - Optional threshold override (defaults to GOOSE_AUTO_COMPACT_THRESHOLD config) -/// * `session_metadata` - Optional session metadata containing actual token counts -/// -/// # Returns -/// * `AutoCompactResult` containing the potentially compacted messages and metadata -pub async fn check_and_compact_messages( - agent: &Agent, - messages: &[Message], - threshold_override: Option, - session_metadata: Option<&crate::session::Session>, -) -> Result { - // First check if compaction is needed - let check_result = - check_compaction_needed(agent, messages, threshold_override, session_metadata).await?; - - // If no compaction is needed, return early - if !check_result.needs_compaction { - debug!( - "No compaction needed (usage: {:.1}% <= {:.1}% threshold)", - check_result.usage_ratio * 100.0, - check_result.percentage_until_compaction - ); - return Ok(AutoCompactResult { - compacted: false, - messages: Conversation::new_unvalidated(messages.to_vec()), - summarization_usage: None, - }); - } - - info!( - "Auto-compacting messages (usage: {:.1}%)", - check_result.usage_ratio * 100.0 - ); - - perform_compaction(agent, messages).await -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::conversation::message::{Message, MessageContent}; - use crate::session::extension_data; - use crate::{ - agents::Agent, - model::ModelConfig, - providers::base::{Provider, ProviderMetadata, ProviderUsage, Usage}, - providers::errors::ProviderError, - }; - use chrono::Utc; - use rmcp::model::{AnnotateAble, RawTextContent, Role, Tool}; - use std::sync::Arc; - - #[derive(Clone)] - struct MockProvider { - model_config: ModelConfig, - } - - #[async_trait::async_trait] - impl Provider for MockProvider { - fn metadata() -> ProviderMetadata { - ProviderMetadata::empty() - } - - fn get_model_config(&self) -> ModelConfig { - self.model_config.clone() - } - - async fn complete_with_model( - &self, - _model_config: &ModelConfig, - _system: &str, - _messages: &[Message], - _tools: &[Tool], - ) -> Result<(Message, ProviderUsage), ProviderError> { - // Return a short summary message - Ok(( - Message::new( - Role::Assistant, - Utc::now().timestamp(), - vec![MessageContent::Text( - RawTextContent { - text: "Summary of conversation".to_string(), - meta: None, - } - .no_annotation(), - )], - ), - ProviderUsage::new("mock".to_string(), Usage::default()), - )) - } - } - - fn create_test_message(text: &str) -> Message { - Message::new( - Role::User, - Utc::now().timestamp(), - vec![MessageContent::text(text.to_string())], - ) - } - - fn create_test_session_metadata( - message_count: usize, - working_dir: &str, - ) -> crate::session::Session { - use crate::conversation::Conversation; - use std::path::PathBuf; - - let mut conversation = Conversation::default(); - for i in 0..message_count { - conversation.push(create_test_message(format!("message {}", i).as_str())); - } - - crate::session::Session { - id: "test_session".to_string(), - working_dir: PathBuf::from(working_dir), - description: "Test session".to_string(), - created_at: Default::default(), - updated_at: Default::default(), - schedule_id: Some("test_job".to_string()), - recipe: None, - total_tokens: Some(100), - input_tokens: Some(50), - output_tokens: Some(50), - accumulated_total_tokens: Some(100), - accumulated_input_tokens: Some(50), - accumulated_output_tokens: Some(50), - extension_data: extension_data::ExtensionData::new(), - conversation: Some(conversation), - message_count, - user_recipe_values: None, - } - } - - #[tokio::test] - async fn test_check_compaction_needed() { - let mock_provider = Arc::new(MockProvider { - model_config: ModelConfig::new("test-model") - .unwrap() - .with_context_limit(Some(100_000)), - }); - - let agent = Agent::new(); - let _ = agent.update_provider(mock_provider).await; - - // Create small messages that won't trigger compaction - let messages = vec![create_test_message("Hello"), create_test_message("World")]; - - let result = check_compaction_needed(&agent, &messages, Some(0.3), None) - .await - .unwrap(); - - assert!(!result.needs_compaction); - assert!(result.current_tokens > 0); - assert!(result.context_limit > 0); - assert!(result.usage_ratio < 0.3); - assert!(result.remaining_tokens > 0); - assert!(result.percentage_until_compaction > 0.0); - } - - #[tokio::test] - async fn test_check_compaction_needed_disabled() { - let mock_provider = Arc::new(MockProvider { - model_config: ModelConfig::new("test-model") - .unwrap() - .with_context_limit(Some(100_000)), - }); - - let agent = Agent::new(); - let _ = agent.update_provider(mock_provider).await; - - let messages = vec![create_test_message("Hello")]; - - // Test with threshold 0 (disabled) - let result = check_compaction_needed(&agent, &messages, Some(0.0), None) - .await - .unwrap(); - - assert!(!result.needs_compaction); - - // Test with threshold 1.0 (disabled) - let result = check_compaction_needed(&agent, &messages, Some(1.0), None) - .await - .unwrap(); - - assert!(!result.needs_compaction); - } - - #[tokio::test] - async fn test_auto_compact_disabled() { - let mock_provider = Arc::new(MockProvider { - model_config: ModelConfig::new("test-model") - .unwrap() - .with_context_limit(Some(10_000)), - }); - - let agent = Agent::new(); - let _ = agent.update_provider(mock_provider).await; - - let messages = vec![create_test_message("Hello"), create_test_message("World")]; - - // Test with threshold 0 (disabled) - let result = check_and_compact_messages(&agent, &messages, Some(0.0), None) - .await - .unwrap(); - - assert!(!result.compacted); - assert_eq!(result.messages.len(), messages.len()); - assert!(result.summarization_usage.is_none()); - - // Test with threshold 1.0 (disabled) - let result = check_and_compact_messages(&agent, &messages, Some(1.0), None) - .await - .unwrap(); - - assert!(!result.compacted); - } - - #[tokio::test] - async fn test_auto_compact_below_threshold() { - let mock_provider = Arc::new(MockProvider { - model_config: ModelConfig::new("test-model") - .unwrap() - .with_context_limit(Some(100_000)), // Increased to ensure overhead doesn't dominate - }); - - let agent = Agent::new(); - let _ = agent.update_provider(mock_provider).await; - - // Create small messages that won't trigger compaction - let messages = vec![create_test_message("Hello"), create_test_message("World")]; - - let result = check_and_compact_messages(&agent, &messages, Some(0.3), None) - .await - .unwrap(); - - assert!(!result.compacted); - assert_eq!(result.messages.len(), messages.len()); - } - - #[tokio::test] - async fn test_auto_compact_above_threshold() { - let mock_provider = Arc::new(MockProvider { - model_config: ModelConfig::new("test-model") - .unwrap() - .with_context_limit(30_000.into()), // Smaller context limit to make threshold easier to hit - }); - - let agent = Agent::new(); - let _ = agent.update_provider(mock_provider).await; - - // Create messages that will exceed 30% of the context limit - // With 30k context limit, 30% is 9k tokens - let mut messages = Vec::new(); - - // Create much longer messages with more content to reach the threshold - for i in 0..300 { - messages.push(create_test_message(&format!( - "This is message number {} with significantly more content to increase token count substantially. \ - We need to ensure that our total token usage exceeds 30% of the available context \ - limit after accounting for system prompt and tools overhead. This message contains \ - multiple sentences to increase the token count substantially. Adding even more text here \ - to make sure we have enough tokens. Lorem ipsum dolor sit amet, consectetur adipiscing elit, \ - sed do eiusmod tempor incididunt ut labore et dolore magna aliqua. Ut enim ad minim veniam, \ - quis nostrud exercitation ullamco laboris nisi ut aliquip ex ea commodo consequat. Duis aute \ - irure dolor in reprehenderit in voluptate velit esse cillum dolore eu fugiat nulla pariatur. \ - Excepteur sint occaecat cupidatat non proident, sunt in culpa qui officia deserunt mollit \ - anim id est laborum. Sed ut perspiciatis unde omnis iste natus error sit voluptatem accusantium \ - doloremque laudantium, totam rem aperiam, eaque ipsa quae ab illo inventore veritatis et quasi \ - architecto beatae vitae dicta sunt explicabo.", - i - ))); - } - - let result = check_and_compact_messages(&agent, &messages, Some(0.3), None) - .await - .unwrap(); - - if !result.compacted { - eprintln!("Test failed - compaction not triggered"); - } - - assert!(result.compacted); - assert!(result.summarization_usage.is_some()); - - // Verify that summarization usage contains token counts - if let Some(usage) = &result.summarization_usage { - assert!(usage.usage.total_tokens.is_some()); - let after = usage.usage.total_tokens.unwrap_or(0) as usize; - assert!( - after > 0, - "Token count after compaction should be greater than 0" - ); - } - - // After visibility implementation, we keep all messages plus summary - // Original messages become user_visible only, summary becomes agent_visible only - assert!(result.messages.len() > messages.len()); - } - - #[tokio::test] - async fn test_auto_compact_respects_config() { - let mock_provider = Arc::new(MockProvider { - model_config: ModelConfig::new("test-model") - .unwrap() - .with_context_limit(Some(30_000)), // Smaller context limit to make threshold easier to hit - }); - - let agent = Agent::new(); - let _ = agent.update_provider(mock_provider).await; - - // Create enough messages to trigger compaction with low threshold - let mut messages = Vec::new(); - // With 30k context limit, after overhead we have ~27k usable tokens - // 10% of 27k = 2.7k tokens, so we need messages that exceed that - for i in 0..200 { - messages.push(create_test_message(&format!( - "Message {} with enough content to ensure we exceed 10% of the context limit. \ - Adding more content to increase token count substantially. This message contains \ - multiple sentences to increase the token count. We need to ensure that our total \ - token usage exceeds 10% of the available context limit after accounting for \ - system prompt and tools overhead.", - i - ))); - } - - // Set config value - let config = Config::global(); - config - .set_param("GOOSE_AUTO_COMPACT_THRESHOLD", serde_json::Value::from(0.1)) - .unwrap(); - - // Should use config value when no override provided - let result = check_and_compact_messages(&agent, &messages, None, None) - .await - .unwrap(); - - // Debug info if not compacted - if !result.compacted { - eprintln!("Test failed - compaction not triggered"); - } - - // With such a low threshold (10%), it should compact - assert!(result.compacted); - - // Clean up config - config - .set_param("GOOSE_AUTO_COMPACT_THRESHOLD", serde_json::Value::from(0.3)) - .unwrap(); - } - - #[tokio::test] - async fn test_auto_compact_uses_session_metadata() { - use crate::session::Session; - - let mock_provider = Arc::new(MockProvider { - model_config: ModelConfig::new("test-model") - .unwrap() - .with_context_limit(10_000.into()), - }); - - let agent = Agent::new(); - let _ = agent.update_provider(mock_provider).await; - - // Create some test messages - let messages = vec![ - create_test_message("First message"), - create_test_message("Second message"), - ]; - - // Create session with specific token counts - #[allow(clippy::field_reassign_with_default)] - let mut session = Session::default(); - { - session.total_tokens = Some(8000); // High token count to trigger compaction - session.accumulated_total_tokens = Some(15000); // Even higher accumulated count - session.input_tokens = Some(5000); - session.output_tokens = Some(3000); - } - - // Test with session - should use total_tokens for compaction (not accumulated) - let result_with_metadata = check_compaction_needed( - &agent, - &messages, - Some(0.3), // 30% threshold - Some(&session), - ) - .await - .unwrap(); - - // With 8000 tokens and context limit around 10000, should trigger compaction - assert!(result_with_metadata.needs_compaction); - assert_eq!(result_with_metadata.current_tokens, 8000); - - // Test without session metadata - should use estimated tokens - let result_without_metadata = check_compaction_needed( - &agent, - &messages, - Some(0.3), // 30% threshold - None, - ) - .await - .unwrap(); - - // Without metadata, should use much lower estimated token count - assert!(!result_without_metadata.needs_compaction); - assert!(result_without_metadata.current_tokens < 8000); - - // Test with session that has only accumulated tokens (no total_tokens) - let mut session_metadata_no_total = Session::default(); - #[allow(clippy::field_reassign_with_default)] - { - session_metadata_no_total.accumulated_total_tokens = Some(7500); - } - - let result_with_no_total = check_compaction_needed( - &agent, - &messages, - Some(0.3), // 30% threshold - Some(&session_metadata_no_total), - ) - .await - .unwrap(); - - // Should fall back to estimation since total_tokens is None - assert!(!result_with_no_total.needs_compaction); - assert!(result_with_no_total.current_tokens < 7500); - - // Test with metadata that has no token counts - should fall back to estimation - let empty_metadata = Session::default(); - - let result_with_empty_metadata = check_compaction_needed( - &agent, - &messages, - Some(0.3), // 30% threshold - Some(&empty_metadata), - ) - .await - .unwrap(); - - // Should fall back to estimation - assert!(!result_with_empty_metadata.needs_compaction); - assert!(result_with_empty_metadata.current_tokens < 7500); - } - - #[tokio::test] - async fn test_auto_compact_end_to_end_with_metadata() { - use crate::session::Session; - - let mock_provider = Arc::new(MockProvider { - model_config: ModelConfig::new("test-model") - .unwrap() - .with_context_limit(10_000.into()), - }); - - let agent = Agent::new(); - let _ = agent.update_provider(mock_provider).await; - - // Create some test messages - let messages = vec![ - create_test_message("First message"), - create_test_message("Second message"), - create_test_message("Third message"), - create_test_message("Fourth message"), - create_test_message("Fifth message"), - ]; - - // Create session metadata with high token count to trigger compaction - let mut session = Session::default(); - #[allow(clippy::field_reassign_with_default)] - { - session.total_tokens = Some(9000); // High enough to trigger compaction - } - - // Test full compaction flow with session metadata - let result = check_and_compact_messages( - &agent, - &messages, - Some(0.3), // 30% threshold - Some(&session), - ) - .await - .unwrap(); - - // Should have triggered compaction - assert!(result.compacted); - assert!(result.summarization_usage.is_some()); - - // Verify the compacted messages are returned - assert!(!result.messages.is_empty()); - - // After visibility implementation, we keep all messages plus summary - // Original messages become user_visible only, summary becomes agent_visible only - assert!(result.messages.len() > messages.len()); - } - - #[tokio::test] - async fn test_auto_compact_with_comprehensive_session_metadata() { - let mock_provider = Arc::new(MockProvider { - model_config: ModelConfig::new("test-model") - .unwrap() - .with_context_limit(8_000.into()), - }); - - let agent = Agent::new(); - let _ = agent.update_provider(mock_provider).await; - - let messages = vec![ - create_test_message("Test message 1"), - create_test_message("Test message 2"), - create_test_message("Test message 3"), - ]; - - // Use the helper function to create comprehensive non-null session metadata - let comprehensive_metadata = create_test_session_metadata(3, "/test/working/dir"); - - // Verify the helper created non-null metadata - assert_eq!( - comprehensive_metadata - .clone() - .conversation - .unwrap_or_default() - .len(), - 3 - ); - assert_eq!( - comprehensive_metadata.working_dir.to_str().unwrap(), - "/test/working/dir" - ); - assert_eq!(comprehensive_metadata.description, "Test session"); - assert_eq!( - comprehensive_metadata.schedule_id, - Some("test_job".to_string()) - ); - assert_eq!(comprehensive_metadata.total_tokens, Some(100)); - assert_eq!(comprehensive_metadata.input_tokens, Some(50)); - assert_eq!(comprehensive_metadata.output_tokens, Some(50)); - assert_eq!(comprehensive_metadata.accumulated_total_tokens, Some(100)); - assert_eq!(comprehensive_metadata.accumulated_input_tokens, Some(50)); - assert_eq!(comprehensive_metadata.accumulated_output_tokens, Some(50)); - - // Test compaction with the comprehensive metadata (low token count, shouldn't compact) - let result_low_tokens = check_compaction_needed( - &agent, - &messages, - Some(0.7), // 70% threshold - Some(&comprehensive_metadata), - ) - .await - .unwrap(); - - assert!(!result_low_tokens.needs_compaction); - assert_eq!(result_low_tokens.current_tokens, 100); // Should use total_tokens from metadata - - // Create a modified version with high token count to trigger compaction - let mut high_token_metadata = create_test_session_metadata(5, "/test/working/dir"); - high_token_metadata.total_tokens = Some(6_000); // High enough to trigger compaction - high_token_metadata.input_tokens = Some(4_000); - high_token_metadata.output_tokens = Some(2_000); - high_token_metadata.accumulated_total_tokens = Some(12_000); - - let result_high_tokens = check_compaction_needed( - &agent, - &messages, - Some(0.7), // 70% threshold - Some(&high_token_metadata), - ) - .await - .unwrap(); - - assert!(result_high_tokens.needs_compaction); - assert_eq!(result_high_tokens.current_tokens, 6_000); // Should use total_tokens, not accumulated - - // Test that metadata fields are preserved correctly in edge cases - let mut edge_case_metadata = create_test_session_metadata(10, "/edge/case/dir"); - edge_case_metadata.total_tokens = None; // No total tokens - edge_case_metadata.accumulated_total_tokens = Some(7_000); // Has accumulated - - let result_edge_case = check_compaction_needed( - &agent, - &messages, - Some(0.5), // 50% threshold - Some(&edge_case_metadata), - ) - .await - .unwrap(); - - // Should fall back to estimation since total_tokens is None - assert!(result_edge_case.current_tokens < 7_000); - // With estimation, likely won't trigger compaction - assert!(!result_edge_case.needs_compaction); - } -} diff --git a/crates/goose/src/context_mgmt/common.rs b/crates/goose/src/context_mgmt/common.rs deleted file mode 100644 index 106f8a28b8b3..000000000000 --- a/crates/goose/src/context_mgmt/common.rs +++ /dev/null @@ -1,99 +0,0 @@ -use std::sync::Arc; - -use rmcp::model::Tool; - -use crate::conversation::message::Message; -use crate::{ - providers::base::Provider, - token_counter::{AsyncTokenCounter, TokenCounter}, -}; - -const ESTIMATE_FACTOR: f32 = 0.7; -pub const SYSTEM_PROMPT_TOKEN_OVERHEAD: usize = 3_000; -pub const TOOLS_TOKEN_OVERHEAD: usize = 5_000; - -pub fn estimate_target_context_limit(provider: Arc) -> usize { - let model_context_limit = provider.get_model_config().context_limit(); - - // Our conservative estimate of the **target** context limit - // Our token count is an estimate since model providers often don't provide the tokenizer (eg. Claude) - let target_limit = (model_context_limit as f32 * ESTIMATE_FACTOR) as usize; - - // subtract out overhead for system prompt and tools, but ensure we don't go negative - let overhead = SYSTEM_PROMPT_TOKEN_OVERHEAD + TOOLS_TOKEN_OVERHEAD; - if target_limit > overhead { - target_limit - overhead - } else { - // If overhead is larger than target limit, return a minimal usable limit - std::cmp::max(target_limit / 2, 1000) - } -} - -pub fn get_messages_token_counts(token_counter: &TokenCounter, messages: &[Message]) -> Vec { - // Calculate current token count of each message, use count_chat_tokens to ensure we - // capture the full content of the message, include ToolRequests and ToolResponses - messages - .iter() - .map(|msg| token_counter.count_chat_tokens("", std::slice::from_ref(msg), &[])) - .collect() -} - -/// Async version of get_messages_token_counts for better performance -pub fn get_messages_token_counts_async( - token_counter: &AsyncTokenCounter, - messages: &[Message], -) -> Vec { - messages - .iter() - .filter(|m| m.is_agent_visible()) - .map(|msg| token_counter.count_chat_tokens("", std::slice::from_ref(msg), &[])) - .collect() -} - -// These are not being used now but could be useful in the future - -#[allow(dead_code)] -pub struct ChatTokenCounts { - pub system: usize, - pub tools: usize, - pub messages: Vec, -} - -#[allow(dead_code)] -pub fn get_token_counts( - token_counter: &TokenCounter, - messages: &mut [Message], - system_prompt: &str, - tools: &mut Vec, -) -> ChatTokenCounts { - // Take into account the system prompt (includes goosehints), and our tools input - let system_prompt_token_count = token_counter.count_tokens(system_prompt); - let tools_token_count = token_counter.count_tokens_for_tools(tools.as_slice()); - let messages_token_count = get_messages_token_counts(token_counter, messages); - - ChatTokenCounts { - system: system_prompt_token_count, - tools: tools_token_count, - messages: messages_token_count, - } -} - -/// Async version of get_token_counts for better performance -#[allow(dead_code)] -pub fn get_token_counts_async( - token_counter: &AsyncTokenCounter, - messages: &mut [Message], - system_prompt: &str, - tools: &mut Vec, -) -> ChatTokenCounts { - // Take into account the system prompt (includes goosehints), and our tools input - let system_prompt_token_count = token_counter.count_tokens(system_prompt); - let tools_token_count = token_counter.count_tokens_for_tools(tools.as_slice()); - let messages_token_count = get_messages_token_counts_async(token_counter, messages); - - ChatTokenCounts { - system: system_prompt_token_count, - tools: tools_token_count, - messages: messages_token_count, - } -} diff --git a/crates/goose/src/context_mgmt/mod.rs b/crates/goose/src/context_mgmt/mod.rs index 00d11d6b871b..41de720a1c11 100644 --- a/crates/goose/src/context_mgmt/mod.rs +++ b/crates/goose/src/context_mgmt/mod.rs @@ -1,6 +1,338 @@ -pub mod auto_compact; -mod common; -pub mod summarize; -pub mod truncate; +use crate::conversation::message::Message; +use crate::conversation::message::MessageMetadata; +use crate::conversation::Conversation; +use crate::prompt_template::render_global_file; +use crate::providers::base::{Provider, ProviderUsage}; +use crate::{agents::Agent, config::Config, token_counter::create_token_counter}; +use anyhow::Result; +use rmcp::model::Role; +use serde::Serialize; +use std::sync::Arc; +use tracing::{debug, info}; -pub use common::*; +pub const DEFAULT_COMPACTION_THRESHOLD: f64 = 0.8; + +/// Result of auto-compaction check +#[derive(Debug)] +pub struct AutoCompactResult { + /// Whether compaction was performed + pub compacted: bool, + /// The messages after potential compaction + pub messages: Conversation, + /// Provider usage from summarization (if compaction occurred) + /// This contains the actual token counts after compaction + pub summarization_usage: Option, +} + +/// Result of checking if compaction is needed +#[derive(Debug)] +pub struct CompactionCheckResult { + /// Whether compaction is needed + pub needs_compaction: bool, + /// Current token count + pub current_tokens: usize, + /// Context limit being used + pub context_limit: usize, + /// Current usage ratio (0.0 to 1.0) + pub usage_ratio: f64, + /// Remaining tokens before compaction threshold + pub remaining_tokens: usize, + /// Percentage until compaction threshold (0.0 to 100.0) + pub percentage_until_compaction: f64, +} + +#[derive(Serialize)] +struct SummarizeContext { + messages: String, +} + +/// Check if messages need compaction and compact them if necessary +/// +/// This function combines checking and compaction. It first checks if compaction +/// is needed based on the threshold, and if so, performs the compaction by +/// summarizing messages and updating their visibility metadata. +/// +/// # Arguments +/// * `agent` - The agent to use for context management +/// * `messages` - The current message history +/// * `force_compact` - If true, skip the threshold check and force compaction +/// * `preserve_last_user_message` - If true and last message is not a user message, copy the most recent user message to the end +/// * `threshold_override` - Optional threshold override (defaults to GOOSE_AUTO_COMPACT_THRESHOLD config) +/// * `session_metadata` - Optional session metadata containing actual token counts +/// +/// # Returns +/// * A tuple containing: +/// - `bool`: Whether compaction was performed +/// - `Conversation`: The potentially compacted messages +/// - `Vec`: Indices of removed messages (empty if no compaction) +/// - `Option`: Provider usage from summarization (if compaction occurred) +pub async fn check_and_compact_messages( + agent: &Agent, + messages_with_user_message: &[Message], + force_compact: bool, + preserve_last_user_message: bool, + threshold_override: Option, + session_metadata: Option<&crate::session::Session>, +) -> std::result::Result<(bool, Conversation, Vec, Option), anyhow::Error> { + // Check if compaction is needed (unless forced) + if !force_compact { + let check_result = check_compaction_needed( + agent, + messages_with_user_message, + threshold_override, + session_metadata, + ) + .await?; + + // If no compaction is needed, return early + if !check_result.needs_compaction { + debug!( + "No compaction needed (usage: {:.1}% <= {:.1}% threshold)", + check_result.usage_ratio * 100.0, + check_result.percentage_until_compaction + ); + return Ok(( + false, + Conversation::new_unvalidated(messages_with_user_message.to_vec()), + Vec::new(), + None, + )); + } + + info!( + "Performing message compaction (usage: {:.1}%)", + check_result.usage_ratio * 100.0 + ); + } else { + info!("Forcing message compaction due to context limit exceeded"); + } + + // Perform the actual compaction + // Check if the most recent message is a user message + let (messages, preserved_user_message) = + if let Some(last_message) = messages_with_user_message.last() { + if matches!(last_message.role, rmcp::model::Role::User) { + // Remove the last user message before compaction + ( + &messages_with_user_message[..messages_with_user_message.len() - 1], + Some(last_message.clone()), + ) + } else if preserve_last_user_message { + // Last message is not a user message, but we want to preserve the most recent user message + // Find the most recent user message and copy it (don't remove from history) + let most_recent_user_message = messages_with_user_message + .iter() + .rev() + .find(|msg| matches!(msg.role, rmcp::model::Role::User)) + .cloned(); + (messages_with_user_message, most_recent_user_message) + } else { + (messages_with_user_message, None) + } + } else { + (messages_with_user_message, None) + }; + + let provider = agent.provider().await?; + let summary = summarize(provider.clone(), messages).await?; + + let (summary_message, summarization_usage) = match summary { + Some((summary_message, provider_usage)) => (summary_message, Some(provider_usage)), + None => { + // No summary was generated (empty input) + tracing::warn!("Summarization failed. Returning empty messages."); + return Ok((false, Conversation::empty(), vec![], None)); + } + }; + + // Create the final message list with updated visibility metadata: + // 1. Original messages become user_visible but not agent_visible + // 2. Summary message becomes agent_visible but not user_visible + // 3. Assistant messages to continue the conversation remain both user_visible and agent_visible + + let mut final_messages = Vec::new(); + let mut final_token_counts = Vec::new(); + + // Add all original messages with updated visibility (preserve user_visible, set agent_visible=false) + for msg in messages.iter().cloned() { + let updated_metadata = msg.metadata.with_agent_invisible(); + let updated_msg = msg.with_metadata(updated_metadata); + final_messages.push(updated_msg); + // Token count doesn't matter for agent_visible=false messages, but we'll use 0 + final_token_counts.push(0); + } + + // Add the compaction marker (user_visible=true, agent_visible=false) + let compaction_marker = Message::assistant() + .with_conversation_compacted("Conversation compacted and summarized") + .with_metadata(MessageMetadata::user_only()); + let compaction_marker_tokens: usize = 0; // Not counted since agent_visible=false + final_messages.push(compaction_marker); + final_token_counts.push(compaction_marker_tokens); + + // Add the summary message (agent_visible=true, user_visible=false) + let summary_msg = summary_message.with_metadata(MessageMetadata::agent_only()); + // For token counting purposes, we use the output tokens (the actual summary content) + // since that's what will be in the context going forward + let summary_tokens = summarization_usage + .as_ref() + .and_then(|usage| usage.usage.output_tokens) + .unwrap_or(0) as usize; + final_messages.push(summary_msg); + final_token_counts.push(summary_tokens); + + // Add an assistant message to continue the conversation (agent_visible=true, user_visible=false) + let assistant_message = Message::assistant() + .with_text( + "The previous message contains a summary that was prepared because a context limit was reached. +Do not mention that you read a summary or that conversation summarization occurred +Just continue the conversation naturally based on the summarized context" + ) + .with_metadata(MessageMetadata::agent_only()); + let assistant_message_tokens: usize = 0; // Not counted since it's for agent context only + final_messages.push(assistant_message); + final_token_counts.push(assistant_message_tokens); + + // Add back the preserved user message if it exists + if let Some(user_message) = preserved_user_message { + final_messages.push(user_message); + } + + Ok(( + true, + Conversation::new_unvalidated(final_messages), + final_token_counts, + summarization_usage, + )) +} + +/// Check if messages need compaction without performing the compaction +/// +/// This function analyzes the current token usage and returns detailed information +/// about whether compaction is needed and how close we are to the threshold. +/// It prioritizes actual token counts from session metadata when available, +/// falling back to estimated counts if needed. +/// +/// # Arguments +/// * `agent` - The agent to use for context management +/// * `messages` - The current message history +/// * `threshold_override` - Optional threshold override (defaults to GOOSE_AUTO_COMPACT_THRESHOLD config) +/// * `session_metadata` - Optional session metadata containing actual token counts +/// +/// # Returns +/// * `CompactionCheckResult` containing detailed information about compaction needs +async fn check_compaction_needed( + agent: &Agent, + messages: &[Message], + threshold_override: Option, + session_metadata: Option<&crate::session::Session>, +) -> Result { + // Get threshold from config or use override + let config = Config::global(); + // TODO(Douwe): check the default here; it seems to reset to 0.3 sometimes + let threshold = threshold_override.unwrap_or_else(|| { + config + .get_param::("GOOSE_AUTO_COMPACT_THRESHOLD") + .unwrap_or(DEFAULT_COMPACTION_THRESHOLD) + }); + + let provider = agent.provider().await?; + let context_limit = provider.get_model_config().context_limit(); + + let (current_tokens, token_source) = match session_metadata.and_then(|m| m.total_tokens) { + Some(tokens) => (tokens as usize, "session metadata"), + None => { + let token_counter = create_token_counter() + .await + .map_err(|e| anyhow::anyhow!("Failed to create token counter: {}", e))?; + + let token_counts: Vec<_> = messages + .iter() + .filter(|m| m.is_agent_visible()) + .map(|msg| token_counter.count_chat_tokens("", std::slice::from_ref(msg), &[])) + .collect(); + + (token_counts.iter().sum(), "estimated") + } + }; + + let usage_ratio = current_tokens as f64 / context_limit as f64; + + let threshold_tokens = (context_limit as f64 * threshold) as usize; + let remaining_tokens = threshold_tokens.saturating_sub(current_tokens); + + let percentage_until_compaction = if usage_ratio < threshold { + (threshold - usage_ratio) * 100.0 + } else { + 0.0 + }; + + let needs_compaction = if threshold <= 0.0 || threshold >= 1.0 { + usage_ratio > DEFAULT_COMPACTION_THRESHOLD + } else { + usage_ratio > threshold + }; + + debug!( + "Compaction check: {} / {} tokens ({:.1}%), threshold: {:.1}%, needs compaction: {}, source: {}", + current_tokens, + context_limit, + usage_ratio * 100.0, + threshold * 100.0, + needs_compaction, + token_source + ); + + Ok(CompactionCheckResult { + needs_compaction, + current_tokens, + context_limit, + usage_ratio, + remaining_tokens, + percentage_until_compaction, + }) +} + +async fn summarize( + provider: Arc, + messages: &[Message], +) -> anyhow::Result, anyhow::Error> { + if messages.is_empty() { + return std::prelude::rust_2015::Ok(None); + } + + // Format all messages as a single string for the summarization prompt + let messages_text = messages + .iter() + .map(|msg| format!("{:?}", msg)) + .collect::>() + .join("\n\n"); + + let context = SummarizeContext { + messages: messages_text, + }; + + // Render the one-shot summarization prompt + let system_prompt = render_global_file("summarize_oneshot.md", &context)?; + + // Create a simple user message requesting summarization + let user_message = Message::user() + .with_text("Please summarize the conversation history provided in the system prompt."); + let summarization_request = vec![user_message]; + + // Send the request to the provider and fetch the response + let (mut response, mut provider_usage) = provider + .complete_fast(&system_prompt, &summarization_request, &[]) + .await?; + + // Set role to user as it will be used in following conversation as user content + response.role = Role::User; + + // Ensure we have token counts, estimating if necessary + provider_usage + .ensure_tokens(&system_prompt, &summarization_request, &response, &[]) + .await + .map_err(|e| anyhow::anyhow!("Failed to ensure usage tokens: {}", e))?; + + std::prelude::rust_2015::Ok(Some((response, provider_usage))) +} diff --git a/crates/goose/src/context_mgmt/summarize.rs b/crates/goose/src/context_mgmt/summarize.rs deleted file mode 100644 index 35e5aef8e293..000000000000 --- a/crates/goose/src/context_mgmt/summarize.rs +++ /dev/null @@ -1,187 +0,0 @@ -use crate::conversation::message::Message; -use crate::prompt_template::render_global_file; -use crate::providers::base::Provider; - -use anyhow::Result; -use rmcp::model::Role; -use serde::Serialize; -use std::sync::Arc; - -#[derive(Serialize)] -struct SummarizeContext { - messages: String, -} - -use crate::providers::base::ProviderUsage; - -/// Summarization function that uses the detailed prompt from the markdown template -pub async fn summarize_messages( - provider: Arc, - messages: &[Message], -) -> Result, anyhow::Error> { - if messages.is_empty() { - return Ok(None); - } - - // Format all messages as a single string for the summarization prompt - let messages_text = messages - .iter() - .map(|msg| format!("{:?}", msg)) - .collect::>() - .join("\n\n"); - - let context = SummarizeContext { - messages: messages_text, - }; - - // Render the one-shot summarization prompt - let system_prompt = render_global_file("summarize_oneshot.md", &context)?; - - // Create a simple user message requesting summarization - let user_message = Message::user() - .with_text("Please summarize the conversation history provided in the system prompt."); - let summarization_request = vec![user_message]; - - // Send the request to the provider and fetch the response - let (mut response, mut provider_usage) = provider - .complete_fast(&system_prompt, &summarization_request, &[]) - .await?; - - // Set role to user as it will be used in following conversation as user content - response.role = Role::User; - - // Ensure we have token counts, estimating if necessary - provider_usage - .ensure_tokens(&system_prompt, &summarization_request, &response, &[]) - .await - .map_err(|e| anyhow::anyhow!("Failed to ensure usage tokens: {}", e))?; - - Ok(Some((response, provider_usage))) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::conversation::message::{Message, MessageContent}; - use crate::model::ModelConfig; - use crate::providers::base::{ProviderMetadata, ProviderUsage, Usage}; - use crate::providers::errors::ProviderError; - use chrono::Utc; - use rmcp::model::Role; - use rmcp::model::Tool; - use rmcp::model::{AnnotateAble, RawTextContent}; - use std::sync::Arc; - - #[derive(Clone)] - struct MockProvider { - model_config: ModelConfig, - } - - #[async_trait::async_trait] - impl Provider for MockProvider { - fn metadata() -> ProviderMetadata { - ProviderMetadata::empty() - } - - fn get_model_config(&self) -> ModelConfig { - self.model_config.clone() - } - - async fn complete_with_model( - &self, - _model_config: &ModelConfig, - _system: &str, - _messages: &[Message], - _tools: &[Tool], - ) -> Result<(Message, ProviderUsage), ProviderError> { - Ok(( - Message::new( - Role::Assistant, - Utc::now().timestamp(), - vec![MessageContent::Text( - RawTextContent { - text: "Summarized content".to_string(), - meta: None, - } - .no_annotation(), - )], - ), - ProviderUsage::new( - "mock".to_string(), - Usage { - input_tokens: Some(100), - output_tokens: Some(50), - total_tokens: Some(150), - }, - ), - )) - } - } - - fn create_mock_provider() -> Result> { - let mock_model_config = ModelConfig::new("test-model")?.with_context_limit(Some(200_000)); - - Ok(Arc::new(MockProvider { - model_config: mock_model_config, - })) - } - - fn create_test_messages() -> Vec { - vec![ - set_up_text_message("Message 1", Role::User), - set_up_text_message("Message 2", Role::Assistant), - set_up_text_message("Message 3", Role::User), - ] - } - - fn set_up_text_message(text: &str, role: Role) -> Message { - Message::new(role, 0, vec![MessageContent::text(text.to_string())]) - } - - #[tokio::test] - async fn test_summarize_messages_basic() { - let provider = create_mock_provider().expect("failed to create mock provider"); - let messages = create_test_messages(); - - let result = summarize_messages(Arc::clone(&provider), &messages).await; - - assert!(result.is_ok(), "The function should return Ok."); - let summary_result = result.unwrap(); - - assert!( - summary_result.is_some(), - "The summary should contain a result." - ); - let (summarized_message, provider_usage) = summary_result.unwrap(); - - assert_eq!( - summarized_message.role, - Role::User, - "The summarized message should be from the user." - ); - assert!( - provider_usage.usage.input_tokens.unwrap_or(0) > 0, - "Should have input token count" - ); - assert!( - provider_usage.usage.output_tokens.unwrap_or(0) > 0, - "Should have output token count" - ); - } - - #[tokio::test] - async fn test_summarize_messages_empty_input() { - let provider = create_mock_provider().expect("failed to create mock provider"); - let messages: Vec = Vec::new(); - - let result = summarize_messages(Arc::clone(&provider), &messages).await; - - assert!(result.is_ok(), "The function should return Ok."); - let summary_result = result.unwrap(); - - assert!( - summary_result.is_none(), - "The summary should be None for empty input." - ); - } -} diff --git a/crates/goose/src/context_mgmt/truncate.rs b/crates/goose/src/context_mgmt/truncate.rs deleted file mode 100644 index b6f74c597225..000000000000 --- a/crates/goose/src/context_mgmt/truncate.rs +++ /dev/null @@ -1,749 +0,0 @@ -use crate::conversation::message::{Message, MessageContent}; -use crate::conversation::Conversation; -use crate::utils::safe_truncate; -use anyhow::{anyhow, Result}; -use rmcp::model::{RawContent, ResourceContents, Role}; -use std::collections::HashSet; -use std::ops::DerefMut; -use tracing::{debug, warn}; - -/// Maximum size for truncated content in characters -const MAX_TRUNCATED_CONTENT_SIZE: usize = 5000; - -/// Handles messages that are individually larger than the context limit -/// by truncating their content rather than removing them entirely -fn handle_oversized_messages( - messages: &[Message], - token_counts: &[usize], - context_limit: usize, - strategy: &dyn TruncationStrategy, -) -> Result<(Conversation, Vec), anyhow::Error> { - let mut truncated_messages = Vec::new(); - let mut truncated_token_counts = Vec::new(); - let mut any_truncated = false; - - // Create a basic token counter for re-estimating truncated content - // Note: This is a rough approximation since we don't have access to the actual tokenizer here - let estimate_tokens = |text: &str| -> usize { - // Rough approximation: 1 token per 4 characters for English text - (text.len() / 4).max(1) - }; - - for (i, (message, &original_tokens)) in messages.iter().zip(token_counts.iter()).enumerate() { - if original_tokens > context_limit { - warn!( - "Message {} has {} tokens, exceeding context limit of {}", - i, original_tokens, context_limit - ); - - // Try to truncate the message content - let truncated_message = truncate_message_content(message, MAX_TRUNCATED_CONTENT_SIZE)?; - let estimated_new_tokens = - estimate_message_tokens(&truncated_message, &estimate_tokens); - - if estimated_new_tokens > context_limit { - // Even truncated message is too large, skip it entirely - warn!("Skipping message {} as even truncated version ({} tokens) exceeds context limit", i, estimated_new_tokens); - any_truncated = true; - continue; - } - - truncated_messages.push(truncated_message); - truncated_token_counts.push(estimated_new_tokens); - any_truncated = true; - } else { - truncated_messages.push(message.clone()); - truncated_token_counts.push(original_tokens); - } - } - - if any_truncated { - debug!("Truncated large message content, now attempting normal truncation"); - // After content truncation, try normal truncation if still needed - return truncate_messages( - &truncated_messages, - &truncated_token_counts, - context_limit, - strategy, - ); - } - - Ok(( - Conversation::new_unvalidated(truncated_messages), - truncated_token_counts, - )) -} - -/// Truncates the content within a message while preserving its structure -fn truncate_message_content(message: &Message, max_content_size: usize) -> Result { - let mut new_message = message.clone(); - - for content in &mut new_message.content { - match content { - MessageContent::Text(text_content) => { - if text_content.text.chars().count() > max_content_size { - let truncated = format!( - "{}\n\n[... content truncated from {} to {} characters ...]", - safe_truncate(&text_content.text, max_content_size), - text_content.text.chars().count(), - max_content_size - ); - text_content.text = truncated; - } - } - MessageContent::ToolResponse(tool_response) => { - if let Ok(ref mut result) = tool_response.tool_result { - for content_item in result { - if let RawContent::Text(ref mut text_content) = content_item.deref_mut() { - if text_content.text.chars().count() > max_content_size { - let truncated = format!( - "{}\n\n[... tool response truncated from {} to {} characters ...]", - safe_truncate(&text_content.text, max_content_size), - text_content.text.chars().count(), - max_content_size - ); - text_content.text = truncated; - } - } - // Handle Resource content which might contain large text - else if let RawContent::Resource(ref mut resource_content) = - content_item.deref_mut() - { - if let ResourceContents::TextResourceContents { text, .. } = - &mut resource_content.resource - { - if text.chars().count() > max_content_size { - let truncated = format!( - "{}\n\n[... resource content truncated from {} to {} characters ...]", - safe_truncate(text, max_content_size), - text.chars().count(), - max_content_size - ); - *text = truncated; - } - } - } - } - } - } - // Other content types are typically smaller, but we could extend this if needed - _ => {} - } - } - - Ok(new_message) -} - -/// Estimates token count for a message using a simple heuristic -fn estimate_message_tokens(message: &Message, estimate_fn: &dyn Fn(&str) -> usize) -> usize { - let mut total_tokens = 10; // Base overhead for message structure - - for content in &message.content { - match content { - MessageContent::Text(text_content) => { - total_tokens += estimate_fn(&text_content.text); - } - MessageContent::ToolResponse(tool_response) => { - if let Ok(ref result) = tool_response.tool_result { - for content_item in result { - match &content_item.raw { - RawContent::Text(text_content) => { - total_tokens += estimate_fn(&text_content.text); - } - RawContent::Resource(resource) => { - match &resource.resource { - ResourceContents::TextResourceContents { text, .. } => { - total_tokens += estimate_fn(text); - } - _ => total_tokens += 5, // Small overhead for other resource types - } - } - _ => { - total_tokens += 5; // Small overhead for other content types - } - } - } - } - } - _ => total_tokens += 5, // Small overhead for other content types - } - } - - total_tokens -} - -/// Truncates the messages to fit within the model's context window. -/// Mutates the input messages and token counts in place. -/// Returns an error if it's impossible to truncate the messages within the context limit. -/// - messages: The vector of messages in the conversation. -/// - token_counts: A parallel vector containing the token count for each message. -/// - context_limit: The maximum allowed context length in tokens. -/// - strategy: The truncation strategy to use. Only option is OldestFirstTruncation. -pub fn truncate_messages( - messages: &[Message], - token_counts: &[usize], - context_limit: usize, - strategy: &dyn TruncationStrategy, -) -> Result<(Conversation, Vec), anyhow::Error> { - let mut messages = messages.to_owned(); - let mut token_counts = token_counts.to_owned(); - - if messages.len() != token_counts.len() { - return Err(anyhow!( - "The vector for messages and token_counts must have same length" - )); - } - - // Step 1: Calculate total tokens - let mut total_tokens: usize = token_counts.iter().sum(); - debug!("Total tokens before truncation: {}", total_tokens); - - // Check if any individual message is larger than the context limit - // First, check for any message that's too large - let max_message_tokens = token_counts.iter().max().copied().unwrap_or(0); - if max_message_tokens > context_limit { - // Try to handle large messages by truncating their content - debug!( - "Found oversized message with {} tokens, attempting content truncation", - max_message_tokens - ); - return handle_oversized_messages(&messages, &token_counts, context_limit, strategy); - } - - let min_user_msg_tokens = messages - .iter() - .zip(token_counts.iter()) - .filter(|(msg, _)| msg.role == Role::User && msg.has_only_text_content()) - .map(|(_, &tokens)| tokens) - .min(); - - // If there are no valid user messages, or the smallest one is too big for the context - if min_user_msg_tokens.is_none() || min_user_msg_tokens.unwrap() > context_limit { - return Err(anyhow!( - "Not possible to truncate messages within context limit: no suitable user messages found" - )); - } - - if total_tokens <= context_limit { - return Ok(( - Conversation::new_unvalidated(messages.to_vec()), - token_counts.to_vec(), - )); // No truncation needed - } - - // Step 2: Determine indices to remove based on strategy - let indices_to_remove = - strategy.determine_indices_to_remove(&messages, &token_counts, context_limit)?; - - // Circuit breaker: if we can't remove enough messages, fail gracefully - let tokens_to_remove: usize = indices_to_remove - .iter() - .map(|&i| token_counts.get(i).copied().unwrap_or(0)) - .sum(); - - if total_tokens - tokens_to_remove > context_limit && !indices_to_remove.is_empty() { - debug!( - "Standard truncation insufficient: {} tokens remain after removing {} tokens", - total_tokens - tokens_to_remove, - tokens_to_remove - ); - // Try more aggressive truncation or content truncation - return handle_oversized_messages(&messages, &token_counts, context_limit, strategy); - } - - if indices_to_remove.is_empty() && total_tokens > context_limit { - return Err(anyhow!( - "Cannot truncate any messages: all messages may be essential or too large individually" - )); - } - - // Step 3: Remove the marked messages - // Vectorize the set and sort in reverse order to avoid shifting indices when removing - let mut indices_to_remove = indices_to_remove.iter().cloned().collect::>(); - indices_to_remove.sort_unstable_by(|a, b| b.cmp(a)); - - for &index in &indices_to_remove { - if index < messages.len() { - let _ = messages.remove(index); - let removed_tokens = token_counts.remove(index); - total_tokens -= removed_tokens; - } - } - - // Step 4: Ensure the last message is a user message with TextContent only - while let Some(last_msg) = messages.last() { - if last_msg.role != Role::User || !last_msg.has_only_text_content() { - let _ = messages.pop().ok_or(anyhow!("Failed to pop message"))?; - let removed_tokens = token_counts - .pop() - .ok_or(anyhow!("Failed to pop token count"))?; - total_tokens -= removed_tokens; - } else { - break; - } - } - - // Step 5: Check first msg is a User message with TextContent only - while let Some(first_msg) = messages.first() { - if first_msg.role != Role::User || !first_msg.has_only_text_content() { - let _ = messages.remove(0); - let removed_tokens = token_counts.remove(0); - total_tokens -= removed_tokens; - } else { - break; - } - } - - debug!("Total tokens after truncation: {}", total_tokens); - - // Ensure we have at least one message remaining and it's within context limit - if messages.is_empty() { - return Err(anyhow!( - "Unable to preserve any messages within context limit" - )); - } - - if total_tokens > context_limit { - return Err(anyhow!( - "Unable to truncate messages within context window." - )); - } - - debug!("Truncation complete. Total tokens: {}", total_tokens); - Ok(( - Conversation::new_unvalidated(messages.to_vec()), - token_counts.to_vec(), - )) -} - -/// Trait representing a truncation strategy -pub trait TruncationStrategy { - /// Determines the indices of messages to remove to fit within the context limit. - /// - /// - `messages`: The list of messages in the conversation. - /// - `token_counts`: A parallel array containing the token count for each message. - /// - `context_limit`: The maximum allowed context length in tokens. - /// - /// Returns a vector of indices to remove. - fn determine_indices_to_remove( - &self, - messages: &[Message], - token_counts: &[usize], - context_limit: usize, - ) -> Result>; -} - -/// Strategy to truncate messages by removing the oldest first -pub struct OldestFirstTruncation; - -impl TruncationStrategy for OldestFirstTruncation { - fn determine_indices_to_remove( - &self, - messages: &[Message], - token_counts: &[usize], - context_limit: usize, - ) -> Result> { - let mut indices_to_remove = HashSet::new(); - let mut total_tokens: usize = token_counts.iter().sum(); - let mut tool_ids_to_remove = HashSet::new(); - - for (i, message) in messages.iter().enumerate() { - if total_tokens <= context_limit { - break; - } - - // Remove the message - indices_to_remove.insert(i); - total_tokens -= token_counts[i]; - debug!( - "OldestFirst: Removing message at index {}. Tokens removed: {}", - i, token_counts[i] - ); - - // If it's a ToolRequest or ToolResponse, mark its pair for removal - if message.is_tool_call() || message.is_tool_response() { - message.get_tool_ids().iter().for_each(|id| { - tool_ids_to_remove.insert((i, id.to_string())); - }); - } - } - - // Now, find and remove paired ToolResponses or ToolRequests - for (i, message) in messages.iter().enumerate() { - let message_tool_ids = message.get_tool_ids(); - // Find the other part of the pair - same tool_id but different message index - for (message_idx, tool_id) in &tool_ids_to_remove { - if message_idx != &i && message_tool_ids.contains(tool_id.as_str()) { - indices_to_remove.insert(i); - // No need to check other tool_ids for this message since it's already marked - break; - } - } - } - - Ok(indices_to_remove) - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::conversation::message::Message; - use anyhow::Result; - use rmcp::model::{CallToolRequestParam, Content}; - use rmcp::object; - - // Helper function to create a user text message with a specified token count - fn user_text(index: usize, tokens: usize) -> (Message, usize) { - let content = format!("User message {}", index); - (Message::user().with_text(content), tokens) - } - - // Helper function to create an assistant text message with a specified token count - fn assistant_text(index: usize, tokens: usize) -> (Message, usize) { - let content = format!("Assistant message {}", index); - (Message::assistant().with_text(content), tokens) - } - - // Helper function to create a tool request message with a specified token count - fn assistant_tool_request( - id: &str, - tool_call: CallToolRequestParam, - tokens: usize, - ) -> (Message, usize) { - ( - Message::assistant().with_tool_request(id, Ok(tool_call)), - tokens, - ) - } - - // Helper function to create a tool response message with a specified token count - fn user_tool_response(id: &str, result: Vec, tokens: usize) -> (Message, usize) { - (Message::user().with_tool_response(id, Ok(result)), tokens) - } - - // Helper function to create a large tool response with massive content - fn large_tool_response(id: &str, large_text: String, tokens: usize) -> (Message, usize) { - ( - Message::user().with_tool_response(id, Ok(vec![Content::text(large_text)])), - tokens, - ) - } - - // Helper function to create messages with alternating user and assistant - // text messages of a fixed token count - fn create_messages_with_counts( - num_pairs: usize, - tokens: usize, - remove_last: bool, - ) -> (Conversation, Vec) { - let mut messages = Conversation::new_unvalidated((0..num_pairs).flat_map(|i| { - vec![ - user_text(i * 2, tokens).0, - assistant_text((i * 2) + 1, tokens).0, - ] - })); - - if remove_last { - messages.pop(); - } - - let token_counts = vec![tokens; messages.len()]; - - (messages, token_counts) - } - - #[test] - fn test_handle_oversized_single_message() -> Result<()> { - // Create a scenario similar to the real issue: one very large tool response - let large_content = "A".repeat(50000); // Very large content - let messages = vec![ - user_text(1, 10).0, - assistant_tool_request( - "tool1", - CallToolRequestParam { - name: "read_file".into(), - arguments: Some(object!({"path": "large_file.txt"})), - }, - 20, - ) - .0, - large_tool_response("tool1", large_content, 100000).0, // Massive tool response - user_text(2, 10).0, - ]; - let token_counts = vec![10, 20, 100000, 10]; // One message is huge - let context_limit = 5000; // Much smaller than the large message - - let result = truncate_messages( - &messages, - &token_counts, - context_limit, - &OldestFirstTruncation, - ); - - // Should succeed by truncating the large content - assert!( - result.is_ok(), - "Should handle oversized message by content truncation" - ); - let (truncated_messages, truncated_counts) = result.unwrap(); - - // Should have some messages remaining - assert!( - !truncated_messages.is_empty(), - "Should have some messages left" - ); - - // Total should be within limit - let total_tokens: usize = truncated_counts.iter().sum(); - assert!( - total_tokens <= context_limit, - "Total tokens {} should be <= context limit {}", - total_tokens, - context_limit - ); - - Ok(()) - } - - #[test] - fn test_oldest_first_no_truncation() -> Result<()> { - let (messages, token_counts) = create_messages_with_counts(1, 10, false); - let context_limit = 25; - - let result = truncate_messages( - messages.messages(), - &token_counts, - context_limit, - &OldestFirstTruncation, - )?; - - assert_eq!(result.0.messages(), messages.messages()); - assert_eq!(result.1, token_counts); - Ok(()) - } - - #[test] - fn test_complex_conversation_with_tools() -> Result<()> { - // Simulating a real conversation with multiple tool interactions - let tool_call1 = CallToolRequestParam { - name: "file_read".into(), - arguments: Some(object!({"path": "/tmp/test.txt"})), - }; - let tool_call2 = CallToolRequestParam { - name: "database_query".into(), - arguments: Some(object!({"query": "SELECT * FROM users"})), - }; - - let messages = vec![ - user_text(1, 15).0, // Initial user query - assistant_tool_request("tool1", tool_call1.clone(), 20).0, - user_tool_response( - "tool1", - vec![Content::text("File contents".to_string())], - 10, - ) - .0, - assistant_text(2, 25).0, // Assistant processes file contents - user_text(3, 10).0, // User follow-up - assistant_tool_request("tool2", tool_call2.clone(), 30).0, - user_tool_response( - "tool2", - vec![Content::text("Query results".to_string())], - 20, - ) - .0, - assistant_text(4, 35).0, // Assistant analyzes query results - user_text(5, 5).0, // Final user confirmation - ]; - - let token_counts = vec![15, 20, 10, 25, 10, 30, 20, 35, 5]; - let context_limit = 100; // Force truncation while preserving some tool interactions - - let result = truncate_messages( - &messages, - &token_counts, - context_limit, - &OldestFirstTruncation, - )?; - let (truncated_messages, truncated_counts) = result; - - // Verify that tool pairs are kept together and the conversation remains coherent - assert!(truncated_messages.len() >= 3); // At least one complete interaction should remain - assert!(truncated_messages.last().unwrap().role == Role::User); // Last message should be from user - - // Verify tool pairs are either both present or both removed - let tool_ids: HashSet<_> = truncated_messages - .iter() - .flat_map(|m| m.get_tool_ids()) - .collect(); - - // Each tool ID should appear 0 or 2 times (request + response) - for id in tool_ids { - let count = truncated_messages - .iter() - .flat_map(|m| m.get_tool_ids().into_iter()) - .filter(|&tool_id| tool_id == id) - .count(); - assert!(count == 0 || count == 2, "Tool pair was split: {}", id); - } - - // Total should be within limit - let total_tokens: usize = truncated_counts.iter().sum(); - assert!(total_tokens <= context_limit); - - Ok(()) - } - - #[test] - fn test_edge_case_context_window() -> Result<()> { - // Test case where we're exactly at the context limit - let (messages, token_counts) = create_messages_with_counts(2, 25, false); - let context_limit = 100; // Exactly matches total tokens - - let result = truncate_messages( - messages.messages(), - &token_counts, - context_limit, - &OldestFirstTruncation, - )?; - let (mut messages, mut token_counts) = result; - - assert_eq!(messages.len(), 4); // No truncation needed - assert_eq!(token_counts.iter().sum::(), 100); - - // Now add one more token to force truncation - messages.push(user_text(5, 1).0); - token_counts.push(1); - - let result = truncate_messages( - messages.messages(), - &token_counts, - context_limit, - &OldestFirstTruncation, - )?; - let (messages, token_counts) = result; - - assert!(token_counts.iter().sum::() <= context_limit); - assert!(messages.last().unwrap().role == Role::User); - - Ok(()) - } - - #[test] - fn test_multi_tool_chain() -> Result<()> { - // Simulate a chain of dependent tool calls - let tool_calls = vec![ - CallToolRequestParam { - name: "git_status".into(), - arguments: Some(object!({})), - }, - CallToolRequestParam { - name: "git_diff".into(), - arguments: Some(object!({"file": "main.rs"})), - }, - CallToolRequestParam { - name: "git_commit".into(), - arguments: Some(object!({"message": "Update"})), - }, - ]; - - let mut messages = Vec::new(); - let mut token_counts = Vec::new(); - - // Build a chain of related tool calls - // 30 tokens each round - for (i, tool_call) in tool_calls.into_iter().enumerate() { - let id = format!("git_{}", i); - messages.push(user_text(i, 10).0); - token_counts.push(10); - - messages.push(assistant_tool_request(&id, tool_call, 15).0); - token_counts.push(20); - } - - let context_limit = 50; // Force partial truncation - - let result = truncate_messages( - &messages, - &token_counts, - context_limit, - &OldestFirstTruncation, - )?; - let (truncated_messages, _) = result; - - // Verify that remaining tool chains are complete - let remaining_tool_ids: HashSet<_> = truncated_messages - .iter() - .flat_map(|m| m.get_tool_ids()) - .collect(); - - for _id in remaining_tool_ids { - // Count request/response pairs - let requests = truncated_messages - .iter() - .flat_map(|m| m.get_tool_request_ids().into_iter()) - .count(); - - let responses = truncated_messages - .iter() - .flat_map(|m| m.get_tool_response_ids().into_iter()) - .count(); - - assert_eq!(requests, 1, "Each remaining tool should have one request"); - assert_eq!(responses, 1, "Each remaining tool should have one response"); - } - - Ok(()) - } - - #[test] - fn test_truncation_with_image_content() -> Result<()> { - // Create a conversation with image content mixed in - let messages = vec![ - Message::user().with_image("base64_data", "image/png"), // 50 tokens - Message::assistant().with_text("I see the image"), // 10 tokens - Message::user().with_text("Can you describe it?"), // 10 tokens - Message::assistant().with_text("It shows..."), // 20 tokens - Message::user().with_text("Thanks!"), // 5 tokens - ]; - let token_counts = vec![50, 10, 10, 20, 5]; - let context_limit = 45; // Force truncation - - let result = truncate_messages( - &messages, - &token_counts, - context_limit, - &OldestFirstTruncation, - )?; - let (messages, token_counts) = result; - - // Verify the conversation still makes sense - assert!(!messages.is_empty()); - assert!(messages.last().unwrap().role == Role::User); - assert!(token_counts.iter().sum::() <= context_limit); - - Ok(()) - } - - #[test] - fn test_error_cases() -> Result<()> { - // Test impossibly small context window - let (messages, token_counts) = create_messages_with_counts(1, 10, false); - let result = truncate_messages( - messages.messages(), - &token_counts, - 5, // Impossibly small context - &OldestFirstTruncation, - ); - assert!(result.is_err()); - - // Test unmatched token counts - let messages = vec![user_text(1, 10).0]; - let token_counts = vec![10, 10]; // Mismatched length - let result = truncate_messages(&messages, &token_counts, 100, &OldestFirstTruncation); - assert!(result.is_err()); - - Ok(()) - } -} diff --git a/crates/goose/src/conversation/message.rs b/crates/goose/src/conversation/message.rs index 8b01a438b181..b47d05774a96 100644 --- a/crates/goose/src/conversation/message.rs +++ b/crates/goose/src/conversation/message.rs @@ -112,12 +112,7 @@ pub struct FrontendToolRequest { } #[derive(Debug, Clone, PartialEq, Serialize, Deserialize, ToSchema)] -pub struct ContextLengthExceeded { - pub msg: String, -} - -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize, ToSchema)] -pub struct SummarizationRequested { +pub struct ConversationCompacted { pub msg: String, } @@ -133,8 +128,7 @@ pub enum MessageContent { FrontendToolRequest(FrontendToolRequest), Thinking(ThinkingContent), RedactedThinking(RedactedThinkingContent), - ContextLengthExceeded(ContextLengthExceeded), - SummarizationRequested(SummarizationRequested), + ConversationCompacted(ConversationCompacted), } impl fmt::Display for MessageContent { @@ -162,10 +156,7 @@ impl fmt::Display for MessageContent { }, MessageContent::Thinking(t) => write!(f, "[Thinking: {}]", t.thinking), MessageContent::RedactedThinking(_r) => write!(f, "[RedactedThinking]"), - MessageContent::ContextLengthExceeded(r) => { - write!(f, "[ContextLengthExceeded: {}]", r.msg) - } - MessageContent::SummarizationRequested(r) => { + MessageContent::ConversationCompacted(r) => { write!(f, "[SummarizationRequested: {}]", r.msg) } } @@ -246,17 +237,13 @@ impl MessageContent { }) } - pub fn context_length_exceeded>(msg: S) -> Self { - MessageContent::ContextLengthExceeded(ContextLengthExceeded { msg: msg.into() }) - } - - pub fn summarization_requested>(msg: S) -> Self { - MessageContent::SummarizationRequested(SummarizationRequested { msg: msg.into() }) + pub fn conversation_compacted>(msg: S) -> Self { + MessageContent::ConversationCompacted(ConversationCompacted { msg: msg.into() }) } // Add this new method to check for summarization requested content - pub fn as_summarization_requested(&self) -> Option<&SummarizationRequested> { - if let MessageContent::SummarizationRequested(ref summarization_requested) = self { + pub fn as_summarization_requested(&self) -> Option<&ConversationCompacted> { + if let MessageContent::ConversationCompacted(ref summarization_requested) = self { Some(summarization_requested) } else { None @@ -390,10 +377,8 @@ impl From for Message { #[serde(rename_all = "camelCase")] pub struct MessageMetadata { /// Whether the message should be visible to the user in the UI - #[serde(default = "default_true")] pub user_visible: bool, /// Whether the message should be included in the agent's context window - #[serde(default = "default_true")] pub agent_visible: bool, } @@ -464,28 +449,18 @@ impl MessageMetadata { } } -fn default_true() -> bool { - true -} - #[derive(ToSchema, Clone, PartialEq, Serialize, Deserialize, Debug)] /// A message to or from an LLM #[serde(rename_all = "camelCase")] pub struct Message { pub id: Option, pub role: Role, - #[serde(default = "default_created")] pub created: i64, #[serde(deserialize_with = "deserialize_sanitized_content")] pub content: Vec, - #[serde(default)] pub metadata: MessageMetadata, } -fn default_created() -> i64 { - 0 // old messages do not have timestamps. -} - impl Message { pub fn new(role: Role, created: i64, content: Vec) -> Self { Message { @@ -605,11 +580,6 @@ impl Message { self.with_content(MessageContent::redacted_thinking(data)) } - /// Add context length exceeded content to the message - pub fn with_context_length_exceeded>(self, msg: S) -> Self { - self.with_content(MessageContent::context_length_exceeded(msg)) - } - /// Get the concatenated text content of the message, separated by newlines pub fn as_concat_text(&self) -> String { self.content @@ -680,9 +650,8 @@ impl Message { .all(|c| matches!(c, MessageContent::Text(_))) } - /// Add summarization requested to the message - pub fn with_summarization_requested>(self, msg: S) -> Self { - self.with_content(MessageContent::summarization_requested(msg)) + pub fn with_conversation_compacted>(self, msg: S) -> Self { + self.with_content(MessageContent::conversation_compacted(msg)) } /// Set the visibility metadata for the message @@ -837,7 +806,8 @@ mod tests { } } } - ] + ], + "metadata": { "agentVisible": true, "userVisible": true } }"#; let message: Message = serde_json::from_str(json_str).unwrap(); @@ -1045,7 +1015,8 @@ mod tests { "data": "base64data", "mimeType": "image/png" }} - ] + ], + "metadata": {{ "agentVisible": true, "userVisible": true }} }}"#, malicious_text ); @@ -1073,7 +1044,8 @@ mod tests { "content": [{ "type": "text", "text": "Hello world 世界 🌍" - }] + }], + "metadata": { "agentVisible": true, "userVisible": true } }"#; let message: Message = serde_json::from_str(clean_json).unwrap(); @@ -1142,20 +1114,6 @@ mod tests { let message: Message = serde_json::from_str(json_with_metadata).unwrap(); assert!(!message.is_user_visible()); assert!(message.is_agent_visible()); - - // Test without metadata (should use defaults) - let json_without_metadata = r#"{ - "role": "user", - "created": 1640995200, - "content": [{ - "type": "text", - "text": "Test" - }] - }"#; - - let message: Message = serde_json::from_str(json_without_metadata).unwrap(); - assert!(message.is_user_visible()); - assert!(message.is_agent_visible()); } #[test] diff --git a/crates/goose/src/providers/formats/anthropic.rs b/crates/goose/src/providers/formats/anthropic.rs index 2bbc0dacfe6d..6142d2188c75 100644 --- a/crates/goose/src/providers/formats/anthropic.rs +++ b/crates/goose/src/providers/formats/anthropic.rs @@ -90,10 +90,7 @@ pub fn format_messages(messages: &[Message]) -> Vec { MessageContent::ToolConfirmationRequest(_tool_confirmation_request) => { // Skip tool confirmation requests } - MessageContent::ContextLengthExceeded(_) => { - // Skip - } - MessageContent::SummarizationRequested(_) => { + MessageContent::ConversationCompacted(_) => { // Skip } MessageContent::Thinking(thinking) => { diff --git a/crates/goose/src/providers/formats/bedrock.rs b/crates/goose/src/providers/formats/bedrock.rs index 8521ec514e72..ae4d37f7eca8 100644 --- a/crates/goose/src/providers/formats/bedrock.rs +++ b/crates/goose/src/providers/formats/bedrock.rs @@ -48,10 +48,7 @@ pub fn to_bedrock_message_content(content: &MessageContent) -> Result { - bail!("ContextLengthExceeded should not get passed to the provider") - } - MessageContent::SummarizationRequested(_) => { + MessageContent::ConversationCompacted(_) => { bail!("SummarizationRequested should not get passed to the provider") } MessageContent::ToolRequest(tool_req) => { diff --git a/crates/goose/src/providers/formats/databricks.rs b/crates/goose/src/providers/formats/databricks.rs index be125b4c7d75..c7a34e54273a 100644 --- a/crates/goose/src/providers/formats/databricks.rs +++ b/crates/goose/src/providers/formats/databricks.rs @@ -127,10 +127,7 @@ fn format_messages(messages: &[Message], image_format: &ImageFormat) -> Vec { - continue; - } - MessageContent::SummarizationRequested(_) => { + MessageContent::ConversationCompacted(_) => { continue; } MessageContent::ToolResponse(response) => { diff --git a/crates/goose/src/providers/formats/openai.rs b/crates/goose/src/providers/formats/openai.rs index cd47f4e51600..fc977ca0b09a 100644 --- a/crates/goose/src/providers/formats/openai.rs +++ b/crates/goose/src/providers/formats/openai.rs @@ -95,10 +95,7 @@ pub fn format_messages(messages: &[Message], image_format: &ImageFormat) -> Vec< // Redacted thinking blocks are not directly used in OpenAI format continue; } - MessageContent::ContextLengthExceeded(_) => { - continue; - } - MessageContent::SummarizationRequested(_) => { + MessageContent::ConversationCompacted(_) => { continue; } MessageContent::ToolRequest(request) => match &request.tool_call { diff --git a/crates/goose/src/providers/formats/snowflake.rs b/crates/goose/src/providers/formats/snowflake.rs index 660f2838db4f..4e52b263086a 100644 --- a/crates/goose/src/providers/formats/snowflake.rs +++ b/crates/goose/src/providers/formats/snowflake.rs @@ -53,10 +53,7 @@ pub fn format_messages(messages: &[Message]) -> Vec { MessageContent::ToolConfirmationRequest(_) => { // Skip tool confirmation requests } - MessageContent::ContextLengthExceeded(_) => { - // Skip - } - MessageContent::SummarizationRequested(_) => { + MessageContent::ConversationCompacted(_) => { // Skip } MessageContent::Thinking(_thinking) => { diff --git a/crates/goose/src/providers/usage_estimator.rs b/crates/goose/src/providers/usage_estimator.rs index e819b6dc8349..9385a57fe376 100644 --- a/crates/goose/src/providers/usage_estimator.rs +++ b/crates/goose/src/providers/usage_estimator.rs @@ -1,6 +1,6 @@ use crate::conversation::message::Message; use crate::providers::base::ProviderUsage; -use crate::token_counter::create_async_token_counter; +use crate::token_counter::create_token_counter; use anyhow::Result; use rmcp::model::Tool; @@ -17,7 +17,7 @@ pub async fn ensure_usage_tokens( return Ok(()); } - let token_counter = create_async_token_counter() + let token_counter = create_token_counter() .await .map_err(|e| anyhow::anyhow!("Failed to create token counter: {}", e))?; diff --git a/crates/goose/src/providers/utils.rs b/crates/goose/src/providers/utils.rs index 9c6ee2aa2a9d..344e7092439d 100644 --- a/crates/goose/src/providers/utils.rs +++ b/crates/goose/src/providers/utils.rs @@ -1,6 +1,8 @@ use super::base::Usage; use super::errors::GoogleErrorCode; +use crate::config::paths::Paths; use crate::model::ModelConfig; +use crate::providers::errors::{OpenAIError, ProviderError}; use anyhow::Result; use base64::Engine; use regex::Regex; @@ -12,8 +14,6 @@ use std::io::Read; use std::path::Path; use std::time::Duration; -use crate::providers::errors::{OpenAIError, ProviderError}; - #[derive(serde::Deserialize)] struct OpenAIErrorResponse { error: OpenAIError, @@ -461,14 +461,32 @@ pub fn emit_debug_trace( T1: ?Sized + Serialize, T2: ?Sized + Serialize, { - tracing::debug!( - model_config = %serde_json::to_string_pretty(model_config).unwrap_or_default(), - input = %serde_json::to_string_pretty(payload).unwrap_or_default(), - output = %serde_json::to_string_pretty(response).unwrap_or_default(), - input_tokens = ?usage.input_tokens.unwrap_or_default(), - output_tokens = ?usage.output_tokens.unwrap_or_default(), - total_tokens = ?usage.total_tokens.unwrap_or_default(), - ); + let logs_dir = Paths::in_state_dir("logs"); + + if let Err(e) = std::fs::create_dir_all(&logs_dir) { + tracing::warn!("Failed to create logs directory: {}", e); + return; + } + + let log_path = |i| logs_dir.join(format!("llm_request.{}.json", i)); + + for i in (0..4).rev() { + let _ = std::fs::rename(log_path(i), log_path(i + 1)); + } + + let data = serde_json::json!({ + "model_config": model_config, + "input": payload, + "output": response, + "usage": usage, + }); + + if let Err(e) = std::fs::write( + log_path(0), + serde_json::to_string_pretty(&data).unwrap_or_default(), + ) { + tracing::warn!("Failed to write log file: {}", e); + } } /// Safely parse a JSON string that may contain doubly-encoded or malformed JSON. diff --git a/crates/goose/src/scheduler.rs b/crates/goose/src/scheduler.rs index 42273229e385..a339731498fe 100644 --- a/crates/goose/src/scheduler.rs +++ b/crates/goose/src/scheduler.rs @@ -1200,7 +1200,7 @@ async fn run_scheduled_job_internal( } if let Some(ref prompt_text) = recipe.prompt { - let mut all_session_messages = + let mut conversation = Conversation::new_unvalidated(vec![Message::user().with_text(prompt_text.clone())]); let session_config = SessionConfig { @@ -1213,11 +1213,7 @@ async fn run_scheduled_job_internal( }; match agent - .reply( - all_session_messages.clone(), - Some(session_config.clone()), - None, - ) + .reply(conversation.clone(), Some(session_config.clone()), None) .await { Ok(mut stream) => { @@ -1231,11 +1227,13 @@ async fn run_scheduled_job_internal( if msg.role == rmcp::model::Role::Assistant { tracing::info!("[Job {}] Assistant: {:?}", job.id, msg.content); } - all_session_messages.push(msg); + conversation.push(msg); } Ok(AgentEvent::McpNotification(_)) => {} Ok(AgentEvent::ModelChange { .. }) => {} - Ok(AgentEvent::HistoryReplaced(_)) => {} + Ok(AgentEvent::HistoryReplaced(updated_conversation)) => { + conversation = updated_conversation; + } Err(e) => { tracing::error!( "[Job {}] Error receiving message from agent: {}", diff --git a/crates/goose/src/session/session_manager.rs b/crates/goose/src/session/session_manager.rs index 283c52f434f1..ddd4d3f513db 100644 --- a/crates/goose/src/session/session_manager.rs +++ b/crates/goose/src/session/session_manager.rs @@ -18,7 +18,7 @@ use tokio::sync::OnceCell; use tracing::{info, warn}; use utoipa::ToSchema; -const CURRENT_SCHEMA_VERSION: i32 = 2; +const CURRENT_SCHEMA_VERSION: i32 = 3; static SESSION_STORAGE: OnceCell> = OnceCell::const_new(); @@ -425,7 +425,8 @@ impl SessionStorage { content_json TEXT NOT NULL, created_timestamp INTEGER NOT NULL, timestamp TIMESTAMP DEFAULT CURRENT_TIMESTAMP, - tokens INTEGER + tokens INTEGER, + metadata_json TEXT ) "#, ) @@ -610,6 +611,15 @@ impl SessionStorage { .execute(&self.pool) .await?; } + 3 => { + sqlx::query( + r#" + ALTER TABLE messages ADD COLUMN metadata_json TEXT + "#, + ) + .execute(&self.pool) + .await?; + } _ => { anyhow::bail!("Unknown migration version: {}", version); } @@ -768,15 +778,15 @@ impl SessionStorage { } async fn get_conversation(&self, session_id: &str) -> Result { - let rows = sqlx::query_as::<_, (String, String, i64)>( - "SELECT role, content_json, created_timestamp FROM messages WHERE session_id = ? ORDER BY timestamp", + let rows = sqlx::query_as::<_, (String, String, i64, Option)>( + "SELECT role, content_json, created_timestamp, metadata_json FROM messages WHERE session_id = ? ORDER BY timestamp", ) .bind(session_id) .fetch_all(&self.pool) .await?; let mut messages = Vec::new(); - for (role_str, content_json, created_timestamp) in rows { + for (role_str, content_json, created_timestamp, metadata_json) in rows { let role = match role_str.as_str() { "user" => Role::User, "assistant" => Role::Assistant, @@ -784,7 +794,12 @@ impl SessionStorage { }; let content = serde_json::from_str(&content_json)?; - let message = Message::new(role, created_timestamp, content); + let metadata = metadata_json + .and_then(|json| serde_json::from_str(&json).ok()) + .unwrap_or_default(); + + let mut message = Message::new(role, created_timestamp, content); + message.metadata = metadata; messages.push(message); } @@ -792,16 +807,19 @@ impl SessionStorage { } async fn add_message(&self, session_id: &str, message: &Message) -> Result<()> { + let metadata_json = serde_json::to_string(&message.metadata)?; + sqlx::query( r#" - INSERT INTO messages (session_id, role, content_json, created_timestamp) - VALUES (?, ?, ?, ?) + INSERT INTO messages (session_id, role, content_json, created_timestamp, metadata_json) + VALUES (?, ?, ?, ?, ?) "#, ) .bind(session_id) .bind(role_to_string(&message.role)) .bind(serde_json::to_string(&message.content)?) .bind(message.created) + .bind(metadata_json) .execute(&self.pool) .await?; @@ -826,16 +844,19 @@ impl SessionStorage { .await?; for message in conversation.messages() { + let metadata_json = serde_json::to_string(&message.metadata)?; + sqlx::query( r#" - INSERT INTO messages (session_id, role, content_json, created_timestamp) - VALUES (?, ?, ?, ?) + INSERT INTO messages (session_id, role, content_json, created_timestamp, metadata_json) + VALUES (?, ?, ?, ?, ?) "#, ) .bind(session_id) .bind(role_to_string(&message.role)) .bind(serde_json::to_string(&message.content)?) .bind(message.created) + .bind(metadata_json) .execute(&mut *tx) .await?; } diff --git a/crates/goose/src/token_counter.rs b/crates/goose/src/token_counter.rs index f61396a416ff..dabb57082b7f 100644 --- a/crates/goose/src/token_counter.rs +++ b/crates/goose/src/token_counter.rs @@ -20,16 +20,12 @@ const ENUM_INIT: isize = -3; const ENUM_ITEM: usize = 3; const FUNC_END: usize = 12; -pub struct AsyncTokenCounter { - tokenizer: Arc, - token_cache: Arc>, -} - pub struct TokenCounter { tokenizer: Arc, + token_cache: Arc>, } -impl AsyncTokenCounter { +impl TokenCounter { pub async fn new() -> Result { let tokenizer = get_tokenizer().await?; Ok(Self { @@ -130,6 +126,9 @@ impl AsyncTokenCounter { } for message in messages { + if !message.metadata.agent_visible { + continue; + } num_tokens += tokens_per_message; for content in &message.content { if let Some(content_text) = content.as_text() { @@ -183,136 +182,6 @@ impl AsyncTokenCounter { } } -impl Default for TokenCounter { - fn default() -> Self { - Self::new() - } -} - -impl TokenCounter { - pub fn new() -> Self { - let tokenizer = get_tokenizer_blocking().expect("Failed to initialize tokenizer"); - Self { tokenizer } - } - - pub fn count_tokens(&self, text: &str) -> usize { - let tokens = self.tokenizer.encode_with_special_tokens(text); - tokens.len() - } - - pub fn count_tokens_for_tools(&self, tools: &[Tool]) -> usize { - let mut func_token_count = 0; - if !tools.is_empty() { - for tool in tools { - func_token_count += FUNC_INIT; - let name = &tool.name; - let description = &tool - .description - .as_ref() - .map(|d| d.as_ref()) - .unwrap_or_default() - .trim_end_matches('.'); - let line = format!("{}:{}", name, description); - func_token_count += self.count_tokens(&line); - - if let Some(serde_json::Value::Object(properties)) = - tool.input_schema.get("properties") - { - if !properties.is_empty() { - func_token_count += PROP_INIT; - for (key, value) in properties { - func_token_count += PROP_KEY; - let p_name = key; - let p_type = value.get("type").and_then(|v| v.as_str()).unwrap_or(""); - let p_desc = value - .get("description") - .and_then(|v| v.as_str()) - .unwrap_or("") - .trim_end_matches('.'); - let line = format!("{}:{}:{}", p_name, p_type, p_desc); - func_token_count += self.count_tokens(&line); - if let Some(enum_values) = value.get("enum").and_then(|v| v.as_array()) - { - func_token_count = - func_token_count.saturating_add_signed(ENUM_INIT); - for item in enum_values { - if let Some(item_str) = item.as_str() { - func_token_count += ENUM_ITEM; - func_token_count += self.count_tokens(item_str); - } - } - } - } - } - } - } - func_token_count += FUNC_END; - } - - func_token_count - } - - pub fn count_chat_tokens( - &self, - system_prompt: &str, - messages: &[Message], - tools: &[Tool], - ) -> usize { - let tokens_per_message = 4; - - let mut num_tokens = 0; - if !system_prompt.is_empty() { - num_tokens += self.count_tokens(system_prompt) + tokens_per_message; - } - - for message in messages { - num_tokens += tokens_per_message; - for content in &message.content { - if let Some(content_text) = content.as_text() { - num_tokens += self.count_tokens(content_text); - } else if let Some(tool_request) = content.as_tool_request() { - if let Ok(tool_call) = tool_request.tool_call.as_ref() { - let text = format!( - "{}:{}:{:?}", - tool_request.id, tool_call.name, tool_call.arguments - ); - num_tokens += self.count_tokens(&text); - } - } else if let Some(tool_response_text) = content.as_tool_response_text() { - num_tokens += self.count_tokens(&tool_response_text); - } else { - continue; - } - } - } - - if !tools.is_empty() { - num_tokens += self.count_tokens_for_tools(tools); - } - - num_tokens += 3; - - num_tokens - } - - pub fn count_everything( - &self, - system_prompt: &str, - messages: &[Message], - tools: &[Tool], - resources: &[String], - ) -> usize { - let mut num_tokens = self.count_chat_tokens(system_prompt, messages, tools); - - if !resources.is_empty() { - for resource in resources { - num_tokens += self.count_tokens(resource); - } - } - num_tokens - } -} - async fn get_tokenizer() -> Result, String> { let tokenizer = TOKENIZER .get_or_init(|| async { @@ -325,99 +194,17 @@ async fn get_tokenizer() -> Result, String> { Ok(tokenizer.clone()) } -fn get_tokenizer_blocking() -> Result, String> { - if let Some(tokenizer) = TOKENIZER.get() { - return Ok(tokenizer.clone()); - } - - match tiktoken_rs::o200k_base() { - Ok(bpe) => { - let tokenizer = Arc::new(bpe); - let _ = TOKENIZER.set(tokenizer.clone()); - Ok(tokenizer) - } - Err(e) => Err(format!("Failed to initialize o200k_base tokenizer: {}", e)), - } -} - -pub async fn create_async_token_counter() -> Result { - AsyncTokenCounter::new().await +pub async fn create_token_counter() -> Result { + TokenCounter::new().await } #[cfg(test)] mod tests { use super::*; - use crate::conversation::message::{Message, MessageContent}; - use rmcp::model::{Role, Tool}; - use rmcp::object; - - #[test] - fn test_count_chat_tokens() { - let counter = TokenCounter::new(); - - let system_prompt = - "You are a helpful assistant that can answer questions about the weather."; - - let messages = vec![ - Message::new( - Role::User, - 0, - vec![MessageContent::text( - "What's the weather like in San Francisco?", - )], - ), - Message::new( - Role::Assistant, - 1, - vec![MessageContent::text( - "Looks like it's 60 degrees Fahrenheit in San Francisco.", - )], - ), - Message::new( - Role::User, - 2, - vec![MessageContent::text("How about New York?")], - ), - ]; - - let tools = vec![Tool::new( - "get_current_weather", - "Get the current weather in a given location", - object!({ - "properties": { - "location": { - "type": "string", - "description": "The city and state, e.g. San Francisco, CA" - }, - "unit": { - "type": "string", - "description": "The unit of temperature to return", - "enum": ["celsius", "fahrenheit"] - } - }, - "required": ["location"] - }), - )]; - - let token_count_without_tools = counter.count_chat_tokens(system_prompt, &messages, &[]); - println!("Total tokens without tools: {}", token_count_without_tools); - - let token_count_with_tools = counter.count_chat_tokens(system_prompt, &messages, &tools); - println!("Total tokens with tools: {}", token_count_with_tools); - - assert!( - token_count_without_tools > 0, - "Should have some tokens without tools" - ); - assert!( - token_count_with_tools > token_count_without_tools, - "Should have more tokens with tools" - ); - } #[tokio::test] - async fn test_async_token_caching() { - let counter = create_async_token_counter().await.unwrap(); + async fn test_token_caching() { + let counter = create_token_counter().await.unwrap(); let text = "This is a test for caching functionality"; @@ -434,8 +221,8 @@ mod tests { } #[tokio::test] - async fn test_async_cache_management() { - let counter = create_async_token_counter().await.unwrap(); + async fn test_cache_management() { + let counter = create_token_counter().await.unwrap(); counter.count_tokens("First text"); counter.count_tokens("Second text"); @@ -454,7 +241,7 @@ mod tests { #[tokio::test] async fn test_concurrent_token_counter_creation() { let handles: Vec<_> = (0..10) - .map(|_| tokio::spawn(async { create_async_token_counter().await.unwrap() })) + .map(|_| tokio::spawn(async { create_token_counter().await.unwrap() })) .collect(); let counters: Vec<_> = futures::future::join_all(handles) @@ -473,7 +260,7 @@ mod tests { #[tokio::test] async fn test_cache_eviction_behavior() { - let counter = create_async_token_counter().await.unwrap(); + let counter = create_token_counter().await.unwrap(); let mut cached_texts = Vec::new(); for i in 0..50 { @@ -493,7 +280,7 @@ mod tests { #[tokio::test] async fn test_concurrent_cache_operations() { - let counter = std::sync::Arc::new(create_async_token_counter().await.unwrap()); + let counter = std::sync::Arc::new(create_token_counter().await.unwrap()); let handles: Vec<_> = (0..20) .map(|i| { @@ -518,24 +305,4 @@ mod tests { assert!(counter.cache_size() > 0); assert!(counter.cache_size() <= MAX_TOKEN_CACHE_SIZE); } - - #[test] - fn test_tokenizer_consistency() { - let sync_counter = TokenCounter::new(); - let text = "This is a test for tokenizer consistency"; - let sync_count = sync_counter.count_tokens(text); - - assert!(sync_count > 0, "Sync tokenizer should produce tokens"); - - let short_text = "Hi"; - let long_text = "This is a much longer text that should produce significantly more tokens than the short text"; - - let short_count = sync_counter.count_tokens(short_text); - let long_count = sync_counter.count_tokens(long_text); - - assert!( - short_count < long_count, - "Longer text should have more tokens" - ); - } } diff --git a/crates/goose/tests/agent.rs b/crates/goose/tests/agent.rs index 510fb30b9362..af5d531c4153 100644 --- a/crates/goose/tests/agent.rs +++ b/crates/goose/tests/agent.rs @@ -117,7 +117,7 @@ async fn run_truncate_test( agent.update_provider(provider).await?; let repeat_count = context_window + 10_000; let large_message_content = "hello ".repeat(repeat_count); - let messages = Conversation::new(vec![ + let conversation = Conversation::new(vec![ Message::user().with_text("hi there. what is 2 + 2?"), Message::assistant().with_text("hey! I think it's 4."), Message::user().with_text(&large_message_content), @@ -130,7 +130,7 @@ async fn run_truncate_test( ]) .unwrap(); - let reply_stream = agent.reply(messages, None, None).await?; + let reply_stream = agent.reply(conversation, None, None).await?; tokio::pin!(reply_stream); let mut responses = Vec::new(); @@ -143,8 +143,8 @@ async fn run_truncate_test( Ok(AgentEvent::ModelChange { .. }) => { // Model change events are informational, just continue } - Ok(AgentEvent::HistoryReplaced(_)) => { - // Handle history replacement events if needed + Ok(AgentEvent::HistoryReplaced(_updated_conversation)) => { + // Should update the conversation here, but we're not reading it } Err(e) => { println!("Error: {:?}", e); @@ -175,14 +175,6 @@ async fn run_truncate_test( assert!(text_content.text.to_lowercase().contains("no")); assert!(!text_content.text.to_lowercase().contains("yes")); } - goose::conversation::message::MessageContent::ContextLengthExceeded(_) => { - // This is an acceptable outcome for providers that don't truncate themselves - // and correctly report that the context length was exceeded. - println!( - "Received ContextLengthExceeded as expected for {:?}", - provider_type - ); - } _ => { panic!( "Unexpected message content type: {:?}", @@ -1119,7 +1111,9 @@ mod max_turns_tests { } Ok(AgentEvent::McpNotification(_)) => {} Ok(AgentEvent::ModelChange { .. }) => {} - Ok(AgentEvent::HistoryReplaced(_)) => {} + Ok(AgentEvent::HistoryReplaced(_updated_conversation)) => { + // We should update the conversation here, but we're not reading it + } Err(e) => { return Err(e); } diff --git a/ui/desktop/openapi.json b/ui/desktop/openapi.json index cb12e582cfd4..a101deb6159c 100644 --- a/ui/desktop/openapi.json +++ b/ui/desktop/openapi.json @@ -2213,30 +2213,14 @@ } ] }, - "ContextLengthExceeded": { - "type": "object", - "required": [ - "msg" - ], - "properties": { - "msg": { - "type": "string" - } - } - }, "ContextManageRequest": { "type": "object", "description": "Request payload for context management operations", "required": [ "messages", - "manageAction", "sessionId" ], "properties": { - "manageAction": { - "type": "string", - "description": "Operation to perform: \"truncation\" or \"summarize\"" - }, "messages": { "type": "array", "items": { @@ -2281,6 +2265,17 @@ "$ref": "#/components/schemas/Message" } }, + "ConversationCompacted": { + "type": "object", + "required": [ + "msg" + ], + "properties": { + "msg": { + "type": "string" + } + } + }, "CreateRecipeRequest": { "type": "object", "required": [ @@ -3098,7 +3093,9 @@ "description": "A message to or from an LLM", "required": [ "role", - "content" + "created", + "content", + "metadata" ], "properties": { "content": { @@ -3296,28 +3293,7 @@ { "allOf": [ { - "$ref": "#/components/schemas/ContextLengthExceeded" - }, - { - "type": "object", - "required": [ - "type" - ], - "properties": { - "type": { - "type": "string", - "enum": [ - "contextLengthExceeded" - ] - } - } - } - ] - }, - { - "allOf": [ - { - "$ref": "#/components/schemas/SummarizationRequested" + "$ref": "#/components/schemas/ConversationCompacted" }, { "type": "object", @@ -3328,7 +3304,7 @@ "type": { "type": "string", "enum": [ - "summarizationRequested" + "conversationCompacted" ] } } @@ -3344,6 +3320,10 @@ "MessageMetadata": { "type": "object", "description": "Metadata for message visibility", + "required": [ + "userVisible", + "agentVisible" + ], "properties": { "agentVisible": { "type": "boolean", @@ -4372,17 +4352,6 @@ "propertyName": "type" } }, - "SummarizationRequested": { - "type": "object", - "required": [ - "msg" - ], - "properties": { - "msg": { - "type": "string" - } - } - }, "TextContent": { "type": "object", "required": [ diff --git a/ui/desktop/src/api/types.gen.ts b/ui/desktop/src/api/types.gen.ts index a20a4f34df29..f8ff7a59a966 100644 --- a/ui/desktop/src/api/types.gen.ts +++ b/ui/desktop/src/api/types.gen.ts @@ -72,18 +72,10 @@ export type ConfigResponse = { export type Content = RawTextContent | RawImageContent | RawEmbeddedResource | RawAudioContent | RawResource; -export type ContextLengthExceeded = { - msg: string; -}; - /** * Request payload for context management operations */ export type ContextManageRequest = { - /** - * Operation to perform: "truncation" or "summarize" - */ - manageAction: string; /** * Collection of messages to be managed */ @@ -110,6 +102,10 @@ export type ContextManageResponse = { export type Conversation = Array; +export type ConversationCompacted = { + msg: string; +}; + export type CreateRecipeRequest = { author?: AuthorRequest | null; session_id: string; @@ -383,9 +379,9 @@ export type LoadedProvider = { */ export type Message = { content: Array; - created?: number; + created: number; id?: string | null; - metadata?: MessageMetadata; + metadata: MessageMetadata; role: Role; }; @@ -408,10 +404,8 @@ export type MessageContent = (TextContent & { type: 'thinking'; }) | (RedactedThinkingContent & { type: 'redactedThinking'; -}) | (ContextLengthExceeded & { - type: 'contextLengthExceeded'; -}) | (SummarizationRequested & { - type: 'summarizationRequested'; +}) | (ConversationCompacted & { + type: 'conversationCompacted'; }); /** @@ -421,11 +415,11 @@ export type MessageMetadata = { /** * Whether the message should be included in the agent's context window */ - agentVisible?: boolean; + agentVisible: boolean; /** * Whether the message should be visible to the user in the UI */ - userVisible?: boolean; + userVisible: boolean; }; /** @@ -788,10 +782,6 @@ export type SuccessCheck = { type: 'Shell'; }; -export type SummarizationRequested = { - msg: string; -}; - export type TextContent = { _meta?: { [key: string]: unknown; diff --git a/ui/desktop/src/components/ProgressiveMessageList.tsx b/ui/desktop/src/components/ProgressiveMessageList.tsx index 9bdd0ab689ab..3c39a97ed5be 100644 --- a/ui/desktop/src/components/ProgressiveMessageList.tsx +++ b/ui/desktop/src/components/ProgressiveMessageList.tsx @@ -172,7 +172,9 @@ export default function ProgressiveMessageList({ const messagesToRender = messages.slice(0, renderedCount); return messagesToRender .map((message, index) => { - // Use custom render function if provided + if (!message.metadata.userVisible) { + return null; + } if (renderMessage) { return renderMessage(message, index); } diff --git a/ui/desktop/src/components/context_management/CompactionMarker.tsx b/ui/desktop/src/components/context_management/CompactionMarker.tsx index f453aa9b0638..870e034cfdba 100644 --- a/ui/desktop/src/components/context_management/CompactionMarker.tsx +++ b/ui/desktop/src/components/context_management/CompactionMarker.tsx @@ -1,5 +1,5 @@ import React from 'react'; -import { Message, SummarizationRequested } from '../../api'; +import { Message, ConversationCompacted } from '../../api'; interface CompactionMarkerProps { message: Message; @@ -7,8 +7,8 @@ interface CompactionMarkerProps { export const CompactionMarker: React.FC = ({ message }) => { const compactionContent = message.content.find( - (content): content is SummarizationRequested & { type: 'summarizationRequested' } => - content.type === 'summarizationRequested' + (content): content is ConversationCompacted & { type: 'conversationCompacted' } => + content.type === 'conversationCompacted' ); const markerText = compactionContent?.msg || 'Conversation compacted'; diff --git a/ui/desktop/src/components/context_management/ContextManager.tsx b/ui/desktop/src/components/context_management/ContextManager.tsx index 54c606bd8852..7ef110aed8b5 100644 --- a/ui/desktop/src/components/context_management/ContextManager.tsx +++ b/ui/desktop/src/components/context_management/ContextManager.tsx @@ -9,12 +9,6 @@ interface ContextManagerState { } interface ContextManagerActions { - handleAutoCompaction: ( - messages: Message[], - setMessages: (messages: Message[]) => void, - append: (message: Message) => void, - sessionId: string - ) => Promise; handleManualCompaction: ( messages: Message[], setMessages: (messages: Message[]) => void, @@ -70,6 +64,7 @@ export const ContextManagerProvider: React.FC<{ children: React.ReactNode }> = ( setIsCompacting(false); } catch (err) { + // TODO(Douwe): move this to the server console.error('Error during compaction:', err); setCompactionError(err instanceof Error ? err.message : 'Unknown error during compaction'); @@ -80,10 +75,11 @@ export const ContextManagerProvider: React.FC<{ children: React.ReactNode }> = ( created: Math.floor(Date.now() / 1000), content: [ { - type: 'summarizationRequested', + type: 'conversationCompacted', msg: 'Compaction failed. Please try again or start a new session.', }, ], + metadata: { userVisible: true, agentVisible: true }, }; setMessages([...messages, errorMarker]); @@ -93,18 +89,6 @@ export const ContextManagerProvider: React.FC<{ children: React.ReactNode }> = ( [] ); - const handleAutoCompaction = useCallback( - async ( - messages: Message[], - setMessages: (messages: Message[]) => void, - append: (message: Message) => void, - sessionId: string - ) => { - await performCompaction(messages, setMessages, append, sessionId, false); - }, - [performCompaction] - ); - const handleManualCompaction = useCallback( async ( messages: Message[], @@ -118,7 +102,7 @@ export const ContextManagerProvider: React.FC<{ children: React.ReactNode }> = ( ); const hasCompactionMarker = useCallback((message: Message): boolean => { - return message.content.some((content) => content.type === 'summarizationRequested'); + return message.content.some((content) => content.type === 'conversationCompacted'); }, []); const value = { @@ -127,7 +111,6 @@ export const ContextManagerProvider: React.FC<{ children: React.ReactNode }> = ( compactionError, // Actions - handleAutoCompaction, handleManualCompaction, hasCompactionMarker, }; diff --git a/ui/desktop/src/components/context_management/__tests__/CompactionMarker.test.tsx b/ui/desktop/src/components/context_management/__tests__/CompactionMarker.test.tsx index 3244bbdfcecc..756d1cf14960 100644 --- a/ui/desktop/src/components/context_management/__tests__/CompactionMarker.test.tsx +++ b/ui/desktop/src/components/context_management/__tests__/CompactionMarker.test.tsx @@ -3,12 +3,20 @@ import { render, screen } from '@testing-library/react'; import { CompactionMarker } from '../CompactionMarker'; import { Message } from '../../../api'; +const default_message: Message = { + metadata: { + agentVisible: false, + userVisible: false, + }, + id: '1', + role: 'assistant', + created: 1000,content: [] +}; + describe('CompactionMarker', () => { - it('should render default message when no summarizationRequested content found', () => { + it('should render default message when no conversationCompacted content found', () => { const message: Message = { - id: '1', - role: 'assistant', - created: 1000, + ...default_message, content: [{ type: 'text', text: 'Regular message' }], }; @@ -17,14 +25,12 @@ describe('CompactionMarker', () => { expect(screen.getByText('Conversation compacted')).toBeInTheDocument(); }); - it('should render custom message from summarizationRequested content', () => { + it('should render custom message from conversationCompacted content', () => { const message: Message = { - id: '1', - role: 'assistant', - created: 1000, + ...default_message, content: [ { type: 'text', text: 'Some other content' }, - { type: 'summarizationRequested', msg: 'Custom compaction message' }, + { type: 'conversationCompacted', msg: 'Custom compaction message' }, ], }; @@ -35,9 +41,7 @@ describe('CompactionMarker', () => { it('should handle empty message content array', () => { const message: Message = { - id: '1', - role: 'assistant', - created: 1000, + ...default_message, content: [], }; @@ -48,10 +52,8 @@ describe('CompactionMarker', () => { it('should handle summarizationRequested content with empty msg', () => { const message: Message = { - id: '1', - role: 'assistant', - created: 1000, - content: [{ type: 'summarizationRequested', msg: '' }], + ...default_message, + content: [{ type: 'conversationCompacted', msg: '' }], }; render(); @@ -62,11 +64,9 @@ describe('CompactionMarker', () => { it('should handle summarizationRequested content with undefined msg', () => { const message: Message = { - id: '1', - role: 'assistant', - created: 1000, + ...default_message, // eslint-disable-next-line @typescript-eslint/no-explicit-any - content: [{ type: 'summarizationRequested' } as any], + content: [{ type: 'conversationCompacted' } as any], }; render(); diff --git a/ui/desktop/src/components/context_management/__tests__/ContextManager.test.tsx b/ui/desktop/src/components/context_management/__tests__/ContextManager.test.tsx index dcddf8137c4e..82887370e321 100644 --- a/ui/desktop/src/components/context_management/__tests__/ContextManager.test.tsx +++ b/ui/desktop/src/components/context_management/__tests__/ContextManager.test.tsx @@ -2,7 +2,18 @@ import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; import { renderHook, act } from '@testing-library/react'; import { ContextManagerProvider, useContextManager } from '../ContextManager'; import * as contextManagement from '../index'; -import { ContextManageResponse, Message } from '../../../api'; +import { Message } from '../../../api'; + +const default_message: Message = { + metadata: { + agentVisible: false, + userVisible: false, + }, + id: '1', + role: 'assistant', + created: 1000, + content: [], +}; // Mock the context management functions vi.mock('../index', () => ({ @@ -14,15 +25,11 @@ const mockManageContextFromBackend = vi.mocked(contextManagement.manageContextFr describe('ContextManager', () => { const mockMessages: Message[] = [ { - id: '1', - role: 'user', - created: 1000, + ...default_message, content: [{ type: 'text', text: 'Hello' }], }, { - id: '2', - role: 'assistant', - created: 2000, + ...default_message, content: [{ type: 'text', text: 'Hi there!' }], }, ]; @@ -51,9 +58,6 @@ describe('ContextManager', () => { expect(result.current.isCompacting).toBe(false); expect(result.current.compactionError).toBe(null); - expect(typeof result.current.handleAutoCompaction).toBe('function'); - expect(typeof result.current.handleManualCompaction).toBe('function'); - expect(typeof result.current.hasCompactionMarker).toBe('function'); }); }); @@ -61,10 +65,8 @@ describe('ContextManager', () => { it('should return true for messages with summarizationRequested content', () => { const { result } = renderContextManager(); const messageWithMarker: Message = { - id: '1', - role: 'assistant', - created: 1000, - content: [{ type: 'summarizationRequested', msg: 'Compaction marker' }], + ...default_message, + content: [{ type: 'conversationCompacted', msg: 'Compaction marker' }], }; expect(result.current.hasCompactionMarker(messageWithMarker)).toBe(true); @@ -73,24 +75,20 @@ describe('ContextManager', () => { it('should return false for messages without summarizationRequested content', () => { const { result } = renderContextManager(); const regularMessage: Message = { - id: '1', - role: 'user', - created: 1000, + ...default_message, content: [{ type: 'text', text: 'Hello' }], }; expect(result.current.hasCompactionMarker(regularMessage)).toBe(false); }); - it('should return true for messages with mixed content including summarizationRequested', () => { + it('should return true for messages with mixed content including conversationCompacted', () => { const { result } = renderContextManager(); const mixedMessage: Message = { - id: '1', - role: 'assistant', - created: 1000, + ...default_message, content: [ { type: 'text', text: 'Some text' }, - { type: 'summarizationRequested', msg: 'Compaction marker' }, + { type: 'conversationCompacted', msg: 'Compaction marker' }, ], }; @@ -98,220 +96,22 @@ describe('ContextManager', () => { }); }); - describe('handleAutoCompaction', () => { - it('should successfully perform auto compaction with server-provided messages', async () => { - // Mock the backend response with 3 messages: marker, summary, continuation - // Note: Server messages may not have id/created, which will be added by the code - mockManageContextFromBackend.mockResolvedValue({ - messages: [ - { - role: 'assistant', - content: [ - { type: 'summarizationRequested', msg: 'Conversation compacted and summarized' }, - ], - } as Message, - { - role: 'assistant', - content: [{ type: 'text', text: 'Summary content' }], - } as Message, - { - role: 'assistant', - content: [ - { - type: 'text', - text: 'The previous message contains a summary that was prepared because a context limit was reached. Do not mention that you read a summary or that conversation summarization occurred Just continue the conversation naturally based on the summarized context', - }, - ], - } as Message, - ], - tokenCounts: [8, 100, 50], - }); - - const { result } = renderContextManager(); - - await act(async () => { - await result.current.handleAutoCompaction( - mockMessages, - mockSetMessages, - mockAppend, - 'test-session-id' - ); - }); - - expect(mockManageContextFromBackend).toHaveBeenCalledWith({ - messages: mockMessages, - manageAction: 'summarize', - sessionId: 'test-session-id', - }); - - // Expect setMessages to be called with all 3 messages from server - // Note: Server doesn't provide id/created fields, so we don't check for them - expect(mockSetMessages).toHaveBeenCalledTimes(1); - const setMessagesCall = mockSetMessages.mock.calls[0][0]; - expect(setMessagesCall).toHaveLength(3); - expect(setMessagesCall[0]).toMatchObject({ - role: 'assistant', - content: [{ type: 'summarizationRequested', msg: 'Conversation compacted and summarized' }], - }); - expect(setMessagesCall[1]).toMatchObject({ - role: 'assistant', - content: [{ type: 'text', text: 'Summary content' }], - }); - expect(setMessagesCall[2]).toMatchObject({ - role: 'assistant', - content: [ - { - type: 'text', - text: 'The previous message contains a summary that was prepared because a context limit was reached. Do not mention that you read a summary or that conversation summarization occurred Just continue the conversation naturally based on the summarized context', - }, - ], - }); - - // Fast-forward timers to trigger the append call - act(() => { - vi.advanceTimersByTime(150); - }); - - // Should append the continuation message (index 2) for auto-compaction - expect(mockAppend).toHaveBeenCalledTimes(1); - const appendedMessage = mockAppend.mock.calls[0][0]; - expect(appendedMessage).toMatchObject({ - role: 'assistant', - content: [ - { - type: 'text', - text: 'The previous message contains a summary that was prepared because a context limit was reached. Do not mention that you read a summary or that conversation summarization occurred Just continue the conversation naturally based on the summarized context', - }, - ], - }); - }); - - it('should handle compaction errors gracefully', async () => { - const error = new Error('Backend error'); - mockManageContextFromBackend.mockRejectedValue(error); - - const { result } = renderContextManager(); - - await act(async () => { - await result.current.handleAutoCompaction( - mockMessages, - mockSetMessages, - mockAppend, - 'test-session-id' - ); - }); - - expect(result.current.compactionError).toBe('Backend error'); - expect(result.current.isCompacting).toBe(false); - - expect(mockSetMessages).toHaveBeenCalledWith([ - ...mockMessages, - expect.objectContaining({ - content: [ - { - type: 'summarizationRequested', - msg: 'Compaction failed. Please try again or start a new session.', - }, - ], - }), - ]); - }); - - it('should set isCompacting state correctly during operation', async () => { - let resolvePromise: (value: ContextManageResponse) => void; - const promise = new Promise((resolve) => { - resolvePromise = resolve; - }); - - mockManageContextFromBackend.mockReturnValue(promise); - - const { result } = renderContextManager(); - - // Start compaction - act(() => { - result.current.handleAutoCompaction( - mockMessages, - mockSetMessages, - mockAppend, - 'test-session-id' - ); - }); - - // Should be compacting - expect(result.current.isCompacting).toBe(true); - expect(result.current.compactionError).toBe(null); - - // Resolve the backend call - resolvePromise!({ - messages: [ - { - role: 'assistant', - content: [{ type: 'text', text: 'Summary content' }], - }, - ], - tokenCounts: [100, 50], - }); - - await act(async () => { - await promise; - }); - - // Should no longer be compacting - expect(result.current.isCompacting).toBe(false); - }); - - it('preserves display: false for ancestor messages', async () => { - mockManageContextFromBackend.mockResolvedValue({ messages: [], tokenCounts: [] }); - - const hiddenMessage: Message = { - id: 'hidden-1', - role: 'user', - created: 1500, - content: [{ type: 'text', text: 'Secret' }], - }; - - const visibleMessage: Message = { - id: 'visible-1', - role: 'assistant', - created: 1600, - content: [{ type: 'text', text: 'Public' }], - }; - - const messages: Message[] = [hiddenMessage, visibleMessage]; - - const { result } = renderContextManager(); - - await act(async () => { - await result.current.handleAutoCompaction( - messages, - mockSetMessages, - mockAppend, - 'test-session-id' - ); - }); - - // No server messages -> setMessages called with empty list - expect(mockSetMessages).toHaveBeenCalledWith([]); - expect(mockAppend).not.toHaveBeenCalled(); - }); - }); - describe('handleManualCompaction', () => { it('should perform compaction with server-provided messages', async () => { mockManageContextFromBackend.mockResolvedValue({ messages: [ { - role: 'assistant', + ...default_message, content: [ - { type: 'summarizationRequested', msg: 'Conversation compacted and summarized' }, + { type: 'conversationCompacted', msg: 'Conversation compacted and summarized' }, ], }, { - role: 'assistant', + ...default_message, content: [{ type: 'text', text: 'Manual summary content' }], }, { - role: 'assistant', + ...default_message, content: [ { type: 'text', @@ -346,7 +146,7 @@ describe('ContextManager', () => { expect(setMessagesCall).toHaveLength(3); expect(setMessagesCall[0]).toMatchObject({ role: 'assistant', - content: [{ type: 'summarizationRequested', msg: 'Conversation compacted and summarized' }], + content: [{ type: 'conversationCompacted', msg: 'Conversation compacted and summarized' }], }); expect(setMessagesCall[1]).toMatchObject({ role: 'assistant', @@ -375,7 +175,7 @@ describe('ContextManager', () => { mockManageContextFromBackend.mockResolvedValue({ messages: [ { - role: 'assistant', + ...default_message, content: [{ type: 'text', text: 'Manual summary content' }], }, ], @@ -408,17 +208,17 @@ describe('ContextManager', () => { mockManageContextFromBackend.mockResolvedValue({ messages: [ { - role: 'assistant', + ...default_message, content: [ - { type: 'summarizationRequested', msg: 'Conversation compacted and summarized' }, + { type: 'conversationCompacted', msg: 'Conversation compacted and summarized' }, ], }, { - role: 'assistant', + ...default_message, content: [{ type: 'text', text: 'Manual summary content' }], }, { - role: 'assistant', + ...default_message, content: [ { type: 'text', @@ -447,7 +247,7 @@ describe('ContextManager', () => { expect(setMessagesCall).toHaveLength(3); expect(setMessagesCall[0]).toMatchObject({ role: 'assistant', - content: [{ type: 'summarizationRequested', msg: 'Conversation compacted and summarized' }], + content: [{ type: 'conversationCompacted', msg: 'Conversation compacted and summarized' }], }); expect(setMessagesCall[1]).toMatchObject({ role: 'assistant', @@ -473,65 +273,6 @@ describe('ContextManager', () => { }); }); - describe('Error Handling', () => { - it('should handle backend errors with unknown error type', async () => { - mockManageContextFromBackend.mockRejectedValue('String error'); - - const { result } = renderContextManager(); - - await act(async () => { - await result.current.handleAutoCompaction( - mockMessages, - mockSetMessages, - mockAppend, - 'test-session-id' - ); - }); - - expect(result.current.compactionError).toBe('Unknown error during compaction'); - }); - - it('should handle missing summary content gracefully with server-provided messages', async () => { - mockManageContextFromBackend.mockResolvedValue({ - messages: [ - { - role: 'assistant', - content: [ - { type: 'toolResponse', id: 'test', toolResult: { content: 'Not text content' } }, - ], - } as Message, - ], - tokenCounts: [100, 50], - }); - - const { result } = renderContextManager(); - - await act(async () => { - await result.current.handleAutoCompaction( - mockMessages, - mockSetMessages, - mockAppend, - 'test-session-id' - ); - }); - - // Should complete without error even if content is not text - expect(result.current.isCompacting).toBe(false); - expect(result.current.compactionError).toBe(null); - - // Should still set messages from server - expect(mockSetMessages).toHaveBeenCalledTimes(1); - const setMessagesCall = mockSetMessages.mock.calls[0][0]; - expect(setMessagesCall).toHaveLength(1); - expect(setMessagesCall[0]).toMatchObject({ - role: 'assistant', - content: [ - { type: 'toolResponse', id: 'test', toolResult: { content: 'Not text content' } }, - ], - }); - }); - }); - describe('Context Provider Error', () => { it('should throw error when useContextManager is used outside provider', () => { expect(() => { diff --git a/ui/desktop/src/hooks/useChatEngine.test.ts b/ui/desktop/src/hooks/useChatEngine.test.ts index 198b37625e8d..f1241a1ff2b2 100644 --- a/ui/desktop/src/hooks/useChatEngine.test.ts +++ b/ui/desktop/src/hooks/useChatEngine.test.ts @@ -71,25 +71,39 @@ describe('useChatEngine', () => { describe('onMessageUpdate', () => { it('should truncate history and append the updated message when a message is edited', () => { // --- 1. ARRANGE --- + const metadata = { + agentVisible: true, + userVisible: true, + }; + const initialMessages: Message[] = [ - { id: '1', role: 'user', content: [{ type: 'text', text: 'First message' }], created: 0 }, + { + id: '1', + role: 'user', + content: [{ type: 'text', text: 'First message' }], + created: 0, + metadata, + }, { id: '2', role: 'assistant', content: [{ type: 'text', text: 'First response' }], created: 1, + metadata, }, { id: '3', role: 'user', content: [{ type: 'text', text: 'Message to be edited' }], created: 2, + metadata, }, { id: '4', role: 'assistant', content: [{ type: 'text', text: 'Response to be deleted' }], created: 3, + metadata, }, ]; diff --git a/ui/desktop/src/hooks/useChatEngine.ts b/ui/desktop/src/hooks/useChatEngine.ts index 6e99a519ad59..1c0f6b7e47a4 100644 --- a/ui/desktop/src/hooks/useChatEngine.ts +++ b/ui/desktop/src/hooks/useChatEngine.ts @@ -340,6 +340,7 @@ export const useChatEngine = ({ role: 'user', created: Date.now(), content: [], + metadata: { userVisible: true, agentVisible: true }, }; const notification = 'Interrupted by the user to make a correction'; diff --git a/ui/desktop/src/hooks/useChatStream.ts b/ui/desktop/src/hooks/useChatStream.ts index 65348804308a..142f60c4b541 100644 --- a/ui/desktop/src/hooks/useChatStream.ts +++ b/ui/desktop/src/hooks/useChatStream.ts @@ -49,6 +49,7 @@ export function useChatStream({ role: 'user', content: [{ type: 'text', text: userMessage }], created: Date.now(), + metadata: { userVisible: true, agentVisible: true }, }; let currentMessages = [...messages, newMessage]; diff --git a/ui/desktop/src/hooks/useMessageStream.ts b/ui/desktop/src/hooks/useMessageStream.ts index 1b3781839e77..6e8febee04f0 100644 --- a/ui/desktop/src/hooks/useMessageStream.ts +++ b/ui/desktop/src/hooks/useMessageStream.ts @@ -1,7 +1,7 @@ import { useCallback, useEffect, useId, useReducer, useRef, useState } from 'react'; import useSWR from 'swr'; import { createUserMessage, hasCompletedToolCalls } from '../types/message'; -import { Message, Role } from '../api'; +import { Conversation, Message, Role } from '../api'; import { getSession, Session } from '../api'; import { ChatState } from '../types/chatState'; @@ -34,6 +34,7 @@ type MessageEvent = | { type: 'Error'; error: string } | { type: 'Finish'; reason: string } | { type: 'ModelChange'; model: string; mode: string } + | { type: 'UpdateConversation'; conversation: Conversation } | NotificationEvent; export interface UseMessageStreamOptions { @@ -328,6 +329,11 @@ export function useMessageStream({ break; } + case 'UpdateConversation': { + setMessages(parsedEvent.conversation); + break; + } + case 'Error': { // Always throw the error so it gets caught and sets the error state // This ensures the retry UI appears for ALL errors @@ -384,6 +390,7 @@ export function useMessageStream({ return currentMessages; }, + // eslint-disable-next-line react-hooks/exhaustive-deps [mutate, mutateChatState, onFinish, onError, forceUpdate, setError] ); @@ -580,6 +587,7 @@ export function useMessageStream({ id: generateMessageId(), role: 'user' as const, created: Math.floor(Date.now() / 1000), + metadata: { userVisible: true, agentVisible: true }, content: [ { type: 'toolResponse' as const, diff --git a/ui/desktop/src/types/message.ts b/ui/desktop/src/types/message.ts index c5273c10c99d..a598556e9096 100644 --- a/ui/desktop/src/types/message.ts +++ b/ui/desktop/src/types/message.ts @@ -1,4 +1,4 @@ -import { Content, Message, ToolConfirmationRequest, ToolRequest, ToolResponse } from '../api'; +import { Message, ToolConfirmationRequest, ToolRequest, ToolResponse } from '../api'; export type ToolRequestMessageContent = ToolRequest & { type: 'toolRequest' }; export type ToolResponseMessageContent = ToolResponse & { type: 'toolResponse' }; @@ -9,58 +9,7 @@ export function createUserMessage(text: string): Message { role: 'user', created: Math.floor(Date.now() / 1000), content: [{ type: 'text', text }], - }; -} - -export function createAssistantMessage(text: string): Message { - return { - id: generateId(), - role: 'assistant', - created: Math.floor(Date.now() / 1000), - content: [{ type: 'text', text }], - }; -} - -export function createToolRequestMessage( - id: string, - toolName: string, - args: Record -): Message { - return { - id: generateId(), - role: 'assistant', - created: Math.floor(Date.now() / 1000), - content: [ - { - type: 'toolRequest', - id, - toolCall: { - status: 'success', - value: { - name: toolName, - arguments: args, - }, - }, - }, - ], - }; -} - -export function createToolResponseMessage(id: string, result: Content[]): Message { - return { - id: generateId(), - role: 'user', - created: Math.floor(Date.now() / 1000), - content: [ - { - type: 'toolResponse', - id, - toolResult: { - status: 'success', - value: result, - }, - }, - ], + metadata: { userVisible: true, agentVisible: true }, }; } @@ -79,6 +28,7 @@ export function createToolErrorResponseMessage(id: string, error: string): Messa }, }, ], + metadata: { userVisible: true, agentVisible: true }, }; } @@ -90,7 +40,6 @@ export function getTextContent(message: Message): string { return message.content .map((content) => { if (content.type === 'text') return content.text; - if (content.type === 'contextLengthExceeded') return content.msg; return ''; }) .join('');