diff --git a/crates/goose-cli/src/scenario_tests/recordings/anthropic/weather_tool.json b/crates/goose-cli/src/scenario_tests/recordings/anthropic/weather_tool.json index 728831c772..1a128963eb 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/anthropic/weather_tool.json +++ b/crates/goose-cli/src/scenario_tests/recordings/anthropic/weather_tool.json @@ -1,325 +1,12 @@ { - "1bc400a528c54b25f4f1f609481e98e44222b3deaf7eee2c9e640e6345c73861": { + "74a3cc715c4c65dcae4ecec084d962947e4ce3cf02a8b639ee23b080e88bf6ec": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:45.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\n\nIf the Extension Manager extension is enabled, you can use the search_available_extensions tool to discover additional\nextensions that can help with your task. To enable or disable extensions, use the manage_extensions tool with the\nextension_name. You should only enable extensions found from the search_available_extensions tool.\nIf Extension Manager is not available, you can only work with currently enabled extensions and cannot dynamically load\nnew ones.\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.", "messages": [ { - "id": null, + "id": "msg_20251124_8_0", "role": "user", - "created": 1760565465, - "content": [ - { - "type": "text", - "text": "tell me what the weather is in Berlin, Germany" - } - ], - "metadata": { - "userVisible": true, - "agentVisible": true - } - } - ], - "tools": [ - { - "name": "weather_extension__get_weather", - "description": "Get the weather for a location", - "inputSchema": { - "properties": { - "location": { - "description": "The city and state, e.g. San Francisco, CA", - "type": "string" - } - }, - "required": [ - "location" - ], - "type": "object" - } - }, - { - "name": "platform__search_available_extensions", - "description": "Searches for additional extensions available to help complete tasks.\n Use this tool when you're unable to find a specific feature or functionality you need to complete your task, or when standard approaches aren't working.\n These extensions might provide the exact tools needed to solve your problem.\n If you find a relevant one, consider using your tools to enable it.", - "inputSchema": { - "properties": {}, - "required": [], - "type": "object" - }, - "annotations": { - "title": "Discover extensions", - "readOnlyHint": true, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": false - } - }, - { - "name": "platform__manage_extensions", - "description": "Tool to manage extensions and tools in goose context.\n Enable or disable extensions to help complete tasks.\n Enable or disable an extension by providing the extension name.\n ", - "inputSchema": { - "properties": { - "action": { - "description": "The action to perform", - "enum": [ - "enable", - "disable" - ], - "type": "string" - }, - "extension_name": { - "description": "The name of the extension to enable", - "type": "string" - } - }, - "required": [ - "action", - "extension_name" - ], - "type": "object" - }, - "annotations": { - "title": "Enable or disable an extension", - "readOnlyHint": false, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": false - } - }, - { - "name": "platform__manage_schedule", - "description": "Manage scheduled recipe execution for this goose instance.\n\nActions:\n- \"list\": List all scheduled jobs\n- \"create\": Create a new scheduled job from a recipe file\n- \"run_now\": Execute a scheduled job immediately \n- \"pause\": Pause a scheduled job\n- \"unpause\": Resume a paused job\n- \"delete\": Remove a scheduled job\n- \"kill\": Terminate a currently running job\n- \"inspect\": Get details about a running job\n- \"sessions\": List execution history for a job\n- \"session_content\": Get the full content (messages) of a specific session\n", - "inputSchema": { - "properties": { - "action": { - "enum": [ - "list", - "create", - "run_now", - "pause", - "unpause", - "delete", - "kill", - "inspect", - "sessions", - "session_content" - ], - "type": "string" - }, - "cron_expression": { - "description": "A cron expression for create action. Supports both 5-field (minute hour day month weekday) and 6-field (second minute hour day month weekday) formats. 5-field expressions are automatically converted to 6-field by prepending '0' for seconds.", - "type": "string" - }, - "execution_mode": { - "default": "background", - "description": "Execution mode for create action: 'foreground' or 'background'", - "enum": [ - "foreground", - "background" - ], - "type": "string" - }, - "job_id": { - "description": "Job identifier for operations on existing jobs", - "type": "string" - }, - "limit": { - "default": 50, - "description": "Limit for sessions list", - "type": "integer" - }, - "recipe_path": { - "description": "Path to recipe file for create action", - "type": "string" - }, - "session_id": { - "description": "Session identifier for session_content action", - "type": "string" - } - }, - "required": [ - "action" - ], - "type": "object" - }, - "annotations": { - "title": "Manage scheduled recipes", - "readOnlyHint": false, - "destructiveHint": true, - "idempotentHint": false, - "openWorldHint": false - } - }, - { - "name": "dynamic_task__create_task", - "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", - "inputSchema": { - "properties": { - "execution_mode": { - "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", - "enum": [ - "sequential", - "parallel" - ], - "type": "string" - }, - "task_parameters": { - "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", - "items": { - "properties": { - "activities": { - "items": { - "type": "string" - }, - "type": "array" - }, - "context": { - "items": { - "type": "string" - }, - "type": "array" - }, - "description": { - "type": "string" - }, - "extensions": { - "items": { - "type": "object" - }, - "type": "array" - }, - "instructions": { - "description": "Task instructions (required if prompt is not provided)", - "type": "string" - }, - "parameters": { - "items": { - "type": "object" - }, - "type": "array" - }, - "prompt": { - "description": "Initial prompt (required if instructions is not provided)", - "type": "string" - }, - "response": { - "type": "object" - }, - "retry": { - "type": "object" - }, - "return_last_only": { - "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", - "type": "boolean" - }, - "settings": { - "type": "object" - }, - "title": { - "type": "string" - } - }, - "type": "object" - }, - "minItems": 1, - "type": "array" - } - }, - "required": [ - "task_parameters" - ], - "type": "object" - }, - "annotations": { - "title": "Create Dynamic Tasks", - "readOnlyHint": false, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": true - } - }, - { - "name": "subagent__execute_task", - "description": "Only use the subagent__execute_task tool when you execute sub recipe task or dynamic task.\n EXECUTION STRATEGY DECISION:\n 1. If the tasks are created with execution_mode, use the execution_mode.\n 2. Execute tasks sequentially unless user explicitly requests parallel execution. PARALLEL: User uses keywords like 'parallel', 'simultaneously', 'at the same time', 'concurrently'\n\n IMPLEMENTATION:\n - Sequential execution: Call this tool multiple times, passing exactly ONE task per call\n - Parallel execution: Call this tool once, passing an ARRAY of all tasks\n\n EXAMPLES:\n User Intent Based:\n - User: 'get weather and tell me a joke' → Sequential (2 separate tool calls, 1 task each)\n - User: 'get weather and joke in parallel' → Parallel (1 tool call with array of 2 tasks)\n - User: 'run these simultaneously' → Parallel (1 tool call with task array)\n - User: 'do task A then task B' → Sequential (2 separate tool calls)", - "inputSchema": { - "properties": { - "execution_mode": { - "default": "sequential", - "description": "Execution strategy for multiple tasks. Use 'sequential' (default) unless user explicitly requests parallel execution with words like 'parallel', 'simultaneously', 'at the same time', or 'concurrently'.", - "enum": [ - "sequential", - "parallel" - ], - "type": "string" - }, - "task_ids": { - "items": { - "description": "Unique identifier for the task", - "type": "string" - }, - "type": "array" - } - }, - "required": [ - "task_ids" - ], - "type": "object" - }, - "annotations": { - "title": "Run tasks in parallel", - "readOnlyHint": false, - "destructiveHint": true, - "idempotentHint": false, - "openWorldHint": true - } - } - ] - }, - "output": { - "message": { - "id": null, - "role": "assistant", - "created": 1760565468, - "content": [ - { - "type": "text", - "text": "I'll check the current weather in Berlin, Germany for you." - }, - { - "type": "toolRequest", - "id": "toolu_013JNt4Fn6i4Ls8HTbn8bQjG", - "toolCall": { - "status": "success", - "value": { - "name": "weather_extension__get_weather", - "arguments": { - "location": "Berlin, Germany" - } - } - } - } - ], - "metadata": { - "userVisible": true, - "agentVisible": true - } - }, - "usage": { - "model": "claude-sonnet-4-20250514", - "usage": { - "input_tokens": 2710, - "output_tokens": 73, - "total_tokens": 2783 - } - } - } - }, - "c6f73996d107f0ca731493c3e18108043693c0e63d377190c9359c599a7ef254": { - "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:45.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", - "messages": [ - { - "id": null, - "role": "user", - "created": 1760565465, + "created": 1763989805, "content": [ { "type": "text", @@ -332,17 +19,13 @@ } }, { - "id": null, + "id": "msg_b4135f6c-9a29-4904-a59f-632ea849ef5a", "role": "assistant", - "created": 1760565468, + "created": 1763989808, "content": [ - { - "type": "text", - "text": "I'll check the current weather in Berlin, Germany for you." - }, { "type": "toolRequest", - "id": "toolu_013JNt4Fn6i4Ls8HTbn8bQjG", + "id": "toolu_01VxSzmLcPK7UNNE4ms8XRa6", "toolCall": { "status": "success", "value": { @@ -360,13 +43,13 @@ } }, { - "id": "msg_17216282-dde0-4344-a137-1e89c85623ad", + "id": "msg_253b4067-0c08-4807-8b36-23b576f138b5", "role": "user", - "created": 1760565468, + "created": 1763989808, "content": [ { "type": "toolResponse", - "id": "toolu_013JNt4Fn6i4Ls8HTbn8bQjG", + "id": "toolu_01VxSzmLcPK7UNNE4ms8XRa6", "toolResult": { "status": "success", "value": [ @@ -386,67 +69,128 @@ ], "tools": [ { - "name": "weather_extension__get_weather", - "description": "Get the weather for a location", + "name": "dynamic_task__create_task", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, activities. Arrays for multiple tasks.", "inputSchema": { - "properties": { - "location": { - "description": "The city and state, e.g. San Francisco, CA", - "type": "string" + "$defs": { + "TaskParameter": { + "description": "Parameters for a single task", + "properties": { + "activities": { + "items": { + "type": "string" + }, + "type": [ + "array", + "null" + ] + }, + "description": { + "type": [ + "string", + "null" + ] + }, + "extensions": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "instructions": { + "type": [ + "string", + "null" + ] + }, + "parameters": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + }, + "response": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "retry": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": [ + "boolean", + "null" + ] + }, + "settings": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "title": { + "type": [ + "string", + "null" + ] + } + }, + "type": "object" } }, - "required": [ - "location" - ], - "type": "object" - } - }, - { - "name": "platform__search_available_extensions", - "description": "Searches for additional extensions available to help complete tasks.\n Use this tool when you're unable to find a specific feature or functionality you need to complete your task, or when standard approaches aren't working.\n These extensions might provide the exact tools needed to solve your problem.\n If you find a relevant one, consider using your tools to enable it.", - "inputSchema": { - "properties": {}, - "required": [], - "type": "object" - }, - "annotations": { - "title": "Discover extensions", - "readOnlyHint": true, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": false - } - }, - { - "name": "platform__manage_extensions", - "description": "Tool to manage extensions and tools in goose context.\n Enable or disable extensions to help complete tasks.\n Enable or disable an extension by providing the extension name.\n ", - "inputSchema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { - "action": { - "description": "The action to perform", - "enum": [ - "enable", - "disable" - ], - "type": "string" + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "type": [ + "string", + "null" + ] }, - "extension_name": { - "description": "The name of the extension to enable", - "type": "string" + "task_parameters": { + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", + "items": { + "$ref": "#/$defs/TaskParameter" + }, + "minItems": 1, + "type": "array" } }, "required": [ - "action", - "extension_name" + "task_parameters" ], + "title": "CreateDynamicTaskParams", "type": "object" }, "annotations": { - "title": "Enable or disable an extension", + "title": "Create Dynamic Tasks", "readOnlyHint": false, "destructiveHint": false, "idempotentHint": false, - "openWorldHint": false + "openWorldHint": true } }, { @@ -473,15 +217,6 @@ "description": "A cron expression for create action. Supports both 5-field (minute hour day month weekday) and 6-field (second minute hour day month weekday) formats. 5-field expressions are automatically converted to 6-field by prepending '0' for seconds.", "type": "string" }, - "execution_mode": { - "default": "background", - "description": "Execution mode for create action: 'foreground' or 'background'", - "enum": [ - "foreground", - "background" - ], - "type": "string" - }, "job_id": { "description": "Job identifier for operations on existing jobs", "type": "string" @@ -513,94 +248,6 @@ "openWorldHint": false } }, - { - "name": "dynamic_task__create_task", - "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", - "inputSchema": { - "properties": { - "execution_mode": { - "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", - "enum": [ - "sequential", - "parallel" - ], - "type": "string" - }, - "task_parameters": { - "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", - "items": { - "properties": { - "activities": { - "items": { - "type": "string" - }, - "type": "array" - }, - "context": { - "items": { - "type": "string" - }, - "type": "array" - }, - "description": { - "type": "string" - }, - "extensions": { - "items": { - "type": "object" - }, - "type": "array" - }, - "instructions": { - "description": "Task instructions (required if prompt is not provided)", - "type": "string" - }, - "parameters": { - "items": { - "type": "object" - }, - "type": "array" - }, - "prompt": { - "description": "Initial prompt (required if instructions is not provided)", - "type": "string" - }, - "response": { - "type": "object" - }, - "retry": { - "type": "object" - }, - "return_last_only": { - "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", - "type": "boolean" - }, - "settings": { - "type": "object" - }, - "title": { - "type": "string" - } - }, - "type": "object" - }, - "minItems": 1, - "type": "array" - } - }, - "required": [ - "task_parameters" - ], - "type": "object" - }, - "annotations": { - "title": "Create Dynamic Tasks", - "readOnlyHint": false, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": true - } - }, { "name": "subagent__execute_task", "description": "Only use the subagent__execute_task tool when you execute sub recipe task or dynamic task.\n EXECUTION STRATEGY DECISION:\n 1. If the tasks are created with execution_mode, use the execution_mode.\n 2. Execute tasks sequentially unless user explicitly requests parallel execution. PARALLEL: User uses keywords like 'parallel', 'simultaneously', 'at the same time', 'concurrently'\n\n IMPLEMENTATION:\n - Sequential execution: Call this tool multiple times, passing exactly ONE task per call\n - Parallel execution: Call this tool once, passing an ARRAY of all tasks\n\n EXAMPLES:\n User Intent Based:\n - User: 'get weather and tell me a joke' → Sequential (2 separate tool calls, 1 task each)\n - User: 'get weather and joke in parallel' → Parallel (1 tool call with array of 2 tasks)\n - User: 'run these simultaneously' → Parallel (1 tool call with task array)\n - User: 'do task A then task B' → Sequential (2 separate tool calls)", @@ -635,6 +282,22 @@ "idempotentHint": false, "openWorldHint": true } + }, + { + "name": "weather_extension__get_weather", + "description": "Get the weather for a location", + "inputSchema": { + "properties": { + "location": { + "description": "The city and state, e.g. San Francisco, CA", + "type": "string" + } + }, + "required": [ + "location" + ], + "type": "object" + } } ] }, @@ -642,11 +305,11 @@ "message": { "id": null, "role": "assistant", - "created": 1760565471, + "created": 1763989811, "content": [ { "type": "text", - "text": "The current weather in **Berlin, Germany** is:\n\n- **Temperature**: 18°C (64°F)\n- **Conditions**: Cloudy\n\nIt's a mild day with overcast skies in Berlin right now." + "text": "The weather in Berlin, Germany is currently **cloudy** with a temperature of **18°C** (64°F)." } ], "metadata": { @@ -657,9 +320,350 @@ "usage": { "model": "claude-sonnet-4-20250514", "usage": { - "input_tokens": 2809, - "output_tokens": 53, - "total_tokens": 2862 + "input_tokens": 2406, + "output_tokens": 29, + "total_tokens": 2435 + } + } + } + }, + "1bc400a528c54b25f4f1f609481e98e44222b3deaf7eee2c9e640e6345c73861": { + "input": { + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\n\nIf the Extension Manager extension is enabled, you can use the search_available_extensions tool to discover additional\nextensions that can help with your task. To enable or disable extensions, use the manage_extensions tool with the\nextension_name. You should only enable extensions found from the search_available_extensions tool.\nIf Extension Manager is not available, you can only work with currently enabled extensions and cannot dynamically load\nnew ones.\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.", + "messages": [ + { + "id": "msg_20251124_8_0", + "role": "user", + "created": 1763989805, + "content": [ + { + "type": "text", + "text": "tell me what the weather is in Berlin, Germany" + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + } + ], + "tools": [ + { + "name": "dynamic_task__create_task", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, activities. Arrays for multiple tasks.", + "inputSchema": { + "$defs": { + "TaskParameter": { + "description": "Parameters for a single task", + "properties": { + "activities": { + "items": { + "type": "string" + }, + "type": [ + "array", + "null" + ] + }, + "description": { + "type": [ + "string", + "null" + ] + }, + "extensions": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "instructions": { + "type": [ + "string", + "null" + ] + }, + "parameters": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + }, + "response": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "retry": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": [ + "boolean", + "null" + ] + }, + "settings": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "title": { + "type": [ + "string", + "null" + ] + } + }, + "type": "object" + } + }, + "$schema": "https://json-schema.org/draft/2020-12/schema", + "properties": { + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "type": [ + "string", + "null" + ] + }, + "task_parameters": { + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", + "items": { + "$ref": "#/$defs/TaskParameter" + }, + "minItems": 1, + "type": "array" + } + }, + "required": [ + "task_parameters" + ], + "title": "CreateDynamicTaskParams", + "type": "object" + }, + "annotations": { + "title": "Create Dynamic Tasks", + "readOnlyHint": false, + "destructiveHint": false, + "idempotentHint": false, + "openWorldHint": true + } + }, + { + "name": "platform__manage_schedule", + "description": "Manage scheduled recipe execution for this goose instance.\n\nActions:\n- \"list\": List all scheduled jobs\n- \"create\": Create a new scheduled job from a recipe file\n- \"run_now\": Execute a scheduled job immediately \n- \"pause\": Pause a scheduled job\n- \"unpause\": Resume a paused job\n- \"delete\": Remove a scheduled job\n- \"kill\": Terminate a currently running job\n- \"inspect\": Get details about a running job\n- \"sessions\": List execution history for a job\n- \"session_content\": Get the full content (messages) of a specific session\n", + "inputSchema": { + "properties": { + "action": { + "enum": [ + "list", + "create", + "run_now", + "pause", + "unpause", + "delete", + "kill", + "inspect", + "sessions", + "session_content" + ], + "type": "string" + }, + "cron_expression": { + "description": "A cron expression for create action. Supports both 5-field (minute hour day month weekday) and 6-field (second minute hour day month weekday) formats. 5-field expressions are automatically converted to 6-field by prepending '0' for seconds.", + "type": "string" + }, + "job_id": { + "description": "Job identifier for operations on existing jobs", + "type": "string" + }, + "limit": { + "default": 50, + "description": "Limit for sessions list", + "type": "integer" + }, + "recipe_path": { + "description": "Path to recipe file for create action", + "type": "string" + }, + "session_id": { + "description": "Session identifier for session_content action", + "type": "string" + } + }, + "required": [ + "action" + ], + "type": "object" + }, + "annotations": { + "title": "Manage scheduled recipes", + "readOnlyHint": false, + "destructiveHint": true, + "idempotentHint": false, + "openWorldHint": false + } + }, + { + "name": "subagent__execute_task", + "description": "Only use the subagent__execute_task tool when you execute sub recipe task or dynamic task.\n EXECUTION STRATEGY DECISION:\n 1. If the tasks are created with execution_mode, use the execution_mode.\n 2. Execute tasks sequentially unless user explicitly requests parallel execution. PARALLEL: User uses keywords like 'parallel', 'simultaneously', 'at the same time', 'concurrently'\n\n IMPLEMENTATION:\n - Sequential execution: Call this tool multiple times, passing exactly ONE task per call\n - Parallel execution: Call this tool once, passing an ARRAY of all tasks\n\n EXAMPLES:\n User Intent Based:\n - User: 'get weather and tell me a joke' → Sequential (2 separate tool calls, 1 task each)\n - User: 'get weather and joke in parallel' → Parallel (1 tool call with array of 2 tasks)\n - User: 'run these simultaneously' → Parallel (1 tool call with task array)\n - User: 'do task A then task B' → Sequential (2 separate tool calls)", + "inputSchema": { + "properties": { + "execution_mode": { + "default": "sequential", + "description": "Execution strategy for multiple tasks. Use 'sequential' (default) unless user explicitly requests parallel execution with words like 'parallel', 'simultaneously', 'at the same time', or 'concurrently'.", + "enum": [ + "sequential", + "parallel" + ], + "type": "string" + }, + "task_ids": { + "items": { + "description": "Unique identifier for the task", + "type": "string" + }, + "type": "array" + } + }, + "required": [ + "task_ids" + ], + "type": "object" + }, + "annotations": { + "title": "Run tasks in parallel", + "readOnlyHint": false, + "destructiveHint": true, + "idempotentHint": false, + "openWorldHint": true + } + }, + { + "name": "weather_extension__get_weather", + "description": "Get the weather for a location", + "inputSchema": { + "properties": { + "location": { + "description": "The city and state, e.g. San Francisco, CA", + "type": "string" + } + }, + "required": [ + "location" + ], + "type": "object" + } + } + ] + }, + "output": { + "message": { + "id": null, + "role": "assistant", + "created": 1763989808, + "content": [ + { + "type": "text", + "text": "I'll get the current weather information for Berlin, Germany." + }, + { + "type": "toolRequest", + "id": "toolu_01VxSzmLcPK7UNNE4ms8XRa6", + "toolCall": { + "status": "success", + "value": { + "name": "weather_extension__get_weather", + "arguments": { + "location": "Berlin, Germany" + } + } + } + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + }, + "usage": { + "model": "claude-sonnet-4-20250514", + "usage": { + "input_tokens": 2320, + "output_tokens": 72, + "total_tokens": 2392 + } + } + } + }, + "8b1b8633232ca390cb3ff37a48daf74168b58f13e2943efcb17957129ee34d1a": { + "input": { + "system": "Reply with only a description in four words or less", + "messages": [ + { + "id": null, + "role": "user", + "created": 1763989805, + "content": [ + { + "type": "text", + "text": "Here are the first few user messages:\ntell me what the weather is in Berlin, Germany\n\nBased on the conversation so far, provide a concise description of this session in 4 words or less. This will be used for finding the session later in a UI with limited space - reply *ONLY* with the description" + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + } + ], + "tools": [] + }, + "output": { + "message": { + "id": null, + "role": "assistant", + "created": 1763989810, + "content": [ + { + "type": "text", + "text": "Berlin weather inquiry" + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + }, + "usage": { + "model": "claude-sonnet-4-20250514", + "usage": { + "input_tokens": 84, + "output_tokens": 6, + "total_tokens": 90 } } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/azure_openai/weather_tool.json b/crates/goose-cli/src/scenario_tests/recordings/azure_openai/weather_tool.json index d6f53bb78f..fffc7b746b 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/azure_openai/weather_tool.json +++ b/crates/goose-cli/src/scenario_tests/recordings/azure_openai/weather_tool.json @@ -1,12 +1,12 @@ { "1bc400a528c54b25f4f1f609481e98e44222b3deaf7eee2c9e640e6345c73861": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:51.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\n\nIf the Extension Manager extension is enabled, you can use the search_available_extensions tool to discover additional\nextensions that can help with your task. To enable or disable extensions, use the manage_extensions tool with the\nextension_name. You should only enable extensions found from the search_available_extensions tool.\nIf Extension Manager is not available, you can only work with currently enabled extensions and cannot dynamically load\nnew ones.\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.", "messages": [ { - "id": null, + "id": "msg_20251124_9_0", "role": "user", - "created": 1760565471, + "created": 1763989812, "content": [ { "type": "text", @@ -21,67 +21,128 @@ ], "tools": [ { - "name": "weather_extension__get_weather", - "description": "Get the weather for a location", + "name": "dynamic_task__create_task", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, activities. Arrays for multiple tasks.", "inputSchema": { - "properties": { - "location": { - "description": "The city and state, e.g. San Francisco, CA", - "type": "string" + "$defs": { + "TaskParameter": { + "description": "Parameters for a single task", + "properties": { + "activities": { + "items": { + "type": "string" + }, + "type": [ + "array", + "null" + ] + }, + "description": { + "type": [ + "string", + "null" + ] + }, + "extensions": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "instructions": { + "type": [ + "string", + "null" + ] + }, + "parameters": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + }, + "response": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "retry": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": [ + "boolean", + "null" + ] + }, + "settings": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "title": { + "type": [ + "string", + "null" + ] + } + }, + "type": "object" } }, - "required": [ - "location" - ], - "type": "object" - } - }, - { - "name": "platform__search_available_extensions", - "description": "Searches for additional extensions available to help complete tasks.\n Use this tool when you're unable to find a specific feature or functionality you need to complete your task, or when standard approaches aren't working.\n These extensions might provide the exact tools needed to solve your problem.\n If you find a relevant one, consider using your tools to enable it.", - "inputSchema": { - "properties": {}, - "required": [], - "type": "object" - }, - "annotations": { - "title": "Discover extensions", - "readOnlyHint": true, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": false - } - }, - { - "name": "platform__manage_extensions", - "description": "Tool to manage extensions and tools in goose context.\n Enable or disable extensions to help complete tasks.\n Enable or disable an extension by providing the extension name.\n ", - "inputSchema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { - "action": { - "description": "The action to perform", - "enum": [ - "enable", - "disable" - ], - "type": "string" + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "type": [ + "string", + "null" + ] }, - "extension_name": { - "description": "The name of the extension to enable", - "type": "string" + "task_parameters": { + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", + "items": { + "$ref": "#/$defs/TaskParameter" + }, + "minItems": 1, + "type": "array" } }, "required": [ - "action", - "extension_name" + "task_parameters" ], + "title": "CreateDynamicTaskParams", "type": "object" }, "annotations": { - "title": "Enable or disable an extension", + "title": "Create Dynamic Tasks", "readOnlyHint": false, "destructiveHint": false, "idempotentHint": false, - "openWorldHint": false + "openWorldHint": true } }, { @@ -108,15 +169,6 @@ "description": "A cron expression for create action. Supports both 5-field (minute hour day month weekday) and 6-field (second minute hour day month weekday) formats. 5-field expressions are automatically converted to 6-field by prepending '0' for seconds.", "type": "string" }, - "execution_mode": { - "default": "background", - "description": "Execution mode for create action: 'foreground' or 'background'", - "enum": [ - "foreground", - "background" - ], - "type": "string" - }, "job_id": { "description": "Job identifier for operations on existing jobs", "type": "string" @@ -148,94 +200,6 @@ "openWorldHint": false } }, - { - "name": "dynamic_task__create_task", - "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", - "inputSchema": { - "properties": { - "execution_mode": { - "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", - "enum": [ - "sequential", - "parallel" - ], - "type": "string" - }, - "task_parameters": { - "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", - "items": { - "properties": { - "activities": { - "items": { - "type": "string" - }, - "type": "array" - }, - "context": { - "items": { - "type": "string" - }, - "type": "array" - }, - "description": { - "type": "string" - }, - "extensions": { - "items": { - "type": "object" - }, - "type": "array" - }, - "instructions": { - "description": "Task instructions (required if prompt is not provided)", - "type": "string" - }, - "parameters": { - "items": { - "type": "object" - }, - "type": "array" - }, - "prompt": { - "description": "Initial prompt (required if instructions is not provided)", - "type": "string" - }, - "response": { - "type": "object" - }, - "retry": { - "type": "object" - }, - "return_last_only": { - "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", - "type": "boolean" - }, - "settings": { - "type": "object" - }, - "title": { - "type": "string" - } - }, - "type": "object" - }, - "minItems": 1, - "type": "array" - } - }, - "required": [ - "task_parameters" - ], - "type": "object" - }, - "annotations": { - "title": "Create Dynamic Tasks", - "readOnlyHint": false, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": true - } - }, { "name": "subagent__execute_task", "description": "Only use the subagent__execute_task tool when you execute sub recipe task or dynamic task.\n EXECUTION STRATEGY DECISION:\n 1. If the tasks are created with execution_mode, use the execution_mode.\n 2. Execute tasks sequentially unless user explicitly requests parallel execution. PARALLEL: User uses keywords like 'parallel', 'simultaneously', 'at the same time', 'concurrently'\n\n IMPLEMENTATION:\n - Sequential execution: Call this tool multiple times, passing exactly ONE task per call\n - Parallel execution: Call this tool once, passing an ARRAY of all tasks\n\n EXAMPLES:\n User Intent Based:\n - User: 'get weather and tell me a joke' → Sequential (2 separate tool calls, 1 task each)\n - User: 'get weather and joke in parallel' → Parallel (1 tool call with array of 2 tasks)\n - User: 'run these simultaneously' → Parallel (1 tool call with task array)\n - User: 'do task A then task B' → Sequential (2 separate tool calls)", @@ -270,6 +234,22 @@ "idempotentHint": false, "openWorldHint": true } + }, + { + "name": "weather_extension__get_weather", + "description": "Get the weather for a location", + "inputSchema": { + "properties": { + "location": { + "description": "The city and state, e.g. San Francisco, CA", + "type": "string" + } + }, + "required": [ + "location" + ], + "type": "object" + } } ] }, @@ -277,11 +257,11 @@ "message": { "id": null, "role": "assistant", - "created": 1760565472, + "created": 1763989813, "content": [ { "type": "toolRequest", - "id": "call_QK2Zg4k9kTxaNXCyOVPbQ8Pc", + "id": "call_5IgyNY33nTougqsNFhizypew", "toolCall": { "status": "success", "value": { @@ -301,21 +281,21 @@ "usage": { "model": "gpt-4o-mini-2024-07-18", "usage": { - "input_tokens": 1708, + "input_tokens": 1492, "output_tokens": 20, - "total_tokens": 1728 + "total_tokens": 1512 } } } }, - "198b76aab4abc29c526ddd37cf906dcf9161c94423d8ca772908bef353ece3e8": { + "2c07b9a6570269fef4ae329889ce5a5243d7b09b52165580be701b67b8b95818": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:51.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\n\nIf the Extension Manager extension is enabled, you can use the search_available_extensions tool to discover additional\nextensions that can help with your task. To enable or disable extensions, use the manage_extensions tool with the\nextension_name. You should only enable extensions found from the search_available_extensions tool.\nIf Extension Manager is not available, you can only work with currently enabled extensions and cannot dynamically load\nnew ones.\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.", "messages": [ { - "id": null, + "id": "msg_20251124_9_0", "role": "user", - "created": 1760565471, + "created": 1763989812, "content": [ { "type": "text", @@ -328,13 +308,13 @@ } }, { - "id": null, + "id": "msg_b87436fe-d710-4fcd-b912-a69ab4ccfe7e", "role": "assistant", - "created": 1760565472, + "created": 1763989813, "content": [ { "type": "toolRequest", - "id": "call_QK2Zg4k9kTxaNXCyOVPbQ8Pc", + "id": "call_5IgyNY33nTougqsNFhizypew", "toolCall": { "status": "success", "value": { @@ -352,13 +332,13 @@ } }, { - "id": "msg_85357fa1-4622-4b5c-bc66-ce79f12038dd", + "id": "msg_9cf859cf-4adc-4464-aba4-1d4ccf9d7d05", "role": "user", - "created": 1760565472, + "created": 1763989813, "content": [ { "type": "toolResponse", - "id": "call_QK2Zg4k9kTxaNXCyOVPbQ8Pc", + "id": "call_5IgyNY33nTougqsNFhizypew", "toolResult": { "status": "success", "value": [ @@ -378,67 +358,128 @@ ], "tools": [ { - "name": "weather_extension__get_weather", - "description": "Get the weather for a location", + "name": "dynamic_task__create_task", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, activities. Arrays for multiple tasks.", "inputSchema": { - "properties": { - "location": { - "description": "The city and state, e.g. San Francisco, CA", - "type": "string" + "$defs": { + "TaskParameter": { + "description": "Parameters for a single task", + "properties": { + "activities": { + "items": { + "type": "string" + }, + "type": [ + "array", + "null" + ] + }, + "description": { + "type": [ + "string", + "null" + ] + }, + "extensions": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "instructions": { + "type": [ + "string", + "null" + ] + }, + "parameters": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + }, + "response": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "retry": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": [ + "boolean", + "null" + ] + }, + "settings": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "title": { + "type": [ + "string", + "null" + ] + } + }, + "type": "object" } }, - "required": [ - "location" - ], - "type": "object" - } - }, - { - "name": "platform__search_available_extensions", - "description": "Searches for additional extensions available to help complete tasks.\n Use this tool when you're unable to find a specific feature or functionality you need to complete your task, or when standard approaches aren't working.\n These extensions might provide the exact tools needed to solve your problem.\n If you find a relevant one, consider using your tools to enable it.", - "inputSchema": { - "properties": {}, - "required": [], - "type": "object" - }, - "annotations": { - "title": "Discover extensions", - "readOnlyHint": true, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": false - } - }, - { - "name": "platform__manage_extensions", - "description": "Tool to manage extensions and tools in goose context.\n Enable or disable extensions to help complete tasks.\n Enable or disable an extension by providing the extension name.\n ", - "inputSchema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { - "action": { - "description": "The action to perform", - "enum": [ - "enable", - "disable" - ], - "type": "string" + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "type": [ + "string", + "null" + ] }, - "extension_name": { - "description": "The name of the extension to enable", - "type": "string" + "task_parameters": { + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", + "items": { + "$ref": "#/$defs/TaskParameter" + }, + "minItems": 1, + "type": "array" } }, "required": [ - "action", - "extension_name" + "task_parameters" ], + "title": "CreateDynamicTaskParams", "type": "object" }, "annotations": { - "title": "Enable or disable an extension", + "title": "Create Dynamic Tasks", "readOnlyHint": false, "destructiveHint": false, "idempotentHint": false, - "openWorldHint": false + "openWorldHint": true } }, { @@ -465,15 +506,6 @@ "description": "A cron expression for create action. Supports both 5-field (minute hour day month weekday) and 6-field (second minute hour day month weekday) formats. 5-field expressions are automatically converted to 6-field by prepending '0' for seconds.", "type": "string" }, - "execution_mode": { - "default": "background", - "description": "Execution mode for create action: 'foreground' or 'background'", - "enum": [ - "foreground", - "background" - ], - "type": "string" - }, "job_id": { "description": "Job identifier for operations on existing jobs", "type": "string" @@ -505,94 +537,6 @@ "openWorldHint": false } }, - { - "name": "dynamic_task__create_task", - "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", - "inputSchema": { - "properties": { - "execution_mode": { - "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", - "enum": [ - "sequential", - "parallel" - ], - "type": "string" - }, - "task_parameters": { - "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", - "items": { - "properties": { - "activities": { - "items": { - "type": "string" - }, - "type": "array" - }, - "context": { - "items": { - "type": "string" - }, - "type": "array" - }, - "description": { - "type": "string" - }, - "extensions": { - "items": { - "type": "object" - }, - "type": "array" - }, - "instructions": { - "description": "Task instructions (required if prompt is not provided)", - "type": "string" - }, - "parameters": { - "items": { - "type": "object" - }, - "type": "array" - }, - "prompt": { - "description": "Initial prompt (required if instructions is not provided)", - "type": "string" - }, - "response": { - "type": "object" - }, - "retry": { - "type": "object" - }, - "return_last_only": { - "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", - "type": "boolean" - }, - "settings": { - "type": "object" - }, - "title": { - "type": "string" - } - }, - "type": "object" - }, - "minItems": 1, - "type": "array" - } - }, - "required": [ - "task_parameters" - ], - "type": "object" - }, - "annotations": { - "title": "Create Dynamic Tasks", - "readOnlyHint": false, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": true - } - }, { "name": "subagent__execute_task", "description": "Only use the subagent__execute_task tool when you execute sub recipe task or dynamic task.\n EXECUTION STRATEGY DECISION:\n 1. If the tasks are created with execution_mode, use the execution_mode.\n 2. Execute tasks sequentially unless user explicitly requests parallel execution. PARALLEL: User uses keywords like 'parallel', 'simultaneously', 'at the same time', 'concurrently'\n\n IMPLEMENTATION:\n - Sequential execution: Call this tool multiple times, passing exactly ONE task per call\n - Parallel execution: Call this tool once, passing an ARRAY of all tasks\n\n EXAMPLES:\n User Intent Based:\n - User: 'get weather and tell me a joke' → Sequential (2 separate tool calls, 1 task each)\n - User: 'get weather and joke in parallel' → Parallel (1 tool call with array of 2 tasks)\n - User: 'run these simultaneously' → Parallel (1 tool call with task array)\n - User: 'do task A then task B' → Sequential (2 separate tool calls)", @@ -627,6 +571,22 @@ "idempotentHint": false, "openWorldHint": true } + }, + { + "name": "weather_extension__get_weather", + "description": "Get the weather for a location", + "inputSchema": { + "properties": { + "location": { + "description": "The city and state, e.g. San Francisco, CA", + "type": "string" + } + }, + "required": [ + "location" + ], + "type": "object" + } } ] }, @@ -634,11 +594,11 @@ "message": { "id": null, "role": "assistant", - "created": 1760565473, + "created": 1763989814, "content": [ { "type": "text", - "text": "The current weather in Berlin, Germany, is cloudy with a temperature of 18°C." + "text": "### Current Weather in Berlin, Germany\n\n- **Condition:** Cloudy\n- **Temperature:** 18°C" } ], "metadata": { @@ -649,9 +609,57 @@ "usage": { "model": "gpt-4o-mini-2024-07-18", "usage": { - "input_tokens": 1750, - "output_tokens": 20, - "total_tokens": 1770 + "input_tokens": 1534, + "output_tokens": 24, + "total_tokens": 1558 + } + } + } + }, + "8b1b8633232ca390cb3ff37a48daf74168b58f13e2943efcb17957129ee34d1a": { + "input": { + "system": "Reply with only a description in four words or less", + "messages": [ + { + "id": null, + "role": "user", + "created": 1763989812, + "content": [ + { + "type": "text", + "text": "Here are the first few user messages:\ntell me what the weather is in Berlin, Germany\n\nBased on the conversation so far, provide a concise description of this session in 4 words or less. This will be used for finding the session later in a UI with limited space - reply *ONLY* with the description" + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + } + ], + "tools": [] + }, + "output": { + "message": { + "id": null, + "role": "assistant", + "created": 1763989813, + "content": [ + { + "type": "text", + "text": "Berlin weather inquiry" + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + }, + "usage": { + "model": "gpt-4o-mini-2024-07-18", + "usage": { + "input_tokens": 84, + "output_tokens": 4, + "total_tokens": 88 } } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/groq/weather_tool.json b/crates/goose-cli/src/scenario_tests/recordings/groq/weather_tool.json index 8eb037f7a2..7a65421844 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/groq/weather_tool.json +++ b/crates/goose-cli/src/scenario_tests/recordings/groq/weather_tool.json @@ -1,12 +1,12 @@ { "1bc400a528c54b25f4f1f609481e98e44222b3deaf7eee2c9e640e6345c73861": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:53.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\n\nIf the Extension Manager extension is enabled, you can use the search_available_extensions tool to discover additional\nextensions that can help with your task. To enable or disable extensions, use the manage_extensions tool with the\nextension_name. You should only enable extensions found from the search_available_extensions tool.\nIf Extension Manager is not available, you can only work with currently enabled extensions and cannot dynamically load\nnew ones.\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.", "messages": [ { - "id": null, + "id": "msg_20251124_10_0", "role": "user", - "created": 1760565473, + "created": 1763989814, "content": [ { "type": "text", @@ -21,67 +21,128 @@ ], "tools": [ { - "name": "weather_extension__get_weather", - "description": "Get the weather for a location", + "name": "dynamic_task__create_task", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, activities. Arrays for multiple tasks.", "inputSchema": { - "properties": { - "location": { - "description": "The city and state, e.g. San Francisco, CA", - "type": "string" + "$defs": { + "TaskParameter": { + "description": "Parameters for a single task", + "properties": { + "activities": { + "items": { + "type": "string" + }, + "type": [ + "array", + "null" + ] + }, + "description": { + "type": [ + "string", + "null" + ] + }, + "extensions": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "instructions": { + "type": [ + "string", + "null" + ] + }, + "parameters": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + }, + "response": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "retry": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": [ + "boolean", + "null" + ] + }, + "settings": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "title": { + "type": [ + "string", + "null" + ] + } + }, + "type": "object" } }, - "required": [ - "location" - ], - "type": "object" - } - }, - { - "name": "platform__search_available_extensions", - "description": "Searches for additional extensions available to help complete tasks.\n Use this tool when you're unable to find a specific feature or functionality you need to complete your task, or when standard approaches aren't working.\n These extensions might provide the exact tools needed to solve your problem.\n If you find a relevant one, consider using your tools to enable it.", - "inputSchema": { - "properties": {}, - "required": [], - "type": "object" - }, - "annotations": { - "title": "Discover extensions", - "readOnlyHint": true, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": false - } - }, - { - "name": "platform__manage_extensions", - "description": "Tool to manage extensions and tools in goose context.\n Enable or disable extensions to help complete tasks.\n Enable or disable an extension by providing the extension name.\n ", - "inputSchema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { - "action": { - "description": "The action to perform", - "enum": [ - "enable", - "disable" - ], - "type": "string" + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "type": [ + "string", + "null" + ] }, - "extension_name": { - "description": "The name of the extension to enable", - "type": "string" + "task_parameters": { + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", + "items": { + "$ref": "#/$defs/TaskParameter" + }, + "minItems": 1, + "type": "array" } }, "required": [ - "action", - "extension_name" + "task_parameters" ], + "title": "CreateDynamicTaskParams", "type": "object" }, "annotations": { - "title": "Enable or disable an extension", + "title": "Create Dynamic Tasks", "readOnlyHint": false, "destructiveHint": false, "idempotentHint": false, - "openWorldHint": false + "openWorldHint": true } }, { @@ -108,15 +169,6 @@ "description": "A cron expression for create action. Supports both 5-field (minute hour day month weekday) and 6-field (second minute hour day month weekday) formats. 5-field expressions are automatically converted to 6-field by prepending '0' for seconds.", "type": "string" }, - "execution_mode": { - "default": "background", - "description": "Execution mode for create action: 'foreground' or 'background'", - "enum": [ - "foreground", - "background" - ], - "type": "string" - }, "job_id": { "description": "Job identifier for operations on existing jobs", "type": "string" @@ -148,94 +200,6 @@ "openWorldHint": false } }, - { - "name": "dynamic_task__create_task", - "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", - "inputSchema": { - "properties": { - "execution_mode": { - "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", - "enum": [ - "sequential", - "parallel" - ], - "type": "string" - }, - "task_parameters": { - "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", - "items": { - "properties": { - "activities": { - "items": { - "type": "string" - }, - "type": "array" - }, - "context": { - "items": { - "type": "string" - }, - "type": "array" - }, - "description": { - "type": "string" - }, - "extensions": { - "items": { - "type": "object" - }, - "type": "array" - }, - "instructions": { - "description": "Task instructions (required if prompt is not provided)", - "type": "string" - }, - "parameters": { - "items": { - "type": "object" - }, - "type": "array" - }, - "prompt": { - "description": "Initial prompt (required if instructions is not provided)", - "type": "string" - }, - "response": { - "type": "object" - }, - "retry": { - "type": "object" - }, - "return_last_only": { - "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", - "type": "boolean" - }, - "settings": { - "type": "object" - }, - "title": { - "type": "string" - } - }, - "type": "object" - }, - "minItems": 1, - "type": "array" - } - }, - "required": [ - "task_parameters" - ], - "type": "object" - }, - "annotations": { - "title": "Create Dynamic Tasks", - "readOnlyHint": false, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": true - } - }, { "name": "subagent__execute_task", "description": "Only use the subagent__execute_task tool when you execute sub recipe task or dynamic task.\n EXECUTION STRATEGY DECISION:\n 1. If the tasks are created with execution_mode, use the execution_mode.\n 2. Execute tasks sequentially unless user explicitly requests parallel execution. PARALLEL: User uses keywords like 'parallel', 'simultaneously', 'at the same time', 'concurrently'\n\n IMPLEMENTATION:\n - Sequential execution: Call this tool multiple times, passing exactly ONE task per call\n - Parallel execution: Call this tool once, passing an ARRAY of all tasks\n\n EXAMPLES:\n User Intent Based:\n - User: 'get weather and tell me a joke' → Sequential (2 separate tool calls, 1 task each)\n - User: 'get weather and joke in parallel' → Parallel (1 tool call with array of 2 tasks)\n - User: 'run these simultaneously' → Parallel (1 tool call with task array)\n - User: 'do task A then task B' → Sequential (2 separate tool calls)", @@ -270,6 +234,22 @@ "idempotentHint": false, "openWorldHint": true } + }, + { + "name": "weather_extension__get_weather", + "description": "Get the weather for a location", + "inputSchema": { + "properties": { + "location": { + "description": "The city and state, e.g. San Francisco, CA", + "type": "string" + } + }, + "required": [ + "location" + ], + "type": "object" + } } ] }, @@ -277,11 +257,11 @@ "message": { "id": null, "role": "assistant", - "created": 1760565474, + "created": 1763989814, "content": [ { "type": "toolRequest", - "id": "04fqdf70k", + "id": "4kpg6v08d", "toolCall": { "status": "success", "value": { @@ -301,21 +281,21 @@ "usage": { "model": "llama-3.3-70b-versatile", "usage": { - "input_tokens": 2698, + "input_tokens": 2258, "output_tokens": 20, - "total_tokens": 2718 + "total_tokens": 2278 } } } }, - "54bbc046e4ff83dd29df1fa575c6d39f6195a259106666dfefd1b10ed662bd45": { + "165258e98e56cdcccdc49bd70ecd2b081b41d35ddb3559ed3bc7b1e420290bed": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:53.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\n\nIf the Extension Manager extension is enabled, you can use the search_available_extensions tool to discover additional\nextensions that can help with your task. To enable or disable extensions, use the manage_extensions tool with the\nextension_name. You should only enable extensions found from the search_available_extensions tool.\nIf Extension Manager is not available, you can only work with currently enabled extensions and cannot dynamically load\nnew ones.\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.", "messages": [ { - "id": null, + "id": "msg_20251124_10_0", "role": "user", - "created": 1760565473, + "created": 1763989814, "content": [ { "type": "text", @@ -328,13 +308,13 @@ } }, { - "id": null, + "id": "msg_9d25f6e9-6d02-4f41-8710-325795e87fbc", "role": "assistant", - "created": 1760565474, + "created": 1763989814, "content": [ { "type": "toolRequest", - "id": "04fqdf70k", + "id": "4kpg6v08d", "toolCall": { "status": "success", "value": { @@ -352,13 +332,13 @@ } }, { - "id": "msg_7256b4f2-18dd-4477-98b7-cde7102e8a46", + "id": "msg_513af900-f4fe-4ed0-aed8-7ffb4c8c384d", "role": "user", - "created": 1760565474, + "created": 1763989814, "content": [ { "type": "toolResponse", - "id": "04fqdf70k", + "id": "4kpg6v08d", "toolResult": { "status": "success", "value": [ @@ -378,67 +358,128 @@ ], "tools": [ { - "name": "weather_extension__get_weather", - "description": "Get the weather for a location", + "name": "dynamic_task__create_task", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, activities. Arrays for multiple tasks.", "inputSchema": { - "properties": { - "location": { - "description": "The city and state, e.g. San Francisco, CA", - "type": "string" + "$defs": { + "TaskParameter": { + "description": "Parameters for a single task", + "properties": { + "activities": { + "items": { + "type": "string" + }, + "type": [ + "array", + "null" + ] + }, + "description": { + "type": [ + "string", + "null" + ] + }, + "extensions": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "instructions": { + "type": [ + "string", + "null" + ] + }, + "parameters": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + }, + "response": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "retry": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": [ + "boolean", + "null" + ] + }, + "settings": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "title": { + "type": [ + "string", + "null" + ] + } + }, + "type": "object" } }, - "required": [ - "location" - ], - "type": "object" - } - }, - { - "name": "platform__search_available_extensions", - "description": "Searches for additional extensions available to help complete tasks.\n Use this tool when you're unable to find a specific feature or functionality you need to complete your task, or when standard approaches aren't working.\n These extensions might provide the exact tools needed to solve your problem.\n If you find a relevant one, consider using your tools to enable it.", - "inputSchema": { - "properties": {}, - "required": [], - "type": "object" - }, - "annotations": { - "title": "Discover extensions", - "readOnlyHint": true, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": false - } - }, - { - "name": "platform__manage_extensions", - "description": "Tool to manage extensions and tools in goose context.\n Enable or disable extensions to help complete tasks.\n Enable or disable an extension by providing the extension name.\n ", - "inputSchema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { - "action": { - "description": "The action to perform", - "enum": [ - "enable", - "disable" - ], - "type": "string" + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "type": [ + "string", + "null" + ] }, - "extension_name": { - "description": "The name of the extension to enable", - "type": "string" + "task_parameters": { + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", + "items": { + "$ref": "#/$defs/TaskParameter" + }, + "minItems": 1, + "type": "array" } }, "required": [ - "action", - "extension_name" + "task_parameters" ], + "title": "CreateDynamicTaskParams", "type": "object" }, "annotations": { - "title": "Enable or disable an extension", + "title": "Create Dynamic Tasks", "readOnlyHint": false, "destructiveHint": false, "idempotentHint": false, - "openWorldHint": false + "openWorldHint": true } }, { @@ -465,15 +506,6 @@ "description": "A cron expression for create action. Supports both 5-field (minute hour day month weekday) and 6-field (second minute hour day month weekday) formats. 5-field expressions are automatically converted to 6-field by prepending '0' for seconds.", "type": "string" }, - "execution_mode": { - "default": "background", - "description": "Execution mode for create action: 'foreground' or 'background'", - "enum": [ - "foreground", - "background" - ], - "type": "string" - }, "job_id": { "description": "Job identifier for operations on existing jobs", "type": "string" @@ -505,94 +537,6 @@ "openWorldHint": false } }, - { - "name": "dynamic_task__create_task", - "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", - "inputSchema": { - "properties": { - "execution_mode": { - "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", - "enum": [ - "sequential", - "parallel" - ], - "type": "string" - }, - "task_parameters": { - "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", - "items": { - "properties": { - "activities": { - "items": { - "type": "string" - }, - "type": "array" - }, - "context": { - "items": { - "type": "string" - }, - "type": "array" - }, - "description": { - "type": "string" - }, - "extensions": { - "items": { - "type": "object" - }, - "type": "array" - }, - "instructions": { - "description": "Task instructions (required if prompt is not provided)", - "type": "string" - }, - "parameters": { - "items": { - "type": "object" - }, - "type": "array" - }, - "prompt": { - "description": "Initial prompt (required if instructions is not provided)", - "type": "string" - }, - "response": { - "type": "object" - }, - "retry": { - "type": "object" - }, - "return_last_only": { - "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", - "type": "boolean" - }, - "settings": { - "type": "object" - }, - "title": { - "type": "string" - } - }, - "type": "object" - }, - "minItems": 1, - "type": "array" - } - }, - "required": [ - "task_parameters" - ], - "type": "object" - }, - "annotations": { - "title": "Create Dynamic Tasks", - "readOnlyHint": false, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": true - } - }, { "name": "subagent__execute_task", "description": "Only use the subagent__execute_task tool when you execute sub recipe task or dynamic task.\n EXECUTION STRATEGY DECISION:\n 1. If the tasks are created with execution_mode, use the execution_mode.\n 2. Execute tasks sequentially unless user explicitly requests parallel execution. PARALLEL: User uses keywords like 'parallel', 'simultaneously', 'at the same time', 'concurrently'\n\n IMPLEMENTATION:\n - Sequential execution: Call this tool multiple times, passing exactly ONE task per call\n - Parallel execution: Call this tool once, passing an ARRAY of all tasks\n\n EXAMPLES:\n User Intent Based:\n - User: 'get weather and tell me a joke' → Sequential (2 separate tool calls, 1 task each)\n - User: 'get weather and joke in parallel' → Parallel (1 tool call with array of 2 tasks)\n - User: 'run these simultaneously' → Parallel (1 tool call with task array)\n - User: 'do task A then task B' → Sequential (2 separate tool calls)", @@ -627,6 +571,22 @@ "idempotentHint": false, "openWorldHint": true } + }, + { + "name": "weather_extension__get_weather", + "description": "Get the weather for a location", + "inputSchema": { + "properties": { + "location": { + "description": "The city and state, e.g. San Francisco, CA", + "type": "string" + } + }, + "required": [ + "location" + ], + "type": "object" + } } ] }, @@ -634,11 +594,11 @@ "message": { "id": null, "role": "assistant", - "created": 1760565474, + "created": 1763989815, "content": [ { "type": "text", - "text": "The weather in Berlin, Germany is cloudy and 18°C." + "text": "The current weather in Berlin, Germany is cloudy with a temperature of 18°C." } ], "metadata": { @@ -649,9 +609,57 @@ "usage": { "model": "llama-3.3-70b-versatile", "usage": { - "input_tokens": 2739, - "output_tokens": 14, - "total_tokens": 2753 + "input_tokens": 2299, + "output_tokens": 18, + "total_tokens": 2317 + } + } + } + }, + "8b1b8633232ca390cb3ff37a48daf74168b58f13e2943efcb17957129ee34d1a": { + "input": { + "system": "Reply with only a description in four words or less", + "messages": [ + { + "id": null, + "role": "user", + "created": 1763989814, + "content": [ + { + "type": "text", + "text": "Here are the first few user messages:\ntell me what the weather is in Berlin, Germany\n\nBased on the conversation so far, provide a concise description of this session in 4 words or less. This will be used for finding the session later in a UI with limited space - reply *ONLY* with the description" + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + } + ], + "tools": [] + }, + "output": { + "message": { + "id": null, + "role": "assistant", + "created": 1763989814, + "content": [ + { + "type": "text", + "text": "Berlin weather query" + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + }, + "usage": { + "model": "llama-3.3-70b-versatile", + "usage": { + "input_tokens": 108, + "output_tokens": 4, + "total_tokens": 112 } } } diff --git a/crates/goose-cli/src/scenario_tests/recordings/openai/weather_tool.json b/crates/goose-cli/src/scenario_tests/recordings/openai/weather_tool.json index ddc200735f..195be6d270 100644 --- a/crates/goose-cli/src/scenario_tests/recordings/openai/weather_tool.json +++ b/crates/goose-cli/src/scenario_tests/recordings/openai/weather_tool.json @@ -1,12 +1,12 @@ { "1bc400a528c54b25f4f1f609481e98e44222b3deaf7eee2c9e640e6345c73861": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:42.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\n\nIf the Extension Manager extension is enabled, you can use the search_available_extensions tool to discover additional\nextensions that can help with your task. To enable or disable extensions, use the manage_extensions tool with the\nextension_name. You should only enable extensions found from the search_available_extensions tool.\nIf Extension Manager is not available, you can only work with currently enabled extensions and cannot dynamically load\nnew ones.\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.", "messages": [ { - "id": null, + "id": "msg_20251124_7_0", "role": "user", - "created": 1760565462, + "created": 1763989801, "content": [ { "type": "text", @@ -21,67 +21,128 @@ ], "tools": [ { - "name": "weather_extension__get_weather", - "description": "Get the weather for a location", + "name": "dynamic_task__create_task", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, activities. Arrays for multiple tasks.", "inputSchema": { - "properties": { - "location": { - "description": "The city and state, e.g. San Francisco, CA", - "type": "string" + "$defs": { + "TaskParameter": { + "description": "Parameters for a single task", + "properties": { + "activities": { + "items": { + "type": "string" + }, + "type": [ + "array", + "null" + ] + }, + "description": { + "type": [ + "string", + "null" + ] + }, + "extensions": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "instructions": { + "type": [ + "string", + "null" + ] + }, + "parameters": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + }, + "response": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "retry": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": [ + "boolean", + "null" + ] + }, + "settings": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "title": { + "type": [ + "string", + "null" + ] + } + }, + "type": "object" } }, - "required": [ - "location" - ], - "type": "object" - } - }, - { - "name": "platform__search_available_extensions", - "description": "Searches for additional extensions available to help complete tasks.\n Use this tool when you're unable to find a specific feature or functionality you need to complete your task, or when standard approaches aren't working.\n These extensions might provide the exact tools needed to solve your problem.\n If you find a relevant one, consider using your tools to enable it.", - "inputSchema": { - "properties": {}, - "required": [], - "type": "object" - }, - "annotations": { - "title": "Discover extensions", - "readOnlyHint": true, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": false - } - }, - { - "name": "platform__manage_extensions", - "description": "Tool to manage extensions and tools in goose context.\n Enable or disable extensions to help complete tasks.\n Enable or disable an extension by providing the extension name.\n ", - "inputSchema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { - "action": { - "description": "The action to perform", - "enum": [ - "enable", - "disable" - ], - "type": "string" + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "type": [ + "string", + "null" + ] }, - "extension_name": { - "description": "The name of the extension to enable", - "type": "string" + "task_parameters": { + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", + "items": { + "$ref": "#/$defs/TaskParameter" + }, + "minItems": 1, + "type": "array" } }, "required": [ - "action", - "extension_name" + "task_parameters" ], + "title": "CreateDynamicTaskParams", "type": "object" }, "annotations": { - "title": "Enable or disable an extension", + "title": "Create Dynamic Tasks", "readOnlyHint": false, "destructiveHint": false, "idempotentHint": false, - "openWorldHint": false + "openWorldHint": true } }, { @@ -108,15 +169,6 @@ "description": "A cron expression for create action. Supports both 5-field (minute hour day month weekday) and 6-field (second minute hour day month weekday) formats. 5-field expressions are automatically converted to 6-field by prepending '0' for seconds.", "type": "string" }, - "execution_mode": { - "default": "background", - "description": "Execution mode for create action: 'foreground' or 'background'", - "enum": [ - "foreground", - "background" - ], - "type": "string" - }, "job_id": { "description": "Job identifier for operations on existing jobs", "type": "string" @@ -148,94 +200,6 @@ "openWorldHint": false } }, - { - "name": "dynamic_task__create_task", - "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", - "inputSchema": { - "properties": { - "execution_mode": { - "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", - "enum": [ - "sequential", - "parallel" - ], - "type": "string" - }, - "task_parameters": { - "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", - "items": { - "properties": { - "activities": { - "items": { - "type": "string" - }, - "type": "array" - }, - "context": { - "items": { - "type": "string" - }, - "type": "array" - }, - "description": { - "type": "string" - }, - "extensions": { - "items": { - "type": "object" - }, - "type": "array" - }, - "instructions": { - "description": "Task instructions (required if prompt is not provided)", - "type": "string" - }, - "parameters": { - "items": { - "type": "object" - }, - "type": "array" - }, - "prompt": { - "description": "Initial prompt (required if instructions is not provided)", - "type": "string" - }, - "response": { - "type": "object" - }, - "retry": { - "type": "object" - }, - "return_last_only": { - "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", - "type": "boolean" - }, - "settings": { - "type": "object" - }, - "title": { - "type": "string" - } - }, - "type": "object" - }, - "minItems": 1, - "type": "array" - } - }, - "required": [ - "task_parameters" - ], - "type": "object" - }, - "annotations": { - "title": "Create Dynamic Tasks", - "readOnlyHint": false, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": true - } - }, { "name": "subagent__execute_task", "description": "Only use the subagent__execute_task tool when you execute sub recipe task or dynamic task.\n EXECUTION STRATEGY DECISION:\n 1. If the tasks are created with execution_mode, use the execution_mode.\n 2. Execute tasks sequentially unless user explicitly requests parallel execution. PARALLEL: User uses keywords like 'parallel', 'simultaneously', 'at the same time', 'concurrently'\n\n IMPLEMENTATION:\n - Sequential execution: Call this tool multiple times, passing exactly ONE task per call\n - Parallel execution: Call this tool once, passing an ARRAY of all tasks\n\n EXAMPLES:\n User Intent Based:\n - User: 'get weather and tell me a joke' → Sequential (2 separate tool calls, 1 task each)\n - User: 'get weather and joke in parallel' → Parallel (1 tool call with array of 2 tasks)\n - User: 'run these simultaneously' → Parallel (1 tool call with task array)\n - User: 'do task A then task B' → Sequential (2 separate tool calls)", @@ -270,6 +234,22 @@ "idempotentHint": false, "openWorldHint": true } + }, + { + "name": "weather_extension__get_weather", + "description": "Get the weather for a location", + "inputSchema": { + "properties": { + "location": { + "description": "The city and state, e.g. San Francisco, CA", + "type": "string" + } + }, + "required": [ + "location" + ], + "type": "object" + } } ] }, @@ -277,11 +257,11 @@ "message": { "id": null, "role": "assistant", - "created": 1760565463, + "created": 1763989803, "content": [ { "type": "toolRequest", - "id": "call_NSF9KSX2tCz8i8qb41WU6GHT", + "id": "call_GmIN2vOnjDg6MSVWWFuZ4FlV", "toolCall": { "status": "success", "value": { @@ -301,21 +281,69 @@ "usage": { "model": "gpt-4o-2024-08-06", "usage": { - "input_tokens": 1708, + "input_tokens": 1492, "output_tokens": 19, - "total_tokens": 1727 + "total_tokens": 1511 } } } }, - "4166a5a4297f1a6c903edc1c3232d12e2503190a3d4f98486d1a61d721bb0952": { + "8b1b8633232ca390cb3ff37a48daf74168b58f13e2943efcb17957129ee34d1a": { "input": { - "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\nThe current date is 2025-10-15 21:57:42.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\nUse the search_available_extensions tool to find additional extensions to enable to help with your task. To enable\nextensions, use the enable_extension tool and provide the extension_name. You should only enable extensions found from\nthe search_available_extensions tool.\n\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n\n## weather_extension\n\n\n\n\n\n\n\n\n\n# Suggestion\n\n\"\"\n\n\n\n\n# sub agents\n\nExecute self contained tasks where step-by-step visibility is not important through subagents.\n\n- Delegate via `dynamic_task__create_task` for: result-only operations, parallelizable work, multi-part requests,\n verification, exploration\n- Parallel subagents for multiple operations, single subagents for independent work\n- Explore solutions in parallel — launch parallel subagents with different approaches (if non-interfering)\n- Provide all needed context — subagents cannot see your context\n- Use extension filters to limit resource access\n- Use return_last_only when only a summary or simple answer is required — inform subagent of this choice.\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.\n\n# Additional Instructions:\n\nRight now you are *NOT* in the chat only mode and have access to tool use and system.", + "system": "Reply with only a description in four words or less", "messages": [ { "id": null, "role": "user", - "created": 1760565462, + "created": 1763989801, + "content": [ + { + "type": "text", + "text": "Here are the first few user messages:\ntell me what the weather is in Berlin, Germany\n\nBased on the conversation so far, provide a concise description of this session in 4 words or less. This will be used for finding the session later in a UI with limited space - reply *ONLY* with the description" + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + } + ], + "tools": [] + }, + "output": { + "message": { + "id": null, + "role": "assistant", + "created": 1763989804, + "content": [ + { + "type": "text", + "text": "Berlin weather inquiry" + } + ], + "metadata": { + "userVisible": true, + "agentVisible": true + } + }, + "usage": { + "model": "gpt-4o-2024-08-06", + "usage": { + "input_tokens": 84, + "output_tokens": 3, + "total_tokens": 87 + } + } + } + }, + "98726e7698e93241dfcba182bdca3a7d0640f0f538f086f4981dca6689b7ab5a": { + "input": { + "system": "You are a general-purpose AI agent called goose, created by Block, the parent company of Square, CashApp, and Tidal.\ngoose is being developed as an open-source software project.\n\ngoose uses LLM providers with tool calling capability. You can be used with different language models (gpt-4o,\nclaude-sonnet-4, o1, llama-3.2, deepseek-r1, etc).\nThese models have varying knowledge cut-off dates depending on when they were trained, but typically it's between 5-10\nmonths prior to the current date.\n\n# Extensions\n\nExtensions allow other applications to provide context to goose. Extensions connect goose to different data sources and\ntools.\nYou are capable of dynamically plugging into new extensions and learning how to use them. You solve higher level\nproblems using the tools in these extensions, and can interact with multiple at once.\n\nIf the Extension Manager extension is enabled, you can use the search_available_extensions tool to discover additional\nextensions that can help with your task. To enable or disable extensions, use the manage_extensions tool with the\nextension_name. You should only enable extensions found from the search_available_extensions tool.\nIf Extension Manager is not available, you can only work with currently enabled extensions and cannot dynamically load\nnew ones.\n\nBecause you dynamically load extensions, your conversation history may refer\nto interactions with extensions that are not currently active. The currently\nactive extensions are below. Each of these extensions provides tools that are\nin your tool specification.\n\n\n## weather_extension\n\n\n\n\n\n\n# Response Guidelines\n\n- Use Markdown formatting for all responses.\n- Follow best practices for Markdown, including:\n - Using headers for organization.\n - Bullet points for lists.\n - Links formatted correctly, either as linked text (e.g., [this is linked text](https://example.com)) or automatic\n links using angle brackets (e.g., ).\n- For code examples, use fenced code blocks by placing triple backticks (` ``` `) before and after the code. Include the\n language identifier after the opening backticks (e.g., ` ```python `) to enable syntax highlighting.\n- Ensure clarity, conciseness, and proper formatting to enhance readability and usability.", + "messages": [ + { + "id": "msg_20251124_7_0", + "role": "user", + "created": 1763989801, "content": [ { "type": "text", @@ -328,13 +356,13 @@ } }, { - "id": null, + "id": "msg_9a0fb3d7-d681-40b9-ab58-bb1cbe03e4b0", "role": "assistant", - "created": 1760565463, + "created": 1763989803, "content": [ { "type": "toolRequest", - "id": "call_NSF9KSX2tCz8i8qb41WU6GHT", + "id": "call_GmIN2vOnjDg6MSVWWFuZ4FlV", "toolCall": { "status": "success", "value": { @@ -352,13 +380,13 @@ } }, { - "id": "msg_926cf617-03cf-4027-ab42-d9bf44831517", + "id": "msg_51ae1986-b095-4163-8d42-004ff1a28c59", "role": "user", - "created": 1760565463, + "created": 1763989803, "content": [ { "type": "toolResponse", - "id": "call_NSF9KSX2tCz8i8qb41WU6GHT", + "id": "call_GmIN2vOnjDg6MSVWWFuZ4FlV", "toolResult": { "status": "success", "value": [ @@ -378,67 +406,128 @@ ], "tools": [ { - "name": "weather_extension__get_weather", - "description": "Get the weather for a location", + "name": "dynamic_task__create_task", + "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, activities. Arrays for multiple tasks.", "inputSchema": { - "properties": { - "location": { - "description": "The city and state, e.g. San Francisco, CA", - "type": "string" + "$defs": { + "TaskParameter": { + "description": "Parameters for a single task", + "properties": { + "activities": { + "items": { + "type": "string" + }, + "type": [ + "array", + "null" + ] + }, + "description": { + "type": [ + "string", + "null" + ] + }, + "extensions": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "instructions": { + "type": [ + "string", + "null" + ] + }, + "parameters": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": [ + "array", + "null" + ] + }, + "prompt": { + "type": [ + "string", + "null" + ] + }, + "response": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "retry": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "return_last_only": { + "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", + "type": [ + "boolean", + "null" + ] + }, + "settings": { + "additionalProperties": true, + "type": [ + "object", + "null" + ] + }, + "title": { + "type": [ + "string", + "null" + ] + } + }, + "type": "object" } }, - "required": [ - "location" - ], - "type": "object" - } - }, - { - "name": "platform__search_available_extensions", - "description": "Searches for additional extensions available to help complete tasks.\n Use this tool when you're unable to find a specific feature or functionality you need to complete your task, or when standard approaches aren't working.\n These extensions might provide the exact tools needed to solve your problem.\n If you find a relevant one, consider using your tools to enable it.", - "inputSchema": { - "properties": {}, - "required": [], - "type": "object" - }, - "annotations": { - "title": "Discover extensions", - "readOnlyHint": true, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": false - } - }, - { - "name": "platform__manage_extensions", - "description": "Tool to manage extensions and tools in goose context.\n Enable or disable extensions to help complete tasks.\n Enable or disable an extension by providing the extension name.\n ", - "inputSchema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { - "action": { - "description": "The action to perform", - "enum": [ - "enable", - "disable" - ], - "type": "string" + "execution_mode": { + "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", + "type": [ + "string", + "null" + ] }, - "extension_name": { - "description": "The name of the extension to enable", - "type": "string" + "task_parameters": { + "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", + "items": { + "$ref": "#/$defs/TaskParameter" + }, + "minItems": 1, + "type": "array" } }, "required": [ - "action", - "extension_name" + "task_parameters" ], + "title": "CreateDynamicTaskParams", "type": "object" }, "annotations": { - "title": "Enable or disable an extension", + "title": "Create Dynamic Tasks", "readOnlyHint": false, "destructiveHint": false, "idempotentHint": false, - "openWorldHint": false + "openWorldHint": true } }, { @@ -465,15 +554,6 @@ "description": "A cron expression for create action. Supports both 5-field (minute hour day month weekday) and 6-field (second minute hour day month weekday) formats. 5-field expressions are automatically converted to 6-field by prepending '0' for seconds.", "type": "string" }, - "execution_mode": { - "default": "background", - "description": "Execution mode for create action: 'foreground' or 'background'", - "enum": [ - "foreground", - "background" - ], - "type": "string" - }, "job_id": { "description": "Job identifier for operations on existing jobs", "type": "string" @@ -505,94 +585,6 @@ "openWorldHint": false } }, - { - "name": "dynamic_task__create_task", - "description": "Create tasks with instructions or prompt. For simple tasks, only include the instructions field. Extensions control: omit field = use all current extensions; empty array [] = no extensions; array with names = only those extensions. Specify extensions as shortnames (the prefixes for your tools). Specify return_last_only as true and have your subagent summarize its work in its last message to conserve your own context. Optional: title, description, extensions, settings, retry, response schema, context, activities. Arrays for multiple tasks.", - "inputSchema": { - "properties": { - "execution_mode": { - "description": "How to execute multiple tasks (default: parallel for multiple tasks, sequential for single task)", - "enum": [ - "sequential", - "parallel" - ], - "type": "string" - }, - "task_parameters": { - "description": "Array of tasks. Each task must have either 'instructions' OR 'prompt' field (at least one is required).", - "items": { - "properties": { - "activities": { - "items": { - "type": "string" - }, - "type": "array" - }, - "context": { - "items": { - "type": "string" - }, - "type": "array" - }, - "description": { - "type": "string" - }, - "extensions": { - "items": { - "type": "object" - }, - "type": "array" - }, - "instructions": { - "description": "Task instructions (required if prompt is not provided)", - "type": "string" - }, - "parameters": { - "items": { - "type": "object" - }, - "type": "array" - }, - "prompt": { - "description": "Initial prompt (required if instructions is not provided)", - "type": "string" - }, - "response": { - "type": "object" - }, - "retry": { - "type": "object" - }, - "return_last_only": { - "description": "If true, return only the last message from the subagent (default: false, returns full conversation)", - "type": "boolean" - }, - "settings": { - "type": "object" - }, - "title": { - "type": "string" - } - }, - "type": "object" - }, - "minItems": 1, - "type": "array" - } - }, - "required": [ - "task_parameters" - ], - "type": "object" - }, - "annotations": { - "title": "Create Dynamic Tasks", - "readOnlyHint": false, - "destructiveHint": false, - "idempotentHint": false, - "openWorldHint": true - } - }, { "name": "subagent__execute_task", "description": "Only use the subagent__execute_task tool when you execute sub recipe task or dynamic task.\n EXECUTION STRATEGY DECISION:\n 1. If the tasks are created with execution_mode, use the execution_mode.\n 2. Execute tasks sequentially unless user explicitly requests parallel execution. PARALLEL: User uses keywords like 'parallel', 'simultaneously', 'at the same time', 'concurrently'\n\n IMPLEMENTATION:\n - Sequential execution: Call this tool multiple times, passing exactly ONE task per call\n - Parallel execution: Call this tool once, passing an ARRAY of all tasks\n\n EXAMPLES:\n User Intent Based:\n - User: 'get weather and tell me a joke' → Sequential (2 separate tool calls, 1 task each)\n - User: 'get weather and joke in parallel' → Parallel (1 tool call with array of 2 tasks)\n - User: 'run these simultaneously' → Parallel (1 tool call with task array)\n - User: 'do task A then task B' → Sequential (2 separate tool calls)", @@ -627,6 +619,22 @@ "idempotentHint": false, "openWorldHint": true } + }, + { + "name": "weather_extension__get_weather", + "description": "Get the weather for a location", + "inputSchema": { + "properties": { + "location": { + "description": "The city and state, e.g. San Francisco, CA", + "type": "string" + } + }, + "required": [ + "location" + ], + "type": "object" + } } ] }, @@ -634,7 +642,7 @@ "message": { "id": null, "role": "assistant", - "created": 1760565465, + "created": 1763989805, "content": [ { "type": "text", @@ -649,9 +657,9 @@ "usage": { "model": "gpt-4o-2024-08-06", "usage": { - "input_tokens": 1750, + "input_tokens": 1534, "output_tokens": 18, - "total_tokens": 1768 + "total_tokens": 1552 } } } diff --git a/crates/goose/src/agents/agent.rs b/crates/goose/src/agents/agent.rs index 4078f98419..ba1b0fc1b7 100644 --- a/crates/goose/src/agents/agent.rs +++ b/crates/goose/src/agents/agent.rs @@ -299,7 +299,7 @@ impl Agent { async fn handle_approved_and_denied_tools( &self, permission_check_result: &PermissionCheckResult, - message_tool_response: Arc>, + request_to_response_map: &HashMap>>, cancel_token: Option, session: &Session, ) -> Result> { @@ -334,18 +334,25 @@ impl Agent { } } - // Handle denied tools - for request in &permission_check_result.denied { - let mut response = message_tool_response.lock().await; - *response = response.clone().with_tool_response( - request.id.clone(), - Ok(vec![rmcp::model::Content::text(DECLINED_RESPONSE)]), - ); - } - + Self::handle_denied_tools(permission_check_result, request_to_response_map).await; Ok(tool_futures) } + async fn handle_denied_tools( + permission_check_result: &PermissionCheckResult, + request_to_response_map: &HashMap>>, + ) { + for request in &permission_check_result.denied { + if let Some(response_msg) = request_to_response_map.get(&request.id) { + let mut response = response_msg.lock().await; + *response = response.clone().with_tool_response( + request.id.clone(), + Ok(vec![rmcp::model::Content::text(DECLINED_RESPONSE)]), + ); + } + } + } + pub async fn set_scheduler(&self, scheduler: Arc) { let mut scheduler_service = self.scheduler_service.lock().await; *scheduler_service = Some(scheduler); @@ -920,336 +927,352 @@ impl Agent { }); Ok(Box::pin(async_stream::try_stream! { - let _ = reply_span.enter(); - let mut turns_taken = 0u32; - let max_turns = session_config.max_turns.unwrap_or(DEFAULT_MAX_TURNS); + let _ = reply_span.enter(); + let mut turns_taken = 0u32; + let max_turns = session_config.max_turns.unwrap_or(DEFAULT_MAX_TURNS); - loop { - if is_token_cancelled(&cancel_token) { - break; + loop { + if is_token_cancelled(&cancel_token) { + break; + } + + if let Some(final_output_tool) = self.final_output_tool.lock().await.as_ref() { + if final_output_tool.final_output.is_some() { + let final_event = AgentEvent::Message( + Message::assistant().with_text(final_output_tool.final_output.clone().unwrap()) + ); + yield final_event; + break; + } + } + + turns_taken += 1; + if turns_taken > max_turns { + yield AgentEvent::Message( + Message::assistant().with_text( + "I've reached the maximum number of actions I can do without user input. Would you like me to continue?" + ) + ); + break; + } + + let conversation_with_moim = super::moim::inject_moim( + conversation.clone(), + &self.extension_manager, + ).await; + + let mut stream = Self::stream_response_from_provider( + self.provider().await?, + &system_prompt, + conversation_with_moim.messages(), + &tools, + &toolshim_tools, + ).await?; + + let mut no_tools_called = true; + let mut messages_to_add = Conversation::default(); + let mut tools_updated = false; + let mut did_recovery_compact_this_iteration = false; + + while let Some(next) = stream.next().await { + if is_token_cancelled(&cancel_token) { + break; + } + + match next { + Ok((response, usage)) => { + // Emit model change event if provider is lead-worker + let provider = self.provider().await?; + if let Some(lead_worker) = provider.as_lead_worker() { + if let Some(ref usage) = usage { + let active_model = usage.model.clone(); + let (lead_model, worker_model) = lead_worker.get_model_info(); + let mode = if active_model == lead_model { + "lead" + } else if active_model == worker_model { + "worker" + } else { + "unknown" + }; + + yield AgentEvent::ModelChange { + model: active_model, + mode: mode.to_string(), + }; + } + } + + if let Some(ref usage) = usage { + Self::update_session_metrics(&session_config, usage, false).await?; + } + + if let Some(response) = response { + let ToolCategorizeResult { + frontend_requests, + remaining_requests, + filtered_response, + } = self.categorize_tools(&response, &tools).await; + let requests_to_record: Vec = frontend_requests.iter().chain(remaining_requests.iter()).cloned().collect(); + self.tool_route_manager + .record_tool_requests(&requests_to_record) + .await; + + yield AgentEvent::Message(filtered_response.clone()); + tokio::task::yield_now().await; + + let num_tool_requests = frontend_requests.len() + remaining_requests.len(); + if num_tool_requests == 0 { + messages_to_add.push(response.clone()); + continue; + } + + let tool_response_messages: Vec>> = (0..num_tool_requests) + .map(|_| Arc::new(Mutex::new(Message::user().with_id( + format!("msg_{}", Uuid::new_v4()) + )))) + .collect(); + + let mut request_to_response_map = HashMap::new(); + for (idx, request) in frontend_requests.iter().chain(remaining_requests.iter()).enumerate() { + request_to_response_map.insert(request.id.clone(), tool_response_messages[idx].clone()); + } + + for (idx, request) in frontend_requests.iter().enumerate() { + let mut frontend_tool_stream = self.handle_frontend_tool_request( + request, + tool_response_messages[idx].clone(), + ); + + while let Some(msg) = frontend_tool_stream.try_next().await? { + yield AgentEvent::Message(msg); } - - if let Some(final_output_tool) = self.final_output_tool.lock().await.as_ref() { - if final_output_tool.final_output.is_some() { - let final_event = AgentEvent::Message( - Message::assistant().with_text(final_output_tool.final_output.clone().unwrap()) + } + if goose_mode == GooseMode::Chat { + // Skip all remaining tool calls in chat mode + for request in remaining_requests.iter() { + if let Some(response_msg) = request_to_response_map.get(&request.id) { + let mut response = response_msg.lock().await; + *response = response.clone().with_tool_response( + request.id.clone(), + Ok(vec![Content::text(CHAT_MODE_TOOL_SKIPPED_RESPONSE)]), ); - yield final_event; - break; + } + } + } else { + // Run all tool inspectors + let inspection_results = self.tool_inspection_manager + .inspect_tools( + &remaining_requests, + conversation.messages(), + ) + .await?; + + let permission_check_result = self.tool_inspection_manager + .process_inspection_results_with_permission_inspector( + &remaining_requests, + &inspection_results, + ) + .unwrap_or_else(|| { + let mut result = PermissionCheckResult { + approved: vec![], + needs_approval: vec![], + denied: vec![], + }; + result.needs_approval.extend(remaining_requests.iter().cloned()); + result + }); + + // Track extension requests + let mut enable_extension_request_ids = vec![]; + for request in &remaining_requests { + if let Ok(tool_call) = &request.tool_call { + if tool_call.name == MANAGE_EXTENSIONS_TOOL_NAME_COMPLETE { + enable_extension_request_ids.push(request.id.clone()); + } } } - turns_taken += 1; - if turns_taken > max_turns { - yield AgentEvent::Message( - Message::assistant().with_text( - "I've reached the maximum number of actions I can do without user input. Would you like me to continue?" - ) - ); - break; - } - - let conversation_with_moim = super::moim::inject_moim( - conversation.clone(), - &self.extension_manager, - ).await; - - let mut stream = Self::stream_response_from_provider( - self.provider().await?, - &system_prompt, - conversation_with_moim.messages(), - &tools, - &toolshim_tools, + let mut tool_futures = self.handle_approved_and_denied_tools( + &permission_check_result, + &request_to_response_map, + cancel_token.clone(), + &session, ).await?; - let mut no_tools_called = true; - let mut messages_to_add = Conversation::default(); - let mut tools_updated = false; - let mut did_recovery_compact_this_iteration = false; + let tool_futures_arc = Arc::new(Mutex::new(tool_futures)); - while let Some(next) = stream.next().await { + let mut tool_approval_stream = self.handle_approval_tool_requests( + &permission_check_result.needs_approval, + tool_futures_arc.clone(), + &request_to_response_map, + cancel_token.clone(), + &session, + &inspection_results, + ); + + while let Some(msg) = tool_approval_stream.try_next().await? { + yield AgentEvent::Message(msg); + } + + tool_futures = { + let mut futures_lock = tool_futures_arc.lock().await; + futures_lock.drain(..).collect::>() + }; + + let with_id = tool_futures + .into_iter() + .map(|(request_id, stream)| { + stream.map(move |item| (request_id.clone(), item)) + }) + .collect::>(); + + let mut combined = stream::select_all(with_id); + let mut all_install_successful = true; + + while let Some((request_id, item)) = combined.next().await { if is_token_cancelled(&cancel_token) { break; } - - match next { - Ok((response, usage)) => { - // Emit model change event if provider is lead-worker - let provider = self.provider().await?; - if let Some(lead_worker) = provider.as_lead_worker() { - if let Some(ref usage) = usage { - let active_model = usage.model.clone(); - let (lead_model, worker_model) = lead_worker.get_model_info(); - let mode = if active_model == lead_model { - "lead" - } else if active_model == worker_model { - "worker" - } else { - "unknown" - }; - - yield AgentEvent::ModelChange { - model: active_model, - mode: mode.to_string(), - }; - } + match item { + ToolStreamItem::Result(output) => { + if enable_extension_request_ids.contains(&request_id) + && output.is_err() + { + all_install_successful = false; } - - if let Some(ref usage) = usage { - Self::update_session_metrics(&session_config, usage, false).await?; - } - - if let Some(response) = response { - messages_to_add.push(response.clone()); - let ToolCategorizeResult { - frontend_requests, - remaining_requests, - filtered_response, - } = self.categorize_tools(&response, &tools).await; - let requests_to_record: Vec = frontend_requests.iter().chain(remaining_requests.iter()).cloned().collect(); - self.tool_route_manager - .record_tool_requests(&requests_to_record) - .await; - - yield AgentEvent::Message(filtered_response.clone()); - tokio::task::yield_now().await; - - let num_tool_requests = frontend_requests.len() + remaining_requests.len(); - if num_tool_requests == 0 { - continue; - } - - let message_tool_response = Arc::new(Mutex::new(Message::user().with_id( - format!("msg_{}", Uuid::new_v4()) - ))); - - let mut frontend_tool_stream = self.handle_frontend_tool_requests( - &frontend_requests, - message_tool_response.clone(), - ); - - while let Some(msg) = frontend_tool_stream.try_next().await? { - yield AgentEvent::Message(msg); - } - - if goose_mode == GooseMode::Chat { - // Skip all tool calls in chat mode - for request in remaining_requests { - let mut response = message_tool_response.lock().await; - *response = response.clone().with_tool_response( - request.id.clone(), - Ok(vec![Content::text(CHAT_MODE_TOOL_SKIPPED_RESPONSE)]), - ); - } - } else { - // Run all tool inspectors (security, repetition, permission, etc.) - let inspection_results = self.tool_inspection_manager - .inspect_tools( - &remaining_requests, - conversation.messages(), - ) - .await?; - - // Process inspection results into permission decisions using the permission inspector - let permission_check_result = self.tool_inspection_manager - .process_inspection_results_with_permission_inspector( - &remaining_requests, - &inspection_results, - ) - .unwrap_or_else(|| { - // Fallback if permission inspector not found - default to needs approval - let mut result = PermissionCheckResult { - approved: vec![], - needs_approval: vec![], - denied: vec![], - }; - result.needs_approval.extend(remaining_requests.iter().cloned()); - result - }); - - // Track extension requests for special handling - let mut enable_extension_request_ids = vec![]; - for request in &remaining_requests { - if let Ok(tool_call) = &request.tool_call { - if tool_call.name == MANAGE_EXTENSIONS_TOOL_NAME_COMPLETE { - enable_extension_request_ids.push(request.id.clone()); - } - } - } - - let mut tool_futures = self.handle_approved_and_denied_tools( - &permission_check_result, - message_tool_response.clone(), - cancel_token.clone(), - &session, - ).await?; - - let tool_futures_arc = Arc::new(Mutex::new(tool_futures)); - - let mut tool_approval_stream = self.handle_approval_tool_requests( - &permission_check_result.needs_approval, - tool_futures_arc.clone(), - message_tool_response.clone(), - cancel_token.clone(), - &session, - &inspection_results, - ); - - while let Some(msg) = tool_approval_stream.try_next().await? { - yield AgentEvent::Message(msg); - } - - tool_futures = { - let mut futures_lock = tool_futures_arc.lock().await; - futures_lock.drain(..).collect::>() - }; - - let with_id = tool_futures - .into_iter() - .map(|(request_id, stream)| { - stream.map(move |item| (request_id.clone(), item)) - }) - .collect::>(); - - let mut combined = stream::select_all(with_id); - let mut all_install_successful = true; - - while let Some((request_id, item)) = combined.next().await { - if is_token_cancelled(&cancel_token) { - break; - } - match item { - ToolStreamItem::Result(output) => { - if enable_extension_request_ids.contains(&request_id) - && output.is_err() - { - all_install_successful = false; - } - let mut response = message_tool_response.lock().await; - *response = - response.clone().with_tool_response(request_id, output); - } - ToolStreamItem::Message(msg) => { - yield AgentEvent::McpNotification(( - request_id, msg, - )); - } - } - } - - if all_install_successful && !enable_extension_request_ids.is_empty() { - if let Err(e) = self.save_extension_state(&session_config).await { - warn!("Failed to save extension state after runtime changes: {}", e); - } - tools_updated = true; - } - } - - let final_message_tool_resp = message_tool_response.lock().await.clone(); - yield AgentEvent::Message(final_message_tool_resp.clone()); - - no_tools_called = false; - messages_to_add.push(final_message_tool_resp); + if let Some(response_msg) = request_to_response_map.get(&request_id) { + let mut response = response_msg.lock().await; + *response = response.clone().with_tool_response(request_id, output); } } - Err(ProviderError::ContextLengthExceeded(_error_msg)) => { - yield AgentEvent::Message( - Message::assistant().with_system_notification( - SystemNotificationType::InlineMessage, - "Context limit reached. Compacting to continue conversation...", - ) - ); - yield AgentEvent::Message( - Message::assistant().with_system_notification( - SystemNotificationType::ThinkingMessage, - COMPACTION_THINKING_TEXT, - ) - ); + ToolStreamItem::Message(msg) => { + yield AgentEvent::McpNotification((request_id, msg)); + } + } + } - match compact_messages(self.provider().await?.as_ref(), &conversation, false).await { - Ok((compacted_conversation, usage)) => { - SessionManager::replace_conversation(&session_config.id, &compacted_conversation).await?; - Self::update_session_metrics(&session_config, &usage, true).await?; - conversation = compacted_conversation; - did_recovery_compact_this_iteration = true; - yield AgentEvent::HistoryReplaced(conversation.clone()); - continue; + if all_install_successful && !enable_extension_request_ids.is_empty() { + if let Err(e) = self.save_extension_state(&session_config).await { + warn!("Failed to save extension state after runtime changes: {}", e); + } + tools_updated = true; + } + } + + for (idx, request) in frontend_requests.iter() + .chain(remaining_requests.iter()).enumerate() { + if request.tool_call.is_ok() { + let request_msg = Message::assistant() + .with_id(format!("msg_{}", Uuid::new_v4())) + .with_tool_request(request.id.clone(), request.tool_call.clone()); + messages_to_add.push(request_msg); + let final_response = tool_response_messages[idx] + .lock().await.clone(); + yield AgentEvent::Message(final_response.clone()); + messages_to_add.push(final_response); + } + } + no_tools_called = false; + } + } + Err(ProviderError::ContextLengthExceeded(_error_msg)) => { + yield AgentEvent::Message( + Message::assistant().with_system_notification( + SystemNotificationType::InlineMessage, + "Context limit reached. Compacting to continue conversation...", + ) + ); + yield AgentEvent::Message( + Message::assistant().with_system_notification( + SystemNotificationType::ThinkingMessage, + COMPACTION_THINKING_TEXT, + ) + ); + + match compact_messages(self.provider().await?.as_ref(), &conversation, false).await { + Ok((compacted_conversation, usage)) => { + SessionManager::replace_conversation(&session_config.id, &compacted_conversation).await?; + Self::update_session_metrics(&session_config, &usage, true).await?; + conversation = compacted_conversation; + did_recovery_compact_this_iteration = true; + yield AgentEvent::HistoryReplaced(conversation.clone()); + continue; + } + Err(e) => { + error!("Error: {}", e); + yield AgentEvent::Message( + Message::assistant().with_text( + format!("Ran into this error trying to compact: {e}.\n\nPlease retry if you think this is a transient or recoverable error.") + ) + ); + break; + } + } } Err(e) => { error!("Error: {}", e); yield AgentEvent::Message( Message::assistant().with_text( - format!("Ran into this error trying to compact: {e}.\n\nPlease retry if you think this is a transient or recoverable error.") + format!("Ran into this error: {e}.\n\nPlease retry if you think this is a transient or recoverable error.") ) ); break; } } } - Err(e) => { - error!("Error: {}", e); - yield AgentEvent::Message( - Message::assistant().with_text( - format!("Ran into this error: {e}.\n\nPlease retry if you think this is a transient or recoverable error.") - ) - ); - break; + if tools_updated { + (tools, toolshim_tools, system_prompt) = + self.prepare_tools_and_prompt(&working_dir).await?; } - } - } - if tools_updated { - (tools, toolshim_tools, system_prompt) = - self.prepare_tools_and_prompt(&working_dir).await?; - } - let mut exit_chat = false; - if no_tools_called { - if let Some(final_output_tool) = self.final_output_tool.lock().await.as_ref() { - if final_output_tool.final_output.is_none() { - warn!("Final output tool has not been called yet. Continuing agent loop."); - let message = Message::user().with_text(FINAL_OUTPUT_CONTINUATION_MESSAGE); - messages_to_add.push(message.clone()); - yield AgentEvent::Message(message); - } else { - let message = Message::assistant().with_text(final_output_tool.final_output.clone().unwrap()); - messages_to_add.push(message.clone()); - yield AgentEvent::Message(message); - exit_chat = true; - } - } else if did_recovery_compact_this_iteration { - // Avoid setting exit_chat; continue from last user message in the conversation - } else { - match self.handle_retry_logic(&mut conversation, &session_config, &initial_messages).await { - Ok(should_retry) => { - if should_retry { - info!("Retry logic triggered, restarting agent loop"); + let mut exit_chat = false; + if no_tools_called { + if let Some(final_output_tool) = self.final_output_tool.lock().await.as_ref() { + if final_output_tool.final_output.is_none() { + warn!("Final output tool has not been called yet. Continuing agent loop."); + let message = Message::user().with_text(FINAL_OUTPUT_CONTINUATION_MESSAGE); + messages_to_add.push(message.clone()); + yield AgentEvent::Message(message); } else { + let message = Message::assistant().with_text(final_output_tool.final_output.clone().unwrap()); + messages_to_add.push(message.clone()); + yield AgentEvent::Message(message); exit_chat = true; } - } - Err(e) => { - error!("Retry logic failed: {}", e); - yield AgentEvent::Message( - Message::assistant().with_text( - format!("Retry logic encountered an error: {}", e) - ) - ); - exit_chat = true; + } else if did_recovery_compact_this_iteration { + // Avoid setting exit_chat; continue from last user message in the conversation + } else { + match self.handle_retry_logic(&mut conversation, &session_config, &initial_messages).await { + Ok(should_retry) => { + if should_retry { + info!("Retry logic triggered, restarting agent loop"); + } else { + exit_chat = true; + } + } + Err(e) => { + error!("Retry logic failed: {}", e); + yield AgentEvent::Message( + Message::assistant().with_text( + format!("Retry logic encountered an error: {}", e) + ) + ); + exit_chat = true; + } + } } } + + for msg in &messages_to_add { + SessionManager::add_message(&session_config.id, msg).await?; + } + conversation.extend(messages_to_add); + if exit_chat { + break; + } + + tokio::task::yield_now().await; } - } - - for msg in &messages_to_add { - SessionManager::add_message(&session_config.id, msg).await?; - } - conversation.extend(messages_to_add); - if exit_chat { - break; - } - - tokio::task::yield_now().await; - } - })) + })) } pub async fn extend_system_prompt(&self, instruction: String) { diff --git a/crates/goose/src/agents/tool_execution.rs b/crates/goose/src/agents/tool_execution.rs index 383b23e297..402bb305ba 100644 --- a/crates/goose/src/agents/tool_execution.rs +++ b/crates/goose/src/agents/tool_execution.rs @@ -1,3 +1,4 @@ +use std::collections::HashMap; use std::future::Future; use std::sync::Arc; @@ -52,97 +53,98 @@ impl Agent { &'a self, tool_requests: &'a [ToolRequest], tool_futures: Arc>>, - message_tool_response: Arc>, + request_to_response_map: &'a HashMap>>, cancellation_token: Option, session: &'a Session, inspection_results: &'a [crate::tool_inspection::InspectionResult], ) -> BoxStream<'a, anyhow::Result> { try_stream! { - for request in tool_requests.iter() { - if let Ok(tool_call) = request.tool_call.clone() { - // Find the corresponding inspection result for this tool request - let security_message = inspection_results.iter() - .find(|result| result.tool_request_id == request.id) - .and_then(|result| { - if let crate::tool_inspection::InspectionAction::RequireApproval(Some(message)) = &result.action { - Some(message.clone()) - } else { - None + for request in tool_requests.iter() { + if let Ok(tool_call) = request.tool_call.clone() { + // Find the corresponding inspection result for this tool request + let security_message = inspection_results.iter() + .find(|result| result.tool_request_id == request.id) + .and_then(|result| { + if let crate::tool_inspection::InspectionAction::RequireApproval(Some(message)) = &result.action { + Some(message.clone()) + } else { + None + } + }); + + let confirmation = Message::assistant() + .with_tool_confirmation_request( + request.id.clone(), + tool_call.name.to_string().clone(), + tool_call.arguments.clone().unwrap_or_default(), + security_message, + ) + .user_only(); + yield confirmation; + + let mut rx = self.confirmation_rx.lock().await; + while let Some((req_id, confirmation)) = rx.recv().await { + if req_id == request.id { + // Log user decision if this was a security alert + if let Some(finding_id) = get_security_finding_id_from_results(&request.id, inspection_results) { + tracing::info!( + counter.goose.prompt_injection_user_decisions = 1, + decision = ?confirmation.permission, + finding_id = %finding_id, + "User security decision" + ); + } + + if confirmation.permission == Permission::AllowOnce || confirmation.permission == Permission::AlwaysAllow { + let (req_id, tool_result) = self.dispatch_tool_call(tool_call.clone(), request.id.clone(), cancellation_token.clone(), session).await; + let mut futures = tool_futures.lock().await; + + futures.push((req_id, match tool_result { + Ok(result) => tool_stream( + result.notification_stream.unwrap_or_else(|| Box::new(stream::empty())), + result.result, + ), + Err(e) => tool_stream( + Box::new(stream::empty()), + futures::future::ready(Err(e)), + ), + })); + + // Update the shared permission manager when user selects "Always Allow" + if confirmation.permission == Permission::AlwaysAllow { + self.tool_inspection_manager + .update_permission_manager(&tool_call.name, PermissionLevel::AlwaysAllow) + .await; } - }); - - let confirmation = Message::assistant() - .with_tool_confirmation_request( - request.id.clone(), - tool_call.name.to_string().clone(), - tool_call.arguments.clone().unwrap_or_default(), - security_message, - ) - .user_only(); - yield confirmation; - - let mut rx = self.confirmation_rx.lock().await; - while let Some((req_id, confirmation)) = rx.recv().await { - if req_id == request.id { - // Log user decision if this was a security alert - if let Some(finding_id) = get_security_finding_id_from_results(&request.id, inspection_results) { - tracing::info!( - counter.goose.prompt_injection_user_decisions = 1, - decision = ?confirmation.permission, - finding_id = %finding_id, - "User security decision" - ); - } - - if confirmation.permission == Permission::AllowOnce || confirmation.permission == Permission::AlwaysAllow { - let (req_id, tool_result) = self.dispatch_tool_call(tool_call.clone(), request.id.clone(), cancellation_token.clone(), session).await; - let mut futures = tool_futures.lock().await; - - futures.push((req_id, match tool_result { - Ok(result) => tool_stream( - result.notification_stream.unwrap_or_else(|| Box::new(stream::empty())), - result.result, - ), - Err(e) => tool_stream( - Box::new(stream::empty()), - futures::future::ready(Err(e)), - ), - })); - - // Update the shared permission manager when user selects "Always Allow" - if confirmation.permission == Permission::AlwaysAllow { - self.tool_inspection_manager - .update_permission_manager(&tool_call.name, PermissionLevel::AlwaysAllow) - .await; - } - } else { - // User declined - add declined response - let mut response = message_tool_response.lock().await; + } else { + // User declined - update the specific response message for this request + if let Some(response_msg) = request_to_response_map.get(&request.id) { + let mut response = response_msg.lock().await; *response = response.clone().with_tool_response( request.id.clone(), Ok(vec![Content::text(DECLINED_RESPONSE)]), ); } - break; // Exit the loop once the matching `req_id` is found } + break; // Exit the loop once the matching `req_id` is found } } } - }.boxed() + } + }.boxed() } - pub(crate) fn handle_frontend_tool_requests<'a>( + pub(crate) fn handle_frontend_tool_request<'a>( &'a self, - tool_requests: &'a [ToolRequest], + tool_request: &'a ToolRequest, message_tool_response: Arc>, ) -> BoxStream<'a, anyhow::Result> { try_stream! { - for request in tool_requests { - if let Ok(tool_call) = request.tool_call.clone() { + if let Ok(tool_call) = tool_request.tool_call.clone() { if self.is_frontend_tool(&tool_call.name).await { // Send frontend tool request and wait for response yield Message::assistant().with_frontend_tool_request( - request.id.clone(), + tool_request.id.clone(), Ok(tool_call.clone()) ); @@ -151,7 +153,6 @@ impl Agent { *response = response.clone().with_tool_response(id, result); } } - } } } .boxed()