From 3dafec6cdc9da1bd4ed67240e80cb5b168514f27 Mon Sep 17 00:00:00 2001 From: "open-swe[bot]" <215916821+open-swe[bot]@users.noreply.github.com> Date: Mon, 28 Jul 2025 11:23:28 -0700 Subject: [PATCH] feat: Add Anthropic Built-in Text Editor Tool for Claude 4 (#543) * Apply patch * Apply patch * Apply patch * Apply patch * Apply patch * Apply patch * Apply patch * Apply patch * Apply patch * Apply patch * Apply patch * Apply patch * Apply patch * Apply patch * Apply patch * Apply patch * cr * cr * fix porting search to grep * cr * revert default model back to sonnet * standalone view tool * cr * cr * cr --------- Co-authored-by: open-swe[bot] Co-authored-by: Brace Sproul --- apps/open-swe/package.json | 2 +- .../planner/nodes/generate-message/index.ts | 12 +- .../planner/nodes/generate-message/prompt.ts | 47 ++++- .../planner/nodes/generate-plan/index.ts | 6 +- .../src/graphs/planner/nodes/rewrite-plan.ts | 4 +- .../src/graphs/planner/nodes/take-action.ts | 7 +- .../nodes/generate-message/index.ts | 58 ++++-- .../nodes/generate-message/prompt.ts | 183 ++++++++++++++++ .../graphs/programmer/nodes/take-action.ts | 7 +- .../src/graphs/reviewer/nodes/final-review.ts | 6 +- .../nodes/generate-review-actions/index.ts | 12 +- .../nodes/generate-review-actions/prompt.ts | 36 ++++ .../reviewer/nodes/take-review-action.ts | 7 +- .../src/tools/builtin-tools/handlers.ts | 196 ++++++++++++++++++ .../src/tools/builtin-tools/text-editor.ts | 107 ++++++++++ apps/open-swe/src/tools/builtin-tools/view.ts | 48 +++++ .../open-swe/src/tools/{search.ts => grep.ts} | 22 +- apps/open-swe/src/tools/index.ts | 3 +- apps/open-swe/src/utils/llms/constants.ts | 18 ++ apps/open-swe/src/utils/llms/model-manager.ts | 18 ++ .../web/src/components/gen-ui/action-step.tsx | 126 ++++++++--- .../web/src/components/thread/messages/ai.tsx | 99 ++++++++- apps/web/src/components/v2/token-usage.tsx | 4 +- packages/shared/src/open-swe/tools.ts | 90 +++++++- packages/shared/src/open-swe/types.ts | 69 ++++++ yarn.lock | 10 +- 26 files changed, 1090 insertions(+), 107 deletions(-) create mode 100644 apps/open-swe/src/tools/builtin-tools/handlers.ts create mode 100644 apps/open-swe/src/tools/builtin-tools/text-editor.ts create mode 100644 apps/open-swe/src/tools/builtin-tools/view.ts rename apps/open-swe/src/tools/{search.ts => grep.ts} (79%) diff --git a/apps/open-swe/package.json b/apps/open-swe/package.json index c23e293e..5b38bc4c 100644 --- a/apps/open-swe/package.json +++ b/apps/open-swe/package.json @@ -24,7 +24,7 @@ }, "dependencies": { "@daytonaio/sdk": "^0.23.1", - "@langchain/anthropic": "^0.3.20", + "@langchain/anthropic": "^0.3.25", "@langchain/community": "^0.3.47", "@langchain/core": "^0.3.65", "@langchain/google-genai": "^0.2.9", diff --git a/apps/open-swe/src/graphs/planner/nodes/generate-message/index.ts b/apps/open-swe/src/graphs/planner/nodes/generate-message/index.ts index c46ed52b..2a9e1ace 100644 --- a/apps/open-swe/src/graphs/planner/nodes/generate-message/index.ts +++ b/apps/open-swe/src/graphs/planner/nodes/generate-message/index.ts @@ -24,7 +24,7 @@ import { SYSTEM_PROMPT } from "./prompt.js"; import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { getMissingMessages } from "../../../../utils/github/issue-messages.js"; import { getPlansFromIssue } from "../../../../utils/github/issue-task.js"; -import { createSearchTool } from "../../../../tools/search.js"; +import { createGrepTool } from "../../../../tools/grep.js"; import { formatCustomRulesPrompt } from "../../../../utils/custom-rules.js"; import { createScratchpadTool } from "../../../../tools/scratchpad.js"; import { getMcpTools } from "../../../../utils/mcp-client.js"; @@ -35,6 +35,7 @@ import { convertMessagesToCacheControlledMessages, trackCachePerformance, } from "../../../../utils/caching.js"; +import { createViewTool } from "../../../../tools/builtin-tools/view.js"; const logger = createLogger(LogLevel.INFO, "GeneratePlanningMessageNode"); @@ -70,18 +71,19 @@ export async function generateAction( state: PlannerGraphState, config: GraphConfig, ): Promise { - const model = await loadModel(config, Task.PROGRAMMER); + const model = await loadModel(config, Task.PLANNER); const modelManager = getModelManager(); - const modelName = modelManager.getModelNameForTask(config, Task.PROGRAMMER); + const modelName = modelManager.getModelNameForTask(config, Task.PLANNER); const modelSupportsParallelToolCallsParam = supportsParallelToolCallsParam( config, - Task.PROGRAMMER, + Task.PLANNER, ); const mcpTools = await getMcpTools(config); const tools = [ - createSearchTool(state), + createGrepTool(state), createShellTool(state), + createViewTool(state), createScratchpadTool( "when generating a final plan, after all context gathering is complete", ), diff --git a/apps/open-swe/src/graphs/planner/nodes/generate-message/prompt.ts b/apps/open-swe/src/graphs/planner/nodes/generate-message/prompt.ts index 84a4acbc..9ab2f601 100644 --- a/apps/open-swe/src/graphs/planner/nodes/generate-message/prompt.ts +++ b/apps/open-swe/src/graphs/planner/nodes/generate-message/prompt.ts @@ -19,11 +19,11 @@ Your sole objective in this phase is to gather comprehensive context about the c - You do not want to have to generate tasks such as 'Locate the XYZ file', 'Examine the structure of the codebase', or 'Do X if Y is true, otherwise to Z'. - To ensure the above does not happen, you should be thorough in your context gathering. Always gather enough context to cover all edge cases, and prevent unclear instructions. 4. Leverage efficient search tools: - - Use \`search\` tool for all file searches. The \`search\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns. - - It's significantly faster results than alternatives like grep or ls -R. + - Use \`grep\` tool for all file searches. The \`grep\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns. + - It wraps the \`ripgrep\` command, which is significantly faster than alternatives like \`grep\` or \`ls -R\`. + - IMPORTANT: Never run \`grep\` via the \`shell\` tool. You should NEVER run \`grep\` commands via the \`shell\` tool as the same functionality is better provided by \`grep\` tool. - When searching for specific file types, use glob patterns - The query field supports both basic strings, and regex - - Always use the \`search\` tools instead calling \`grep\` via the \`shell\` tool. You should NEVER call \`grep\` as the same functionality is better provided by \`search\`. - If the user passes a URL, you should use the \`get_url_content\` tool to fetch the contents of the URL. - You should only use this tool to fetch the contents of a URL the user has provided, or that you've discovered during your context searching, which you believe is vital to gathering context for the user's request. 5. Format shell commands precisely: Ensure all shell commands include proper quoting and escaping. Well-formatted commands prevent errors and provide reliable results. @@ -33,8 +33,49 @@ Your sole objective in this phase is to gather comprehensive context about the c 8. Only search for what is necessary: Your goal is to gather the minimum amount of context necessary to generate a plan. You should not gather context or perform searches that are not necessary to generate a plan. - You will always be able to gather more context after the planning phase, so ensure that the actions you perform in this planning phase are only the most necessary and targeted actions to gather context. - Avoid rabbit holes for gathering context. You should always first consider whether or not the action you're about to take is necessary to generate a plan for the user's request. If it is not, do not take it. + 9. Try to maintain your current working directory throughout the session by using absolute paths and avoiding usage of cd. You may use cd if the User explicitly requests it. + + ### Grep search tool + - Use the \`grep\` tool for all file searches. The \`grep\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns. + - It accepts a query string, or regex to search for. + - It can search for specific file types using glob patterns. + - Returns a list of results, including file paths and line numbers + - It wraps the \`ripgrep\` command, which is significantly faster than alternatives like \`grep\` or \`ls -R\`. + - IMPORTANT: Never run \`grep\` via the \`shell\` tool. You should NEVER run \`grep\` commands via the \`shell\` tool as the same functionality is better provided by \`grep\` tool. + + ### Shell tool + The \`shell\` tool allows Claude to execute shell commands. + Parameters: + - \`command\`: The shell command to execute. Accepts a list of strings which are joined with spaces to form the command to execute. + - \`workdir\` (optional): The working directory for the command. Defaults to the root of the repository. + - \`timeout\` (optional): The timeout for the command in seconds. Defaults to 60 seconds. + + ### View file tool + The \`view\` tool allows Claude to examine the contents of a file or list the contents of a directory. It can read the entire file or a specific range of lines. + Parameters: + - \`command\`: Must be “view” + - \`path\`: The path to the file or directory to view + - \`view_range\` (optional): An array of two integers specifying the start and end line numbers to view. Line numbers are 1-indexed, and -1 for the end line means read to the end of the file. This parameter only applies when viewing files, not directories. + + ### Scratchpad tool + The \`scratchpad\` tool allows Claude to write to a scratchpad. This is used for writing down findings, and other context which will be useful for the final review. + Parameters: + - \`scratchpad\`: A list of strings containing the text to write to the scratchpad. + + ### Get URL content tool + The \`get_url_content\` tool allows Claude to fetch the contents of a URL. If the total character count of the URL contents exceeds the limit, the \`get_url_content\` tool will return a summarized version of the contents. + Parameters: + - \`url\`: The URL to fetch the contents of + + ### Search document for tool + The \`search_document_for\` tool allows Claude to search for specific content within a document/url contents. + Parameters: + - \`url\`: The URL to fetch the contents of + - \`query\`: The query to search for within the document. This should be a natural language query. The query will be passed to a separate LLM and prompted to extract context from the document which answers this query. + + {CURRENT_WORKING_DIRECTORY} Already cloned and accessible in the current directory diff --git a/apps/open-swe/src/graphs/planner/nodes/generate-plan/index.ts b/apps/open-swe/src/graphs/planner/nodes/generate-plan/index.ts index ee9c81e0..cb9748c0 100644 --- a/apps/open-swe/src/graphs/planner/nodes/generate-plan/index.ts +++ b/apps/open-swe/src/graphs/planner/nodes/generate-plan/index.ts @@ -54,12 +54,12 @@ export async function generatePlan( state: PlannerGraphState, config: GraphConfig, ): Promise { - const model = await loadModel(config, Task.PROGRAMMER); + const model = await loadModel(config, Task.PLANNER); const modelManager = getModelManager(); - const modelName = modelManager.getModelNameForTask(config, Task.PROGRAMMER); + const modelName = modelManager.getModelNameForTask(config, Task.PLANNER); const modelSupportsParallelToolCallsParam = supportsParallelToolCallsParam( config, - Task.SUMMARIZER, + Task.PLANNER, ); const sessionPlanTool = createSessionPlanToolFields(); const modelWithTools = model.bindTools([sessionPlanTool], { diff --git a/apps/open-swe/src/graphs/planner/nodes/rewrite-plan.ts b/apps/open-swe/src/graphs/planner/nodes/rewrite-plan.ts index c98f2766..d9ae0f4a 100644 --- a/apps/open-swe/src/graphs/planner/nodes/rewrite-plan.ts +++ b/apps/open-swe/src/graphs/planner/nodes/rewrite-plan.ts @@ -265,10 +265,10 @@ export async function rewritePlan( throw new Error("No plan change request found."); } - const model = await loadModel(config, Task.PROGRAMMER); + const model = await loadModel(config, Task.PLANNER); const modelSupportsParallelToolCallsParam = supportsParallelToolCallsParam( config, - Task.PROGRAMMER, + Task.PLANNER, ); const tasksToModify = await identifyTasksToModify( state, diff --git a/apps/open-swe/src/graphs/planner/nodes/take-action.ts b/apps/open-swe/src/graphs/planner/nodes/take-action.ts index 767fe268..fc68ced7 100644 --- a/apps/open-swe/src/graphs/planner/nodes/take-action.ts +++ b/apps/open-swe/src/graphs/planner/nodes/take-action.ts @@ -20,7 +20,7 @@ import { safeBadArgsError, } from "../../../utils/zod-to-string.js"; -import { createSearchTool } from "../../../tools/search.js"; +import { createGrepTool } from "../../../tools/grep.js"; import { getChangedFilesStatus, stashAndClearChanges, @@ -34,6 +34,7 @@ import { Command } from "@langchain/langgraph"; import { filterHiddenMessages } from "../../../utils/message/filter-hidden.js"; import { DO_NOT_RENDER_ID_PREFIX } from "@open-swe/shared/constants"; import { processToolCallContent } from "../../../utils/tool-output-processing.js"; +import { createViewTool } from "../../../tools/builtin-tools/view.js"; const logger = createLogger(LogLevel.INFO, "TakeAction"); @@ -48,8 +49,9 @@ export async function takeActions( throw new Error("Last message is not an AI message with tool calls."); } + const viewTool = createViewTool(state); const shellTool = createShellTool(state); - const searchTool = createSearchTool(state); + const searchTool = createGrepTool(state); const scratchpadTool = createScratchpadTool(""); const getURLContentTool = createGetURLContentTool(state); const searchDocumentForTool = createSearchDocumentForTool(state, config); @@ -62,6 +64,7 @@ export async function takeActions( ]; const allTools = [ + viewTool, shellTool, searchTool, scratchpadTool, diff --git a/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts b/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts index 5f2aa788..d1e6988e 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts @@ -29,12 +29,13 @@ import { DEPENDENCIES_INSTALLED_PROMPT, DEPENDENCIES_NOT_INSTALLED_PROMPT, DYNAMIC_SYSTEM_PROMPT, + STATIC_ANTHROPIC_SYSTEM_INSTRUCTIONS, STATIC_SYSTEM_INSTRUCTIONS, } from "./prompt.js"; import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { getMissingMessages } from "../../../../utils/github/issue-messages.js"; import { getPlansFromIssue } from "../../../../utils/github/issue-task.js"; -import { createSearchTool } from "../../../../tools/search.js"; +import { createGrepTool } from "../../../../tools/grep.js"; import { createInstallDependenciesTool } from "../../../../tools/install-dependencies.js"; import { formatCustomRulesPrompt } from "../../../../utils/custom-rules.js"; import { getMcpTools } from "../../../../utils/mcp-client.js"; @@ -75,21 +76,32 @@ const formatDynamicContextPrompt = (state: GraphState) => { ); }; -const formatStaticInstructionsPrompt = (state: GraphState) => { - return STATIC_SYSTEM_INSTRUCTIONS.replaceAll( - "{REPO_DIRECTORY}", - getRepoAbsolutePath(state.targetRepository), - ).replaceAll("{CUSTOM_RULES}", formatCustomRulesPrompt(state.customRules)); +const formatStaticInstructionsPrompt = ( + state: GraphState, + isAnthropicModel: boolean, +) => { + return isAnthropicModel + ? STATIC_ANTHROPIC_SYSTEM_INSTRUCTIONS + : STATIC_SYSTEM_INSTRUCTIONS.replaceAll( + "{REPO_DIRECTORY}", + getRepoAbsolutePath(state.targetRepository), + ).replaceAll( + "{CUSTOM_RULES}", + formatCustomRulesPrompt(state.customRules), + ); }; -const formatCacheablePrompt = (state: GraphState): CacheablePromptSegment[] => { +const formatCacheablePrompt = ( + state: GraphState, + isAnthropicModel: boolean, +): CacheablePromptSegment[] => { const codeReview = getCodeReviewFields(state.internalMessages); const segments: CacheablePromptSegment[] = [ // Cache Breakpoint 2: Static Instructions { type: "text", - text: formatStaticInstructionsPrompt(state), + text: formatStaticInstructionsPrompt(state, isAnthropicModel), cache_control: { type: "ephemeral" }, }, @@ -150,11 +162,10 @@ export async function generateAction( ); const mcpTools = await getMcpTools(config); const markTaskCompletedTool = createMarkTaskCompletedToolFields(); - - const tools = [ - createSearchTool(state), + const isAnthropicModel = modelName.includes("claude-"); + const sharedTools = [ + createGrepTool(state), createShellTool(state), - createApplyPatchTool(state), createRequestHumanHelpToolFields(), createUpdatePlanToolFields(), createGetURLContentTool(state), @@ -163,6 +174,18 @@ export async function generateAction( createSearchDocumentForTool(state, config), ...mcpTools, ]; + const anthropicModelTools = [ + { + type: "text_editor_20250429", + name: "str_replace_based_edit_tool", + }, + ]; + const nonAnthropicModelTools = [createApplyPatchTool(state)]; + + const tools = [ + ...sharedTools, + ...(isAnthropicModel ? anthropicModelTools : nonAnthropicModelTools), + ]; logger.info( `MCP tools added to Programmer: ${mcpTools.map((t) => t.name).join(", ")}`, ); @@ -199,10 +222,13 @@ export async function generateAction( const response = await modelWithTools.invoke([ { role: "system", - content: formatCacheablePrompt({ - ...state, - taskPlan: latestTaskPlan ?? state.taskPlan, - }), + content: formatCacheablePrompt( + { + ...state, + taskPlan: latestTaskPlan ?? state.taskPlan, + }, + isAnthropicModel, + ), }, ...inputMessagesWithCache, formatSpecificPlanPrompt(state), diff --git a/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts b/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts index fd4cd81f..9a9bc460 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts @@ -1,3 +1,183 @@ +export const STATIC_ANTHROPIC_SYSTEM_INSTRUCTIONS = ` +You are a terminal-based agentic coding assistant built by LangChain. You wrap LLM models to enable natural language interaction with local codebases. You are precise, safe, and helpful. + + + + You are currently executing a specific task from a pre-generated plan. You have access to: + - Project context and files + - Shell commands and code editing tools + - A sandboxed, git-backed workspace with rollback support + + + + + - Persistence: Keep working until the current task is completely resolved. Only terminate when you are certain the task is complete. + - Accuracy: Never guess or make up information. Always use tools to gather accurate data about files and codebase structure. + - Planning: Leverage the plan context and task summaries heavily - they contain critical information about completed work and the overall strategy. + + + + - You are executing a task from the plan. + - Previous completed tasks and their summaries contain crucial context - always review them first + - Condensed context messages in conversation history summarize previous work - read these to avoid duplication + - The plan generation summary provides important codebase insights + - After some tasks are completed, you may be provided with a code review and additional tasks. Ensure you inspect the code review (if present) and new tasks to ensure the work you're doing satisfies the user's request. + - Only modify the code outlined in the current task. You should always AVOID modifying code which is unrelated to the current tasks. + + + + {REPO_DIRECTORY} + {REPO_DIRECTORY} + - All changes are auto-committed - no manual commits needed, and you should never create backup files. + - Work only within the existing Git repository + - Use \`install_dependencies\` to install dependencies (skip if installation fails). IMPORTANT: You should only call this tool if you're executing a task which REQUIRES installing dependencies. Keep in mind that not all tasks will require installing dependencies. + + + + ### Grep search tool + - Use the \`grep\` tool for all file searches. The \`grep\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns. + - It accepts a query string, or regex to search for. + - It can search for specific file types using glob patterns. + - Returns a list of results, including file paths and line numbers + - It wraps the \`ripgrep\` command, which is significantly faster than alternatives like \`grep\` or \`ls -R\`. + - IMPORTANT: Never run \`grep\` via the \`shell\` tool. You should NEVER run \`grep\` commands via the \`shell\` tool as the same functionality is better provided by \`grep\` tool. + + ### View file command + The \`view\` command allows Claude to examine the contents of a file or list the contents of a directory. It can read the entire file or a specific range of lines. + Parameters: + - \`command\`: Must be “view” + - \`path\`: The path to the file or directory to view + - \`view_range\` (optional): An array of two integers specifying the start and end line numbers to view. Line numbers are 1-indexed, and -1 for the end line means read to the end of the file. This parameter only applies when viewing files, not directories. + + ### Str replace command + The \`str_replace\` command allows Claude to replace a specific string in a file with a new string. This is used for making precise edits. + Parameters: + - \`command\`: Must be “str_replace” + - \`path\`: The path to the file to modify + - \`old_str\`: The text to replace (must match exactly, including whitespace and indentation) + - \`new_str\`: The new text to insert in place of the old text + + ### Create command + The \`create\` command allows Claude to create a new file with specified content. + Parameters: + - \`command\`: Must be “create” + - \`path\`: The path where the new file should be created + - \`file_text\`: The content to write to the new file + + ### Insert command + The \`insert\` command allows Claude to insert text at a specific location in a file. + Parameters: + - \`command\`: Must be “insert” + - \`path\`: The path to the file to modify + - \`insert_line\`: The line number after which to insert the text (0 for beginning of file) + - \`new_str\`: The text to insert + + ### Shell tool + The \`shell\` tool allows Claude to execute shell commands. + Parameters: + - \`command\`: The shell command to execute. Accepts a list of strings which are joined with spaces to form the command to execute. + - \`workdir\` (optional): The working directory for the command. Defaults to the root of the repository. + - \`timeout\` (optional): The timeout for the command in seconds. Defaults to 60 seconds. + + ### Request human help tool + The \`request_human_help\` tool allows Claude to request human help if all possible tools/actions have been exhausted, and Claude is unable to complete the task. + Parameters: + - \`help_request\`: The message to send to the human + + ### Update plan tool + The \`update_plan\` tool allows Claude to update the plan if it notices issues with the current plan which requires modifications. + Parameters: + - \`update_plan_reasoning\`: The reasoning for why you are updating the plan. This should include context which will be useful when actually updating the plan, such as what plan items to update, edit, or remove, along with any other context that would be useful when updating the plan. + + ### Get URL content tool + The \`get_url_content\` tool allows Claude to fetch the contents of a URL. If the total character count of the URL contents exceeds the limit, the \`get_url_content\` tool will return a summarized version of the contents. + Parameters: + - \`url\`: The URL to fetch the contents of + + ### Search document for tool + The \`search_document_for\` tool allows Claude to search for specific content within a document/url contents. + Parameters: + - \`url\`: The URL to fetch the contents of + - \`query\`: The query to search for within the document. This should be a natural language query. The query will be passed to a separate LLM and prompted to extract context from the document which answers this query. + + ### Install dependencies tool + The \`install_dependencies\` tool allows Claude to install dependencies for a project. This should only be called if dependencies have not been installed yet. + Parameters: + - \`command\`: The dependencies install command to execute. Ensure this command is properly formatted, using the correct package manager for this project, and the correct command to install dependencies. It accepts a list of strings which are joined with spaces to form the command to execute. + - \`workdir\` (optional): The working directory for the command. Defaults to the root of the repository. + - \`timeout\` (optional): The timeout for the command in seconds. Defaults to 60 seconds. + + ### Mark task completed tool + The \`mark_task_completed\` tool allows Claude to mark a task as completed. + Parameters: + - \`completed_task_summary\`: A summary of the completed task. This summary should include high level context about the actions you took to complete the task, and any other context which would be useful to another developer reviewing the actions you took. Ensure this is properly formatted using markdown. + + + + - Search: Use the \`grep\` tool for all file searches. The \`grep\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns. + - When searching for specific file types, use glob patterns + - The query field supports both basic strings, and regex + - Dependencies: Use the correct package manager; skip if installation fails + - Use the \`install_dependencies\` tool to install dependencies (skip if installation fails). IMPORTANT: You should only call this tool if you're executing a task which REQUIRES installing dependencies. Keep in mind that not all tasks will require installing dependencies. + - Pre-commit: Run \`pre-commit run --files ...\` if .pre-commit-config.yaml exists + - History: Use \`git log\` and \`git blame\` for additional context when needed + - Parallel Tool Calling: You're allowed, and encouraged to call multiple tools at once, as long as they do not conflict, or depend on each other. + - URL Content: Use the \`get_url_content\` tool to fetch the contents of a URL. You should only use this tool to fetch the contents of a URL the user has provided, or that you've discovered during your context searching, which you believe is vital to gathering context for the user's request. + - Scripts may require dependencies to be installed: Remember that sometimes scripts may require dependencies to be installed before they can be run. + - Always ensure you've installed dependencies before running a script which might require them. + + + + - When modifying files: + - Read files before modifying them + - Fix root causes, not symptoms + - Maintain existing code style + - Update documentation as needed + - Remove unnecessary inline comments after completion + - Comments should only be included if a core maintainer of the codebase would not be able to understand the code without them (this means most of the time, you should not include comments) + - Never add copyright/license headers unless requested + - Ignore unrelated bugs or broken tests + - Write concise and clear code. Do not write overly verbose code + - Any tests written should always be executed after creating them to ensure they pass. + - If you've created a new test, ensure the plan has an explicit step to run this new test. If the plan does not include a step to run the tests, ensure you call the \`update_plan\` tool to add a step to run the tests. + - When running a test, ensure you include the proper flags/environment variables to exclude colors/text formatting. This can cause the output to be unreadable. For example, when running Jest tests you pass the \`--no-colors\` flag. In PyTest you set the \`NO_COLOR\` environment variable (prefix the command with \`export NO_COLOR=1\`) + - Only install trusted, well-maintained packages. If installing a new dependency which is not explicitly requested by the user, ensure it is a well-maintained, and widely used package. + - Ensure package manager files are updated to include the new dependency. + - If a command you run fails (e.g. a test, build, lint, etc.), and you make changes to fix the issue, ensure you always re-run the command after making the changes to ensure the fix was successful. + - IMPORTANT: You are NEVER allowed to create backup files. All changes in the codebase are tracked by git, so never create file copies, or backups. + + + + - For coding tasks: Focus on implementation and provide brief summaries + - When generating text which will be shown to the user, ensure you always use markdown formatting to make the text easy to read and understand. + - Avoid using title tags in the markdown (e.g. # or ##) as this will clog up the output space. + - You should however use other valid markdown syntax, and smaller heading tags (e.g. ### or ####), bold/italic text, code blocks and inline code, and so on, to make the text easy to read and understand. + + + + request_human_help + Use only after exhausting all attempts to gather context + + update_plan + Use this tool to add or remove tasks from the plan, or to update the plan in any other way + + + + - When you believe you've completed a task, you may call the \`mark_task_completed\` tool to mark the task as complete. + - The \`mark_task_completed\` tool should NEVER be called in parallel with any other tool calls. Ensure it's the only tool you're calling in this message, if you do determine the task is completed. + - Carefully read over the actions you've taken, and the current task (listed below) to ensure the task is complete. You want to avoid prematurely marking a task as complete. + - If the current task involves fixing an issue, such as a failing test, a broken build, etc., you must validate the issue is ACTUALLY fixed before marking it as complete. + - To verify a fix, ensure you run the test, build, or other command first to validate the fix. + - If you do not believe the task is complete, you do not need to call the \`mark_task_completed\` tool. You can continue working on the task, until you determine it is complete. + + + + + + {CUSTOM_RULES} + +`; + export const STATIC_SYSTEM_INSTRUCTIONS = ` You are a terminal-based agentic coding assistant built by LangChain. You wrap LLM models to enable natural language interaction with local codebases. You are precise, safe, and helpful. @@ -76,6 +256,9 @@ You are a terminal-based agentic coding assistant built by LangChain. You wrap L - For coding tasks: Focus on implementation and provide brief summaries + - When generating text which will be shown to the user, ensure you always use markdown formatting to make the text easy to read and understand. + - Avoid using title tags in the markdown (e.g. # or ##) as this will clog up the output space. + - You should however use other valid markdown syntax, and smaller heading tags (e.g. ### or ####), bold/italic text, code blocks and inline code, and so on, to make the text easy to read and understand. diff --git a/apps/open-swe/src/graphs/programmer/nodes/take-action.ts b/apps/open-swe/src/graphs/programmer/nodes/take-action.ts index 6ce641a7..d40ebaf5 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/take-action.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/take-action.ts @@ -4,6 +4,7 @@ import { createLogger, LogLevel } from "../../../utils/logger.js"; import { createApplyPatchTool, createGetURLContentTool, + createTextEditorTool, createShellTool, createSearchDocumentForTool, } from "../../../tools/index.js"; @@ -30,7 +31,7 @@ import { } from "../../../utils/tree.js"; import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { createInstallDependenciesTool } from "../../../tools/install-dependencies.js"; -import { createSearchTool } from "../../../tools/search.js"; +import { createGrepTool } from "../../../tools/grep.js"; import { getMcpTools } from "../../../utils/mcp-client.js"; import { shouldDiagnoseError } from "../../../utils/tool-message-error.js"; import { getGitHubTokensFromConfig } from "../../../utils/github-tokens.js"; @@ -52,7 +53,8 @@ export async function takeAction( const applyPatchTool = createApplyPatchTool(state); const shellTool = createShellTool(state); - const searchTool = createSearchTool(state); + const searchTool = createGrepTool(state); + const textEditorTool = createTextEditorTool(state); const installDependenciesTool = createInstallDependenciesTool(state); const getURLContentTool = createGetURLContentTool(state); const searchDocumentForTool = createSearchDocumentForTool(state, config); @@ -67,6 +69,7 @@ export async function takeAction( const allTools = [ shellTool, searchTool, + textEditorTool, installDependenciesTool, applyPatchTool, getURLContentTool, diff --git a/apps/open-swe/src/graphs/reviewer/nodes/final-review.ts b/apps/open-swe/src/graphs/reviewer/nodes/final-review.ts index dc414d4c..2b73867e 100644 --- a/apps/open-swe/src/graphs/reviewer/nodes/final-review.ts +++ b/apps/open-swe/src/graphs/reviewer/nodes/final-review.ts @@ -117,12 +117,12 @@ export async function finalReview( const completedTool = createCodeReviewMarkTaskCompletedFields(); const incompleteTool = createCodeReviewMarkTaskNotCompleteFields(); const tools = [completedTool, incompleteTool]; - const model = await loadModel(config, Task.PROGRAMMER); + const model = await loadModel(config, Task.REVIEWER); const modelManager = getModelManager(); - const modelName = modelManager.getModelNameForTask(config, Task.PROGRAMMER); + const modelName = modelManager.getModelNameForTask(config, Task.REVIEWER); const modelSupportsParallelToolCallsParam = supportsParallelToolCallsParam( config, - Task.PROGRAMMER, + Task.REVIEWER, ); const modelWithTools = model.bindTools(tools, { tool_choice: "any", diff --git a/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/index.ts b/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/index.ts index 496836cd..e30d3801 100644 --- a/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/index.ts +++ b/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/index.ts @@ -14,7 +14,7 @@ import { getMessageContentString } from "@open-swe/shared/messages"; import { PREVIOUS_REVIEW_PROMPT, SYSTEM_PROMPT } from "./prompt.js"; import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { - createSearchTool, + createGrepTool, createShellTool, createInstallDependenciesTool, } from "../../../../tools/index.js"; @@ -34,6 +34,7 @@ import { trackCachePerformance, } from "../../../../utils/caching.js"; import { createScratchpadTool } from "../../../../tools/scratchpad.js"; +import { createViewTool } from "../../../../tools/builtin-tools/view.js"; const logger = createLogger(LogLevel.INFO, "GenerateReviewActionsNode"); @@ -112,16 +113,17 @@ export async function generateReviewActions( state: ReviewerGraphState, config: GraphConfig, ): Promise { - const model = await loadModel(config, Task.PROGRAMMER); + const model = await loadModel(config, Task.REVIEWER); const modelManager = getModelManager(); - const modelName = modelManager.getModelNameForTask(config, Task.PROGRAMMER); + const modelName = modelManager.getModelNameForTask(config, Task.REVIEWER); const modelSupportsParallelToolCallsParam = supportsParallelToolCallsParam( config, - Task.PROGRAMMER, + Task.REVIEWER, ); const tools = [ - createSearchTool(state), + createGrepTool(state), createShellTool(state), + createViewTool(state), createInstallDependenciesTool(state), createScratchpadTool( "when generating a final review, after all context gathering and reviewing is complete", diff --git a/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/prompt.ts b/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/prompt.ts index 77446828..b623d689 100644 --- a/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/prompt.ts +++ b/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/prompt.ts @@ -99,6 +99,42 @@ By reviewing these actions, and comparing them to the plan and original user req Only gather context right now in order to inform your final review, and to provide any additional steps to take after the review. + + ### Grep search tool + - Use the \`grep\` tool for all file searches. The \`grep\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns. + - It accepts a query string, or regex to search for. + - It can search for specific file types using glob patterns. + - Returns a list of results, including file paths and line numbers + - It wraps the \`ripgrep\` command, which is significantly faster than alternatives like \`grep\` or \`ls -R\`. + - IMPORTANT: Never run \`grep\` via the \`shell\` tool. You should NEVER run \`grep\` commands via the \`shell\` tool as the same functionality is better provided by \`grep\` tool. + + ### Shell tool + The \`shell\` tool allows Claude to execute shell commands. + Parameters: + - \`command\`: The shell command to execute. Accepts a list of strings which are joined with spaces to form the command to execute. + - \`workdir\` (optional): The working directory for the command. Defaults to the root of the repository. + - \`timeout\` (optional): The timeout for the command in seconds. Defaults to 60 seconds. + + ### View file tool + The \`view\` tool allows Claude to examine the contents of a file or list the contents of a directory. It can read the entire file or a specific range of lines. + Parameters: + - \`command\`: Must be “view” + - \`path\`: The path to the file or directory to view + - \`view_range\` (optional): An array of two integers specifying the start and end line numbers to view. Line numbers are 1-indexed, and -1 for the end line means read to the end of the file. This parameter only applies when viewing files, not directories. + + ### Install dependencies tool + The \`install_dependencies\` tool allows Claude to install dependencies for a project. This should only be called if dependencies have not been installed yet. + Parameters: + - \`command\`: The dependencies install command to execute. Ensure this command is properly formatted, using the correct package manager for this project, and the correct command to install dependencies. It accepts a list of strings which are joined with spaces to form the command to execute. + - \`workdir\` (optional): The working directory for the command. Defaults to the root of the repository. + - \`timeout\` (optional): The timeout for the command in seconds. Defaults to 60 seconds. + + ### Scratchpad tool + The \`scratchpad\` tool allows Claude to write to a scratchpad. This is used for writing down findings, and other context which will be useful for the final review. + Parameters: + - \`scratchpad\`: A list of strings containing the text to write to the scratchpad. + + {CURRENT_WORKING_DIRECTORY} Already cloned and accessible in the current directory diff --git a/apps/open-swe/src/graphs/reviewer/nodes/take-review-action.ts b/apps/open-swe/src/graphs/reviewer/nodes/take-review-action.ts index d0bccc49..94e410b4 100644 --- a/apps/open-swe/src/graphs/reviewer/nodes/take-review-action.ts +++ b/apps/open-swe/src/graphs/reviewer/nodes/take-review-action.ts @@ -17,7 +17,7 @@ import { createLogger, LogLevel } from "../../../utils/logger.js"; import { zodSchemaToString } from "../../../utils/zod-to-string.js"; import { formatBadArgsError } from "../../../utils/zod-to-string.js"; import { truncateOutput } from "../../../utils/truncate-outputs.js"; -import { createSearchTool } from "../../../tools/search.js"; +import { createGrepTool } from "../../../tools/grep.js"; import { checkoutBranchAndCommit, getChangedFilesStatus, @@ -31,6 +31,7 @@ import { getGitHubTokensFromConfig } from "../../../utils/github-tokens.js"; import { createScratchpadTool } from "../../../tools/scratchpad.js"; import { getActiveTask } from "@open-swe/shared/open-swe/tasks"; import { createPullRequestToolCallMessage } from "../../../utils/message/create-pr-message.js"; +import { createViewTool } from "../../../tools/builtin-tools/view.js"; const logger = createLogger(LogLevel.INFO, "TakeReviewAction"); @@ -46,12 +47,14 @@ export async function takeReviewerActions( } const shellTool = createShellTool(state); - const searchTool = createSearchTool(state); + const searchTool = createGrepTool(state); + const viewTool = createViewTool(state); const installDependenciesTool = createInstallDependenciesTool(state); const scratchpadTool = createScratchpadTool(""); const allTools = [ shellTool, searchTool, + viewTool, installDependenciesTool, scratchpadTool, ]; diff --git a/apps/open-swe/src/tools/builtin-tools/handlers.ts b/apps/open-swe/src/tools/builtin-tools/handlers.ts new file mode 100644 index 00000000..51d4ff72 --- /dev/null +++ b/apps/open-swe/src/tools/builtin-tools/handlers.ts @@ -0,0 +1,196 @@ +import { Sandbox } from "@daytonaio/sdk"; +import { readFile, writeFile } from "../../utils/read-write.js"; +import { getSandboxErrorFields } from "../../utils/sandbox-error-fields.js"; + +export async function handleViewCommand( + sandbox: Sandbox, + path: string, + workDir: string, + viewRange?: [number, number], +): Promise { + try { + // Check if path is a directory + const statOutput = await sandbox.process.executeCommand( + `stat -c %F "${path}"`, + workDir, + ); + + if (statOutput.exitCode === 0 && statOutput.result?.includes("directory")) { + // List directory contents + const lsOutput = await sandbox.process.executeCommand( + `ls -la "${path}"`, + workDir, + ); + + if (lsOutput.exitCode !== 0) { + throw new Error(`Failed to list directory: ${lsOutput.result}`); + } + + return `Directory listing for ${path}:\n${lsOutput.result}`; + } + + // Read file contents + const { success, output } = await readFile({ + sandbox, + filePath: path, + workDir, + }); + + if (!success) { + throw new Error(output); + } + + // Apply view range if specified + if (viewRange) { + const lines = output.split("\n"); + const [start, end] = viewRange; + const startIndex = Math.max(0, start - 1); // Convert to 0-indexed + const endIndex = end === -1 ? lines.length : Math.min(lines.length, end); + + const selectedLines = lines.slice(startIndex, endIndex); + const numberedLines = selectedLines.map( + (line, index) => `${startIndex + index + 1}: ${line}`, + ); + + return numberedLines.join("\n"); + } + + // Return full file with line numbers + const lines = output.split("\n"); + const numberedLines = lines.map((line, index) => `${index + 1}: ${line}`); + return numberedLines.join("\n"); + } catch (e) { + const errorFields = getSandboxErrorFields(e); + if (errorFields) { + throw new Error(`Failed to view ${path}: ${errorFields.result}`); + } + throw new Error( + `Failed to view ${path}: ${e instanceof Error ? e.message : String(e)}`, + ); + } +} + +export async function handleStrReplaceCommand( + sandbox: Sandbox, + path: string, + workDir: string, + oldStr: string, + newStr: string, +): Promise { + const { success: readSuccess, output: fileContent } = await readFile({ + sandbox, + filePath: path, + workDir, + }); + + if (!readSuccess) { + throw new Error(`Failed to read file ${path}: ${fileContent}`); + } + + // Count occurrences of old string + const occurrences = ( + fileContent.match( + new RegExp(oldStr.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"), "g"), + ) || [] + ).length; + + if (occurrences === 0) { + throw new Error( + `No match found for replacement text in ${path}. Please check your text and try again.`, + ); + } + + if (occurrences > 1) { + throw new Error( + `Found ${occurrences} matches for replacement text in ${path}. Please provide more context to make a unique match.`, + ); + } + + // Perform replacement + const newContent = fileContent.replace(oldStr, newStr); + + const { success: writeSuccess, output: writeOutput } = await writeFile({ + sandbox, + filePath: path, + content: newContent, + workDir, + }); + + if (!writeSuccess) { + throw new Error(`Failed to write file ${path}: ${writeOutput}`); + } + + return `Successfully replaced text in ${path} at exactly one location.`; +} + +export async function handleCreateCommand( + sandbox: Sandbox, + path: string, + workDir: string, + fileText: string, +): Promise { + // Check if file already exists + const { success: readSuccess } = await readFile({ + sandbox, + filePath: path, + workDir, + }); + + if (readSuccess) { + throw new Error( + `File ${path} already exists. Use str_replace to modify existing files.`, + ); + } + + const { success: writeSuccess, output: writeOutput } = await writeFile({ + sandbox, + filePath: path, + content: fileText, + workDir, + }); + + if (!writeSuccess) { + throw new Error(`Failed to create file ${path}: ${writeOutput}`); + } + + return `Successfully created file ${path}.`; +} + +export async function handleInsertCommand( + sandbox: Sandbox, + path: string, + workDir: string, + insertLine: number, + newStr: string, +): Promise { + const { success: readSuccess, output: fileContent } = await readFile({ + sandbox, + filePath: path, + workDir, + }); + + if (!readSuccess) { + throw new Error(`Failed to read file ${path}: ${fileContent}`); + } + + const lines = fileContent.split("\n"); + + // Insert at specified line (0 = beginning, 1 = after first line, etc.) + const insertIndex = Math.max(0, Math.min(lines.length, insertLine)); + lines.splice(insertIndex, 0, newStr); + + const newContent = lines.join("\n"); + + const { success: writeSuccess, output: writeOutput } = await writeFile({ + sandbox, + filePath: path, + content: newContent, + workDir, + }); + + if (!writeSuccess) { + throw new Error(`Failed to write file ${path}: ${writeOutput}`); + } + + return `Successfully inserted text in ${path} at line ${insertLine}.`; +} diff --git a/apps/open-swe/src/tools/builtin-tools/text-editor.ts b/apps/open-swe/src/tools/builtin-tools/text-editor.ts new file mode 100644 index 00000000..e6f05944 --- /dev/null +++ b/apps/open-swe/src/tools/builtin-tools/text-editor.ts @@ -0,0 +1,107 @@ +import { tool } from "@langchain/core/tools"; +import { GraphState } from "@open-swe/shared/open-swe/types"; +import { createLogger, LogLevel } from "../../utils/logger.js"; +import { getRepoAbsolutePath } from "@open-swe/shared/git"; +import { getSandboxSessionOrThrow } from "../utils/get-sandbox-id.js"; +import { createTextEditorToolFields } from "@open-swe/shared/open-swe/tools"; +import { + handleViewCommand, + handleStrReplaceCommand, + handleCreateCommand, + handleInsertCommand, +} from "./handlers.js"; + +const logger = createLogger(LogLevel.INFO, "TextEditorTool"); + +export function createTextEditorTool( + state: Pick, +) { + const textEditorTool = tool( + async (input): Promise<{ result: string; status: "success" | "error" }> => { + try { + const sandbox = await getSandboxSessionOrThrow(input); + const workDir = getRepoAbsolutePath(state.targetRepository); + + const { + command, + path, + view_range, + old_str, + new_str, + file_text, + insert_line, + } = input; + + let result: string; + + switch (command) { + case "view": + result = await handleViewCommand( + sandbox, + path, + workDir, + view_range, + ); + break; + case "str_replace": + if (!old_str || new_str === undefined) { + throw new Error( + "str_replace command requires both old_str and new_str parameters", + ); + } + result = await handleStrReplaceCommand( + sandbox, + path, + workDir, + old_str, + new_str, + ); + break; + case "create": + if (!file_text) { + throw new Error("create command requires file_text parameter"); + } + result = await handleCreateCommand( + sandbox, + path, + workDir, + file_text, + ); + break; + case "insert": + if (insert_line === undefined || new_str === undefined) { + throw new Error( + "insert command requires both insert_line and new_str parameters", + ); + } + result = await handleInsertCommand( + sandbox, + path, + workDir, + insert_line, + new_str, + ); + break; + default: + throw new Error(`Unknown command: ${command}`); + } + + logger.info( + `Text editor command '${command}' executed successfully on ${path}`, + ); + return { result, status: "success" }; + } catch (error) { + const errorMessage = + error instanceof Error ? error.message : String(error); + logger.error(`Text editor command failed: ${errorMessage}`); + return { + result: `Error: ${errorMessage}`, + status: "error", + }; + } + }, + createTextEditorToolFields(state.targetRepository), + ); + + return textEditorTool; +} diff --git a/apps/open-swe/src/tools/builtin-tools/view.ts b/apps/open-swe/src/tools/builtin-tools/view.ts new file mode 100644 index 00000000..1485af63 --- /dev/null +++ b/apps/open-swe/src/tools/builtin-tools/view.ts @@ -0,0 +1,48 @@ +import { tool } from "@langchain/core/tools"; +import { GraphState } from "@open-swe/shared/open-swe/types"; +import { createLogger, LogLevel } from "../../utils/logger.js"; +import { getRepoAbsolutePath } from "@open-swe/shared/git"; +import { getSandboxSessionOrThrow } from "../utils/get-sandbox-id.js"; +import { createViewToolFields } from "@open-swe/shared/open-swe/tools"; +import { handleViewCommand } from "./handlers.js"; + +const logger = createLogger(LogLevel.INFO, "ViewTool"); + +export function createViewTool( + state: Pick, +) { + const viewTool = tool( + async (input): Promise<{ result: string; status: "success" | "error" }> => { + try { + const sandbox = await getSandboxSessionOrThrow(input); + const workDir = getRepoAbsolutePath(state.targetRepository); + + const { command, path, view_range } = input; + if (command !== "view") { + throw new Error(`Unknown command: ${command}`); + } + + const result = await handleViewCommand( + sandbox, + path, + workDir, + view_range, + ); + + logger.info(`View command executed successfully on ${path}`); + return { result, status: "success" }; + } catch (error) { + const errorMessage = + error instanceof Error ? error.message : String(error); + logger.error(`View command failed: ${errorMessage}`); + return { + result: `Error: ${errorMessage}`, + status: "error", + }; + } + }, + createViewToolFields(state.targetRepository), + ); + + return viewTool; +} diff --git a/apps/open-swe/src/tools/search.ts b/apps/open-swe/src/tools/grep.ts similarity index 79% rename from apps/open-swe/src/tools/search.ts rename to apps/open-swe/src/tools/grep.ts index e9ad669d..6907f9dc 100644 --- a/apps/open-swe/src/tools/search.ts +++ b/apps/open-swe/src/tools/grep.ts @@ -4,31 +4,31 @@ import { getSandboxErrorFields } from "../utils/sandbox-error-fields.js"; import { createLogger, LogLevel } from "../utils/logger.js"; import { TIMEOUT_SEC } from "@open-swe/shared/constants"; import { - createSearchToolFields, - formatSearchCommand, + createGrepToolFields, + formatGrepCommand, } from "@open-swe/shared/open-swe/tools"; import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { wrapScript } from "../utils/wrap-script.js"; import { getSandboxSessionOrThrow } from "./utils/get-sandbox-id.js"; -const logger = createLogger(LogLevel.INFO, "SearchTool"); +const logger = createLogger(LogLevel.INFO, "GrepTool"); const DEFAULT_ENV = { // Prevents corepack from showing a y/n download prompt which causes the command to hang COREPACK_ENABLE_DOWNLOAD_PROMPT: "0", }; -export function createSearchTool( +export function createGrepTool( state: Pick, ) { - const searchTool = tool( + const grepTool = tool( async (input): Promise<{ result: string; status: "success" | "error" }> => { try { const sandbox = await getSandboxSessionOrThrow(input); const repoRoot = getRepoAbsolutePath(state.targetRepository); - const command = formatSearchCommand(input); - logger.info("Running search command", { + const command = formatGrepCommand(input); + logger.info("Running grep search command", { command: command.join(" "), repoRoot, }); @@ -50,7 +50,7 @@ export function createSearchTool( } else if (response.exitCode > 1) { const errorResult = response.result ?? response.artifacts?.stdout; throw new Error( - `Failed to run search command. Exit code: ${response.exitCode}\nError: ${errorResult}`, + `Failed to run grep search command. Exit code: ${response.exitCode}\nError: ${errorResult}`, ); } @@ -64,15 +64,15 @@ export function createSearchTool( const errorResult = errorFields.result ?? errorFields.artifacts?.stdout; throw new Error( - `Failed to run search command. Exit code: ${errorFields.exitCode}\nError: ${errorResult}`, + `Failed to run grep search command. Exit code: ${errorFields.exitCode}\nError: ${errorResult}`, ); } throw e; } }, - createSearchToolFields(state.targetRepository), + createGrepToolFields(state.targetRepository), ); - return searchTool; + return grepTool; } diff --git a/apps/open-swe/src/tools/index.ts b/apps/open-swe/src/tools/index.ts index 03c4d11a..a9940253 100644 --- a/apps/open-swe/src/tools/index.ts +++ b/apps/open-swe/src/tools/index.ts @@ -1,5 +1,6 @@ export * from "./apply-patch.js"; export * from "./shell.js"; +export * from "./builtin-tools/text-editor.js"; export * from "./url-content.js"; export * from "./search-documents-for/index.js"; export { @@ -7,6 +8,6 @@ export { createSessionPlanToolFields, createRequestHumanHelpToolFields, } from "@open-swe/shared/open-swe/tools"; -export * from "./search.js"; +export * from "./grep.js"; export * from "./install-dependencies.js"; export * from "./planner-notes.js"; diff --git a/apps/open-swe/src/utils/llms/constants.ts b/apps/open-swe/src/utils/llms/constants.ts index af971b9a..e8f7e138 100644 --- a/apps/open-swe/src/utils/llms/constants.ts +++ b/apps/open-swe/src/utils/llms/constants.ts @@ -1,4 +1,9 @@ export enum Task { + /** + * Used for programmer tasks. This includes: writing code, + * generating plans, taking context gathering actions, etc. + */ + PLANNER = "planner", /** * Used for programmer tasks. This includes: writing code, * generating plans, taking context gathering actions, etc. @@ -9,6 +14,11 @@ export enum Task { * routing to different agents. */ ROUTER = "router", + /** + * Used for reviewer tasks. This includes: reviewing code, + * generating plans, taking context gathering actions, etc. + */ + REVIEWER = "reviewer", /** * Used for summarizing tasks. This includes: summarizing * the conversation history, summarizing actions taken during @@ -18,10 +28,18 @@ export enum Task { } export const TASK_TO_CONFIG_DEFAULTS_MAP = { + [Task.PLANNER]: { + modelName: "anthropic:claude-sonnet-4-0", + temperature: 0, + }, [Task.PROGRAMMER]: { modelName: "anthropic:claude-sonnet-4-0", temperature: 0, }, + [Task.REVIEWER]: { + modelName: "anthropic:claude-sonnet-4-0", + temperature: 0, + }, [Task.ROUTER]: { modelName: "anthropic:claude-3-5-haiku-latest", temperature: 0, diff --git a/apps/open-swe/src/utils/llms/model-manager.ts b/apps/open-swe/src/utils/llms/model-manager.ts index cd638673..42b5b928 100644 --- a/apps/open-swe/src/utils/llms/model-manager.ts +++ b/apps/open-swe/src/utils/llms/model-manager.ts @@ -248,12 +248,24 @@ export class ModelManager { task: Task, ): ModelLoadConfig { const taskMap = { + [Task.PLANNER]: { + modelName: + config.configurable?.[`${task}ModelName`] ?? + TASK_TO_CONFIG_DEFAULTS_MAP[task].modelName, + temperature: config.configurable?.[`${task}Temperature`] ?? 0, + }, [Task.PROGRAMMER]: { modelName: config.configurable?.[`${task}ModelName`] ?? TASK_TO_CONFIG_DEFAULTS_MAP[task].modelName, temperature: config.configurable?.[`${task}Temperature`] ?? 0, }, + [Task.REVIEWER]: { + modelName: + config.configurable?.[`${task}ModelName`] ?? + TASK_TO_CONFIG_DEFAULTS_MAP[task].modelName, + temperature: config.configurable?.[`${task}Temperature`] ?? 0, + }, [Task.ROUTER]: { modelName: config.configurable?.[`${task}ModelName`] ?? @@ -304,17 +316,23 @@ export class ModelManager { ): ModelLoadConfig | null { const defaultModels: Record> = { anthropic: { + [Task.PLANNER]: "claude-sonnet-4-0", [Task.PROGRAMMER]: "claude-sonnet-4-0", + [Task.REVIEWER]: "claude-sonnet-4-0", [Task.ROUTER]: "claude-3-5-haiku-latest", [Task.SUMMARIZER]: "claude-sonnet-4-0", }, "google-genai": { + [Task.PLANNER]: "gemini-2.5-flash", [Task.PROGRAMMER]: "gemini-2.5-pro", + [Task.REVIEWER]: "gemini-2.5-flash", [Task.ROUTER]: "gemini-2.5-flash", [Task.SUMMARIZER]: "gemini-2.5-pro", }, openai: { + [Task.PLANNER]: "o3", [Task.PROGRAMMER]: "gpt-4o", + [Task.REVIEWER]: "o3", [Task.ROUTER]: "gpt-4o-mini", [Task.SUMMARIZER]: "gpt-4.1-mini", }, diff --git a/apps/web/src/components/gen-ui/action-step.tsx b/apps/web/src/components/gen-ui/action-step.tsx index 98e71825..ef64e09e 100644 --- a/apps/web/src/components/gen-ui/action-step.tsx +++ b/apps/web/src/components/gen-ui/action-step.tsx @@ -1,6 +1,6 @@ "use client"; -import { useState } from "react"; +import { JSX, useState } from "react"; import { Terminal, FileText, @@ -23,8 +23,9 @@ import { createInstallDependenciesToolFields, createScratchpadFields, createGetURLContentToolFields, - createSearchToolFields, + createGrepToolFields, createSearchDocumentForToolFields, + createTextEditorToolFields, } from "@open-swe/shared/open-swe/tools"; import { z } from "zod"; import { @@ -50,10 +51,12 @@ const scratchpadTool = createScratchpadFields(""); type ScratchpadToolArgs = z.infer; const getURLContentTool = createGetURLContentToolFields(); type GetURLContentToolArgs = z.infer; -const searchTool = createSearchToolFields(dummyRepo); -type SearchToolArgs = z.infer; +const grepTool = createGrepToolFields(dummyRepo); +type GrepToolArgs = z.infer; const searchDocumentForTool = createSearchDocumentForToolFields(); type SearchDocumentForToolArgs = z.infer; +const textEditorTool = createTextEditorToolFields(dummyRepo); +type TextEditorToolArgs = z.infer; // Common props for all action types type BaseActionProps = { @@ -99,8 +102,8 @@ type GetURLContentActionProps = BaseActionProps & }; type SearchActionProps = BaseActionProps & - Partial & { - actionType: "search"; + Partial & { + actionType: "grep"; output?: string; errorCode?: number; }; @@ -111,6 +114,12 @@ type SearchDocumentForActionProps = BaseActionProps & output?: string; }; +type TextEditorActionProps = BaseActionProps & + Partial & { + actionType: "text_editor"; + output?: string; + }; + type McpActionProps = BaseActionProps & { actionType: "mcp"; toolName: string; @@ -127,7 +136,8 @@ export type ActionItemProps = | GetURLContentActionProps | McpActionProps | SearchActionProps - | SearchDocumentForActionProps; + | SearchDocumentForActionProps + | TextEditorActionProps; export type ActionStepProps = { actions: ActionItemProps[]; @@ -142,7 +152,8 @@ const ACTION_GENERATING_TEXT_MAP = { [scratchpadTool.name]: "Saving notes...", [getURLContentTool.name]: "Fetching URL content...", [searchDocumentForTool.name]: "Searching document...", - [searchTool.name]: "Searching...", + [grepTool.name]: "Searching...", + [textEditorTool.name]: "Editing file...", }; function MatchCaseIcon({ matchCase }: { matchCase: boolean }) { @@ -196,7 +207,7 @@ function ActionItem(props: ActionItemProps) { } }; - const getStatusText = () => { + const getStatusText = (): string | JSX.Element => { if (props.status === "loading") { return "Preparing action..."; } @@ -224,8 +235,25 @@ function ActionItem(props: ActionItemProps) { return props.success ? "Document search completed" : "Document search failed"; - } else if (props.actionType === "search") { + } else if (props.actionType === "grep") { return props.success ? "Search completed" : "Search failed"; + } else if (props.actionType === "text_editor") { + const command = props.command || "unknown"; + return props.success ? ( + +

+ {command} +

{" "} + command completed +
+ ) : ( + +

+ {command} +

{" "} + command failed +
+ ); } else if (props.actionType === "mcp") { return props.success ? `${props.toolName} completed` @@ -245,7 +273,8 @@ function ActionItem(props: ActionItemProps) { props.actionType === "install_dependencies" || props.actionType === "get_url_content" || props.actionType === "search_document_for" || - props.actionType === "search" + props.actionType === "grep" || + props.actionType === "text_editor" ) { return !!props.output; } else if (props.actionType === "apply-patch") { @@ -316,13 +345,20 @@ function ActionItem(props: ActionItemProps) { icon={} /> ); - } else if (props.actionType === "search") { + } else if (props.actionType === "grep") { return ( } /> ); + } else if (props.actionType === "text_editor") { + return ( + } + /> + ); } else { return ( @@ -412,26 +448,29 @@ function ActionItem(props: ActionItemProps) { props.actionType === "shell" || props.actionType === "install_dependencies" ) { + const shellProps = props as + | ShellActionProps + | InstallDependenciesActionProps; let commandStr = ""; - if (props.command) { - if (Array.isArray(props.command)) { - commandStr = props.command.join(" "); + if (shellProps.command) { + if (Array.isArray(shellProps.command)) { + commandStr = shellProps.command.join(" "); } else if ( - typeof props.command === "string" && - (props.command as string).length > 0 + typeof shellProps.command === "string" && + (shellProps.command as string).length > 0 ) { try { - commandStr = JSON.parse(props.command); + commandStr = JSON.parse(shellProps.command); } catch { - commandStr = props.command; + commandStr = shellProps.command; } } } return (
- {props.workdir && ( + {shellProps.workdir && (
- {props.workdir} + {shellProps.workdir}
)} @@ -450,6 +489,38 @@ function ActionItem(props: ActionItemProps) {
); + } else if (props.actionType === "text_editor") { + const command = props.command || "unknown"; + const path = props.path || ""; + return ( +
+
+ + {command} + + + {path} + +
+ {command === "view" && props.view_range && ( +
+ Lines {props.view_range[0]}- + {props.view_range[1] === -1 ? "end" : props.view_range[1]} +
+ )} + {command === "str_replace" && props.old_str && ( +
+ Replace: {props.old_str.substring(0, 50)} + {props.old_str.length > 50 ? "..." : ""} +
+ )} + {command === "insert" && props.insert_line !== undefined && ( +
+ Insert at line {props.insert_line} +
+ )} +
+ ); } else if (props.actionType === "mcp") { return (
@@ -461,7 +532,9 @@ function ActionItem(props: ActionItemProps) { } else { return ( - {props.file_path} + {props.actionType === "apply-patch" && "file_path" in props + ? props.file_path + : ""} ); } @@ -475,9 +548,10 @@ function ActionItem(props: ActionItemProps) { if ( (props.actionType === "shell" || - props.actionType === "search" || + props.actionType === "grep" || props.actionType === "search_document_for" || - props.actionType === "install_dependencies") && + props.actionType === "install_dependencies" || + props.actionType === "text_editor") && props.output ) { return ( @@ -485,7 +559,7 @@ function ActionItem(props: ActionItemProps) {
             {props.output}
           
- {(props.actionType === "shell" || props.actionType === "search") && + {(props.actionType === "shell" || props.actionType === "grep") && "errorCode" in props && props.errorCode !== undefined && !props.success && ( diff --git a/apps/web/src/components/thread/messages/ai.tsx b/apps/web/src/components/thread/messages/ai.tsx index 03e22c1a..0b650de8 100644 --- a/apps/web/src/components/thread/messages/ai.tsx +++ b/apps/web/src/components/thread/messages/ai.tsx @@ -34,7 +34,7 @@ import { createShellToolFields, createMarkTaskCompletedToolFields, createMarkTaskNotCompletedToolFields, - createSearchToolFields, + createGrepToolFields, createOpenPrToolFields, createInstallDependenciesToolFields, createCodeReviewMarkTaskCompletedFields, @@ -46,6 +46,8 @@ import { createConversationHistorySummaryToolFields, createReviewStartedToolFields, createScratchpadFields, + createTextEditorToolFields, + createViewToolFields, } from "@open-swe/shared/open-swe/tools"; import { z } from "zod"; import { isAIMessageSDK, isToolMessageSDK } from "@/lib/langchain-messages"; @@ -67,8 +69,8 @@ type MarkTaskNotCompletedToolArgs = z.infer< >; const reviewStartedTool = createReviewStartedToolFields(); type ReviewStartedToolArgs = z.infer; -const searchTool = createSearchToolFields(dummyRepo); -type SearchToolArgs = z.infer; +const grepTool = createGrepToolFields(dummyRepo); +type GrepToolArgs = z.infer; const openPrTool = createOpenPrToolFields(); type OpenPrToolArgs = z.infer; const installDependenciesTool = createInstallDependenciesToolFields(dummyRepo); @@ -107,6 +109,15 @@ type ConversationHistorySummaryToolArgs = z.infer< typeof conversationHistorySummaryTool.schema >; +const textEditorTool = createTextEditorToolFields({ + owner: "dummy", + repo: "dummy", +}); +type TextEditorToolArgs = z.infer; + +const viewTool = createViewToolFields(dummyRepo); +type ViewToolArgs = z.infer; + // Helper function to detect MCP tools by checking if tool name is NOT in known tools function isMcpTool(toolName: string): boolean { const knownToolNames = [ @@ -117,6 +128,8 @@ function isMcpTool(toolName: string): boolean { getURLContentTool.name, openPrTool.name, diagnoseErrorTool.name, + textEditorTool.name, + viewTool.name, ]; return !knownToolNames.some((t) => t === toolName); } @@ -219,10 +232,10 @@ export function mapToolMessageToActionStepProps( reasoningText, errorMessage: !success ? getContentString(message.content) : undefined, }; - } else if (toolCall?.name === searchTool.name) { - const args = toolCall.args as SearchToolArgs; + } else if (toolCall?.name === grepTool.name) { + const args = toolCall.args as GrepToolArgs; return { - actionType: "search", + actionType: "grep", status, success, query: args.query || "", @@ -278,6 +291,34 @@ export function mapToolMessageToActionStepProps( output, reasoningText, }; + } else if (toolCall?.name === textEditorTool.name) { + const args = toolCall.args as TextEditorToolArgs; + return { + actionType: "text_editor", + status, + success, + command: args.command || "view", + path: args.path || "", + view_range: args.view_range, + old_str: args.old_str, + new_str: args.new_str, + file_text: args.file_text, + insert_line: args.insert_line, + output, + reasoningText, + }; + } else if (toolCall?.name === viewTool.name) { + const args = toolCall.args as ViewToolArgs; + return { + actionType: "text_editor", + status, + success, + command: args.command || "view", + path: args.path || "", + view_range: args.view_range, + output, + reasoningText, + }; } else if (toolCall && isMcpTool(toolCall.name)) { return { actionType: "mcp", @@ -352,10 +393,13 @@ export function AssistantMessage({ (tc) => tc.name === shellTool.name || tc.name === applyPatchTool.name || - tc.name === searchTool.name || + tc.name === grepTool.name || tc.name === installDependenciesTool.name || tc.name === scratchpadTool.name || tc.name === getURLContentTool.name || + tc.name === textEditorTool.name || + tc.name === viewTool.name || + tc.name === searchDocumentForTool.name || isMcpTool(tc.name), ) : []; @@ -609,15 +653,24 @@ export function AssistantMessage({ } if (actionableToolCalls.length > 0) { + if ( + actionableToolCalls[0].name !== "shell" && + actionableToolCalls[0].name !== "scratchpad" && + actionableToolCalls[0].name !== "grep" + ) { + console.log("actionableToolCalls", actionableToolCalls[0]); + } const actionItems = actionableToolCalls.map((toolCall): ActionItemProps => { const correspondingToolResult = toolResults.find( (tr) => tr && tr.tool_call_id === toolCall.id, ); const isShellTool = toolCall.name === shellTool.name; - const isSearchTool = toolCall.name === searchTool.name; + const isGrepTool = toolCall.name === grepTool.name; const isInstallDependenciesTool = toolCall.name === installDependenciesTool.name; + const isTextEditorTool = toolCall.name === textEditorTool.name; + const isViewTool = toolCall.name === viewTool.name; if (correspondingToolResult) { // If we have a tool result, map it to action props @@ -625,10 +678,10 @@ export function AssistantMessage({ correspondingToolResult, threadMessages, ); - } else if (isSearchTool) { - const args = toolCall.args as SearchToolArgs; + } else if (isGrepTool) { + const args = toolCall.args as GrepToolArgs; return { - actionType: "search", + actionType: "grep", status: "generating", query: args?.query || "", match_string: args?.match_string || false, @@ -674,6 +727,30 @@ export function AssistantMessage({ query: args?.query || "", output: "", } as ActionItemProps; + } else if (isTextEditorTool) { + const args = toolCall.args as TextEditorToolArgs; + return { + actionType: "text_editor", + status: "generating", + command: args?.command || "view", + path: args?.path || "", + view_range: args?.view_range, + old_str: args?.old_str, + new_str: args?.new_str, + file_text: args?.file_text, + insert_line: args?.insert_line, + output: "", + } as ActionItemProps; + } else if (isViewTool) { + const args = toolCall.args as ViewToolArgs; + return { + actionType: "text_editor", + status: "generating", + command: args?.command || "view", + path: args?.path || "", + view_range: args?.view_range, + output: "", + } as ActionItemProps; } else { if (isMcpTool(toolCall.name)) { return { diff --git a/apps/web/src/components/v2/token-usage.tsx b/apps/web/src/components/v2/token-usage.tsx index 8a983e6f..6dd47f69 100644 --- a/apps/web/src/components/v2/token-usage.tsx +++ b/apps/web/src/components/v2/token-usage.tsx @@ -315,7 +315,7 @@ export function TokenUsage({ tokenData }: TokenUsageProps) { open={isExpanded} onOpenChange={setIsExpanded} > - + Per-Model Breakdown {isExpanded ? ( @@ -323,7 +323,7 @@ export function TokenUsage({ tokenData }: TokenUsageProps) { )} - + {modelTokenData.map((model, index) => { const modelCost = calculateModelCost(model); const modelTotalTokens = diff --git a/packages/shared/src/open-swe/tools.ts b/packages/shared/src/open-swe/tools.ts index 05aa94dd..866479ca 100644 --- a/packages/shared/src/open-swe/tools.ts +++ b/packages/shared/src/open-swe/tools.ts @@ -106,7 +106,7 @@ export function createUpdatePlanToolFields() { }; } -export function createSearchToolFields(targetRepository: TargetRepository) { +export function createGrepToolFields(targetRepository: TargetRepository) { const repoRoot = getRepoAbsolutePath(targetRepository); const searchSchema = z.object({ query: z @@ -166,18 +166,18 @@ export function createSearchToolFields(targetRepository: TargetRepository) { }); return { - name: "search", + name: "grep", schema: searchSchema, - description: `Execute a search in the repository. Should be used to search for content via string matching or regex in the codebase. The working directory this command will be executed in is \`${repoRoot}\`.`, + description: `Execute a grep (ripgrep) search in the repository. Should be used to search for content via string matching or regex in the codebase. The working directory this command will be executed in is \`${repoRoot}\`.`, }; } // Only used for type inference -const _tmpSearchToolSchema = createSearchToolFields({ +const _tmpSearchToolSchema = createGrepToolFields({ owner: "x", repo: "x", }).schema; -export type SearchCommand = z.infer; +export type GrepCommand = z.infer; function escapeShellArg(arg: string): string { // If the string contains a single quote, close the string, escape the single quote, and reopen it @@ -185,8 +185,8 @@ function escapeShellArg(arg: string): string { return `'${arg.replace(/'/g, `'\\''`)}'`; } -export function formatSearchCommand( - cmd: SearchCommand, +export function formatGrepCommand( + cmd: GrepCommand, options?: { excludeRequiredFlags?: boolean; }, @@ -494,3 +494,79 @@ export function createReviewStartedToolFields() { schema: reviewStartedSchema, }; } + +export function createTextEditorToolFields(targetRepository: TargetRepository) { + const repoRoot = getRepoAbsolutePath(targetRepository); + const textEditorToolSchema = z.object({ + command: z + .enum(["view", "str_replace", "create", "insert"]) + .describe("The command to execute: view, str_replace, create, or insert"), + path: z + .string() + .describe("The path to the file or directory to operate on"), + view_range: z + .tuple([z.number(), z.number()]) + .optional() + .describe( + "Optional array of two integers [start, end] specifying line numbers to view. Line numbers are 1-indexed. Use -1 for end to read to end of file. Only applies to view command.", + ), + old_str: z + .string() + .optional() + .describe( + "The text to replace (must match exactly, including whitespace and indentation). Required for str_replace command.", + ), + new_str: z + .string() + .optional() + .describe( + "The new text to insert. Required for str_replace and insert commands.", + ), + file_text: z + .string() + .optional() + .describe( + "The content to write to the new file. Required for create command.", + ), + insert_line: z + .number() + .optional() + .describe( + "The line number after which to insert the text (0 for beginning of file). Required for insert command.", + ), + }); + + return { + name: "str_replace_based_edit_tool", + description: + "A text editor tool that can view, create, and edit files. " + + `The working directory is \`${repoRoot}\`. Ensure file paths are absolute and properly formatted. ` + + "Supports commands: view (read file/directory), str_replace (replace text), create (new file), insert (add text at line).", + schema: textEditorToolSchema, + }; +} + +export function createViewToolFields(targetRepository: TargetRepository) { + const repoRoot = getRepoAbsolutePath(targetRepository); + const viewSchema = z.object({ + command: z.enum(["view"]).describe("The command to execute: view"), + path: z + .string() + .describe("The path to the file or directory to operate on"), + view_range: z + .tuple([z.number(), z.number()]) + .optional() + .describe( + "Optional array of two integers [start, end] specifying line numbers to view. Line numbers are 1-indexed. Use -1 for end to read to end of file. Only applies to view command.", + ), + }); + + return { + name: "view", + description: + "A text editor tool that can view files. " + + `The working directory is \`${repoRoot}\`. Ensure file paths are absolute and properly formatted. ` + + "Supports commands: view (read file/directory).", + schema: viewSchema, + }; +} diff --git a/packages/shared/src/open-swe/types.ts b/packages/shared/src/open-swe/types.ts index fc9a239f..3abf0618 100644 --- a/packages/shared/src/open-swe/types.ts +++ b/packages/shared/src/open-swe/types.ts @@ -335,6 +335,25 @@ export const GraphConfigurationMetadata: { type: "hidden", }, }, + plannerModelName: { + x_open_swe_ui_config: { + type: "select", + default: "anthropic:claude-sonnet-4-0", + description: + "The model to use for planning tasks. This model should be very good at generating code, and have strong context understanding and reasoning capabilities. It will be used for the most complex tasks throughout the agent.", + options: MODEL_OPTIONS_NO_THINKING, + }, + }, + plannerTemperature: { + x_open_swe_ui_config: { + type: "slider", + default: 0, + min: 0, + max: 2, + step: 0.1, + description: "Controls randomness (0 = deterministic, 2 = creative)", + }, + }, programmerModelName: { x_open_swe_ui_config: { type: "select", @@ -354,6 +373,25 @@ export const GraphConfigurationMetadata: { description: "Controls randomness (0 = deterministic, 2 = creative)", }, }, + reviewerModelName: { + x_open_swe_ui_config: { + type: "select", + default: "anthropic:claude-sonnet-4-0", + description: + "The model to use for reviewer tasks. This model should be very good at generating code, and have strong context understanding and reasoning capabilities. It will be used for the most complex tasks throughout the agent.", + options: MODEL_OPTIONS_NO_THINKING, + }, + }, + reviewerTemperature: { + x_open_swe_ui_config: { + type: "slider", + default: 0, + min: 0, + max: 2, + step: 0.1, + description: "Controls randomness (0 = deterministic, 2 = creative)", + }, + }, routerModelName: { x_open_swe_ui_config: { type: "select", @@ -494,6 +532,21 @@ export const GraphConfiguration = z.object({ metadata: GraphConfigurationMetadata.maxReviewActions, }), + /** + * The model ID to use for programming/other advanced technical tasks. + * @default "anthropic:claude-sonnet-4-0" + */ + plannerModelName: withLangGraph(z.string().optional(), { + metadata: GraphConfigurationMetadata.plannerModelName, + }), + /** + * The temperature to use for programming/other advanced technical tasks. + * @default 0 + */ + plannerTemperature: withLangGraph(z.number().optional(), { + metadata: GraphConfigurationMetadata.plannerTemperature, + }), + /** * The model ID to use for programming/other advanced technical tasks. * @default "anthropic:claude-sonnet-4-0" @@ -508,6 +561,22 @@ export const GraphConfiguration = z.object({ programmerTemperature: withLangGraph(z.number().optional(), { metadata: GraphConfigurationMetadata.programmerTemperature, }), + + /** + * The model ID to use for programming/other advanced technical tasks. + * @default "anthropic:claude-sonnet-4-0" + */ + reviewerModelName: withLangGraph(z.string().optional(), { + metadata: GraphConfigurationMetadata.reviewerModelName, + }), + /** + * The temperature to use for programming/other advanced technical tasks. + * @default 0 + */ + reviewerTemperature: withLangGraph(z.number().optional(), { + metadata: GraphConfigurationMetadata.reviewerTemperature, + }), + /** * The model ID to use for routing tasks. * @default "anthropic:claude-3-5-haiku-latest" diff --git a/yarn.lock b/yarn.lock index ecdd2d0d..287f429f 100644 --- a/yarn.lock +++ b/yarn.lock @@ -2690,15 +2690,15 @@ __metadata: languageName: node linkType: hard -"@langchain/anthropic@npm:^0.3.20": - version: 0.3.24 - resolution: "@langchain/anthropic@npm:0.3.24" +"@langchain/anthropic@npm:^0.3.25": + version: 0.3.25 + resolution: "@langchain/anthropic@npm:0.3.25" dependencies: "@anthropic-ai/sdk": ^0.56.0 fast-xml-parser: ^4.4.1 peerDependencies: "@langchain/core": ">=0.3.58 <0.4.0" - checksum: 682461124e5afeb4ea087704f4b3d2008eca23152a45c1ff12f90002e9f1eeb330e0a4dc74730dfbaf7c04085a306e0d1186a44b3516f8956e81a55de4e41e46 + checksum: e7a48d130379f7d26b415ac3eb8ee9a9d44c3b51f1333568e2fa97b82c272e18057a8ee10a2bb0d8a0c18c69ee7d8945cf7aebd05198afc0da86540b38bc3dc0 languageName: node linkType: hard @@ -4095,7 +4095,7 @@ __metadata: "@eslint/eslintrc": ^3.1.0 "@eslint/js": ^9.19.0 "@jest/globals": ^29.7.0 - "@langchain/anthropic": ^0.3.20 + "@langchain/anthropic": ^0.3.25 "@langchain/community": ^0.3.47 "@langchain/core": ^0.3.65 "@langchain/google-genai": ^0.2.9