diff --git a/apps/open-swe/package.json b/apps/open-swe/package.json index ee323a00..7df339cd 100644 --- a/apps/open-swe/package.json +++ b/apps/open-swe/package.json @@ -28,7 +28,7 @@ "@langchain/community": "^0.3.47", "@langchain/core": "^0.3.56", "@langchain/google-genai": "^0.2.9", - "@langchain/langgraph": "^0.3.3", + "@langchain/langgraph": "^0.3.8", "@langchain/langgraph-sdk": "^0.0.92", "@langchain/mcp-adapters": "^0.5.2", "@langchain/openai": "^0.5.10", diff --git a/apps/open-swe/src/graphs/planner/index.ts b/apps/open-swe/src/graphs/planner/index.ts index 040fbaee..f5c494dc 100644 --- a/apps/open-swe/src/graphs/planner/index.ts +++ b/apps/open-swe/src/graphs/planner/index.ts @@ -18,7 +18,7 @@ import { } from "./nodes/index.js"; import { isAIMessage } from "@langchain/core/messages"; import { initializeSandbox } from "../shared/initialize-sandbox.js"; -import { diagnoseError } from "./nodes/diagnose-error.js"; +import { diagnoseError } from "../shared/diagnose-error.js"; function takeActionOrGeneratePlan( state: PlannerGraphState, diff --git a/apps/open-swe/src/graphs/planner/nodes/generate-message/index.ts b/apps/open-swe/src/graphs/planner/nodes/generate-message/index.ts index 562be677..f31fa725 100644 --- a/apps/open-swe/src/graphs/planner/nodes/generate-message/index.ts +++ b/apps/open-swe/src/graphs/planner/nodes/generate-message/index.ts @@ -19,10 +19,9 @@ import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { getMissingMessages } from "../../../../utils/github/issue-messages.js"; import { filterHiddenMessages } from "../../../../utils/message/filter-hidden.js"; import { getPlansFromIssue } from "../../../../utils/github/issue-task.js"; -import { createRgTool } from "../../../../tools/rg.js"; +import { createSearchTool } from "../../../../tools/search.js"; import { formatCustomRulesPrompt } from "../../../../utils/custom-rules.js"; import { createPlannerNotesTool } from "../../../../tools/planner-notes.js"; -import { createFindInstancesOfTool } from "../../../../tools/find-instances-of.js"; import { getMcpTools } from "../../../../utils/mcp-client.js"; const logger = createLogger(LogLevel.INFO, "GeneratePlanningMessageNode"); @@ -55,9 +54,8 @@ export async function generateAction( const mcpTools = await getMcpTools(config); const tools = [ - createRgTool(state), + createSearchTool(state), createShellTool(state), - createFindInstancesOfTool(state), createPlannerNotesTool(), createGetURLContentTool(), ...mcpTools, diff --git a/apps/open-swe/src/graphs/planner/nodes/generate-message/prompt.ts b/apps/open-swe/src/graphs/planner/nodes/generate-message/prompt.ts index ed31b71b..2a183ec2 100644 --- a/apps/open-swe/src/graphs/planner/nodes/generate-message/prompt.ts +++ b/apps/open-swe/src/graphs/planner/nodes/generate-message/prompt.ts @@ -20,11 +20,11 @@ Your sole objective in this phase is to gather comprehensive context about the c - To ensure the above does not happen, you should be thorough in your context gathering. Always gather enough context to cover all edge cases, and prevent unclear instructions. 4. **Leverage efficient search tools**: - - Use the \`find_instances_of\` tool when searching for specific keywords/strings in files. This tool utilizes \`rg\` (ripgrep) under the hood, but it is optimized for searching for specific keywords/strings in files. - - Use the \`rg\` (ripgrep) tool when performing more complex searches (e.g. regex searches). - - When searching for specific file types, use glob patterns: \`rg -i pattern -g **/*.tsx project-directory/\` - - This explicit pattern matching ensures accurate results across all file extensions - - Always use \`rg\` or \`find_instances_of\` tools instead calling \`grep\` via the \`shell\` tool. You should NEVER call \`grep\` as the same functionality is better provided by \`rg\` or \`find_instances_of\`. + - Use \`search\` tool for all file searches. The \`search\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns. + - It's significantly faster results than alternatives like grep or ls -R. + - When searching for specific file types, use glob patterns + - The pattern field supports both basic strings, and regex + - Always use the \`search\` tools instead calling \`grep\` via the \`shell\` tool. You should NEVER call \`grep\` as the same functionality is better provided by \`search\`. - If the user passes a URL, you should use the \`get_url_content\` tool to fetch the contents of the URL. - You should only use this tool to fetch the contents of a URL the user has provided, or that you've discovered during your context searching, which you believe is vital to gathering context for the user's request. diff --git a/apps/open-swe/src/graphs/planner/nodes/take-action.ts b/apps/open-swe/src/graphs/planner/nodes/take-action.ts index 816c47dd..ba7aa852 100644 --- a/apps/open-swe/src/graphs/planner/nodes/take-action.ts +++ b/apps/open-swe/src/graphs/planner/nodes/take-action.ts @@ -14,12 +14,11 @@ import { safeBadArgsError, } from "../../../utils/zod-to-string.js"; import { truncateOutput } from "../../../utils/truncate-outputs.js"; -import { createRgTool } from "../../../tools/rg.js"; +import { createSearchTool } from "../../../tools/search.js"; import { getChangedFilesStatus, stashAndClearChanges, } from "../../../utils/github/git.js"; -import { createFindInstancesOfTool } from "../../../tools/find-instances-of.js"; import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { createPlannerNotesTool } from "../../../tools/planner-notes.js"; import { getMcpTools } from "../../../utils/mcp-client.js"; @@ -41,18 +40,16 @@ export async function takeActions( } const shellTool = createShellTool(state); - const rgTool = createRgTool(state); + const searchTool = createSearchTool(state); const plannerNotesTool = createPlannerNotesTool(); - const findInstancesOfTool = createFindInstancesOfTool(state); const getURLContentTool = createGetURLContentTool(); const mcpTools = await getMcpTools(config); const allTools = [ shellTool, - rgTool, + searchTool, plannerNotesTool, getURLContentTool, - findInstancesOfTool, ...mcpTools, ]; const toolsMap = Object.fromEntries( @@ -129,7 +126,7 @@ export async function takeActions( : { error: e }), }); const errMessage = e instanceof Error ? e.message : "Unknown error"; - result = `FAILED TO CALL TOOL: "${toolCall.name}"\n\nError: ${errMessage}`; + result = `FAILED TO CALL TOOL: "${toolCall.name}"\n\n${errMessage}`; } } diff --git a/apps/open-swe/src/graphs/programmer/index.ts b/apps/open-swe/src/graphs/programmer/index.ts index 574b28f6..4ef92469 100644 --- a/apps/open-swe/src/graphs/programmer/index.ts +++ b/apps/open-swe/src/graphs/programmer/index.ts @@ -15,28 +15,40 @@ import { updatePlan, summarizeHistory, } from "./nodes/index.js"; -import { isAIMessage } from "@langchain/core/messages"; +import { BaseMessage, isAIMessage } from "@langchain/core/messages"; import { initializeSandbox } from "../shared/initialize-sandbox.js"; +import { graph as reviewerGraph } from "../reviewer/index.js"; import { getRemainingPlanItems } from "../../utils/current-task.js"; import { getActivePlanItems } from "@open-swe/shared/open-swe/tasks"; +function lastMessagesMissingToolCalls( + messages: BaseMessage[], + threshold: number, +) { + const lastMessages = messages.slice(-threshold); + if (!lastMessages.every(isAIMessage)) { + // If some of the last messages are not AI messages, we should return false. + return false; + } + return lastMessages.every((m) => !m.tool_calls?.length); +} + /** * Routes to the next appropriate node after taking action. * If the last message is an AI message with tool calls, it routes to "take-action". * Otherwise, it ends the process. * * @param {GraphState} state - The current graph state. - * @returns {"generate-conclusion" | "take-action" | "request-help" | "generate-action" | Send} The next node to execute, or END if the process should stop. + * @returns {"reviewer-subgraph" | "take-action" | "request-help" | "generate-action" | Send} The next node to execute, or END if the process should stop. */ -async function routeGeneratedAction( +function routeGeneratedAction( state: GraphState, -): Promise< - | "generate-conclusion" +): + | "reviewer-subgraph" | "take-action" | "request-help" | "generate-action" - | Send -> { + | Send { const { internalMessages } = state; const lastMessage = internalMessages[internalMessages.length - 1]; @@ -64,12 +76,29 @@ async function routeGeneratedAction( const activePlanItems = getActivePlanItems(state.taskPlan); const hasRemainingTasks = getRemainingPlanItems(activePlanItems).length > 0; // If the model did not generate a tool call, but there are remaining tasks, we should route back to the generate action step. - if (hasRemainingTasks) { + // Also add a check ensuring that the last to messages generated have tool calls. Otherwise we can end. + if (hasRemainingTasks && !lastMessagesMissingToolCalls(internalMessages, 2)) { return "generate-action"; } - // No tool calls, generate a conclusion. - return "generate-conclusion"; + // No tool calls, route to reviewer subgraph + return "reviewer-subgraph"; +} + +/** + * Conditional edge called after the reviewer. If there are no more actions to take, then open a PR. + * Otherwise, route to generate actions to continue with the new tasks. + */ +function routeGenerateActionsOrEnd( + state: GraphState, +): "generate-conclusion" | "generate-action" { + const activePlanItems = getActivePlanItems(state.taskPlan); + const allCompleted = activePlanItems.every((p) => p.completed); + if (allCompleted) { + return "generate-conclusion"; + } + + return "generate-action"; } const workflow = new StateGraph(GraphAnnotation, GraphConfiguration) @@ -80,12 +109,13 @@ const workflow = new StateGraph(GraphAnnotation, GraphConfiguration) }) .addNode("update-plan", updatePlan) .addNode("progress-plan-step", progressPlanStep, { - ends: ["summarize-history", "generate-action", "generate-conclusion"], + ends: ["summarize-history", "generate-action", "reviewer-subgraph"], }) .addNode("generate-conclusion", generateConclusion) .addNode("request-help", requestHelp, { ends: ["generate-action", END], }) + .addNode("reviewer-subgraph", reviewerGraph) .addNode("open-pr", openPullRequest) .addNode("diagnose-error", diagnoseError) .addNode("summarize-history", summarizeHistory) @@ -94,14 +124,18 @@ const workflow = new StateGraph(GraphAnnotation, GraphConfiguration) .addConditionalEdges("generate-action", routeGeneratedAction, [ "take-action", "request-help", - "generate-conclusion", + "reviewer-subgraph", "update-plan", "generate-action", ]) .addEdge("update-plan", "generate-action") - .addEdge("generate-conclusion", "open-pr") .addEdge("diagnose-error", "generate-action") + .addConditionalEdges("reviewer-subgraph", routeGenerateActionsOrEnd, [ + "generate-conclusion", + "generate-action", + ]) .addEdge("summarize-history", "generate-action") + .addEdge("generate-conclusion", "open-pr") .addEdge("open-pr", END); // Zod types are messed up diff --git a/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts b/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts index 4e8e7f21..b18c980e 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts @@ -18,6 +18,7 @@ import { getCurrentPlanItem } from "../../../../utils/current-task.js"; import { getMessageContentString } from "@open-swe/shared/messages"; import { getActivePlanItems } from "@open-swe/shared/open-swe/tasks"; import { + CODE_REVIEW_PROMPT, DEPENDENCIES_INSTALLED_PROMPT, INSTALL_DEPENDENCIES_TOOL_PROMPT, SYSTEM_PROMPT, @@ -25,10 +26,14 @@ import { import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { getMissingMessages } from "../../../../utils/github/issue-messages.js"; import { getPlansFromIssue } from "../../../../utils/github/issue-task.js"; -import { createRgTool } from "../../../../tools/rg.js"; +import { createSearchTool } from "../../../../tools/search.js"; import { createInstallDependenciesTool } from "../../../../tools/install-dependencies.js"; import { formatCustomRulesPrompt } from "../../../../utils/custom-rules.js"; import { getMcpTools } from "../../../../utils/mcp-client.js"; +import { + formatCodeReviewPrompt, + getCodeReviewFields, +} from "../../../../utils/review.js"; const logger = createLogger(LogLevel.INFO, "GenerateMessageNode"); @@ -38,6 +43,7 @@ const formatPrompt = (state: GraphState): string => { const currentPlanItem = activePlanItems .filter((p) => !p.completed) .sort((a, b) => a.index - b.index)[0]; + const codeReview = getCodeReviewFields(state.internalMessages); return SYSTEM_PROMPT.replaceAll( "{PLAN_PROMPT_WITH_SUMMARIES}", formatPlanPrompt(getActivePlanItems(state.taskPlan), { @@ -65,7 +71,16 @@ const formatPrompt = (state: GraphState): string => { ? INSTALL_DEPENDENCIES_TOOL_PROMPT : DEPENDENCIES_INSTALLED_PROMPT, ) - .replaceAll("{CUSTOM_RULES}", formatCustomRulesPrompt(state.customRules)); + .replaceAll("{CUSTOM_RULES}", formatCustomRulesPrompt(state.customRules)) + .replaceAll( + "{CODE_REVIEW_PROMPT}", + codeReview + ? formatCodeReviewPrompt(CODE_REVIEW_PROMPT, { + review: codeReview.review, + newActions: codeReview.newActions, + }) + : "", + ); }; export async function generateAction( @@ -76,7 +91,7 @@ export async function generateAction( const mcpTools = await getMcpTools(config); const tools = [ - createRgTool(state), + createSearchTool(state), createShellTool(state), createApplyPatchTool(state), createRequestHumanHelpToolFields(), diff --git a/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts b/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts index d73c1393..f4458c4f 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts @@ -1,6 +1,21 @@ export const INSTALL_DEPENDENCIES_TOOL_PROMPT = `* Use \`install_dependencies\` to install dependencies (skip if installation fails). IMPORTANT: You should only call this tool if you're executing a task which REQUIRES installing dependencies. Keep in mind that not all tasks will require installing dependencies.`; export const DEPENDENCIES_INSTALLED_PROMPT = `* Dependencies have already been installed. *`; +export const CODE_REVIEW_PROMPT = `# Code Review & New Actions + +The code changes you've made have been reviewed by a code reviewer. The code review has determined that the changes do _not_ satisfy the user's request, and have outlined a list of additional actions to take in order to successfully complete the user's request. + +The code review has provided this review of the changes: + +## Code Review +{CODE_REVIEW} + +The code review has outlined the following actions to take: + +## Actions to Take +{CODE_REVIEW_ACTIONS} +`; + export const SYSTEM_PROMPT = `# Identity You are a terminal-based agentic coding assistant built by LangChain. You wrap LLM models to enable natural language interaction with local codebases. You are precise, safe, and helpful. @@ -26,6 +41,7 @@ You are currently executing a specific task from a pre-generated plan. You have * Previous completed tasks and their summaries contain crucial context - always review them first * Condensed context messages in conversation history summarize previous work - read these to avoid duplication * The plan generation summary provides important codebase insights +* After some tasks are completed, you may be provided with a code review and additional tasks. Ensure you inspect the code review (if present) and new tasks to ensure the work you're doing satisfies the user's request. ### File and Code Management @@ -40,8 +56,10 @@ You are currently executing a specific task from a pre-generated plan. You have ### Tool Usage Best Practices -* **Search**: Use the \`find_instances_of\` tool when searching for specific keywords/strings in files. When performing more complex searches (e.g. regex searches), use the \`rg\` tool (ripgrep) (not grep/ls -R) with glob patterns (e.g., \`rg -i pattern -g **/*.tsx\`). - * Both \`find_instances_of\` and \`rg\` are optimized for searching files, and should be used in place of \`grep\` or \`ls -R\`. +* **Search**: Use \`search\` tool for all file searches. The \`search\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns. + * It's significantly faster results than alternatives like grep or ls -R. + * When searching for specific file types, use glob patterns + * The pattern field supports both basic strings, and regex * **Dependencies**: Use the correct package manager; skip if installation fails * **Pre-commit**: Run \`pre-commit run --files ...\` if .pre-commit-config.yaml exists * **History**: Use \`git log\` and \`git blame\` for additional context when needed @@ -52,7 +70,6 @@ You are currently executing a specific task from a pre-generated plan. You have * **Scripts may require dependencies to be installed**: Remember that sometimes scripts may require dependencies to be installed before they can be run. * Always ensure you've installed dependencies before running a script which might require them. - ### Coding Standards When modifying files: @@ -70,6 +87,7 @@ When modifying files: * When running a test, ensure you include the proper flags/environment variables to exclude colors/text formatting. This can cause the output to be unreadable. For example, when running Jest tests you pass the \`--no-colors\` flag. In PyTest you set the \`NO_COLOR\` environment variable (prefix the command with \`export NO_COLOR=1\`) * Only install trusted, well-maintained packages. If installing a new dependency which is not explicitly requested by the user, ensure it is a well-maintained, and widely used package. * Ensure package manager files are updated to include the new dependency. +* If a command you run fails (e.g. a test, build, lint, etc.), and you make changes to fix the issue, ensure you always re-run the command after making the changes to ensure the fix was successful. ### Communication Guidelines @@ -102,4 +120,6 @@ Location: {REPO_DIRECTORY} {CODEBASE_TREE} +{CODE_REVIEW_PROMPT} + {CUSTOM_RULES}`; diff --git a/apps/open-swe/src/graphs/programmer/nodes/progress-plan-step.ts b/apps/open-swe/src/graphs/programmer/nodes/progress-plan-step.ts index 60b9f265..ca8b575c 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/progress-plan-step.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/progress-plan-step.ts @@ -43,6 +43,8 @@ Here is the plan, along with the summaries of each completed task: Analyze the tasks you've completed, the tasks which are remaining, and the current task you just took an action on. In addition to this, you're also provided the full conversation history between you and the user. All of the messages in this conversation are from the previous steps/actions you've taken, and any user input. +If the task you're working on is to fix a failing command (e.g. a test, build, lint, etc.), and you've made changes to fix the issue, you must re-run the command to ensure the fix was successful before you can mark the task as complete. + For example: If you have a failing test, and you've applied an update to the file to fix the test, you MUST re-run the test before you can mark the task as complete. Take all of this information, and determine if the current task is complete, or if you still have work left to do. Once you've determined the status of the current task, call either: @@ -178,7 +180,7 @@ Once you've determined the status of the current task, call either the \`mark_ta taskPlan: updatedPlanTasks, }; return new Command({ - goto: "generate-conclusion", + goto: "reviewer-subgraph", update: commandUpdate, }); } diff --git a/apps/open-swe/src/graphs/programmer/nodes/take-action.ts b/apps/open-swe/src/graphs/programmer/nodes/take-action.ts index aa0c30d0..68c07974 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/take-action.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/take-action.ts @@ -24,7 +24,7 @@ import { getSandboxWithErrorHandling } from "../../../utils/sandbox.js"; import { getCodebaseTree } from "../../../utils/tree.js"; import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { createInstallDependenciesTool } from "../../../tools/install-dependencies.js"; -import { createRgTool } from "../../../tools/rg.js"; +import { createSearchTool } from "../../../tools/search.js"; import { getMcpTools } from "../../../utils/mcp-client.js"; import { shouldDiagnoseError } from "../../../utils/tool-message-error.js"; @@ -42,7 +42,7 @@ export async function takeAction( const applyPatchTool = createApplyPatchTool(state); const shellTool = createShellTool(state); - const rgTool = createRgTool(state); + const searchTool = createSearchTool(state); const installDependenciesTool = createInstallDependenciesTool(state); const getURLContentTool = createGetURLContentTool(); @@ -50,7 +50,7 @@ export async function takeAction( const allTools = [ shellTool, - rgTool, + searchTool, installDependenciesTool, applyPatchTool, getURLContentTool, @@ -122,7 +122,7 @@ export async function takeAction( : { error: e }), }); const errMessage = e instanceof Error ? e.message : "Unknown error"; - result = `FAILED TO CALL TOOL: "${toolCall.name}"\n\nError: ${errMessage}`; + result = `FAILED TO CALL TOOL: "${toolCall.name}"\n\n${errMessage}`; } } diff --git a/apps/open-swe/src/graphs/reviewer/index.ts b/apps/open-swe/src/graphs/reviewer/index.ts new file mode 100644 index 00000000..3f84b92f --- /dev/null +++ b/apps/open-swe/src/graphs/reviewer/index.ts @@ -0,0 +1,59 @@ +import { END, START, StateGraph } from "@langchain/langgraph"; +import { + ReviewerGraphState, + ReviewerGraphStateObj, +} from "@open-swe/shared/open-swe/reviewer/types"; +import { + GraphConfig, + GraphConfiguration, +} from "@open-swe/shared/open-swe/types"; +import { + finalReview, + generateReviewActions, + initializeState, + takeReviewerActions, +} from "./nodes/index.js"; +import { isAIMessage } from "@langchain/core/messages"; +import { diagnoseError } from "../shared/diagnose-error.js"; + +function takeReviewActionsOrFinalReview( + state: ReviewerGraphState, + config: GraphConfig, +): "take-review-actions" | "final-review" { + const { reviewerMessages } = state; + const lastMessage = reviewerMessages[reviewerMessages.length - 1]; + + const maxReviewActions = config.configurable?.maxReviewActions ?? 30; + const maxActionsCount = maxReviewActions * 2 + 1; + if ( + isAIMessage(lastMessage) && + lastMessage.tool_calls?.length && + reviewerMessages.length < maxActionsCount + ) { + return "take-review-actions"; + } + + // If the last message does not have tool calls, continue to generate the final review. + return "final-review"; +} + +const workflow = new StateGraph(ReviewerGraphStateObj, GraphConfiguration) + .addNode("initialize-state", initializeState) + .addNode("generate-review-actions", generateReviewActions) + .addNode("take-review-actions", takeReviewerActions, { + ends: ["generate-review-actions", "diagnose-reviewer-error"], + }) + .addNode("diagnose-reviewer-error", diagnoseError) + .addNode("final-review", finalReview) + .addEdge(START, "initialize-state") + .addEdge("initialize-state", "generate-review-actions") + .addConditionalEdges( + "generate-review-actions", + takeReviewActionsOrFinalReview, + ["take-review-actions", "final-review"], + ) + .addEdge("diagnose-reviewer-error", "generate-review-actions") + .addEdge("final-review", END); + +export const graph = workflow.compile(); +graph.name = "Open SWE - Reviewer"; diff --git a/apps/open-swe/src/graphs/reviewer/nodes/final-review.ts b/apps/open-swe/src/graphs/reviewer/nodes/final-review.ts new file mode 100644 index 00000000..dfdc41d6 --- /dev/null +++ b/apps/open-swe/src/graphs/reviewer/nodes/final-review.ts @@ -0,0 +1,158 @@ +import { + ReviewerGraphState, + ReviewerGraphUpdate, +} from "@open-swe/shared/open-swe/reviewer/types"; +import { getUserRequest } from "../../../utils/user-request.js"; +import { formatPlanPromptWithSummaries } from "../../../utils/plan-prompt.js"; +import { + getActivePlanItems, + getActiveTask, + updateTaskPlanItems, +} from "@open-swe/shared/open-swe/tasks"; +import { + createCodeReviewMarkTaskCompletedFields, + createCodeReviewMarkTaskNotCompleteFields, +} from "@open-swe/shared/open-swe/tools"; +import { loadModel, Task } from "../../../utils/load-model.js"; +import { GraphConfig, PlanItem } from "@open-swe/shared/open-swe/types"; +import { z } from "zod"; +import { addTaskPlanToIssue } from "../../../utils/github/issue-task.js"; +import { getMessageString } from "../../../utils/message/content.js"; +import { ToolMessage } from "@langchain/core/messages"; + +const SYSTEM_PROMPT = `You are a code reviewer for a software engineer working on a large codebase. + + +You've just finished reviewing the actions taken by the Programmer Assistant, and are ready to provide a final review. In this final review, you are to either: +1. Determine all of the necessary actions have been taken which completed the user's request, and all of the individual tasks outlined in the plan. +or +2. Determine that the actions taken are insufficient, and do not fully complete the user's request, and all of the individual tasks outlined in the plan. + +If you determine that the task is completed, you may call the \`{COMPLETE_TOOL_NAME}\` tool, providing your final review. +If you determine that the task has not been fully completed, you may call the \`{NOT_COMPLETE_TOOL_NAME}\` tool, providing your review, and a list of additional actions to take which will successfully satisfy your review, and complete the task. + + + +Here is the full list of actions you took during your review: +{REVIEW_ACTIONS} + +Here is the user's original request: +{USER_REQUEST} + +And here are the tasks which were outlined in the plan, and completed by the Programmer Assistant: +{PLANNED_TASKS} + + + +If you determine that the task is not completed, keep the following in mind when generating your review: +- Formatting/linting scripts should always be executed last, since any changes made after them could cause the codebase to no longer be properly formatted/linted. + +Carefully read over all of the provided context above, and if you determine that the task has NOT been completed, call the \`{NOT_COMPLETE_TOOL_NAME}\` tool. +Otherwise, if you determine that the task has been successfully completed, call the \`{COMPLETE_TOOL_NAME}\` tool. +`; + +const formatSystemPrompt = (state: ReviewerGraphState) => { + const markCompletedToolName = createCodeReviewMarkTaskCompletedFields().name; + const markNotCompleteToolName = + createCodeReviewMarkTaskNotCompleteFields().name; + const userRequest = getUserRequest(state.messages); + const activePlan = getActivePlanItems(state.taskPlan); + const tasksString = formatPlanPromptWithSummaries(activePlan); + const messagesString = state.reviewerMessages + .map(getMessageString) + .join("\n"); + return SYSTEM_PROMPT.replaceAll("{REVIEW_ACTIONS}", messagesString) + .replaceAll("{USER_REQUEST}", userRequest) + .replaceAll("{PLANNED_TASKS}", tasksString) + .replaceAll("{COMPLETE_TOOL_NAME}", markCompletedToolName) + .replaceAll("{NOT_COMPLETE_TOOL_NAME}", markNotCompleteToolName); +}; + +export async function finalReview( + state: ReviewerGraphState, + config: GraphConfig, +): Promise { + const completedTool = createCodeReviewMarkTaskCompletedFields(); + const incompleteTool = createCodeReviewMarkTaskNotCompleteFields(); + const tools = [completedTool, incompleteTool]; + const model = await loadModel(config, Task.PLANNER); + const modelWithTools = model.bindTools(tools, { + tool_choice: "any", + parallel_tool_calls: false, + }); + + const response = await modelWithTools.invoke([ + { + role: "user", + content: formatSystemPrompt(state), + }, + ]); + + const toolCall = response.tool_calls?.[0]; + if (!toolCall) { + throw new Error("No tool call review generated"); + } + + if (toolCall.name === completedTool.name) { + // Marked as completed. No further actions necessary. + const toolMessage = new ToolMessage({ + tool_call_id: toolCall.id ?? "", + content: "Marked task as completed.", + }); + const messagesUpdate = [response, toolMessage]; + return { + messages: messagesUpdate, + internalMessages: messagesUpdate, + reviewerMessages: messagesUpdate, + }; + } + + if (toolCall.name !== incompleteTool.name) { + throw new Error("Invalid tool call"); + } + + // Not done. Add the new plan items to the task, then return. + const newActions = (toolCall.args as z.infer) + .additional_actions; + const activeTask = getActiveTask(state.taskPlan); + const activePlanItems = getActivePlanItems(state.taskPlan); + const completedPlanItems = activePlanItems.filter((p) => p.completed); + const newPlanItemsList: PlanItem[] = [ + // Only include completed plan items from the previous task plan in the update. + ...completedPlanItems, + ...newActions.map((a, index) => ({ + index: completedPlanItems.length + index, + plan: a, + completed: false, + summary: undefined, + })), + ]; + const updatedTaskPlan = updateTaskPlanItems( + state.taskPlan, + activeTask.id, + newPlanItemsList, + "agent", + ); + + await addTaskPlanToIssue( + { + githubIssueId: state.githubIssueId, + targetRepository: state.targetRepository, + }, + config, + updatedTaskPlan, + ); + + const toolMessage = new ToolMessage({ + tool_call_id: toolCall.id ?? "", + content: "Marked task as incomplete.", + }); + + const messagesUpdate = [response, toolMessage]; + + return { + taskPlan: updatedTaskPlan, + messages: messagesUpdate, + internalMessages: messagesUpdate, + }; +} diff --git a/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/index.ts b/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/index.ts new file mode 100644 index 00000000..2814a8f1 --- /dev/null +++ b/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/index.ts @@ -0,0 +1,113 @@ +import { loadModel, Task } from "../../../../utils/load-model.js"; +import { + ReviewerGraphState, + ReviewerGraphUpdate, +} from "@open-swe/shared/open-swe/reviewer/types"; +import { GraphConfig } from "@open-swe/shared/open-swe/types"; +import { createLogger, LogLevel } from "../../../../utils/logger.js"; +import { getMessageContentString } from "@open-swe/shared/messages"; +import { PREVIOUS_REVIEW_PROMPT, SYSTEM_PROMPT } from "./prompt.js"; +import { getRepoAbsolutePath } from "@open-swe/shared/git"; +import { + createSearchTool, + createShellTool, + createInstallDependenciesTool, +} from "../../../../tools/index.js"; +import { formatCustomRulesPrompt } from "../../../../utils/custom-rules.js"; +import { getUserRequest } from "../../../../utils/user-request.js"; +import { getActivePlanItems } from "@open-swe/shared/open-swe/tasks"; +import { formatPlanPromptWithSummaries } from "../../../../utils/plan-prompt.js"; +import { + formatCodeReviewPrompt, + getCodeReviewFields, +} from "../../../../utils/review.js"; +import { BaseMessage } from "@langchain/core/messages"; +import { getMessageString } from "../../../../utils/message/content.js"; + +const logger = createLogger(LogLevel.INFO, "GenerateReviewActionsNode"); + +function formatSystemPrompt(state: ReviewerGraphState): string { + const userRequest = getUserRequest(state.messages); + const activePlan = getActivePlanItems(state.taskPlan); + const tasksString = formatPlanPromptWithSummaries(activePlan); + const codeReview = getCodeReviewFields(state.internalMessages); + + return SYSTEM_PROMPT.replaceAll( + "{CODEBASE_TREE}", + state.codebaseTree || "No codebase tree generated yet.", + ) + .replaceAll( + "{CURRENT_WORKING_DIRECTORY}", + getRepoAbsolutePath(state.targetRepository), + ) + .replaceAll("{CUSTOM_RULES}", formatCustomRulesPrompt(state.customRules)) + .replaceAll("{CHANGED_FILES}", state.changedFiles) + .replaceAll("{BASE_BRANCH_NAME}", state.baseBranchName) + .replaceAll("{COMPLETED_TASKS_AND_SUMMARIES}", tasksString) + .replaceAll( + "{DEPENDENCIES_INSTALLED}", + state.dependenciesInstalled ? "Yes" : "No", + ) + .replaceAll("{USER_REQUEST}", userRequest) + .replaceAll( + "{PREVIOUS_REVIEW_PROMPT}", + codeReview + ? formatCodeReviewPrompt(PREVIOUS_REVIEW_PROMPT, { + review: codeReview.review, + newActions: codeReview.newActions, + }) + : "", + ); +} + +function formatUserConversationHistoryMessage(messages: BaseMessage[]): string { + return `Here is the full conversation history of the programmer. This includes all of the actions taken by the programmer, as well as any user input. +If the history has been truncated, it is because the conversation was too long. In this case, you should only consider the most recent messages. + + +${messages.map(getMessageString).join("\n")} +`; +} + +export async function generateReviewActions( + state: ReviewerGraphState, + config: GraphConfig, +): Promise { + const model = await loadModel(config, Task.ACTION_GENERATOR); + const tools = [ + createSearchTool(state), + createShellTool(state), + createInstallDependenciesTool(state), + ]; + const modelWithTools = model.bindTools(tools, { + tool_choice: "auto", + parallel_tool_calls: true, + }); + + const response = await modelWithTools.invoke([ + { + role: "system", + content: formatSystemPrompt(state), + }, + { + role: "user", + content: formatUserConversationHistoryMessage(state.internalMessages), + }, + ...state.reviewerMessages, + ]); + + logger.info("Generated review actions", { + ...(getMessageContentString(response.content) && { + content: getMessageContentString(response.content), + }), + ...response.tool_calls?.map((tc) => ({ + name: tc.name, + args: tc.args, + })), + }); + + return { + messages: [response], + reviewerMessages: [response], + }; +} diff --git a/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/prompt.ts b/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/prompt.ts new file mode 100644 index 00000000..0ae08847 --- /dev/null +++ b/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/prompt.ts @@ -0,0 +1,120 @@ +export const PREVIOUS_REVIEW_PROMPT = ` +You've already generated a review of the changes, and since then the programmer has implemented fixes. +The review you left is as follows: + +{CODE_REVIEW} + + +The actions you outlined to take are as follows: + +{CODE_REVIEW_ACTIONS} + + +Given this review and the actions you requested be completed to successfully complete the user's request, you should now review the changes again. +You do not need to provide an extensive review of the entire codebase. You should focus your new review on the actions you outlined above to take, and the changes since the previous review. +`; + +export const SYSTEM_PROMPT = `You are a terminal-based agentic coding assistant built by LangChain that enables natural language interaction with local codebases. You excel at being precise, safe, and helpful in your analysis. + + +Reviewer Assistant - Read-Only Phase + + + +Your sole objective in this phase is to review the actions taken by the Programmer Assistant which were based on the plan generated by the Planner Assistant. +By reviewing these actions, and comparing them to the plan and original user request, you will eventually determine if the actions taken are sufficient to complete the user's request, or if more actions need to be taken. + + + +1. **Use only read operations**: Execute commands that inspect and analyze the codebase without modifying any files. This ensures we understand the current state before making changes. + +2. **Make high-quality, targeted tool calls**: Each command should have a clear purpose in reviewing the actions taken by the Programmer Assistant. + +3. **Use git commands to gather context**: Below you're provided with a section '', which lists all of the files that were modified/created/deleted in the current branch. + - Ensure you use this, paired with commands such as 'git diff {BASE_BRANCH_NAME} ' to inspect a diff of a file to gather context about the changes made by the Programmer Assistant. + +3. **Gather all of the context necessary**: Ensure you gather all of the context necessary to provide a review of the changes made by the Programmer Assistant. + +4. **Leverage \`search\` tool**: Use \`search\` tool for all file searches. The \`search\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns. + - It's significantly faster results than alternatives like grep or ls -R. + - When searching for specific file types, use glob patterns + - The pattern field supports both basic strings, and regex + +5. **Format shell commands precisely**: Ensure all shell commands include proper quoting and escaping. Well-formatted commands prevent errors and provide reliable results. + +6. **Only take necessary actions**: You should only take actions which are absolutely necessary to provide a quality review of ONLY the changes in the current branch & the user's request. + - Think about whether or not the request you're reviewing is a simple one, which would warrant less review actions to take, or a more complex request, which would require a more detailed review. + +7. **Parallel tool calling**: It is highly recommended that you use parallel tool calling to gather context as quickly and efficiently as possible. + - When you know ahead of time there are multiple commands you want to run to gather context, of which they are independent and can be run in parallel, you should use parallel tool calling. + +8. **Always use the correct package manager**: If taking an action which requires a package manager (e.g. npm/yarn or pip/poetry, etc.), ensure you always search for the package manager used by the codebase, and use that one. + - Using a package manager that is different from the one used by the codebase may result in unexpected behavior, or errors. + +9. **Prefer using pre-made scripts**: If taking an action like running tests, formatting, linting, etc., always prefer using pre-made scripts over running commands manually. + - If you want to run a command like this, but are unsure if a pre-made script exists, always search for it first. + +10. **Signal completion clearly**: When you have gathered sufficient context, respond with exactly 'done' without any tool calls. This indicates readiness to proceed to the final review phase. + + + +You should inspect each of the files modified by the programmer (see the section below), and confirm they properly implement the plan (see the section below), and that the user's request has been fully implemented. +You should be reviewing them from the perspective of a quality assurance engineer, ensuring the code written is of the highest quality, fully implements the user's request, and all actions have been taken for the PR to be accepted. + +You're also provided with the conversation history of the actions the programmer has taken, and any user input they've received. The first user message below contains this information. +Ensure you carefully read over all of these messages to ensure you have the proper context and do not duplicate actions the programmer has already taken. + +Common tasks you should always confirm were executed: +- Linter/formatter scripts were executed +- Unit tests were executed +- If no tests for the code written/updated exists, confirm whether or not tests should be written +- Documentation was updated, if applicable + +**IMPORTANT**: +Keep in mind that not all requests/changes will need tests to be written, or documentation to be added/updated. Ensure you consider whether or not the standard engineering organization would write tests, or documentation for the changes you're reviewing. +After considering this, you may not need to check if tests should be written, or documentation should be added/updated. + +Based on the generated plan, the actions taken and files changed, you should review the modified code and determine if it properly completes the overall task, or if more changes need to be made/existing changes should be modified. +On top of inspecting the changed files, you should also look to see if the programmer missed anything, made changes which do not respect the custom rules, or if the changes are otherwise insufficient to complete the task. + +You do not want to do more work than required, but you always should complete tasks which you believe are necessary to complete the user's request, and merge the pull request without further action. + +After you're satisfied with the context you've gathered, and are ready to provide a final review, respond with exactly 'done' without any tool calls. +This will redirect you to a final review step where you'll submit your final review, and optionally provide a list of additional actions to take. + +**REMINDER**: +You are ONLY gathering context. Any non-read actions you believe are necessary to take can be executed after you've provided your final review. +Only gather context right now in order to inform your final review, and to provide any additional steps to take after the review. + + + +**Current Working Directory**: {CURRENT_WORKING_DIRECTORY} +**Repository Status**: Already cloned and accessible in the current directory +**Base Branch Name**: {BASE_BRANCH_NAME} +**Dependencies Installed**: {DEPENDENCIES_INSTALLED} + +**Codebase Structure** (3 levels deep, respecting .gitignore): +Generated via: \`git ls-files | tree --fromfile -L 3\` + +{CODEBASE_TREE} + + +**Changed Files**: +Generated via: \`git diff {BASE_BRANCH_NAME} --name-only\` + +{CHANGED_FILES} + + + +{CUSTOM_RULES} + + +{COMPLETED_TASKS_AND_SUMMARIES} + + +{PREVIOUS_REVIEW_PROMPT} + + +The user's request is as follows (it's also included in the conversation history below). +{USER_REQUEST} +`; diff --git a/apps/open-swe/src/graphs/reviewer/nodes/index.ts b/apps/open-swe/src/graphs/reviewer/nodes/index.ts new file mode 100644 index 00000000..b8da89bc --- /dev/null +++ b/apps/open-swe/src/graphs/reviewer/nodes/index.ts @@ -0,0 +1,4 @@ +export * from "./generate-review-actions/index.js"; +export * from "./take-review-action.js"; +export * from "./initialize-state.js"; +export * from "./final-review.js"; diff --git a/apps/open-swe/src/graphs/reviewer/nodes/initialize-state.ts b/apps/open-swe/src/graphs/reviewer/nodes/initialize-state.ts new file mode 100644 index 00000000..1583f94c --- /dev/null +++ b/apps/open-swe/src/graphs/reviewer/nodes/initialize-state.ts @@ -0,0 +1,60 @@ +import { + ReviewerGraphState, + ReviewerGraphUpdate, +} from "@open-swe/shared/open-swe/reviewer/types"; +import { getSandboxWithErrorHandling } from "../../../utils/sandbox.js"; +import { getRepoAbsolutePath } from "@open-swe/shared/git"; +import { createLogger, LogLevel } from "../../../utils/logger.js"; +import { GraphConfig } from "@open-swe/shared/open-swe/types"; + +const logger = createLogger(LogLevel.INFO, "InitializeStateNode"); + +export async function initializeState( + state: ReviewerGraphState, + config: GraphConfig, +): Promise { + const repoRoot = getRepoAbsolutePath(state.targetRepository); + logger.info("Initializing state for reviewer"); + // get the base branch name, then get the changed files + const { sandbox, codebaseTree, dependenciesInstalled } = + await getSandboxWithErrorHandling( + state.sandboxSessionId, + state.targetRepository, + state.branchName, + config, + ); + + let baseBranchName = state.targetRepository.branch; + if (!baseBranchName) { + const baseBranchNameRes = await sandbox.process.executeCommand( + "git config init.defaultBranch", + repoRoot, + ); + if (baseBranchNameRes.exitCode !== 0) { + throw new Error( + `Failed to get base branch name: ${JSON.stringify(baseBranchNameRes, null, 2)}`, + ); + } + baseBranchName = baseBranchNameRes.result.trim(); + } + + const changedFilesRes = await sandbox.process.executeCommand( + `git diff ${baseBranchName} --name-only`, + repoRoot, + ); + if (changedFilesRes.exitCode !== 0) { + throw new Error( + `Failed to get changed files: ${JSON.stringify(changedFilesRes, null, 2)}`, + ); + } + const changedFiles = changedFilesRes.result.trim(); + + logger.info("Finished getting state for reviewer"); + + return { + baseBranchName, + changedFiles, + ...(codebaseTree ? { codebaseTree } : {}), + ...(dependenciesInstalled !== null ? { dependenciesInstalled } : {}), + }; +} diff --git a/apps/open-swe/src/graphs/reviewer/nodes/take-review-action.ts b/apps/open-swe/src/graphs/reviewer/nodes/take-review-action.ts new file mode 100644 index 00000000..24446d4e --- /dev/null +++ b/apps/open-swe/src/graphs/reviewer/nodes/take-review-action.ts @@ -0,0 +1,188 @@ +import { v4 as uuidv4 } from "uuid"; +import { isAIMessage, ToolMessage } from "@langchain/core/messages"; +import { + createInstallDependenciesTool, + createShellTool, +} from "../../../tools/index.js"; +import { GraphConfig } from "@open-swe/shared/open-swe/types"; +import { + ReviewerGraphState, + ReviewerGraphUpdate, +} from "@open-swe/shared/open-swe/reviewer/types"; +import { createLogger, LogLevel } from "../../../utils/logger.js"; +import { zodSchemaToString } from "../../../utils/zod-to-string.js"; +import { formatBadArgsError } from "../../../utils/zod-to-string.js"; +import { truncateOutput } from "../../../utils/truncate-outputs.js"; +import { createSearchTool } from "../../../tools/search.js"; +import { + checkoutBranchAndCommit, + getChangedFilesStatus, +} from "../../../utils/github/git.js"; +import { getRepoAbsolutePath } from "@open-swe/shared/git"; +import { getSandboxWithErrorHandling } from "../../../utils/sandbox.js"; +import { Command } from "@langchain/langgraph"; +import { shouldDiagnoseError } from "../../../utils/tool-message-error.js"; + +const logger = createLogger(LogLevel.INFO, "TakeReviewAction"); + +export async function takeReviewerActions( + state: ReviewerGraphState, + config: GraphConfig, +): Promise { + const { reviewerMessages } = state; + const lastMessage = reviewerMessages[reviewerMessages.length - 1]; + + if (!isAIMessage(lastMessage) || !lastMessage.tool_calls?.length) { + throw new Error("Last message is not an AI message with tool calls."); + } + + const shellTool = createShellTool(state); + const searchTool = createSearchTool(state); + const installDependenciesTool = createInstallDependenciesTool(state); + const allTools = [shellTool, searchTool, installDependenciesTool]; + const toolsMap = Object.fromEntries( + allTools.map((tool) => [tool.name, tool]), + ); + + const toolCalls = lastMessage.tool_calls; + if (!toolCalls?.length) { + throw new Error("No tool calls found."); + } + + const { sandbox, codebaseTree, dependenciesInstalled } = + await getSandboxWithErrorHandling( + state.sandboxSessionId, + state.targetRepository, + state.branchName, + config, + ); + + const toolCallResultsPromise = toolCalls.map(async (toolCall) => { + const tool = toolsMap[toolCall.name]; + if (!tool) { + logger.error(`Unknown tool: ${toolCall.name}`); + const toolMessage = new ToolMessage({ + id: uuidv4(), + tool_call_id: toolCall.id ?? "", + content: `Unknown tool: ${toolCall.name}`, + name: toolCall.name, + status: "error", + }); + + return toolMessage; + } + + logger.info("Executing review action", { + ...toolCall, + }); + + let result = ""; + let toolCallStatus: "success" | "error" = "success"; + try { + const toolResult = + // @ts-expect-error tool.invoke types are weird here... + (await tool.invoke({ + ...toolCall.args, + // Pass in the existing/new sandbox session ID to the tool call. + // use `x` prefix to avoid name conflicts with tool args. + xSandboxSessionId: sandbox.id, + })) as { + result: string; + status: "success" | "error"; + }; + result = toolResult.result; + toolCallStatus = toolResult.status; + } catch (e) { + toolCallStatus = "error"; + if ( + e instanceof Error && + e.message === "Received tool input did not match expected schema" + ) { + logger.error("Received tool input did not match expected schema", { + toolCall, + expectedSchema: zodSchemaToString(tool.schema), + }); + result = formatBadArgsError(tool.schema, toolCall.args); + } else { + logger.error("Failed to call tool", { + ...(e instanceof Error + ? { name: e.name, message: e.message, stack: e.stack } + : { error: e }), + }); + const errMessage = e instanceof Error ? e.message : "Unknown error"; + result = `FAILED TO CALL TOOL: "${toolCall.name}"\n\n${errMessage}`; + } + } + + const toolMessage = new ToolMessage({ + id: uuidv4(), + tool_call_id: toolCall.id ?? "", + content: truncateOutput(result), + name: toolCall.name, + status: toolCallStatus, + }); + return toolMessage; + }); + + const toolCallResults = await Promise.all(toolCallResultsPromise); + const repoPath = getRepoAbsolutePath(state.targetRepository); + const changedFiles = await getChangedFilesStatus(repoPath, sandbox); + let branchName: string | undefined = state.branchName; + if (changedFiles.length > 0) { + logger.info(`Has ${changedFiles.length} changed files. Committing.`, { + changedFiles, + }); + branchName = await checkoutBranchAndCommit( + config, + state.targetRepository, + sandbox, + { + branchName, + }, + ); + } + + let wereDependenciesInstalled: boolean | null = null; + toolCallResults.forEach((toolCallResult) => { + if (toolCallResult.name === installDependenciesTool.name) { + wereDependenciesInstalled = toolCallResult.status === "success"; + } + }); + + // Prioritize wereDependenciesInstalled over dependenciesInstalled + const dependenciesInstalledUpdate = + wereDependenciesInstalled !== null + ? wereDependenciesInstalled + : dependenciesInstalled !== null + ? dependenciesInstalled + : null; + + logger.info("Completed review action", { + ...toolCallResults.map((tc) => ({ + tool_call_id: tc.tool_call_id, + status: tc.status, + })), + }); + + const commandUpdate: ReviewerGraphUpdate = { + messages: toolCallResults, + reviewerMessages: toolCallResults, + ...(branchName && { branchName }), + ...(codebaseTree ? { codebaseTree } : {}), + ...(dependenciesInstalledUpdate !== null && { + dependenciesInstalled: dependenciesInstalledUpdate, + }), + }; + + const shouldRouteDiagnoseNode = shouldDiagnoseError([ + ...state.reviewerMessages, + ...toolCallResults, + ]); + + return new Command({ + goto: shouldRouteDiagnoseNode + ? "diagnose-reviewer-error" + : "generate-review-actions", + update: commandUpdate, + }); +} diff --git a/apps/open-swe/src/graphs/planner/nodes/diagnose-error.ts b/apps/open-swe/src/graphs/shared/diagnose-error.ts similarity index 88% rename from apps/open-swe/src/graphs/planner/nodes/diagnose-error.ts rename to apps/open-swe/src/graphs/shared/diagnose-error.ts index a192aa0a..7b1d9c25 100644 --- a/apps/open-swe/src/graphs/planner/nodes/diagnose-error.ts +++ b/apps/open-swe/src/graphs/shared/diagnose-error.ts @@ -5,18 +5,14 @@ import { } from "@langchain/core/messages"; import { createDiagnoseErrorToolFields } from "@open-swe/shared/open-swe/tools"; -import { getMessageString } from "../../../utils/message/content.js"; -import { loadModel, Task } from "../../../utils/load-model.js"; import { z } from "zod"; -import { createLogger, LogLevel } from "../../../utils/logger.js"; -import { getAllLastFailedActions } from "../../../utils/tool-message-error.js"; -import { - PlannerGraphState, - PlannerGraphUpdate, -} from "@open-swe/shared/open-swe/planner/types"; import { GraphConfig } from "@open-swe/shared/open-swe/types"; +import { createLogger, LogLevel } from "../../utils/logger.js"; +import { getAllLastFailedActions } from "../../utils/tool-message-error.js"; +import { getMessageString } from "../../utils/message/content.js"; +import { loadModel, Task } from "../../utils/load-model.js"; -const logger = createLogger(LogLevel.INFO, "DiagnoseError"); +const logger = createLogger(LogLevel.INFO, "SharedDiagnoseError"); const systemPrompt = `You are operating as a terminal-based agentic coding assistant built by LangChain. It wraps LLM models to enable natural language interaction with a local codebase. You are expected to be precise, safe, and helpful. @@ -71,10 +67,17 @@ const formatUserPrompt = (messages: BaseMessage[]): string => { ); }; +interface DiagnoseErrorInputs { + messages: BaseMessage[]; + codebaseTree: string; +} + +type DiagnoseErrorUpdate = Partial; + export async function diagnoseError( - state: PlannerGraphState, + state: DiagnoseErrorInputs, config: GraphConfig, -): Promise { +): Promise { const lastFailedAction = state.messages.findLast( (m) => isToolMessage(m) && m.status === "error", ); @@ -82,7 +85,7 @@ export async function diagnoseError( throw new Error("No failed action found in messages"); } - logger.info("The last two tool calls resulted in errors. Diagnosing error."); + logger.info("The last few tool calls resulted in errors. Diagnosing error."); const model = await loadModel(config, Task.SUMMARIZER); const modelWithTools = model.bindTools([diagnoseErrorTool], { diff --git a/apps/open-swe/src/tools/apply-patch.ts b/apps/open-swe/src/tools/apply-patch.ts index 64ea6ee9..14fa1819 100644 --- a/apps/open-swe/src/tools/apply-patch.ts +++ b/apps/open-swe/src/tools/apply-patch.ts @@ -7,9 +7,77 @@ import { createLogger, LogLevel } from "../utils/logger.js"; import { createApplyPatchToolFields } from "@open-swe/shared/open-swe/tools"; import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { getSandboxSessionOrThrow } from "./utils/get-sandbox-id.js"; +import * as fs from "fs/promises"; +import * as path from "path"; +import * as os from "os"; const logger = createLogger(LogLevel.INFO, "ApplyPatchTool"); +/** + * Attempts to apply a patch using Git CLI + * @param sandbox The sandbox session + * @param workDir The working directory + * @param diffContent The diff content + * @returns Object with success status and output or error message + */ +async function applyPatchWithGit( + sandbox: any, + workDir: string, + diffContent: string, +): Promise<{ success: boolean; output: string }> { + let tempDir = ""; + try { + // Create a temporary file to store the diff + tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "git-patch-")); + const tempPatchFile = path.join(tempDir, "patch.diff"); + + // Write the diff to the temporary file + await fs.writeFile(tempPatchFile, diffContent, "utf8"); + + // Execute git apply with --verbose for detailed error messages + const response = await sandbox.process.executeCommand( + `git apply --verbose "${tempPatchFile}"`, + workDir, + {}, + 30, // 30 seconds timeout + ); + + if (response.exitCode !== 0) { + return { + success: false, + output: `Git apply failed with exit code ${response.exitCode}:\n${response.result || response.artifacts?.stdout || "No error output"}`, + }; + } + + return { + success: true, + output: response.result || "Patch applied successfully", + }; + } catch (error) { + return { + success: false, + output: + error instanceof Error + ? error.message + : "Unknown error applying patch with git", + }; + } finally { + // Clean up the temporary file after git apply has completed + if (tempDir) { + try { + await fs.rm(tempDir, { recursive: true, force: true }); + } catch (cleanupError) { + logger.warn(`Failed to clean up temporary directory: ${tempDir}`, { + error: + cleanupError instanceof Error + ? cleanupError.message + : String(cleanupError), + }); + } + } + } +} + export function createApplyPatchTool(state: GraphState) { const applyPatchTool = tool( async (input): Promise<{ result: string; status: "success" | "error" }> => { @@ -25,51 +93,87 @@ export function createApplyPatchTool(state: GraphState) { workDir, }); if (!readFileSuccess) { - logger.error(readFileOutput); throw new Error(readFileOutput); } + // First try to apply the patch using Git CLI for better error messages + logger.info(`Attempting to apply patch to ${file_path} using Git CLI`); + const gitResult = await applyPatchWithGit(sandbox, workDir, diff); + + // If Git successfully applied the patch, read the updated file and return success + if (gitResult.success) { + const { success: readUpdatedFileSuccess, output: updatedContent } = + await readFile({ + sandbox, + filePath: file_path, + workDir, + }); + + if (!readUpdatedFileSuccess) { + throw new Error( + `Failed to read updated file after applying patch: ${updatedContent}`, + ); + } + + logger.info(`Successfully applied diff to ${file_path} using Git CLI`); + return { + result: `Successfully applied diff to \`${file_path}\` and saved changes.`, + status: "success", + }; + } + + // If Git failed, fall back to the diff library with detailed error capture + logger.warn( + `Git CLI patch application failed: ${gitResult.output}. Falling back to diff library.`, + ); + let patchedContent: string | false; let fixedDiff: string | false = false; let errorApplyingPatchMessage: string | undefined; + try { - logger.info(`Applying patch to file ${file_path}`); + logger.info(`Applying patch to file ${file_path} using diff library`); patchedContent = applyPatch(readFileOutput, diff); } catch (e) { - errorApplyingPatchMessage = e instanceof Error ? e.message : undefined; + errorApplyingPatchMessage = + e instanceof Error ? e.message : "Unknown error"; try { - logger.warn("Failed to apply patch, trying to fix diff", { - error: e, - }); + logger.warn( + "Failed to apply patch: Invalid diff. Attempting to fix", + { + ...(e instanceof Error + ? { name: e.name, message: e.message, stack: e.stack } + : { error: e }), + }, + ); const fixedDiff_ = fixGitPatch(diff, { [file_path]: readFileOutput, }); patchedContent = applyPatch(readFileOutput, fixedDiff_); - logger.info("Successfully fixed diff and applied patch to file", { - file_path, - }); if (patchedContent) { + logger.info("Successfully fixed diff and applied patch to file", { + file_path, + }); fixedDiff = fixedDiff_; } } catch (_) { - logger.error("Failed to apply patch", { - ...(e instanceof Error - ? { name: e.name, message: e.message, stack: e.stack } - : { error: e }), - }); - const errMessage = e instanceof Error ? e.message : "Unknown error"; + // Combine both Git and diff library error messages for maximum context + const diffErrMessage = + e instanceof Error ? e.message : "Unknown error"; throw new Error( - `FAILED TO APPLY PATCH: The diff could not be applied to file '${file_path}'.\n\nError: ${errMessage}`, + `FAILED TO APPLY PATCH: The diff could not be applied to file '${file_path}'.\n\n` + + `Git Error: ${gitResult.output}\n\n` + + `Diff Library Error: ${diffErrMessage}`, ); } } if (patchedContent === false) { - logger.error( - `FAILED TO APPLY PATCH: The diff could not be applied to file '${file_path}'. This may be due to an invalid diff format or conflicting changes with the file's current content. Original content length: ${readFileOutput.length}, Diff: ${diff.substring(0, 100)}...`, - ); throw new Error( - `FAILED TO APPLY PATCH: The diff could not be applied to file '${file_path}'. This may be due to an invalid diff format or conflicting changes with the file's current content. Original content length: ${readFileOutput.length}, Diff: ${diff.substring(0, 100)}...`, + `FAILED TO APPLY PATCH: The diff could not be applied to file '${file_path}'.\n\n` + + `Git Error: ${gitResult.output}\n\n` + + `This may be due to an invalid diff format or conflicting changes with the file's current content. ` + + `Original content length: ${readFileOutput.length}, Diff: ${diff.substring(0, 100)}...`, ); } @@ -81,9 +185,6 @@ export function createApplyPatchTool(state: GraphState) { workDir, }); if (!writeFileSuccess) { - logger.error("Failed to write file", { - writeFileOutput, - }); throw new Error(writeFileOutput); } @@ -95,6 +196,10 @@ export function createApplyPatchTool(state: GraphState) { `\nHere is the error that was thrown when your generated diff was applied:\n\n${errorApplyingPatchMessage}\n` + `\nThe diff which was applied is:\n\n${fixedDiff}\n`; } + + // Include Git error for context even on success + resultMessage += `\n\nGit apply attempt failed with message:\n\n${gitResult.output}\n`; + return { result: resultMessage, status: "success", diff --git a/apps/open-swe/src/tools/find-instances-of.ts b/apps/open-swe/src/tools/find-instances-of.ts deleted file mode 100644 index 0b40b8ee..00000000 --- a/apps/open-swe/src/tools/find-instances-of.ts +++ /dev/null @@ -1,137 +0,0 @@ -import { tool } from "@langchain/core/tools"; -import { GraphState } from "@open-swe/shared/open-swe/types"; -import { getSandboxErrorFields } from "../utils/sandbox-error-fields.js"; -import { createLogger, LogLevel } from "../utils/logger.js"; -import { TIMEOUT_SEC } from "@open-swe/shared/constants"; -import { createFindInstancesOfToolFields } from "@open-swe/shared/open-swe/tools"; -import { getRepoAbsolutePath } from "@open-swe/shared/git"; -import { z } from "zod"; -import { wrapScript } from "../utils/wrap-script.js"; -import { getSandboxSessionOrThrow } from "./utils/get-sandbox-id.js"; - -const logger = createLogger(LogLevel.INFO, "FindInstancesOfTool"); - -const DEFAULT_ENV = { - // Prevents corepack from showing a y/n download prompt which causes the command to hang - COREPACK_ENABLE_DOWNLOAD_PROMPT: "0", -}; - -export function createFindInstancesOfTool( - state: Pick, -) { - const findInstancesOfFields = createFindInstancesOfToolFields( - state.targetRepository, - ); - const formatFindInstancesOfCommand = ( - input: z.infer, - ): string[] => { - const args = ["rg"]; - - // Always include these flags for consistent output - args.push("--color", "never", "--line-number", "--heading"); - - // Add context lines (3 above and 3 below) - args.push("-A", "3", "-B", "3"); - - // Handle case sensitivity - if (!input.case_sensitive) { - args.push("-i"); - } - - // Handle word matching - if (input.match_word) { - args.push("--word-regexp"); - } - - // Handle file inclusion/exclusion patterns - if (input.exclude_files) { - args.push("-g", `!${input.exclude_files}`); - } - - if (input.include_files) { - args.push("-g", input.include_files); - } - - // For literal string matching (not regex) - args.push("--fixed-strings"); - - // Add the search query as the last argument (ensure it's properly quoted) - const formattedQuery = `'${input.query.replace(/^'|'$/g, "")}'`; - args.push(formattedQuery); - - return args; - }; - - const findInstancesOfTool = tool( - async ( - input: z.infer, - ): Promise<{ result: string; status: "success" | "error" }> => { - try { - const sandbox = await getSandboxSessionOrThrow(input); - - const repoRoot = getRepoAbsolutePath(state.targetRepository); - const command = formatFindInstancesOfCommand(input); - logger.info("Running find_instances_of command", { - command: command.join(" "), - repoRoot, - }); - - const response = await sandbox.process.executeCommand( - wrapScript(command.join(" ")), - repoRoot, - DEFAULT_ENV, - TIMEOUT_SEC, - ); - - let successResult = response.result; - - if ( - response.exitCode === 1 || - (response.exitCode === 127 && response.result.startsWith("sh: 1: ")) - ) { - logger.info("Exit code 1. no results found", { - ...response, - }); - successResult = `Exit code 1. No results found.\n\n${response.result}`; - } else if (response.exitCode > 1) { - logger.error("Failed to run find_instances_of command", { - error: response.result, - error_result: response, - input, - }); - throw new Error("Command failed. Exit code: " + response.exitCode); - } - - return { - result: successResult, - status: "success", - }; - } catch (e) { - const errorFields = getSandboxErrorFields(e); - if (errorFields) { - logger.error("Failed to run find_instances_of command", { - input, - error: errorFields, - }); - throw new Error("Command failed. Exit code: " + errorFields.exitCode); - } - - logger.error( - "Failed to run find_instances_of command: " + - (e instanceof Error ? e.message : "Unknown error"), - { - error: e, - input, - }, - ); - throw new Error( - "FAILED TO RUN FIND_INSTANCES_OF COMMAND: " + - (e instanceof Error ? e.message : "Unknown error"), - ); - } - }, - findInstancesOfFields, - ); - - return findInstancesOfTool; -} diff --git a/apps/open-swe/src/tools/index.ts b/apps/open-swe/src/tools/index.ts index 0dd82f54..c7d45258 100644 --- a/apps/open-swe/src/tools/index.ts +++ b/apps/open-swe/src/tools/index.ts @@ -6,3 +6,6 @@ export { createSessionPlanToolFields, createRequestHumanHelpToolFields, } from "@open-swe/shared/open-swe/tools"; +export * from "./search.js"; +export * from "./install-dependencies.js"; +export * from "./planner-notes.js"; diff --git a/apps/open-swe/src/tools/install-dependencies.ts b/apps/open-swe/src/tools/install-dependencies.ts index 8a08070f..efb72077 100644 --- a/apps/open-swe/src/tools/install-dependencies.ts +++ b/apps/open-swe/src/tools/install-dependencies.ts @@ -37,13 +37,9 @@ export function createInstallDependenciesTool( ); if (response.exitCode !== 0) { - logger.error("Failed to install dependencies", { - error: response.result, - error_result: response, - input, - }); + const errorResult = response.result ?? response.artifacts?.stdout; throw new Error( - `Command failed. Exit code: ${response.exitCode}\nResult: ${response.result}\nStdout:\n${response.artifacts?.stdout}`, + `Failed to install dependencies. Exit code: ${response.exitCode}\nError: ${errorResult}`, ); } @@ -54,27 +50,14 @@ export function createInstallDependenciesTool( } catch (e) { const errorFields = getSandboxErrorFields(e); if (errorFields) { - logger.error("Failed to install dependencies", { - input, - error: errorFields, - }); + const errorResult = + errorFields.result ?? errorFields.artifacts?.stdout; throw new Error( - `Command failed. Exit code: ${errorFields.exitCode}\nError: ${errorFields.result ?? errorFields.artifacts?.stdout}`, + `Failed to install dependencies. Exit code: ${errorFields.exitCode}\nError: ${errorResult}`, ); } - logger.error( - "Failed to install dependencies: " + - (e instanceof Error ? e.message : "Unknown error"), - { - error: e, - input, - }, - ); - throw new Error( - "FAILED TO INSTALL DEPENDENCIES: " + - (e instanceof Error ? e.message : "Unknown error"), - ); + throw e; } }, createInstallDependenciesToolFields(state.targetRepository), diff --git a/apps/open-swe/src/tools/rg.ts b/apps/open-swe/src/tools/search.ts similarity index 55% rename from apps/open-swe/src/tools/rg.ts rename to apps/open-swe/src/tools/search.ts index 0592046d..e9ad669d 100644 --- a/apps/open-swe/src/tools/rg.ts +++ b/apps/open-swe/src/tools/search.ts @@ -4,35 +4,31 @@ import { getSandboxErrorFields } from "../utils/sandbox-error-fields.js"; import { createLogger, LogLevel } from "../utils/logger.js"; import { TIMEOUT_SEC } from "@open-swe/shared/constants"; import { - createRgToolFields, - formatRgCommand, + createSearchToolFields, + formatSearchCommand, } from "@open-swe/shared/open-swe/tools"; import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { wrapScript } from "../utils/wrap-script.js"; import { getSandboxSessionOrThrow } from "./utils/get-sandbox-id.js"; -const logger = createLogger(LogLevel.INFO, "RgTool"); +const logger = createLogger(LogLevel.INFO, "SearchTool"); const DEFAULT_ENV = { // Prevents corepack from showing a y/n download prompt which causes the command to hang COREPACK_ENABLE_DOWNLOAD_PROMPT: "0", }; -export function createRgTool( +export function createSearchTool( state: Pick, ) { - const rgTool = tool( + const searchTool = tool( async (input): Promise<{ result: string; status: "success" | "error" }> => { try { const sandbox = await getSandboxSessionOrThrow(input); const repoRoot = getRepoAbsolutePath(state.targetRepository); - const command = formatRgCommand({ - pattern: input.pattern, - paths: input.paths, - flags: input.flags, - }); - logger.info("Running rg command", { + const command = formatSearchCommand(input); + logger.info("Running search command", { command: command.join(" "), repoRoot, }); @@ -49,18 +45,12 @@ export function createRgTool( response.exitCode === 1 || (response.exitCode === 127 && response.result.startsWith("sh: 1: ")) ) { - logger.info("Exit code 1. no results found", { - ...response, - }); - successResult = `Exit code 1. No results found.\n\n${response.result}`; + const errorResult = response.result ?? response.artifacts?.stdout; + successResult = `Exit code 1. No results found.\n\n${errorResult}`; } else if (response.exitCode > 1) { - logger.error("Failed to run rg command", { - error: response.result, - error_result: response, - input, - }); + const errorResult = response.result ?? response.artifacts?.stdout; throw new Error( - `Command failed. Exit code: ${response.exitCode}\nResult: ${response.result}\nStdout:\n${response.artifacts?.stdout}`, + `Failed to run search command. Exit code: ${response.exitCode}\nError: ${errorResult}`, ); } @@ -71,31 +61,18 @@ export function createRgTool( } catch (e) { const errorFields = getSandboxErrorFields(e); if (errorFields) { - logger.error("Failed to run rg command", { - input, - error: errorFields, - }); + const errorResult = + errorFields.result ?? errorFields.artifacts?.stdout; throw new Error( - `Command failed. Exit code: ${errorFields.exitCode}\nError: ${errorFields.result ?? errorFields.artifacts?.stdout}`, + `Failed to run search command. Exit code: ${errorFields.exitCode}\nError: ${errorResult}`, ); } - logger.error( - "Failed to run rg command: " + - (e instanceof Error ? e.message : "Unknown error"), - { - error: e, - input, - }, - ); - throw new Error( - "FAILED TO RUN RG COMMAND: " + - (e instanceof Error ? e.message : "Unknown error"), - ); + throw e; } }, - createRgToolFields(state.targetRepository), + createSearchToolFields(state.targetRepository), ); - return rgTool; + return searchTool; } diff --git a/apps/open-swe/src/tools/shell.ts b/apps/open-swe/src/tools/shell.ts index a1b22dae..7c467d6c 100644 --- a/apps/open-swe/src/tools/shell.ts +++ b/apps/open-swe/src/tools/shell.ts @@ -1,13 +1,10 @@ import { tool } from "@langchain/core/tools"; import { GraphState } from "@open-swe/shared/open-swe/types"; import { getSandboxErrorFields } from "../utils/sandbox-error-fields.js"; -import { createLogger, LogLevel } from "../utils/logger.js"; import { TIMEOUT_SEC } from "@open-swe/shared/constants"; import { createShellToolFields } from "@open-swe/shared/open-swe/tools"; import { getSandboxSessionOrThrow } from "./utils/get-sandbox-id.js"; -const logger = createLogger(LogLevel.INFO, "ShellTool"); - const DEFAULT_ENV = { // Prevents corepack from showing a y/n download prompt which causes the command to hang COREPACK_ENABLE_DOWNLOAD_PROMPT: "0", @@ -30,13 +27,9 @@ export function createShellTool( ); if (response.exitCode !== 0) { - logger.error("Failed to run command", { - error: response.result, - error_result: response, - input, - }); + const errorResult = response.result ?? response.artifacts?.stdout; throw new Error( - `Command failed. Exit code: ${response.exitCode}\nResult: ${response.result}\nStdout:\n${response.artifacts?.stdout}`, + `Command failed. Exit code: ${response.exitCode}\nResult: ${errorResult}`, ); } @@ -47,27 +40,14 @@ export function createShellTool( } catch (e) { const errorFields = getSandboxErrorFields(e); if (errorFields) { - logger.error("Failed to run command", { - input, - error: errorFields, - }); + const errorResult = + errorFields.result ?? errorFields.artifacts?.stdout; throw new Error( - `Command failed. Exit code: ${errorFields.exitCode}\nError: ${errorFields.result}\nStdout:\n${errorFields.artifacts?.stdout}`, + `Command failed. Exit code: ${errorFields.exitCode}\nError: ${errorResult}`, ); } - logger.error( - "Failed to run command: " + - (e instanceof Error ? e.message : "Unknown error"), - { - error: e, - input, - }, - ); - throw new Error( - "FAILED TO RUN COMMAND: " + - (e instanceof Error ? e.message : "Unknown error"), - ); + throw e; } }, createShellToolFields(state.targetRepository), diff --git a/apps/open-swe/src/utils/read-write.ts b/apps/open-swe/src/utils/read-write.ts index c0413eab..d4e1ef4d 100644 --- a/apps/open-swe/src/utils/read-write.ts +++ b/apps/open-swe/src/utils/read-write.ts @@ -49,12 +49,10 @@ async function readFileFunc(inputs: { ); if (readOutput.exitCode !== 0) { - logger.error(`Error reading file '${filePath}' from sandbox via cat:`, { - readOutput, - }); + const errorResult = readOutput.result ?? readOutput.artifacts?.stdout; return { success: false, - output: `FAILED TO READ FILE from sandbox '${filePath}'. Exit code: ${readOutput.exitCode}.\nResult: ${readOutput.result}\nStdout: ${readOutput.artifacts?.stdout}`, + output: `FAILED TO READ FILE from sandbox '${filePath}'. Exit code: ${readOutput.exitCode}.\nResult: ${errorResult}`, }; } @@ -89,7 +87,9 @@ async function readFileFunc(inputs: { let outputMessage = `FAILED TO EXECUTE READ COMMAND for sandbox '${filePath}'.`; const errorFields = getSandboxErrorFields(e); if (errorFields) { - outputMessage += `\nExit code: ${errorFields.exitCode}\nResult: ${errorFields.result}\nStdout: ${errorFields.artifacts?.stdout}`; + const errorResult = errorFields.result ?? errorFields.artifacts?.stdout; + + outputMessage += `\nExit code: ${errorFields.exitCode}\nResult: ${errorResult}`; } else { outputMessage += ` Error: ${(e as Error).message || String(e)}`; } @@ -134,12 +134,10 @@ ${delimiter}`; ); if (writeOutput.exitCode !== 0) { - logger.error(`Error writing file '${filePath}' to sandbox via cat:`, { - writeOutput, - }); + const errorResult = writeOutput.result ?? writeOutput.artifacts?.stdout; return { success: false, - output: `FAILED TO WRITE FILE to sandbox '${filePath}'. Exit code: ${writeOutput.exitCode}\nResult: ${writeOutput.result}\nStdout: ${writeOutput.artifacts?.stdout}`, + output: `FAILED TO WRITE FILE to sandbox '${filePath}'. Exit code: ${writeOutput.exitCode}\nResult: ${errorResult}`, }; } return { @@ -159,7 +157,8 @@ ${delimiter}`; let outputMessage = `FAILED TO EXECUTE WRITE COMMAND for sandbox '${filePath}'.`; const errorFields = getSandboxErrorFields(e); if (errorFields) { - outputMessage += `\nExit code: ${errorFields.exitCode}\nResult: ${errorFields.result}\nStdout: ${errorFields.artifacts?.stdout}`; + const errorResult = errorFields.result ?? errorFields.artifacts?.stdout; + outputMessage += `\nExit code: ${errorFields.exitCode}\nResult: ${errorResult}`; } else { outputMessage += ` Error: ${(e as Error).message || String(e)}`; } diff --git a/apps/open-swe/src/utils/review.ts b/apps/open-swe/src/utils/review.ts new file mode 100644 index 00000000..12656098 --- /dev/null +++ b/apps/open-swe/src/utils/review.ts @@ -0,0 +1,45 @@ +import { BaseMessage, isAIMessage } from "@langchain/core/messages"; +import { createCodeReviewMarkTaskNotCompleteFields } from "@open-swe/shared/open-swe/tools"; +import { z } from "zod"; + +export function getCodeReviewFields( + messages: BaseMessage[], +): { review: string; newActions: string[] } | null { + const codeReviewToolFields = createCodeReviewMarkTaskNotCompleteFields(); + const codeReviewMessage = messages + .filter(isAIMessage) + .findLast( + (m) => + m.tool_calls?.length && + m.tool_calls.some((tc) => tc.name === codeReviewToolFields.name), + ); + const codeReviewToolCall = codeReviewMessage?.tool_calls?.find( + (tc) => tc.name === codeReviewToolFields.name, + ); + if (!codeReviewMessage || !codeReviewToolCall) return null; + const codeReviewArgs = codeReviewToolCall.args as z.infer< + typeof codeReviewToolFields.schema + >; + if (!codeReviewArgs.review || !codeReviewArgs.additional_actions?.length) + return null; + + return { + review: codeReviewArgs.review, + newActions: codeReviewArgs.additional_actions, + }; +} + +export function formatCodeReviewPrompt( + reviewPrompt: string, + inputs: { + review: string; + newActions: string[]; + }, +): string { + return reviewPrompt + .replaceAll("{CODE_REVIEW}", inputs.review) + .replaceAll( + "{CODE_REVIEW_ACTIONS}", + inputs.newActions.map((a) => `* ${a}`).join("\n"), + ); +} diff --git a/apps/web/package.json b/apps/web/package.json index 263f6916..a2450d75 100644 --- a/apps/web/package.json +++ b/apps/web/package.json @@ -20,7 +20,7 @@ }, "dependencies": { "@langchain/core": "^0.3.57", - "@langchain/langgraph": "^0.3.3", + "@langchain/langgraph": "^0.3.8", "@langchain/langgraph-sdk": "^0.0.92", "@octokit/app": "^16.0.1", "@open-swe/shared": "*", diff --git a/apps/web/src/components/gen-ui/action-step.tsx b/apps/web/src/components/gen-ui/action-step.tsx index 3378b17a..d05777d8 100644 --- a/apps/web/src/components/gen-ui/action-step.tsx +++ b/apps/web/src/components/gen-ui/action-step.tsx @@ -1,35 +1,28 @@ "use client"; -import { JSX, useState } from "react"; +import { useState } from "react"; import { Terminal, FileText, ChevronDown, - ChevronRight, ChevronUp, MessageSquare, Search, - AlertCircle, CheckCircle, XCircle, Loader2, Globe, - Pencil, - Package, FileCode, CloudDownload, - Hash, } from "lucide-react"; -import { MarkdownText } from "../thread/markdown-text"; +import { BasicMarkdownText } from "../thread/markdown-text"; import { createApplyPatchToolFields, createShellToolFields, createInstallDependenciesToolFields, createTakePlannerNotesFields, createGetURLContentToolFields, - createFindInstancesOfToolFields, - formatRgCommand, - RipgrepCommand, + createSearchToolFields, } from "@open-swe/shared/open-swe/tools"; import { z } from "zod"; import { @@ -55,8 +48,8 @@ const plannerNotesTool = createTakePlannerNotesFields(); type PlannerNotesToolArgs = z.infer; const getURLContentTool = createGetURLContentToolFields(); type GetURLContentToolArgs = z.infer; -const findInstancesOfTool = createFindInstancesOfToolFields(dummyRepo); -type FindInstancesOfToolArgs = z.infer; +const searchTool = createSearchToolFields(dummyRepo); +type SearchToolArgs = z.infer; // Common props for all action types type BaseActionProps = { @@ -83,13 +76,6 @@ type PatchActionProps = BaseActionProps & fixedDiff?: string; }; -type RgActionProps = BaseActionProps & - Partial & { - actionType: "rg"; - output?: string; - errorCode?: number; - }; - type InstallDependenciesActionProps = BaseActionProps & Partial & { actionType: "install_dependencies"; @@ -108,9 +94,9 @@ type GetURLContentActionProps = BaseActionProps & output?: string; }; -type FindInstancesOfActionProps = BaseActionProps & - Partial & { - actionType: "find_instances_of"; +type SearchActionProps = BaseActionProps & + Partial & { + actionType: "search"; output?: string; errorCode?: number; }; @@ -119,11 +105,10 @@ export type ActionItemProps = | (BaseActionProps & { status: "loading" }) | ShellActionProps | PatchActionProps - | RgActionProps | InstallDependenciesActionProps | PlannerNotesActionProps | GetURLContentActionProps - | FindInstancesOfActionProps; + | SearchActionProps; export type ActionStepProps = { actions: ActionItemProps[]; @@ -134,11 +119,10 @@ export type ActionStepProps = { const ACTION_GENERATING_TEXT_MAP = { [shellTool.name]: "Executing...", [applyPatchTool.name]: "Applying patch...", - ["rg"]: "Searching...", [installDependenciesTool.name]: "Installing dependencies...", [plannerNotesTool.name]: "Saving notes...", [getURLContentTool.name]: "Fetching URL content...", - [findInstancesOfTool.name]: "Finding instances...", + [searchTool.name]: "Searching...", }; function MatchCaseIcon({ matchCase }: { matchCase: boolean }) { @@ -237,8 +221,6 @@ function ActionItem(props: ActionItemProps) { return props.success ? "Command completed" : "Command failed"; } else if (props.actionType === "apply-patch") { return props.success ? "Patch applied" : "Patch failed"; - } else if (props.actionType === "rg") { - return props.success ? "Search completed" : "Search failed"; } else if (props.actionType === "install_dependencies") { return props.success ? "Dependencies installed" : "Installation failed"; } else if (props.actionType === "planner_notes") { @@ -247,7 +229,7 @@ function ActionItem(props: ActionItemProps) { return props.success ? "URL content fetched" : "Failed to fetch URL content"; - } else if (props.actionType === "find_instances_of") { + } else if (props.actionType === "search") { return props.success ? "Search completed" : "Search failed"; } } @@ -261,10 +243,9 @@ function ActionItem(props: ActionItemProps) { if ( props.actionType === "shell" || - props.actionType === "rg" || props.actionType === "install_dependencies" || props.actionType === "get_url_content" || - props.actionType === "find_instances_of" + props.actionType === "search" ) { return !!props.output; } else if (props.actionType === "apply-patch") { @@ -310,13 +291,6 @@ function ActionItem(props: ActionItemProps) { icon={} /> ); - } else if (props.actionType === "rg") { - return ( - } - /> - ); } else if (props.actionType === "get_url_content") { return ( } /> ); - } else if (props.actionType === "find_instances_of") { + } else if (props.actionType === "search") { return ( } + toolNamePretty="Search" + icon={} /> ); } else { @@ -353,7 +327,7 @@ function ActionItem(props: ActionItemProps) { if (props.actionType === "planner_notes") { return ( -
+
Planner Notes @@ -363,7 +337,7 @@ function ActionItem(props: ActionItemProps) { if (props.actionType === "get_url_content") { return ( -
+
{props.url} @@ -371,28 +345,44 @@ function ActionItem(props: ActionItemProps) { ); } - if (props.actionType === "find_instances_of") { + if (props.actionType === "search") { + const castProps = props as SearchActionProps; return ( -
-
- - {props.query} - -
- - -
- {(props.include_files || props.exclude_files) && ( -
- {props.include_files && ( - Include: {props.include_files} - )} - {props.include_files && props.exclude_files && | } - {props.exclude_files && ( - Exclude: {props.exclude_files} +
+
+
+ + {castProps.pattern} + +
+ + {castProps.regex && ( + + regex + )}
- )} +
+
+ {castProps.include_files && ( + Include: {castProps.include_files} + )} + {castProps.exclude_files && ( + Exclude: {castProps.exclude_files} + )} + {castProps.context_lines !== undefined && + castProps.context_lines > 0 && ( + Context: {castProps.context_lines} lines + )} + {castProps.max_results !== undefined && + castProps.max_results > 0 && ( + Max results: {castProps.max_results} + )} + {castProps.file_types && castProps.file_types.length > 0 && ( + File types: {castProps.file_types.join(", ")} + )} + {castProps.follow_symlinks && Follow symlinks} +
); } @@ -417,7 +407,7 @@ function ActionItem(props: ActionItemProps) { } } return ( -
+
{props.workdir && (
{props.workdir} @@ -428,31 +418,9 @@ function ActionItem(props: ActionItemProps) {
); - } else if (props.actionType === "rg") { - let formattedRgCommand = ""; - try { - formattedRgCommand = - formatRgCommand( - { - pattern: props.pattern, - paths: props.paths, - flags: props.flags, - }, - { excludeRequiredFlags: true }, - )?.join(" ") ?? ""; - } catch { - // no-op - } - return ( -
- - {formattedRgCommand} - -
- ); } else { return ( - + {props.file_path} ); @@ -467,8 +435,7 @@ function ActionItem(props: ActionItemProps) { if ( (props.actionType === "shell" || - props.actionType === "rg" || - props.actionType === "find_instances_of" || + props.actionType === "search" || props.actionType === "install_dependencies") && props.output ) { @@ -547,10 +514,10 @@ function ActionItem(props: ActionItemProps) { return (
-
+
{renderHeaderIcon()} {renderHeaderContent()} -
+
{getStatusText()} @@ -558,7 +525,7 @@ function ActionItem(props: ActionItemProps) { {shouldShowToggle() && ( {showSummary && ( -

- {summaryText} -

+ + {summaryText} + )}
)} diff --git a/apps/web/src/components/gen-ui/pull-request-opened.tsx b/apps/web/src/components/gen-ui/pull-request-opened.tsx index 5dc3167c..971a4e8e 100644 --- a/apps/web/src/components/gen-ui/pull-request-opened.tsx +++ b/apps/web/src/components/gen-ui/pull-request-opened.tsx @@ -35,10 +35,12 @@ export function PullRequestOpened({ switch (status) { case "loading": return ( -
+
); case "generating": - return ; + return ( + + ); case "done": return ; } @@ -62,28 +64,28 @@ export function PullRequestOpened({ }; return ( -
-
- +
+
+
{title && status === "done" && ( -
+
{title}
)} {branch && status === "done" && ( -
+
{branch} → {targetBranch}
)} {!title && ( - + {getStatusText()} )}
- + {getStatusText()} {getStatusIcon()} @@ -92,7 +94,7 @@ export function PullRequestOpened({ href={url} target="_blank" rel="noopener noreferrer" - className="text-gray-500 hover:text-gray-700" + className="text-gray-500 hover:text-gray-700 dark:text-gray-400 dark:hover:text-gray-300" title="Open pull request" > @@ -101,7 +103,7 @@ export function PullRequestOpened({ {shouldShowToggle() && (
{expanded && description && status === "done" && ( -
-

+
+

Description

-
+
{description}
diff --git a/apps/web/src/components/gen-ui/push-changes.tsx b/apps/web/src/components/gen-ui/push-changes.tsx index daf613e0..165a496c 100644 --- a/apps/web/src/components/gen-ui/push-changes.tsx +++ b/apps/web/src/components/gen-ui/push-changes.tsx @@ -12,6 +12,7 @@ import { MessageSquare, FileText, } from "lucide-react"; +import { BasicMarkdownText } from "../thread/markdown-text"; type PushChangesProps = { status: "loading" | "generating" | "done"; @@ -35,8 +36,8 @@ export function PushChanges({ const [expanded, setExpanded] = useState( Boolean(status === "done" && gitStatus), ); - const [showReasoning, setShowReasoning] = useState(false); - const [showSummary, setShowSummary] = useState(false); + const [showReasoning, setShowReasoning] = useState(true); + const [showSummary, setShowSummary] = useState(true); const getStatusIcon = () => { switch (status) { @@ -152,9 +153,9 @@ export function PushChanges({ {showSummary ? "Hide summary" : "Show summary"} {showSummary && ( -

+ {summaryText} -

+ )}
)} diff --git a/apps/web/src/components/gen-ui/replanning-step.tsx b/apps/web/src/components/gen-ui/replanning-step.tsx index 70571b63..982ddd69 100644 --- a/apps/web/src/components/gen-ui/replanning-step.tsx +++ b/apps/web/src/components/gen-ui/replanning-step.tsx @@ -9,6 +9,7 @@ import { FileText, } from "lucide-react"; import { useState } from "react"; +import { BasicMarkdownText } from "../thread/markdown-text"; type ReplanningStepProps = { status: "loading" | "generating" | "done"; @@ -21,8 +22,8 @@ export function ReplanningStep({ reasoningText, summaryText, }: ReplanningStepProps) { - const [showReasoning, setShowReasoning] = useState(false); - const [showSummary, setShowSummary] = useState(false); + const [showReasoning, setShowReasoning] = useState(true); + const [showSummary, setShowSummary] = useState(true); const getStatusIcon = () => { switch (status) { @@ -85,9 +86,9 @@ export function ReplanningStep({ {showSummary ? "Hide summary" : "Show summary"} {showSummary && ( -

+ {summaryText} -

+ )}

)} diff --git a/apps/web/src/components/gen-ui/task-review.tsx b/apps/web/src/components/gen-ui/task-review.tsx new file mode 100644 index 00000000..2f44a97b --- /dev/null +++ b/apps/web/src/components/gen-ui/task-review.tsx @@ -0,0 +1,290 @@ +"use client"; + +import { useState } from "react"; +import { + CheckCircle, + XCircle, + Loader2, + ChevronDown, + ChevronUp, + MessageSquare, + FileText, +} from "lucide-react"; +import { BasicMarkdownText } from "../thread/markdown-text"; + +type MarkTaskCompletedProps = { + status: "loading" | "generating" | "done"; + review?: string; + reasoningText?: string; + summaryText?: string; +}; + +export function MarkTaskCompleted({ + status, + review, + reasoningText, + summaryText, +}: MarkTaskCompletedProps) { + const [expanded, setExpanded] = useState(!!(status === "done" && review)); + const [showReasoning, setShowReasoning] = useState(true); + const [showSummary, setShowSummary] = useState(true); + + const getStatusIcon = () => { + switch (status) { + case "loading": + return ( +
+ ); + case "generating": + return ( + + ); + case "done": + return ( + + ); + } + }; + + const getStatusText = () => { + switch (status) { + case "loading": + return "Preparing task review..."; + case "generating": + return "Reviewing task completion..."; + case "done": + return "Task marked as completed"; + } + }; + + return ( +
+ {reasoningText && ( +
+ + {showReasoning && ( +

+ {reasoningText} +

+ )} +
+ )} + +
setExpanded((prev) => !prev) + : undefined + } + > + + + {getStatusText()} + +
+ {getStatusIcon()} + {status === "done" && review && ( + + )} +
+
+ + {expanded && review && status === "done" && ( +
+

+ Final Review +

+ + {review} + +
+ )} + + {summaryText && status === "done" && ( +
+ + {showSummary && ( + + {summaryText} + + )} +
+ )} +
+ ); +} + +type MarkTaskIncompleteProps = { + status: "loading" | "generating" | "done"; + review?: string; + additionalActions?: string[]; + reasoningText?: string; + summaryText?: string; +}; + +export function MarkTaskIncomplete({ + status, + review, + additionalActions, + reasoningText, + summaryText, +}: MarkTaskIncompleteProps) { + const [expanded, setExpanded] = useState( + !!(status === "done" && (review || additionalActions)), + ); + const [showReasoning, setShowReasoning] = useState(true); + const [showSummary, setShowSummary] = useState(true); + + const getStatusIcon = () => { + switch (status) { + case "loading": + return ( +
+ ); + case "generating": + return ; + case "done": + return ; + } + }; + + const getStatusText = () => { + switch (status) { + case "loading": + return "Preparing task review..."; + case "generating": + return "Reviewing task completion..."; + case "done": + return "Task marked as incomplete"; + } + }; + + return ( +
+ {reasoningText && ( +
+ + {showReasoning && ( +

+ {reasoningText} +

+ )} +
+ )} + +
setExpanded((prev) => !prev) + : undefined + } + > + + + {getStatusText()} + +
+ {getStatusIcon()} + {status === "done" && (review || additionalActions) && ( + + )} +
+
+ + {expanded && status === "done" && (review || additionalActions) && ( +
+ {review && ( +
+

+ Final Review +

+ + {review} + +
+ )} + + {additionalActions && additionalActions.length > 0 && ( +
+

+ Additional Actions Required ({additionalActions.length}) +

+
    + {additionalActions.map((action, index) => ( +
  1. +
    + + {index + 1} + +
    + + {action} + +
  2. + ))} +
+
+ )} +
+ )} + + {summaryText && status === "done" && ( +
+ + {showSummary && ( + + {summaryText} + + )} +
+ )} +
+ ); +} diff --git a/apps/web/src/components/gen-ui/task-summary.tsx b/apps/web/src/components/gen-ui/task-summary.tsx index 087513cb..2aa02857 100644 --- a/apps/web/src/components/gen-ui/task-summary.tsx +++ b/apps/web/src/components/gen-ui/task-summary.tsx @@ -9,22 +9,22 @@ import { FileText, MinusCircle, } from "lucide-react"; +import { cn } from "@/lib/utils"; +import { BasicMarkdownText } from "../thread/markdown-text"; type TaskSummaryProps = { status: "loading" | "generating" | "done"; completed?: boolean; - summary?: string; summaryText?: string; }; export function TaskSummary({ status, completed, - summary, summaryText, }: TaskSummaryProps) { const [expanded, setExpanded] = useState(false); - const [showSummary, setShowSummary] = useState(false); + const [showSummary, setShowSummary] = useState(true); const getStatusIcon = () => { switch (status) { @@ -59,9 +59,9 @@ export function TaskSummary({ return (
setExpanded(!expanded) : undefined } @@ -70,26 +70,8 @@ export function TaskSummary({ {getStatusText()} - {status === "done" && summary && ( - - )}
- {expanded && summary && status === "done" && ( -
-

- Task Summary -

-

{summary}

-
- )} - {summaryText && status === "done" && (
{showSummary && ( -

{summaryText} -

+ )}
)} diff --git a/apps/web/src/components/github/repo-branch-selectors/branch-selector.tsx b/apps/web/src/components/github/repo-branch-selectors/branch-selector.tsx index 737cf61c..c2fa38e0 100644 --- a/apps/web/src/components/github/repo-branch-selectors/branch-selector.tsx +++ b/apps/web/src/components/github/repo-branch-selectors/branch-selector.tsx @@ -193,7 +193,7 @@ export function BranchSelector({ {selectedBranch || placeholder}
- + @@ -230,13 +230,13 @@ export function BranchSelector({ {branch.name} {isDefault && ( - + default )} {branch.protected && (
- +
)}
diff --git a/apps/web/src/components/github/repo-branch-selectors/index.tsx b/apps/web/src/components/github/repo-branch-selectors/index.tsx index 4091ce49..134bf6a4 100644 --- a/apps/web/src/components/github/repo-branch-selectors/index.tsx +++ b/apps/web/src/components/github/repo-branch-selectors/index.tsx @@ -6,12 +6,12 @@ export function RepositoryBranchSelectors() { const [threadId] = useQueryState("threadId"); const chatStarted = !!threadId; const defaultButtonStyles = - "bg-inherit border-none text-gray-500 hover:text-black dark:hover:text-gray-300 text-xs p-0 px-0 py-0 !p-0 !px-0 !py-0 h-fit hover:bg-inherit shadow-none"; + "bg-inherit border-none text-muted-foreground hover:text-muted-foreground/70 text-xs p-0 px-0 py-0 !p-0 !px-0 !py-0 h-fit hover:bg-inherit shadow-none"; const defaultStylesChatStarted = - "hover:bg-inherit cursor-default hover:cursor-default hover:text-black dark:hover:text-gray-300 hover:border-gray-300 hover:ring-inherit shadow-none p-0 px-0 py-0 !p-0 !px-0 !py-0"; + "hover:bg-inherit cursor-default hover:cursor-default text-muted-foreground hover:border-gray-300 hover:ring-inherit shadow-none p-0 px-0 py-0 !p-0 !px-0 !py-0"; return ( -
+
- + diff --git a/apps/web/src/components/thread/markdown-text.tsx b/apps/web/src/components/thread/markdown-text.tsx index a0d51663..be0f7a17 100644 --- a/apps/web/src/components/thread/markdown-text.tsx +++ b/apps/web/src/components/thread/markdown-text.tsx @@ -114,7 +114,7 @@ const defaultComponents: any = { ), p: ({ className, ...props }: { className?: string }) => (

), @@ -135,13 +135,13 @@ const defaultComponents: any = { ), ul: ({ className, ...props }: { className?: string }) => (

    li]:mt-2", className)} + className={cn("my-2 ml-6 list-disc [&>li]:mt-1", className)} {...props} /> ), ol: ({ className, ...props }: { className?: string }) => (
      li]:mt-2", className)} + className={cn("my-2 ml-6 list-decimal [&>li]:mt-2", className)} {...props} /> ), @@ -243,9 +243,12 @@ const defaultComponents: any = { }, }; -const MarkdownTextImpl: FC<{ children: string }> = ({ children }) => { +const MarkdownTextImpl: FC<{ children: string; className?: string }> = ({ + children, + className, +}) => { return ( -
      +
      = ({ children }) => { }; export const MarkdownText = memo(MarkdownTextImpl); + +const BasicMarkdownTextImpl: FC<{ children: string; className?: string }> = ({ + children, + className, +}) => { + const basicMarkdownComponents = { ...defaultComponents }; + // Don't render headers, instead render them as bold text + delete basicMarkdownComponents.h1; + delete basicMarkdownComponents.h2; + delete basicMarkdownComponents.h3; + delete basicMarkdownComponents.h4; + delete basicMarkdownComponents.h5; + delete basicMarkdownComponents.h6; + + return ( +
      + + {children} + +
      + ); +}; + +export const BasicMarkdownText = memo(BasicMarkdownTextImpl); diff --git a/apps/web/src/components/thread/messages/ai.tsx b/apps/web/src/components/thread/messages/ai.tsx index dba1dce5..5f769e1d 100644 --- a/apps/web/src/components/thread/messages/ai.tsx +++ b/apps/web/src/components/thread/messages/ai.tsx @@ -21,6 +21,10 @@ import { Interrupt } from "./interrupt"; import { ActionStep, ActionItemProps } from "@/components/gen-ui/action-step"; import { TaskSummary } from "@/components/gen-ui/task-summary"; import { PullRequestOpened } from "@/components/gen-ui/pull-request-opened"; +import { + MarkTaskCompleted, + MarkTaskIncomplete, +} from "@/components/gen-ui/task-review"; import { DiagnoseErrorAction } from "@/components/v2/diagnose-error-action"; import { WriteTechnicalNotes } from "@/components/gen-ui/write-technical-notes"; import { ToolCall } from "@langchain/core/messages/tool"; @@ -29,13 +33,14 @@ import { createShellToolFields, createMarkTaskCompletedToolFields, createMarkTaskNotCompletedToolFields, - createRgToolFields, + createSearchToolFields, createOpenPrToolFields, createInstallDependenciesToolFields, createTakePlannerNotesFields, + createCodeReviewMarkTaskCompletedFields, + createCodeReviewMarkTaskNotCompleteFields, createDiagnoseErrorToolFields, createGetURLContentToolFields, - createFindInstancesOfToolFields, createWriteTechnicalNotesToolFields, createConversationHistorySummaryToolFields, } from "@open-swe/shared/open-swe/tools"; @@ -56,8 +61,8 @@ const markTaskNotCompletedTool = createMarkTaskNotCompletedToolFields(); type MarkTaskNotCompletedToolArgs = z.infer< typeof markTaskNotCompletedTool.schema >; -const rgTool = createRgToolFields(dummyRepo); -type RgToolArgs = z.infer; +const searchTool = createSearchToolFields(dummyRepo); +type SearchToolArgs = z.infer; const openPrTool = createOpenPrToolFields(); type OpenPrToolArgs = z.infer; const installDependenciesTool = createInstallDependenciesToolFields(dummyRepo); @@ -66,6 +71,16 @@ type InstallDependenciesToolArgs = z.infer< >; const plannerNotesTool = createTakePlannerNotesFields(); type PlannerNotesToolArgs = z.infer; +const markFinalReviewTaskCompletedTool = + createCodeReviewMarkTaskCompletedFields(); +type MarkFinalReviewTaskCompletedToolArgs = z.infer< + typeof markFinalReviewTaskCompletedTool.schema +>; +const markFinalReviewTaskIncompleteTool = + createCodeReviewMarkTaskNotCompleteFields(); +type MarkFinalReviewTaskIncompleteToolArgs = z.infer< + typeof markFinalReviewTaskIncompleteTool.schema +>; const diagnoseErrorTool = createDiagnoseErrorToolFields(); type DiagnoseErrorToolArgs = z.infer; @@ -73,9 +88,6 @@ type DiagnoseErrorToolArgs = z.infer; const getURLContentTool = createGetURLContentToolFields(); type GetURLContentToolArgs = z.infer; -const findInstancesOfTool = createFindInstancesOfToolFields(dummyRepo); -type FindInstancesOfToolArgs = z.infer; - const writeTechnicalNotesTool = createWriteTechnicalNotesToolFields(); type WriteTechnicalNotesToolArgs = z.infer< typeof writeTechnicalNotesTool.schema @@ -182,14 +194,21 @@ export function mapToolMessageToActionStepProps( reasoningText, errorMessage: !success ? getContentString(message.content) : undefined, }; - } else if (toolCall?.name === rgTool.name) { - const args = toolCall.args as RgToolArgs; + } else if (toolCall?.name === searchTool.name) { + const args = toolCall.args as SearchToolArgs; return { - actionType: "rg", + actionType: "search", status, success, pattern: args.pattern || "", - paths: args.paths || [], + regex: args.regex || false, + case_sensitive: args.case_sensitive || false, + context_lines: args.context_lines || 0, + max_results: args.max_results || 0, + follow_symlinks: args.follow_symlinks || false, + exclude_files: args.exclude_files || "", + include_files: args.include_files || "", + file_types: args.file_types || [], output: getContentString(message.content), reasoningText, }; @@ -223,24 +242,6 @@ export function mapToolMessageToActionStepProps( output: getContentString(message.content), reasoningText, }; - } else if (toolCall?.name === findInstancesOfTool.name) { - const args = toolCall.args as FindInstancesOfToolArgs; - // case_sensitive and match_word both default to true. - const caseSensitive = - args.case_sensitive === undefined ? true : args.case_sensitive; - const matchWord = args.match_word === undefined ? true : args.match_word; - return { - actionType: "find_instances_of", - status, - success, - query: args.query || "", - case_sensitive: caseSensitive, - match_word: matchWord, - include_files: args.include_files, - exclude_files: args.exclude_files, - output: getContentString(message.content), - reasoningText, - }; } return { status: "loading", @@ -305,11 +306,10 @@ export function AssistantMessage({ (tc) => tc.name === shellTool.name || tc.name === applyPatchTool.name || - tc.name === rgTool.name || + tc.name === searchTool.name || tc.name === installDependenciesTool.name || tc.name === plannerNotesTool.name || - tc.name === getURLContentTool.name || - tc.name === findInstancesOfTool.name, + tc.name === getURLContentTool.name, ) : []; @@ -325,6 +325,18 @@ export function AssistantMessage({ ? aiToolCalls.find((tc) => tc.name === openPrTool.name) : undefined; + const markFinalReviewTaskCompletedToolCall = message + ? aiToolCalls.find( + (tc) => tc.name === markFinalReviewTaskCompletedTool.name, + ) + : undefined; + + const markFinalReviewTaskIncompleteToolCall = message + ? aiToolCalls.find( + (tc) => tc.name === markFinalReviewTaskIncompleteTool.name, + ) + : undefined; + const diagnoseErrorToolCall = message ? aiToolCalls.find((tc) => tc.name === diagnoseErrorTool.name) : undefined; @@ -472,6 +484,50 @@ export function AssistantMessage({ ); } + // If task completed review tool call is present, render the task review component + if (markFinalReviewTaskCompletedToolCall) { + const args = + markFinalReviewTaskCompletedToolCall.args as MarkFinalReviewTaskCompletedToolArgs; + const correspondingToolResult = toolResults.find( + (tr) => tr && tr.tool_call_id === markFinalReviewTaskCompletedToolCall.id, + ); + + const status = correspondingToolResult ? "done" : "generating"; + + return ( +
      + +
      + ); + } + + // If task incomplete review tool call is present, render the task review component + if (markFinalReviewTaskIncompleteToolCall) { + const args = + markFinalReviewTaskIncompleteToolCall.args as MarkFinalReviewTaskIncompleteToolArgs; + const correspondingToolResult = toolResults.find( + (tr) => + tr && tr.tool_call_id === markFinalReviewTaskIncompleteToolCall.id, + ); + + const status = correspondingToolResult ? "done" : "generating"; + + return ( +
      + +
      + ); + } + if (actionableToolCalls.length > 0) { const actionItems = actionableToolCalls.map((toolCall): ActionItemProps => { const correspondingToolResult = toolResults.find( @@ -479,7 +535,7 @@ export function AssistantMessage({ ); const isShellTool = toolCall.name === shellTool.name; - const isRgTool = toolCall.name === rgTool.name; + const isSearchTool = toolCall.name === searchTool.name; const isInstallDependenciesTool = toolCall.name === installDependenciesTool.name; @@ -489,13 +545,20 @@ export function AssistantMessage({ correspondingToolResult, threadMessages, ); - } else if (isRgTool) { - const args = toolCall.args as RgToolArgs; + } else if (isSearchTool) { + const args = toolCall.args as SearchToolArgs; return { - actionType: "rg", + actionType: "search", status: "generating", pattern: args?.pattern || "", - paths: args?.paths || [], + regex: args?.regex || false, + case_sensitive: args?.case_sensitive || false, + context_lines: args?.context_lines || 0, + max_results: args?.max_results || 0, + follow_symlinks: args?.follow_symlinks || false, + exclude_files: args?.exclude_files || [], + include_files: args?.include_files || [], + file_types: args?.file_types || [], output: "", } as ActionItemProps; } else if (isInstallDependenciesTool) { @@ -522,23 +585,6 @@ export function AssistantMessage({ url: args?.url || "", output: "", } as ActionItemProps; - } else if (toolCall.name === findInstancesOfTool.name) { - const args = toolCall.args as FindInstancesOfToolArgs; - // case_sensitive and match_word both default to true. - const caseSensitive = - args.case_sensitive === undefined ? true : args.case_sensitive; - const matchWord = - args.match_word === undefined ? true : args.match_word; - return { - actionType: "find_instances_of", - status: "generating", - query: args?.query || "", - case_sensitive: caseSensitive, - match_word: matchWord, - include_files: args?.include_files, - exclude_files: args?.exclude_files, - output: "", - } as ActionItemProps; } else { if (isShellTool) { const args = toolCall.args as ShellToolArgs; diff --git a/apps/web/src/components/v2/actions-renderer.tsx b/apps/web/src/components/v2/actions-renderer.tsx index 2eec3466..3d1db5c9 100644 --- a/apps/web/src/components/v2/actions-renderer.tsx +++ b/apps/web/src/components/v2/actions-renderer.tsx @@ -2,7 +2,10 @@ import { isAIMessageSDK, isHumanMessageSDK } from "@/lib/langchain-messages"; import { UseStream, useStream } from "@langchain/langgraph-sdk/react"; import { AssistantMessage } from "../thread/messages/ai"; import { Dispatch, SetStateAction, useEffect, useRef, useState } from "react"; -import { ManagerGraphState } from "@open-swe/shared/open-swe/manager/types"; +import { + ManagerGraphState, + ManagerGraphUpdate, +} from "@open-swe/shared/open-swe/manager/types"; import { useCancelStream } from "@/hooks/useCancelStream"; import { isCustomNodeEvent, @@ -102,9 +105,22 @@ function addMessagesToState( newMessages: Message[], ): Message[] { const existingIds = new Set(existingMessages.map((message) => message.id)); - const uniqueNewMessages = newMessages.filter( - (message) => !message.id || !existingIds.has(message.id), - ); + + // First deduplicate within newMessages array itself + const seenNewIds = new Set(); + const uniqueNewMessages = newMessages.filter((message) => { + // Skip messages without IDs or those already in existingMessages + if (message.id && existingIds.has(message.id)) return false; + + // Handle duplicates within newMessages + if (message.id) { + if (seenNewIds.has(message.id)) return false; + seenNewIds.add(message.id); + } + + return true; + }); + return [...existingMessages, ...uniqueNewMessages]; } @@ -123,9 +139,28 @@ function isNodeEndMessagesUpdate( ); } +function isNodeEndCommandUpdate(data: unknown): data is { + output: { lg_name: string; goto: string; update: ManagerGraphUpdate }; +} { + return !!( + typeof data === "object" && + data !== null && + "output" in data && + data.output && + typeof data.output === "object" && + "lg_name" in data.output && + "goto" in data.output && + "update" in data.output && + typeof data.output.lg_name === "string" && + (typeof data.output.goto === "string" || Array.isArray(data.output.goto)) && + typeof data.output.update === "object" + ); +} + const REVIEWER_NODE_IDS = [ - "take-review-actions", "generate-review-actions", + "take-review-actions", + "diagnose-reviewer-error", "final-review", ]; @@ -144,6 +179,11 @@ export function ActionsRenderer({ const joinedRunId = useRef(undefined); const [streamLoading, setStreamLoading] = useState(false); const [mergedMessages, setMergedMessages] = useState([]); + const debouncedSetMessages = useRef( + debounce((messages: Message[]) => { + setMergedMessages((prev) => addMessagesToState(prev, messages)); + }, 100), + ).current; const stream = useStream({ apiUrl: process.env.NEXT_PUBLIC_API_URL, @@ -160,11 +200,18 @@ export function ActionsRenderer({ data.event === "on_chain_end" && data.metadata?.langgraph_node && REVIEWER_NODE_IDS.includes(data.metadata.langgraph_node as string) && - data.data && - isNodeEndMessagesUpdate(data.data) + data.data ) { - const outputMessages = data.data.output.messages; - setMergedMessages((prev) => [...prev, ...outputMessages]); + if (isNodeEndCommandUpdate(data.data)) { + const outputMessages = data.data.output.update + .messages as unknown as Message[]; + console.log("outputMessages", outputMessages); + debouncedSetMessages(outputMessages); + } else if (isNodeEndMessagesUpdate(data.data)) { + const outputMessages = data.data.output.messages; + console.log("outputMessages", outputMessages); + debouncedSetMessages(outputMessages); + } } }, fetchStateHistory: false, @@ -290,12 +337,6 @@ export function ActionsRenderer({ } }, [stream.values, graphId]); - const debouncedSetMessages = useRef( - debounce((messages: Message[]) => { - setMergedMessages((prev) => addMessagesToState(prev, messages)); - }, 100), - ).current; - useEffect(() => { debouncedSetMessages(stream.messages); return () => { diff --git a/apps/web/src/components/v2/diagnose-error-action.tsx b/apps/web/src/components/v2/diagnose-error-action.tsx index 0319d4fe..dd43461f 100644 --- a/apps/web/src/components/v2/diagnose-error-action.tsx +++ b/apps/web/src/components/v2/diagnose-error-action.tsx @@ -19,7 +19,7 @@ export function DiagnoseErrorAction({ diagnosis, reasoningText, }: DiagnoseErrorActionProps) { - const [showReasoning, setShowReasoning] = useState(false); + const [showReasoning, setShowReasoning] = useState(true); const getStatusIcon = () => { switch (status) { diff --git a/apps/web/src/components/v2/manager-chat.tsx b/apps/web/src/components/v2/manager-chat.tsx index 6778f0fe..728dbf8f 100644 --- a/apps/web/src/components/v2/manager-chat.tsx +++ b/apps/web/src/components/v2/manager-chat.tsx @@ -12,6 +12,7 @@ import { useStream } from "@langchain/langgraph-sdk/react"; import { ManagerGraphState } from "@open-swe/shared/open-swe/manager/types"; import { cn } from "@/lib/utils"; import { isAIMessageSDK } from "@/lib/langchain-messages"; +import { BasicMarkdownText } from "../thread/markdown-text"; function MessageCopyButton({ content }: { content: string }) { const [copied, setCopied] = useState(false); @@ -121,17 +122,17 @@ export function ManagerChat({ )}
      -
      +
      {message.type === "human" ? "You" : "Agent"} +
      + +
      -
      + {messageContentString} -
      -
      - -
      +
      ); diff --git a/apps/web/src/components/v2/terminal-input.tsx b/apps/web/src/components/v2/terminal-input.tsx index 6f3ba7ca..fbc1fe2a 100644 --- a/apps/web/src/components/v2/terminal-input.tsx +++ b/apps/web/src/components/v2/terminal-input.tsx @@ -145,7 +145,7 @@ export function TerminalInput({ return (
      -
      +
      open-swe @ github diff --git a/apps/web/src/components/v2/thread-view-loading.tsx b/apps/web/src/components/v2/thread-view-loading.tsx index 79259b97..321fb72e 100644 --- a/apps/web/src/components/v2/thread-view-loading.tsx +++ b/apps/web/src/components/v2/thread-view-loading.tsx @@ -66,7 +66,9 @@ export function LoadingActionsCard() {
      - +
      + +
      ); } diff --git a/packages/shared/package.json b/packages/shared/package.json index 855c8879..74959ebd 100644 --- a/packages/shared/package.json +++ b/packages/shared/package.json @@ -18,7 +18,7 @@ }, "dependencies": { "@langchain/core": "^0.3.56", - "@langchain/langgraph": "^0.3.3", + "@langchain/langgraph": "^0.3.8", "@langchain/langgraph-sdk": "^0.0.92", "@octokit/rest": "^22.0.0", "zod": "^3.25.32" diff --git a/packages/shared/src/open-swe/reviewer/types.ts b/packages/shared/src/open-swe/reviewer/types.ts new file mode 100644 index 00000000..2a19051b --- /dev/null +++ b/packages/shared/src/open-swe/reviewer/types.ts @@ -0,0 +1,106 @@ +import "@langchain/langgraph/zod"; +import { z } from "zod"; +import { + Messages, + messagesStateReducer, + MessagesZodState, +} from "@langchain/langgraph"; +import { CustomRules, TargetRepository, TaskPlan } from "../types.js"; +import { withLangGraph } from "@langchain/langgraph/zod"; +import { BaseMessage } from "@langchain/core/messages"; + +export const ReviewerGraphStateObj = MessagesZodState.extend({ + /** + * We must include the internal messages so that the reviewer has an + * accurate picture of the conversation. + */ + internalMessages: withLangGraph(z.custom(), { + reducer: { + schema: z.custom(), + fn: messagesStateReducer, + }, + jsonSchemaExtra: { + langgraph_type: "messages", + }, + default: () => [], + }), + /** + * A separate list of messages for the reviewer. Used to track both + * internal messages which do not need to be shown to the user/propagated + * back to the programmer, and to determine how many reviewer actions have + * been executed. + */ + reviewerMessages: withLangGraph(z.custom(), { + reducer: { + schema: z.custom(), + fn: messagesStateReducer, + }, + jsonSchemaExtra: { + langgraph_type: "messages", + }, + default: () => [], + }), + sandboxSessionId: withLangGraph(z.string(), { + reducer: { + schema: z.string(), + fn: (_state, update) => update, + }, + }), + targetRepository: withLangGraph(z.custom(), { + reducer: { + schema: z.custom(), + fn: (_state, update) => update, + }, + }), + githubIssueId: withLangGraph(z.custom(), { + reducer: { + schema: z.custom(), + fn: (_state, update) => update, + }, + }), + codebaseTree: withLangGraph(z.string(), { + reducer: { + schema: z.string(), + fn: (_state, update) => update, + }, + }), + taskPlan: withLangGraph(z.custom(), { + reducer: { + schema: z.custom(), + fn: (_state, update) => update, + }, + }), + branchName: withLangGraph(z.string(), { + reducer: { + schema: z.string(), + fn: (_state, update) => update, + }, + }), + baseBranchName: withLangGraph(z.string(), { + reducer: { + schema: z.string(), + fn: (_state, update) => update, + }, + }), + changedFiles: withLangGraph(z.string(), { + reducer: { + schema: z.string(), + fn: (_state, update) => update, + }, + }), + customRules: withLangGraph(z.custom().optional(), { + reducer: { + schema: z.custom().optional(), + fn: (_state, update) => update, + }, + }), + dependenciesInstalled: withLangGraph(z.boolean(), { + reducer: { + schema: z.boolean(), + fn: (_state, update) => update, + }, + }), +}); + +export type ReviewerGraphState = z.infer; +export type ReviewerGraphUpdate = Partial; diff --git a/packages/shared/src/open-swe/tools.ts b/packages/shared/src/open-swe/tools.ts index c82dc1aa..b9ae1da2 100644 --- a/packages/shared/src/open-swe/tools.ts +++ b/packages/shared/src/open-swe/tools.ts @@ -106,113 +106,33 @@ export function createUpdatePlanToolFields() { }; } -export function createRgToolFields(targetRepository: TargetRepository) { +export function createSearchToolFields(targetRepository: TargetRepository) { const repoRoot = getRepoAbsolutePath(targetRepository); - // Main ripgrep command schema - const ripgrepCommandSchema = z.object({ + const searchSchema = z.object({ pattern: z .string() + .describe("The string or regex to search the codebase for."), + regex: z + .boolean() .optional() + .default(false) .describe( - "The search pattern (regex). Leave empty when using flags like --files or --type-list", - ), - - paths: z - .array(z.string()) - .optional() - .describe( - "Files or directories to search. If empty, searches current directory", - ), - - flags: z - .array(z.string()) - .optional() - .describe( - 'Array of flags with their values. Examples: ["-i", "--type=rust", "-A", "3", "--files"]. Short flags like -i can be standalone, flags with values can be separate strings or use = for long flags', - ), - }); - - return { - name: "rg", - schema: ripgrepCommandSchema, - description: `Call this tool to run the rg command (ripgrep). This should ONLY be called if you want to search for files in the repository. The working directory this command will be executed in is \`${repoRoot}\`.`, - }; -} - -// Only used for type inference -const _tmpRgToolSchema = createRgToolFields({ owner: "x", repo: "x" }).schema; -export type RipgrepCommand = z.infer; - -export function formatRgCommand( - cmd: RipgrepCommand, - options?: { - excludeRequiredFlags?: boolean; - }, -): string[] { - const args = ["rg"]; - - // Always include these flags - const requiredFlags = ["--color", "never", "--line-number", "--heading"]; - - // Add user-provided flags, ensuring we don't duplicate the required ones - if (cmd.flags) { - // Filter out any flags that would duplicate our required flags - const filteredFlags = cmd.flags.filter((flag) => { - // Check for exact matches or flags that start with our required prefixes - return ( - !requiredFlags.includes(flag) && - !flag.startsWith("--color=") && - flag !== "--line-number" && - flag !== "-n" && - flag !== "--heading" - ); - }); - - args.push(...filteredFlags); - } - - if (!options?.excludeRequiredFlags) { - // Add the required flags - args.push(...requiredFlags); - } - - if (cmd.pattern) { - args.push(cmd.pattern); - } - - if (cmd.paths) { - args.push(...cmd.paths); - } - - return args; -} - -export function createFindInstancesOfToolFields( - targetRepository: TargetRepository, -) { - const repoRoot = getRepoAbsolutePath(targetRepository); - const findInstancesOfSchema = z.object({ - query: z - .string() - .describe( - "The query/keyword to search for. This should be a literal string, not a regex.", + "Whether or not to treat the pattern as a regex. Defaults to false.", ), case_sensitive: z .boolean() .optional() - .default(true) + .default(false) .describe( - "Whether or not to make the query search case sensitive. Defaults to true", + "Whether or not to make the search case sensitive. Defaults to false.", ), - match_word: z - .boolean() + context_lines: z + .number() .optional() - .default(true) - .describe( - "Whether or not to only show results which match the exact keyword. Defaults to true", - ), + .default(0) + .describe("Number of lines of context to include before/after matches."), exclude_files: z .string() @@ -223,15 +143,109 @@ export function createFindInstancesOfToolFields( .string() .optional() .describe("Glob pattern of files to include"), + + max_results: z + .number() + .optional() + .default(0) + .describe( + "Maximum number of results to return. Defaults to 0, which returns all results.", + ), + file_types: z + .array(z.string()) + .optional() + .describe("Restrict to certain file extensions (e.g., ['.js', '.ts'])."), + follow_symlinks: z + .boolean() + .optional() + .default(false) + .describe("Whether or not to follow symlinks. Defaults to false."), }); return { - name: "find_instances_of", - schema: findInstancesOfSchema, - description: `Find all instances of a string in the repository. Returns results with 3 lines of context above and below each match, absolute file paths, and total result count. The working directory this command will be executed in is \`${repoRoot}\`.`, + name: "search", + schema: searchSchema, + description: `Execute a search in the repository. The working directory this command will be executed in is \`${repoRoot}\`.`, }; } +// Only used for type inference +const _tmpSearchToolSchema = createSearchToolFields({ + owner: "x", + repo: "x", +}).schema; +export type SearchCommand = z.infer; + +function escapeShellArg(arg: string): string { + // If the string contains a single quote, close the string, escape the single quote, and reopen it + // Example: foo'bar → 'foo'\''bar' + return `'${arg.replace(/'/g, `'\\''`)}'`; +} + +export function formatSearchCommand( + cmd: SearchCommand, + options?: { + excludeRequiredFlags?: boolean; + }, +): string[] { + const args = ["rg"]; + + // Required flags to keep formatting and output consistent + const requiredFlags = ["--color=never", "--line-number", "--heading"]; + + if (!options?.excludeRequiredFlags) { + args.push(...requiredFlags); + } + + // Case sensitivity + if (!cmd.case_sensitive) { + args.push("-i"); + } + + // Regex vs fixed string + if (!cmd.regex) { + args.push("--fixed-strings"); + } + + // Context lines + if (cmd.context_lines && cmd.context_lines > 0) { + args.push(`-C`, String(cmd.context_lines)); + } + + // File globs + if (cmd.include_files) { + args.push("--glob", cmd.include_files); + } + + if (cmd.exclude_files) { + args.push("--glob", `!${cmd.exclude_files}`); + } + + // File types + if (cmd.file_types && cmd.file_types.length > 0) { + for (const ext of cmd.file_types) { + args.push("--glob", `**/*${ext}`); + } + } + + // Follow symlinks + if (cmd.follow_symlinks) { + args.push("-L"); + } + + // Max results (0 = unlimited) + if (cmd.max_results && cmd.max_results > 0) { + args.push("--max-count", String(cmd.max_results)); + } + + // The pattern (must come after flags) + if (cmd.pattern) { + args.push(escapeShellArg(cmd.pattern)); + } + + return args; +} + export function createMarkTaskNotCompletedToolFields() { const markTaskNotCompletedToolSchema = z.object({ reasoning: z @@ -396,3 +410,42 @@ export function createConversationHistorySummaryToolFields() { schema: conversationHistorySummarySchema, }; } + +export function createCodeReviewMarkTaskCompletedFields() { + const markTaskCompletedSchema = z.object({ + review: z + .string() + .describe( + "Your final review for the completed task. This should be concise, but descriptive.", + ), + }); + + return { + name: "code_review_mark_task_completed", + schema: markTaskCompletedSchema, + description: + "Use this tool to mark a task as completed. This should be called if you determine that the task has been successfully completed.", + }; +} + +export function createCodeReviewMarkTaskNotCompleteFields() { + const markTaskNotCompleteSchema = z.object({ + review: z + .string() + .describe( + "Your final review for the completed task. This should be concise, but descriptive.", + ), + additional_actions: z + .array(z.string()) + .describe( + "A list of additional actions to take which will successfully satisfy your review, and complete the task.", + ), + }); + + return { + name: "code_review_mark_task_not_complete", + schema: markTaskNotCompleteSchema, + description: + "Use this tool to mark a task as not complete. This should be called if you determine that the task has not been successfully completed, and you have additional tasks the programmer should take to successfully complete the task.", + }; +} diff --git a/packages/shared/src/open-swe/types.ts b/packages/shared/src/open-swe/types.ts index c8c13ff0..304f11fa 100644 --- a/packages/shared/src/open-swe/types.ts +++ b/packages/shared/src/open-swe/types.ts @@ -240,6 +240,16 @@ export const GraphAnnotation = MessagesZodState.extend({ fn: (_state, update) => update, }, }), + /** + * The review generated by the reviewer subgraph + */ + review: withLangGraph(z.custom(), { + reducer: { + schema: z.custom(), + fn: (_state, update) => update, + }, + default: () => "", + }), // ---NOT USED--- ui: z @@ -272,6 +282,15 @@ export const GraphConfigurationMetadata: { "Maximum number of context gathering actions during planning", }, }, + maxReviewActions: { + x_open_swe_ui_config: { + type: "number", + default: 30, + min: 1, + max: 250, + description: "Maximum number of review actions during planning", + }, + }, plannerModelName: { x_open_swe_ui_config: { type: "select", @@ -441,6 +460,16 @@ export const GraphConfiguration = z.object({ .number() .optional() .langgraph.metadata(GraphConfigurationMetadata.maxContextActions), + /** + * The maximum number of context gathering actions to take during review. + * Each action consists of 2 messages (request & result), plus 1 human message. + * Total messages = maxReviewActions * 2 + 1 + * @default 30 + */ + maxReviewActions: z + .number() + .optional() + .langgraph.metadata(GraphConfigurationMetadata.maxReviewActions), /** * The model ID to use for the planning step. diff --git a/yarn.lock b/yarn.lock index 55525c32..41af88f7 100644 --- a/yarn.lock +++ b/yarn.lock @@ -2597,12 +2597,12 @@ __metadata: languageName: node linkType: hard -"@langchain/langgraph@npm:^0.3.3": - version: 0.3.7 - resolution: "@langchain/langgraph@npm:0.3.7" +"@langchain/langgraph@npm:^0.3.8": + version: 0.3.8 + resolution: "@langchain/langgraph@npm:0.3.8" dependencies: "@langchain/langgraph-checkpoint": ~0.0.18 - "@langchain/langgraph-sdk": ~0.0.90 + "@langchain/langgraph-sdk": ~0.0.92 uuid: ^10.0.0 zod: ^3.25.32 peerDependencies: @@ -2611,7 +2611,7 @@ __metadata: peerDependenciesMeta: zod-to-json-schema: optional: true - checksum: 6a924940d92d4c0c97c18665a1d1f70bfa83c2d5f8b221ea6fe90d3d47bfa6518a6032058bb9b722199ce0c9fb511e9fe2818e2d877ffb7c43510715c1b82d84 + checksum: 08089b5f151d43b556dba604699c52cb4d323e1fc17ab24572429db582c90ad2da477f90070b2812b03e6aec9d080c09a292c420943b65aeaba858cdf0d12e3f languageName: node linkType: hard @@ -3461,7 +3461,7 @@ __metadata: "@langchain/community": ^0.3.47 "@langchain/core": ^0.3.56 "@langchain/google-genai": ^0.2.9 - "@langchain/langgraph": ^0.3.3 + "@langchain/langgraph": ^0.3.8 "@langchain/langgraph-cli": latest "@langchain/langgraph-sdk": ^0.0.92 "@langchain/mcp-adapters": ^0.5.2 @@ -3512,7 +3512,7 @@ __metadata: "@eslint/eslintrc": ^3.1.0 "@eslint/js": ^9.19.0 "@langchain/core": ^0.3.56 - "@langchain/langgraph": ^0.3.3 + "@langchain/langgraph": ^0.3.8 "@langchain/langgraph-sdk": ^0.0.92 "@octokit/rest": ^22.0.0 "@octokit/types": ^12.0.0 @@ -3537,7 +3537,7 @@ __metadata: dependencies: "@eslint/js": ^9.19.0 "@langchain/core": ^0.3.57 - "@langchain/langgraph": ^0.3.3 + "@langchain/langgraph": ^0.3.8 "@langchain/langgraph-sdk": ^0.0.92 "@octokit/app": ^16.0.1 "@octokit/types": ^14.1.0