diff --git a/apps/open-swe/src/graphs/manager/nodes/classify-message.ts b/apps/open-swe/src/graphs/manager/nodes/classify-message.ts deleted file mode 100644 index 3323eed9..00000000 --- a/apps/open-swe/src/graphs/manager/nodes/classify-message.ts +++ /dev/null @@ -1,426 +0,0 @@ -import { GraphConfig, TaskPlan } from "@open-swe/shared/open-swe/types"; -import { - ManagerGraphState, - ManagerGraphUpdate, -} from "@open-swe/shared/open-swe/manager/types"; -import { createLangGraphClient } from "../../../utils/langgraph-client.js"; -import { - BaseMessage, - HumanMessage, - isHumanMessage, - RemoveMessage, -} from "@langchain/core/messages"; -import { z } from "zod"; -import { removeLastHumanMessage } from "../../../utils/message/modify-array.js"; -import { formatPlanPrompt } from "../../../utils/plan-prompt.js"; -import { getActivePlanItems } from "@open-swe/shared/open-swe/tasks"; -import { getMessageString } from "../../../utils/message/content.js"; -import { loadModel, Task } from "../../../utils/load-model.js"; -import { Command, END } from "@langchain/langgraph"; -import { getMessageContentString } from "@open-swe/shared/messages"; -import { createIssue, createIssueComment } from "../../../utils/github/api.js"; -import { getGitHubTokensFromConfig } from "../../../utils/github-tokens.js"; -import { createIssueTitleAndBodyFromMessages } from "../utils/generate-issue-fields.js"; -import { ThreadStatus } from "@langchain/langgraph-sdk"; -import { - extractIssueTitleAndContentFromMessage, - formatContentForIssueBody, -} from "../../../utils/github/issue-messages.js"; -import { getDefaultHeaders } from "../../../utils/default-headers.js"; - -// This should not be shown to the user if the programmer is running -const PLAN_ROUTING_OPTION = `- plan: Call this route if the user's message is a complete request which you can use to kickoff a new planning session (only if one is not already running), or it's an entirely new request which you should also start a new planning session for (only if both the planner and programmer are not running). You may also call this route if the planner is running, and the user's message contains updated instructions, or additional context which may be relevant/helpful to the planner.`; - -// This should only be included in the state when the programmer is running. -const CODE_ROUTING_OPTION = `- code: Call this route if the user's message should be added to the programmer's currently running session. This should be called if you determine the user is trying to provide extra context to the programmer.`; - -// This should only be included when the programmer/planner is running. -const CREATE_ISSUE_ROUTING_OPTION = `- create_new_issue: Call this route if the user's request should create a new GitHub issue, and should be executed independently from the current request. This should only be called if the new request does not depend on the current request.`; - -// This should only be included if the task plan exists. -const TASK_PLAN_PROMPT = `# Task Plan -The following is the current state of the task plan generated by the planner. You should use this as context when determining where to route the user's message, and how to reply to them. -{TASK_PLAN} -\n\n`; - -const CONVERSATION_HISTORY_PROMPT = `# Conversation History -The following is the conversation history between the user and you. This does not include their most recent message, which is the one you are currently classifying. You should use this as context when determining where to route the user's message, and how to reply to them. -{CONVERSATION_HISTORY} -\n\n`; - -// This prompt does not generate the route, it only generates the response. -const CLASSIFICATION_SYSTEM_PROMPT = `# Identity -You're a highly intelligent AI software engineering manager, tasked with identifying the user's intent, and responding to their message, and determining how you'll route it to the proper AI assistant. -Your overall system is an AI coding agent, tasked with completing user's requests to improve their codebase. - -# Instructions -Carefully examine the user's message, along with the conversation history provided (or none, if it's the first message they sent) to you in this system message below. -Using their most recent request, the conversation history, and the current status of your two AI assistants (programmer and planner), generate a response to send to the user. -Below you're provided with routes to take given the user's request. You should not select a route in this step, but your response should make it clear which route you'll take. (the routing will handle in a step which is not exposed to the user). -Ensure your response is clear, and concise. You should not explicitly state which route you're taking, but it should be obvious to anyone who also knows what routes are available. -Although you're only supposed to classify & respond to the latest message, this does not mean you should look at it in isolation. You should consider the conversation history as a whole, and the current status of your two AI assistants (programmer and planner) to determine how to respond to the user's new message. - -# Assistant Statuses -The planner's current status is: {PLANNER_STATUS} -The programmer's current status is: {PROGRAMMER_STATUS} - -{TASK_PLAN_PROMPT} -{CONVERSATION_HISTORY_PROMPT} - -# Routing Options -Based on all of the context provided above, generate a response to send to the user, including messaging about the route you'll select from the below options in your next step. -Your routing options are: -- no_op: This should be called when the user's message does not warrant starting a new planning session, or updating the running session, or the same with the programmer if it's already running. -{PLAN_ROUTING_OPTION} -{CREATE_ISSUE_ROUTING_OPTION} -{CODE_ROUTING_OPTION} - -# Response -Your response should be clear, concise and straight to the point. Do NOT include any additional context, such as an idea for how to implement their request. -You're only acting as a manager, and thus you should only respond with a short message about which route you'll take. You do not need to explain why you're taking that route. -Your response will not exceed two sentences. You will be rewarded for being concise. -`; - -// This prompt uses the response to generate the route. -const ROUTING_SYSTEM_PROMPT = `# Identity -You're a highly intelligent AI software engineering manager, tasked with identifying the user's intent, and responding to their message, plus routing it to the proper AI assistant. -Your overall system is an AI coding agent, tasked with completing user's requests to improve their codebase. - -# Instructions -Carefully examine the user's message, the conversation history provided (or none, if it's the first message they sent) to you in this system message below, and the response you just generated in the previous step. -Using their most recent request, and your response to the user (which will contain context about how you should route the user's message), and the current status of your two AI assistants (programmer and planner), call the \`respond_and_route\` tool to route the user's response. - -# Assistant Statuses -The planner's current status is: {PLANNER_STATUS} -The programmer's current status is: {PROGRAMMER_STATUS} - -{TASK_PLAN_PROMPT} -{CONVERSATION_HISTORY_PROMPT} - -# Response -This is the response you just generated which was sent to the user. It will include context about how/what route you should take. Pay careful attention to this response, and ensure you're following it. -{ROUTING_RESPONSE} - -# Routing Options -Based on all of the context provided above, generate a response to send to the user, including messaging about the route you'll select from the below options in your next step. -Your routing options are: -- no_op: This should be called when the user's message does not warrant starting a new planning session, or updating the running session, or the same with the programmer if it's already running. -{PLAN_ROUTING_OPTION} -{CREATE_ISSUE_ROUTING_OPTION} -{CODE_ROUTING_OPTION}`; - -const baseClassificationSchema = z.object({ - response: z - .string() - .describe( - "The response to send to the user. This should be clear, concise, and include any additional context the user may need to know about how/why you're handling their new message.", - ), - route: z - .enum(["no_op"]) - .describe("The route to take to handle the user's new message."), -}); - -const createClassificationPromptAndToolSchema = (inputs: { - programmerStatus: ThreadStatus | "not_started"; - plannerStatus: ThreadStatus | "not_started"; - messages: BaseMessage[]; - taskPlan: TaskPlan; - routingResponse?: string; -}): { - prompt: string; - schema: z.ZodTypeAny; -} => { - const conversationHistoryWithoutLatest = removeLastHumanMessage( - inputs.messages, - ); - const formattedTaskPlanPrompt = inputs.taskPlan - ? TASK_PLAN_PROMPT.replaceAll( - "{TASK_PLAN}", - formatPlanPrompt(getActivePlanItems(inputs.taskPlan)), - ) - : null; - const formattedConversationHistoryPrompt = - conversationHistoryWithoutLatest?.length - ? CONVERSATION_HISTORY_PROMPT.replaceAll( - "{CONVERSATION_HISTORY}", - conversationHistoryWithoutLatest.map(getMessageString).join("\n"), - ) - : null; - - const programmerIsRunning = inputs.programmerStatus === "busy"; - const showCreateIssueRoutingOption = - inputs.programmerStatus !== "not_started" || - inputs.plannerStatus !== "not_started"; - - const systemPrompt = inputs.routingResponse - ? ROUTING_SYSTEM_PROMPT - : CLASSIFICATION_SYSTEM_PROMPT; - - const prompt = systemPrompt - .replaceAll("{PROGRAMMER_STATUS}", inputs.programmerStatus) - .replaceAll("{PLANNER_STATUS}", inputs.plannerStatus) - .replaceAll( - "{CODE_ROUTING_OPTION}", - programmerIsRunning ? CODE_ROUTING_OPTION : "", - ) - // Only show the planner option if the programmer is not running - .replaceAll( - "{PLAN_ROUTING_OPTION}", - !programmerIsRunning ? PLAN_ROUTING_OPTION : "", - ) - .replaceAll( - "{CREATE_ISSUE_ROUTING_OPTION}", - // Do not show the create new issue option if both the planner & programmer have not started - // if either have started/currently running/completed, show the option - showCreateIssueRoutingOption ? CREATE_ISSUE_ROUTING_OPTION : "", - ) - .replaceAll("{TASK_PLAN_PROMPT}", formattedTaskPlanPrompt ?? "") - .replaceAll( - "{CONVERSATION_HISTORY_PROMPT}", - formattedConversationHistoryPrompt ?? "", - ) - .replaceAll("{ROUTING_RESPONSE}", inputs.routingResponse ?? ""); - - const schema = baseClassificationSchema.extend({ - route: z - .enum([ - "no_op", - ...(programmerIsRunning ? ["code"] : []), - ...(!programmerIsRunning ? ["plan"] : []), - ...(showCreateIssueRoutingOption ? ["create_new_issue"] : []), - ]) - .describe("The route to take to handle the user's new message."), - }); - - return { - prompt, - schema, - }; -}; - -/** - * Classify the latest human message to determine how to route the request. - * Requests can be routed to: - * 1. reply - dont need to plan, just reply. This could be if the user sends a message which is not classified as a request, or if the programmer/planner is already running. - * a. if the planner/programmer is already running, we'll simply reply with - */ -export async function classifyMessage( - state: ManagerGraphState, - config: GraphConfig, -): Promise { - const userMessage = state.messages.findLast(isHumanMessage); - if (!userMessage) { - throw new Error("No human message found."); - } - - const langGraphClient = createLangGraphClient({ - defaultHeaders: getDefaultHeaders(config), - }); - - const [programmerThread, plannerThread] = await Promise.all([ - state.programmerSession?.threadId - ? langGraphClient.threads.get(state.programmerSession.threadId) - : undefined, - state.plannerSession?.threadId - ? langGraphClient.threads.get(state.plannerSession.threadId) - : undefined, - ]); - const programmerStatus = programmerThread?.status ?? "not_started"; - const plannerStatus = plannerThread?.status ?? "not_started"; - - const { prompt: responsePrompt } = createClassificationPromptAndToolSchema({ - programmerStatus, - plannerStatus, - messages: state.messages, - taskPlan: state.taskPlan, - }); - - const model = await loadModel(config, Task.CLASSIFICATION); - const response = await model.invoke([ - { - role: "system", - content: responsePrompt, - }, - userMessage, - ]); - - // Get the new prompt, this time passing the `routingResponse` to the prompt. - const { prompt: routingPrompt, schema } = - createClassificationPromptAndToolSchema({ - programmerStatus, - plannerStatus, - messages: state.messages, - taskPlan: state.taskPlan, - routingResponse: getMessageContentString(response.content), - }); - const respondAndRouteTool = { - name: "respond_and_route", - description: "Respond to the user's message and determine how to route it.", - schema, - }; - const modelWithTools = model.bindTools([respondAndRouteTool], { - tool_choice: respondAndRouteTool.name, - parallel_tool_calls: false, - }); - - const toolCallingResponse = await modelWithTools - .withConfig({ tags: ["nostream"] }) - .invoke([ - { - role: "system", - content: routingPrompt, - }, - userMessage, - ]); - - const toolCall = toolCallingResponse.tool_calls?.[0]; - if (!toolCall) { - throw new Error("No tool call found."); - } - const toolCallArgs = toolCall.args as z.infer< - typeof baseClassificationSchema - >; - - if (toolCallArgs.route === "no_op") { - // If it's a no_op, just add the message to the state and return. - const commandUpdate: ManagerGraphUpdate = { - messages: [response], - }; - return new Command({ - update: commandUpdate, - goto: END, - }); - } - - if ((toolCallArgs.route as string) === "create_new_issue") { - // Route to node which kicks off new manager run, passing in the full conversation history. - const commandUpdate: ManagerGraphUpdate = { - messages: [response], - }; - return new Command({ - update: commandUpdate, - goto: "create-new-session", - }); - } - - const { githubAccessToken } = getGitHubTokensFromConfig(config); - let githubIssueId = state.githubIssueId; - - const newMessages: BaseMessage[] = [response]; - - // If it's not a no_op, ensure there is a GitHub issue with the user's request. - if (!githubIssueId) { - // If there are multiple human messages in the state, generate a github issue with an LLM. - // Otherwise extract it from the user's message. - let title = ""; - let body = ""; - if (state.messages.filter(isHumanMessage).length > 1) { - const titleAndContent = await createIssueTitleAndBodyFromMessages( - state.messages, - config, - ); - title = titleAndContent.title; - body = titleAndContent.body; - } else { - const titleAndContent = extractIssueTitleAndContentFromMessage( - getMessageContentString(userMessage.content), - ); - title = titleAndContent.title; - body = titleAndContent.content; - } - - const newIssue = await createIssue({ - owner: state.targetRepository.owner, - repo: state.targetRepository.repo, - title, - body: formatContentForIssueBody(body), - githubAccessToken, - }); - if (!newIssue) { - throw new Error("Failed to create issue."); - } - githubIssueId = newIssue.number; - // Ensure we remove the old message, and replace it with an exact copy, - // but with the issue ID & isOriginalIssue set in additional_kwargs. - newMessages.push( - ...[ - new RemoveMessage({ - id: userMessage.id ?? "", - }), - new HumanMessage({ - ...userMessage, - additional_kwargs: { - githubIssueId: githubIssueId, - isOriginalIssue: true, - }, - }), - ], - ); - } else if ( - githubIssueId && - state.messages.filter(isHumanMessage).length > 1 - ) { - // If there already is a GitHub issue ID in state, and multiple human messages, add any - // human messages to the issue which weren't already added. - const messagesNotInIssue = state.messages - .filter(isHumanMessage) - .filter((message) => { - // If the message doesn't contain `githubIssueId` in additional kwargs, it hasn't been added to the issue. - return !message.additional_kwargs?.githubIssueId; - }); - - const createCommentsPromise = messagesNotInIssue.map(async (message) => { - const createdIssue = await createIssueComment({ - owner: state.targetRepository.owner, - repo: state.targetRepository.repo, - issueNumber: githubIssueId, - body: getMessageContentString(message.content), - githubToken: githubAccessToken, - }); - if (!createdIssue?.id) { - throw new Error("Failed to create issue comment"); - } - newMessages.push( - ...[ - new RemoveMessage({ - id: message.id ?? "", - }), - new HumanMessage({ - ...message, - additional_kwargs: { - githubIssueId: githubIssueId, - githubIssueCommentId: createdIssue.id, - }, - }), - ], - ); - }); - - await Promise.all(createCommentsPromise); - } - - // Issue has been created, and any missing human messages have been added to it. - - const commandUpdate: ManagerGraphUpdate = { - messages: newMessages, - ...(githubIssueId ? { githubIssueId } : {}), - }; - - if ((toolCallArgs.route as any) === "code") { - // If the route was code, we don't need to do anything since the issue now contains the new messages, and the coding agent will handle pulling them in. - return new Command({ - update: commandUpdate, - goto: END, - }); - } - - if (toolCallArgs.route === "plan") { - // Always kickoff a new start planner node. This will enqueue new runs on the planner graph. - return new Command({ - update: commandUpdate, - goto: "start-planner", - }); - } - - throw new Error(`Invalid route: ${toolCallArgs.route}`); -} diff --git a/apps/open-swe/src/graphs/manager/nodes/classify-message/index.ts b/apps/open-swe/src/graphs/manager/nodes/classify-message/index.ts new file mode 100644 index 00000000..b8b3cb88 --- /dev/null +++ b/apps/open-swe/src/graphs/manager/nodes/classify-message/index.ts @@ -0,0 +1,293 @@ +import { GraphConfig, GraphState } from "@open-swe/shared/open-swe/types"; +import { + ManagerGraphState, + ManagerGraphUpdate, +} from "@open-swe/shared/open-swe/manager/types"; +import { createLangGraphClient } from "../../../../utils/langgraph-client.js"; +import { + BaseMessage, + HumanMessage, + isHumanMessage, + RemoveMessage, +} from "@langchain/core/messages"; +import { z } from "zod"; +import { loadModel, Task } from "../../../../utils/load-model.js"; +import { Command, END } from "@langchain/langgraph"; +import { getMessageContentString } from "@open-swe/shared/messages"; +import { + createIssue, + createIssueComment, +} from "../../../../utils/github/api.js"; +import { getGitHubTokensFromConfig } from "../../../../utils/github-tokens.js"; +import { createIssueFieldsFromMessages } from "../../utils/generate-issue-fields.js"; +import { + extractIssueTitleAndContentFromMessage, + formatContentForIssueBody, +} from "../../../../utils/github/issue-messages.js"; +import { getDefaultHeaders } from "../../../../utils/default-headers.js"; +import { BASE_CLASSIFICATION_SCHEMA } from "./schemas.js"; +import { getPlansFromIssue } from "../../../../utils/github/issue-task.js"; +import { HumanResponse } from "@langchain/langgraph/prebuilt"; +import { PLANNER_GRAPH_ID } from "@open-swe/shared/constants"; +import { createLogger, LogLevel } from "../../../../utils/logger.js"; +import { PlannerGraphState } from "@open-swe/shared/open-swe/planner/types"; +import { createClassificationPromptAndToolSchema } from "./utils.js"; + +const logger = createLogger(LogLevel.INFO, "ClassifyMessage"); + +/** + * Classify the latest human message to determine how to route the request. + * Requests can be routed to: + * 1. reply - dont need to plan, just reply. This could be if the user sends a message which is not classified as a request, or if the programmer/planner is already running. + * a. if the planner/programmer is already running, we'll simply reply with + */ +export async function classifyMessage( + state: ManagerGraphState, + config: GraphConfig, +): Promise { + const userMessage = state.messages.findLast(isHumanMessage); + if (!userMessage) { + throw new Error("No human message found."); + } + + const langGraphClient = createLangGraphClient({ + defaultHeaders: getDefaultHeaders(config), + }); + + const plannerThread = state.plannerSession?.threadId + ? await langGraphClient.threads.get( + state.plannerSession.threadId, + ) + : undefined; + const plannerThreadValues = plannerThread?.values; + const programmerThread = plannerThreadValues?.programmerSession?.threadId + ? await langGraphClient.threads.get( + plannerThreadValues.programmerSession.threadId, + ) + : undefined; + + const programmerStatus = programmerThread?.status ?? "not_started"; + const plannerStatus = plannerThread?.status ?? "not_started"; + + // If the githubIssueId is defined, fetch the most recent task plan (if exists). Otherwise fallback to state task plan + const issuePlans = state.githubIssueId + ? await getPlansFromIssue(state, config) + : null; + const taskPlan = issuePlans?.taskPlan ?? state.taskPlan; + + const { prompt, schema } = createClassificationPromptAndToolSchema({ + programmerStatus, + plannerStatus, + messages: state.messages, + taskPlan, + proposedPlan: issuePlans?.proposedPlan ?? undefined, + }); + const respondAndRouteTool = { + name: "respond_and_route", + description: "Respond to the user's message and determine how to route it.", + schema, + }; + const model = await loadModel(config, Task.CLASSIFICATION); + const modelWithTools = model.bindTools([respondAndRouteTool], { + tool_choice: respondAndRouteTool.name, + parallel_tool_calls: false, + }); + + const response = await modelWithTools.invoke([ + { + role: "system", + content: prompt, + }, + userMessage, + ]); + + const toolCall = response.tool_calls?.[0]; + if (!toolCall) { + throw new Error("No tool call found."); + } + const toolCallArgs = toolCall.args as z.infer< + typeof BASE_CLASSIFICATION_SCHEMA + >; + + if (toolCallArgs.route === "no_op") { + // If it's a no_op, just add the message to the state and return. + const commandUpdate: ManagerGraphUpdate = { + messages: [response], + }; + return new Command({ + update: commandUpdate, + goto: END, + }); + } + + if ((toolCallArgs.route as string) === "create_new_issue") { + // Route to node which kicks off new manager run, passing in the full conversation history. + const commandUpdate: ManagerGraphUpdate = { + messages: [response], + }; + return new Command({ + update: commandUpdate, + goto: "create-new-session", + }); + } + + const { githubAccessToken } = getGitHubTokensFromConfig(config); + let githubIssueId = state.githubIssueId; + + const newMessages: BaseMessage[] = [response]; + + // If it's not a no_op, ensure there is a GitHub issue with the user's request. + if (!githubIssueId) { + const { title } = await createIssueFieldsFromMessages( + state.messages, + config.configurable, + ); + const { content: body } = extractIssueTitleAndContentFromMessage( + getMessageContentString(userMessage.content), + ); + + const newIssue = await createIssue({ + owner: state.targetRepository.owner, + repo: state.targetRepository.repo, + title, + body: formatContentForIssueBody(body), + githubAccessToken, + }); + if (!newIssue) { + throw new Error("Failed to create issue."); + } + githubIssueId = newIssue.number; + // Ensure we remove the old message, and replace it with an exact copy, + // but with the issue ID & isOriginalIssue set in additional_kwargs. + newMessages.push( + ...[ + new RemoveMessage({ + id: userMessage.id ?? "", + }), + new HumanMessage({ + ...userMessage, + additional_kwargs: { + githubIssueId: githubIssueId, + isOriginalIssue: true, + }, + }), + ], + ); + } else if ( + githubIssueId && + state.messages.filter(isHumanMessage).length > 1 + ) { + // If there already is a GitHub issue ID in state, and multiple human messages, add any + // human messages to the issue which weren't already added. + const messagesNotInIssue = state.messages + .filter(isHumanMessage) + .filter((message) => { + // If the message doesn't contain `githubIssueId` in additional kwargs, it hasn't been added to the issue. + return !message.additional_kwargs?.githubIssueId; + }); + + const createCommentsPromise = messagesNotInIssue.map(async (message) => { + const createdIssue = await createIssueComment({ + owner: state.targetRepository.owner, + repo: state.targetRepository.repo, + issueNumber: githubIssueId, + body: getMessageContentString(message.content), + githubToken: githubAccessToken, + }); + if (!createdIssue?.id) { + throw new Error("Failed to create issue comment"); + } + newMessages.push( + ...[ + new RemoveMessage({ + id: message.id ?? "", + }), + new HumanMessage({ + ...message, + additional_kwargs: { + githubIssueId: githubIssueId, + githubIssueCommentId: createdIssue.id, + }, + }), + ], + ); + }); + + await Promise.all(createCommentsPromise); + + let newPlannerId: string | undefined; + if (plannerStatus === "interrupted") { + if (!state.plannerSession?.threadId) { + throw new Error("No planner session found. Unable to resume planner."); + } + // We need to resume the planner session via a 'response' so that it can re-plan + const plannerResume: HumanResponse = { + type: "response", + args: "resume planner", + }; + logger.info("Resuming planner session"); + const newPlannerRun = await langGraphClient.runs.create( + state.plannerSession?.threadId, + PLANNER_GRAPH_ID, + { + command: { + resume: plannerResume, + }, + }, + ); + newPlannerId = newPlannerRun.run_id; + logger.info("Planner session resumed", { + runId: newPlannerRun.run_id, + threadId: state.plannerSession.threadId, + }); + } + + // After creating the new comment, we can add the message to state and end. + const commandUpdate: ManagerGraphUpdate = { + messages: newMessages, + ...(newPlannerId && state.plannerSession?.threadId + ? { + plannerSession: { + threadId: state.plannerSession.threadId, + runId: newPlannerId, + }, + } + : {}), + }; + return new Command({ + update: commandUpdate, + goto: END, + }); + } + + // Issue has been created, and any missing human messages have been added to it. + + const commandUpdate: ManagerGraphUpdate = { + messages: newMessages, + ...(githubIssueId ? { githubIssueId } : {}), + }; + + if ( + (toolCallArgs.route as any) === "update_programmer" || + (toolCallArgs.route as any) === "update_planner" || + (toolCallArgs.route as any) === "resume_and_update_planner" + ) { + // If the route is one of the above, we don't need to do anything since the issue now contains + // the new messages, and the coding agent will handle pulling them in. This should never be + // reachable since we should return early after adding the Github comment, but include anyways... + return new Command({ + update: commandUpdate, + goto: END, + }); + } + + if (toolCallArgs.route === "start_planner") { + // Always kickoff a new start planner node. This will enqueue new runs on the planner graph. + return new Command({ + update: commandUpdate, + goto: "start-planner", + }); + } + + throw new Error(`Invalid route: ${toolCallArgs.route}`); +} diff --git a/apps/open-swe/src/graphs/manager/nodes/classify-message/prompts.ts b/apps/open-swe/src/graphs/manager/nodes/classify-message/prompts.ts new file mode 100644 index 00000000..aa0f88e2 --- /dev/null +++ b/apps/open-swe/src/graphs/manager/nodes/classify-message/prompts.ts @@ -0,0 +1,71 @@ +export const UPDATE_PROGRAMMER_ROUTING_OPTION = `- update_programmer: You should call this route if the user's message should be added to the programmer's currently running session. This should be called if you determine the user is trying to provide extra context to the programmer's current session.\n`; + +export const START_PLANNER_ROUTING_OPTION = `- start_planner: You should call this route if the user's message is a complete request you can send to the planner, which it can use to generate a plan. This route may be called when the planner has not started yet.\n`; + +export const UPDATE_PLANNER_ROUTING_OPTION = `- update_planner: You should call this route if the user sends a new message containing anything from a related request that the planner should plan for, additional context about their previous request/the codebase, or something which the planner should be aware of.\n`; + +export const RESUME_AND_UPDATE_PLANNER_ROUTING_OPTION = `- resume_and_update_planner: You should call this route if the planner is currently interrupted, and the user's message includes additional context/related requests the which require updates to the plan. This will resume the planner so that it can handle the user's new request.\n`; + +export const CREATE_NEW_ISSUE_ROUTING_OPTION = `- create_new_issue: Call this route if the user's request should create a new GitHub issue, and should be executed independently from the current request. This should only be called if the new request does not depend on the current request.\n`; + +// This should only be included if the task plan exists. +export const TASK_PLAN_PROMPT = `# Task Plan +The following is the current state of the task plan generated by the planner. You should use this as context when determining where to route the user's message, and how to reply to them. +{TASK_PLAN} +\n\n`; + +// This should only be included if the proposed plan exists, and the task plan does NOT exist. +export const PROPOSED_PLAN_PROMPT = `# Proposed Plan +The following is the proposed plan the planner agent generated, and the user has yet to accept. You should use this as context when determining where to route the user's message, and how to reply to them. +{PROPOSED_PLAN} +\n\n`; + +export const CONVERSATION_HISTORY_PROMPT = `# Conversation History +The following is the conversation history between the user and you. This does not include their most recent message, which is the one you are currently classifying. You should use this as context when determining where to route the user's message, and how to reply to them. +{CONVERSATION_HISTORY} +\n\n`; + +// This prompt does not generate the route, it only generates the response. +export const CLASSIFICATION_SYSTEM_PROMPT = `# Identity +You're a highly intelligent AI software engineering manager, tasked with identifying the user's intent, and responding to their message, and determining how you'll route it to the proper AI assistant. +You're acting as the manager in a larger AI coding agent system, tasked with responding, routing and taking management actions based on the user's requests. + +# Instructions +Carefully examine the user's message, along with the conversation history provided (or none, if it's the first message they sent) to you in this system message below. +Using their most recent request, the conversation history, and the current status of your two AI assistants (programmer and planner), generate a response to send to the user, and a route to take. + +Below you're provided with routes you may take given the user's request. Your response should not explicitly mention the route you want to take, but it should be able to be inferred by your response. +Ensure your response is clear, and concise. + +Although you're only supposed to classify & respond to the latest message, this does not mean you should look at it in isolation. You should consider the conversation history as a whole, and the current status of your two AI assistants (programmer and planner) to determine how to respond & route the user's new message. + +# Context +Although it's not shown here, you do have access to the full repository contents the user is referencing. Because of this, you should always assume you'll have access to any/all files or folders the user is referencing. + +# Assistant Statuses +The planner's current status is: {PLANNER_STATUS} +The programmer's current status is: {PROGRAMMER_STATUS} + +{TASK_PLAN_PROMPT} +{CONVERSATION_HISTORY_PROMPT} + +# Routing Options +Based on all of the context provided above, generate a response to send to the user, including messaging about the route you'll select from the below options in your next step. +Your routing options are: +- no_op: This should be called when the user's message is not a new request, additional context, or a new issue to create. This should only be called when none of the routing options are appropriate. +{UPDATE_PROGRAMMER_ROUTING_OPTION}{START_PLANNER_ROUTING_OPTION}{UPDATE_PLANNER_ROUTING_OPTION}{RESUME_AND_UPDATE_PLANNER_ROUTING_OPTION}{CREATE_NEW_ISSUE_ROUTING_OPTION} + +# Response +Your response should be clear, concise and straight to the point. Do NOT include any additional context, such as an idea for how to implement their request. + +**IMPORTANT**: +Remember, you are ONLY allowed to route to one of: {ROUTING_OPTIONS} +You should NEVER try to route to an option which is not listed above, even if the conversation history shows you calling a route that's not shown above. +Routes are not always available to be called, so ensure you only call one of the options shown above. + +You're only acting as a manager, and thus your response to the user's message should be a short message about which route you'll take, WITHOUT actually referencing the route you'll take. +Your manager will be very happy with you if you're able to articulate the route you plan to take, without actually mentioning the route! Ensure each response to the user is slightly different too. You should never repeat responses. + +You do not need to explain why you're taking that route to the user. +Your response will not exceed two sentences. You will be rewarded for being concise. +`; diff --git a/apps/open-swe/src/graphs/manager/nodes/classify-message/schemas.ts b/apps/open-swe/src/graphs/manager/nodes/classify-message/schemas.ts new file mode 100644 index 00000000..262a1130 --- /dev/null +++ b/apps/open-swe/src/graphs/manager/nodes/classify-message/schemas.ts @@ -0,0 +1,27 @@ +import { z } from "zod"; + +export const BASE_CLASSIFICATION_SCHEMA = z.object({ + internal_reasoning: z + .string() + .describe( + "The reasoning being the decision of the route you're going to take. This is internal, and not shown to the user, so you may be technical in your reasoning. Please include all the reasoning, and context which led you to choose this route.", + ), + response: z + .string() + .describe( + "The response to send to the user. This should be clear, concise, and include any additional context the user may need to know about how/why you're handling their new message.", + ), + route: z + .enum(["no_op"]) + .describe("The route to take to handle the user's new message."), +}); + +export function createClassificationSchema(enumOptions: [string, ...string[]]) { + const schema = BASE_CLASSIFICATION_SCHEMA.extend({ + route: z + .enum(enumOptions) + .describe("The route to take to handle the user's new message."), + }); + + return schema; +} diff --git a/apps/open-swe/src/graphs/manager/nodes/classify-message/utils.ts b/apps/open-swe/src/graphs/manager/nodes/classify-message/utils.ts new file mode 100644 index 00000000..575ad5d9 --- /dev/null +++ b/apps/open-swe/src/graphs/manager/nodes/classify-message/utils.ts @@ -0,0 +1,168 @@ +import { TaskPlan } from "@open-swe/shared/open-swe/types"; +import { + AIMessage, + BaseMessage, + isAIMessage, + isHumanMessage, + isToolMessage, + ToolMessage, +} from "@langchain/core/messages"; +import { z } from "zod"; +import { removeLastHumanMessage } from "../../../../utils/message/modify-array.js"; +import { formatPlanPrompt } from "../../../../utils/plan-prompt.js"; +import { getActivePlanItems } from "@open-swe/shared/open-swe/tasks"; +import { + getHumanMessageString, + getToolMessageString, + getUnknownMessageString, +} from "../../../../utils/message/content.js"; +import { getMessageContentString } from "@open-swe/shared/messages"; +import { ThreadStatus } from "@langchain/langgraph-sdk"; +import { + CLASSIFICATION_SYSTEM_PROMPT, + CONVERSATION_HISTORY_PROMPT, + CREATE_NEW_ISSUE_ROUTING_OPTION, + UPDATE_PLANNER_ROUTING_OPTION, + UPDATE_PROGRAMMER_ROUTING_OPTION, + PROPOSED_PLAN_PROMPT, + RESUME_AND_UPDATE_PLANNER_ROUTING_OPTION, + START_PLANNER_ROUTING_OPTION, + TASK_PLAN_PROMPT, +} from "./prompts.js"; +import { createClassificationSchema } from "./schemas.js"; + +const THREAD_STATUS_READABLE_STRING_MAP = { + not_started: "not started", + busy: "currently running", + idle: "not running", + interrupted: "interrupted -- awaiting human response", + error: "error", +}; + +function formatMessageForClassification(message: BaseMessage): string { + if (isHumanMessage(message)) { + return getHumanMessageString(message); + } + + // Special formatting for the AI messages as we don't want to show what status was called since the available statuses are dynamic. + if (isAIMessage(message)) { + const aiMessage = message as AIMessage; + const toolCallName = aiMessage.tool_calls?.[0]?.name; + const toolCallResponseStr = aiMessage.tool_calls?.[0]?.args?.response; + const toolCallStr = + toolCallName && toolCallResponseStr + ? `Tool call: ${toolCallName}\nArgs: ${JSON.stringify({ response: toolCallResponseStr }, null)}\n` + : ""; + const content = getMessageContentString(aiMessage.content); + return `\nContent: ${content}\n${toolCallStr}`; + } + + if (isToolMessage(message)) { + const toolMessage = message as ToolMessage; + return getToolMessageString(toolMessage); + } + + return getUnknownMessageString(message); +} + +export function createClassificationPromptAndToolSchema(inputs: { + programmerStatus: ThreadStatus | "not_started"; + plannerStatus: ThreadStatus | "not_started"; + messages: BaseMessage[]; + taskPlan: TaskPlan; + proposedPlan?: string[]; +}): { + prompt: string; + schema: z.ZodTypeAny; +} { + const conversationHistoryWithoutLatest = removeLastHumanMessage( + inputs.messages, + ); + const formattedTaskPlanPrompt = inputs.taskPlan + ? TASK_PLAN_PROMPT.replaceAll( + "{TASK_PLAN}", + formatPlanPrompt(getActivePlanItems(inputs.taskPlan)), + ) + : null; + const formattedProposedPlanPrompt = inputs.proposedPlan?.length + ? PROPOSED_PLAN_PROMPT.replace( + "{PROPOSED_PLAN}", + inputs.proposedPlan + .map((p, index) => ` ${index + 1}: ${p}`) + .join("\n"), + ) + : null; + + const formattedConversationHistoryPrompt = + conversationHistoryWithoutLatest?.length + ? CONVERSATION_HISTORY_PROMPT.replaceAll( + "{CONVERSATION_HISTORY}", + conversationHistoryWithoutLatest + .map(formatMessageForClassification) + .join("\n"), + ) + : null; + + const programmerRunning = inputs.programmerStatus === "busy"; + const plannerRunning = inputs.plannerStatus === "busy"; + const plannerInterrupted = inputs.plannerStatus === "interrupted"; + const plannerNotStarted = inputs.plannerStatus === "not_started"; + + const showCreateIssueOption = + inputs.programmerStatus !== "not_started" || + inputs.plannerStatus !== "not_started"; + + const routingOptions: [string, ...string[]] = [ + "no_op", + ...(programmerRunning ? ["update_programmer"] : []), + ...(plannerNotStarted ? ["start_planner"] : []), + ...(plannerRunning ? ["update_planner"] : []), + ...(plannerInterrupted ? ["resume_and_update_planner"] : []), + ...(showCreateIssueOption ? ["create_new_issue"] : []), + ]; + + const prompt = CLASSIFICATION_SYSTEM_PROMPT.replaceAll( + "{PROGRAMMER_STATUS}", + THREAD_STATUS_READABLE_STRING_MAP[inputs.programmerStatus], + ) + .replaceAll( + "{PLANNER_STATUS}", + THREAD_STATUS_READABLE_STRING_MAP[inputs.plannerStatus], + ) + .replaceAll("{ROUTING_OPTIONS}", routingOptions.join(", ")) + .replaceAll( + "{UPDATE_PROGRAMMER_ROUTING_OPTION}", + programmerRunning ? UPDATE_PROGRAMMER_ROUTING_OPTION : "", + ) + .replaceAll( + "{START_PLANNER_ROUTING_OPTION}", + plannerNotStarted ? START_PLANNER_ROUTING_OPTION : "", + ) + .replaceAll( + "{UPDATE_PLANNER_ROUTING_OPTION}", + plannerRunning ? UPDATE_PLANNER_ROUTING_OPTION : "", + ) + .replaceAll( + "{RESUME_AND_UPDATE_PLANNER_ROUTING_OPTION}", + plannerNotStarted ? RESUME_AND_UPDATE_PLANNER_ROUTING_OPTION : "", + ) + .replaceAll( + "{CREATE_NEW_ISSUE_ROUTING_OPTION}", + showCreateIssueOption ? CREATE_NEW_ISSUE_ROUTING_OPTION : "", + ) + .replaceAll( + "{TASK_PLAN_PROMPT}", + formattedTaskPlanPrompt ?? formattedProposedPlanPrompt ?? "", + ) + .replaceAll( + "{CONVERSATION_HISTORY_PROMPT}", + formattedConversationHistoryPrompt ?? "", + ); + + const schema = createClassificationSchema(routingOptions); + + return { + prompt, + schema, + }; +} diff --git a/apps/open-swe/src/graphs/manager/nodes/create-new-session.ts b/apps/open-swe/src/graphs/manager/nodes/create-new-session.ts index 01b6c0ec..835eaed0 100644 --- a/apps/open-swe/src/graphs/manager/nodes/create-new-session.ts +++ b/apps/open-swe/src/graphs/manager/nodes/create-new-session.ts @@ -4,7 +4,7 @@ import { ManagerGraphState, ManagerGraphUpdate, } from "@open-swe/shared/open-swe/manager/types"; -import { createIssueTitleAndBodyFromMessages } from "../utils/generate-issue-fields.js"; +import { createIssueFieldsFromMessages } from "../utils/generate-issue-fields.js"; import { MANAGER_GRAPH_ID } from "@open-swe/shared/constants"; import { createLangGraphClient } from "../../../utils/langgraph-client.js"; import { createIssue } from "../../../utils/github/api.js"; @@ -30,9 +30,9 @@ export async function createNewSession( state: ManagerGraphState, config: GraphConfig, ): Promise { - const titleAndContent = await createIssueTitleAndBodyFromMessages( + const titleAndContent = await createIssueFieldsFromMessages( state.messages, - config, + config.configurable, ); const { githubAccessToken } = getGitHubTokensFromConfig(config); const newIssue = await createIssue({ diff --git a/apps/open-swe/src/graphs/manager/nodes/index.ts b/apps/open-swe/src/graphs/manager/nodes/index.ts index a1751bad..99a795cb 100644 --- a/apps/open-swe/src/graphs/manager/nodes/index.ts +++ b/apps/open-swe/src/graphs/manager/nodes/index.ts @@ -1,4 +1,4 @@ export * from "./initialize-github-issue.js"; -export * from "./classify-message.js"; +export * from "./classify-message/index.js"; export * from "./start-planner.js"; export * from "./create-new-session.js"; diff --git a/apps/open-swe/src/graphs/manager/utils/generate-issue-fields.ts b/apps/open-swe/src/graphs/manager/utils/generate-issue-fields.ts index 45d47fbd..bd8f7e2f 100644 --- a/apps/open-swe/src/graphs/manager/utils/generate-issue-fields.ts +++ b/apps/open-swe/src/graphs/manager/utils/generate-issue-fields.ts @@ -1,15 +1,14 @@ import { BaseMessage } from "@langchain/core/messages"; import { GraphConfig } from "@open-swe/shared/open-swe/types"; -import { traceable } from "langsmith/traceable"; import { z } from "zod"; import { loadModel, Task } from "../../../utils/load-model.js"; import { getMessageString } from "../../../utils/message/content.js"; -async function createIssueTitleAndBodyFromMessagesFunc( +export async function createIssueFieldsFromMessages( messages: BaseMessage[], - config: GraphConfig, + configurable: GraphConfig["configurable"], ): Promise<{ title: string; body: string }> { - const model = await loadModel(config, Task.ACTION_GENERATOR); + const model = await loadModel({ configurable }, Task.ACTION_GENERATOR); const githubIssueTool = { name: "create_github_issue", description: "Create a new GitHub issue with the given title and body.", @@ -31,7 +30,7 @@ async function createIssueTitleAndBodyFromMessagesFunc( tool_choice: githubIssueTool.name, parallel_tool_calls: false, }) - .withConfig({ tags: ["nostream"] }); + .withConfig({ tags: ["nostream"], runName: "create-issue-fields" }); const prompt = `You're an AI programmer, tasked with taking the conversation history provided below, and creating a new GitHub issue. Ensure the issue title and body are both clear and concise. Do not hallucinate any information not found in the conversation history. @@ -54,8 +53,3 @@ With the above conversation history in mind, please call the ${githubIssueTool.n } return toolCall.args as z.infer; } - -export const createIssueTitleAndBodyFromMessages = traceable( - createIssueTitleAndBodyFromMessagesFunc, - { name: "create-issue-title-and-body-from-messages" }, -); diff --git a/apps/open-swe/src/graphs/planner/index.ts b/apps/open-swe/src/graphs/planner/index.ts index 524b27f5..0a2e4da8 100644 --- a/apps/open-swe/src/graphs/planner/index.ts +++ b/apps/open-swe/src/graphs/planner/index.ts @@ -14,6 +14,7 @@ import { prepareGraphState, notetaker, takeActions, + determineNeedsContext, } from "./nodes/index.js"; import { isAIMessage } from "@langchain/core/messages"; import { initializeSandbox } from "../shared/initialize-sandbox.js"; @@ -50,7 +51,12 @@ const workflow = new StateGraph(PlannerGraphStateObj, GraphConfiguration) .addNode("take-plan-actions", takeActions) .addNode("generate-plan", generatePlan) .addNode("notetaker", notetaker) - .addNode("interrupt-proposed-plan", interruptProposedPlan) + .addNode("interrupt-proposed-plan", interruptProposedPlan, { + ends: [END, "determine-needs-context"], + }) + .addNode("determine-needs-context", determineNeedsContext, { + ends: ["generate-plan-context-action", "generate-plan"], + }) .addEdge(START, "prepare-graph-state") .addEdge("initialize-sandbox", "generate-plan-context-action") .addConditionalEdges( @@ -60,8 +66,7 @@ const workflow = new StateGraph(PlannerGraphStateObj, GraphConfiguration) ) .addEdge("take-plan-actions", "generate-plan-context-action") .addEdge("generate-plan", "notetaker") - .addEdge("notetaker", "interrupt-proposed-plan") - .addEdge("interrupt-proposed-plan", END); + .addEdge("notetaker", "interrupt-proposed-plan"); export const graph = workflow.compile(); graph.name = "Open SWE - Planner"; diff --git a/apps/open-swe/src/graphs/planner/nodes/determine-needs-context.ts b/apps/open-swe/src/graphs/planner/nodes/determine-needs-context.ts new file mode 100644 index 00000000..5a1e7eb1 --- /dev/null +++ b/apps/open-swe/src/graphs/planner/nodes/determine-needs-context.ts @@ -0,0 +1,163 @@ +import { Command } from "@langchain/langgraph"; +import { + PlannerGraphState, + PlannerGraphUpdate, +} from "@open-swe/shared/open-swe/planner/types"; +import { GraphConfig } from "@open-swe/shared/open-swe/types"; +import { z } from "zod"; +import { loadModel, Task } from "../../../utils/load-model.js"; +import { getMissingMessages } from "../../../utils/github/issue-messages.js"; +import { getMessageString } from "../../../utils/message/content.js"; +import { isHumanMessage } from "@langchain/core/messages"; +import { getMessageContentString } from "@open-swe/shared/messages"; +import { filterHiddenMessages } from "../../../utils/message/filter-hidden.js"; +import { createLogger, LogLevel } from "../../../utils/logger.js"; + +const logger = createLogger(LogLevel.INFO, "DetermineNeedsContext"); + +const SYSTEM_PROMPT = `You are a terminal-based agentic coding assistant built by LangChain that enables natural language interaction with local codebases. You excel at being precise, safe, and helpful in your analysis. + + +Context Gathering Assistant - Read-Only Phase + + + +Your sole objective in this step is to determine whether or not the user's followup request requires additional context to be gathered in order to update the plan/add additional steps to the plan. + + + +You're provided with these main pieces of information: +- **Conversation history**: This is the full conversation history between you, the user, and including any actions you took while gathering context. +- **Context gathering notes**: This is the notes you took while gathering context. Includes the most relevant context you discovered while gathering context for the plan. +- **Proposed plan**: This is the plan you generated for the user's request, which the user is likely trying to follow up on (e.g. modify it in some way, or add new step(s)). +- **User followup request**: This is the specific followup request made by the user (the conversation history will also include this). This is the message you should look at when determining whether or not you need to gather more context before you can update the proposed plan. + +Given this information, carefully read over it all and determine whether or not you need to gather more context before you can update the proposed plan. +You may already have enough context from the conversation history and the actions you executed, or the notes you took while gathering context, to update the proposed plan. + +The state of the repository has NOT changed since you last gathered context & proposed the plan. + +To make your decision, you must first provide reasoning for why you need to gather more context, or why you already have enough context. Then, make your decision. +Both of these steps should be executed by calling the \`determine_context\` tool. + + + +{CONVERSATION_HISTORY} + + + +{CONTEXT_GATHERING_NOTES} + + + +{PROPOSED_PLAN} + + + +{USER_FOLLOWUP_REQUEST} + + + +Once again, with all of the above information, determine whether or not you need to gather more context before you can accurately update the proposed plan. + +`; + +function formatSystemPrompt(state: PlannerGraphState): string { + const formattedConversationHistoryPrompt = state.messages + .map(getMessageString) + .join("\n"); + const formattedProposedPlan = state.proposedPlan + .map((p, index) => ` ${index + 1}. ${p}`) + .join("\n"); + const userFollowupRequestMsg = state.messages.findLast(isHumanMessage); + if (!userFollowupRequestMsg) { + throw new Error("User followup request not found."); + } + const userFollowupRequestStr = getMessageContentString( + userFollowupRequestMsg.content, + ); + + return SYSTEM_PROMPT.replace( + "{CONVERSATION_HISTORY}", + formattedConversationHistoryPrompt, + ) + .replace("{CONTEXT_GATHERING_NOTES}", state.contextGatheringNotes) + .replace("{PROPOSED_PLAN}", formattedProposedPlan) + .replace("{USER_FOLLOWUP_REQUEST}", userFollowupRequestStr); +} + +const determineContextSchema = z.object({ + reasoning: z + .string() + .describe( + "The reasoning for whether or not you have enough context to update the proposed plan, or why you need to gather more context before you can update the proposed plan.", + ), + decision: z + .enum(["have_context", "need_context"]) + .describe( + "Whether or not you have enough context to update the proposed plan, or if you need to gather more context before you can accurately update the proposed plan. " + + "If you have enough context to update the plan, respond with 'have_context'. " + + "If you need to gather more context, respond with 'need_context'.", + ), +}); +const determineContextTool = { + name: "determine_context", + description: + "Determine whether or not you have enough context to update the proposed plan, or if you need to gather more context before you can accurately update the proposed plan.", + schema: determineContextSchema, +}; + +export async function determineNeedsContext( + state: PlannerGraphState, + config: GraphConfig, +): Promise { + const [missingMessages, model] = await Promise.all([ + getMissingMessages(state, config), + loadModel(config, Task.CLASSIFICATION), + ]); + if (!missingMessages.length) { + throw new Error( + "Can not determine if more context is needed if there are no missing messages.", + ); + } + const modelWithTools = model.bindTools([determineContextTool], { + tool_choice: determineContextTool.name, + parallel_tool_calls: false, + }); + + const response = await modelWithTools.invoke([ + { + role: "user", + content: formatSystemPrompt({ + ...state, + messages: [...filterHiddenMessages(state.messages), ...missingMessages], + }), + }, + ]); + + const toolCall = response.tool_calls?.[0]; + if (!toolCall) { + throw new Error("No tool call found."); + } + + const commandUpdate: PlannerGraphUpdate = { + messages: missingMessages, + }; + + const shouldGatherContext = + (toolCall.args as z.infer).decision === + "need_context"; + logger.info( + "Determined whether or not additional context is needed to update plan", + { + ...toolCall.args, + }, + ); + + return new Command({ + goto: shouldGatherContext + ? "generate-plan-context-action" + : "generate-plan", + update: commandUpdate, + }); +} diff --git a/apps/open-swe/src/graphs/planner/nodes/generate-message/index.ts b/apps/open-swe/src/graphs/planner/nodes/generate-message/index.ts index a58b8a2a..562be677 100644 --- a/apps/open-swe/src/graphs/planner/nodes/generate-message/index.ts +++ b/apps/open-swe/src/graphs/planner/nodes/generate-message/index.ts @@ -18,7 +18,7 @@ import { SYSTEM_PROMPT } from "./prompt.js"; import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { getMissingMessages } from "../../../../utils/github/issue-messages.js"; import { filterHiddenMessages } from "../../../../utils/message/filter-hidden.js"; -import { getTaskPlanFromIssue } from "../../../../utils/github/issue-task.js"; +import { getPlansFromIssue } from "../../../../utils/github/issue-task.js"; import { createRgTool } from "../../../../tools/rg.js"; import { formatCustomRulesPrompt } from "../../../../utils/custom-rules.js"; import { createPlannerNotesTool } from "../../../../tools/planner-notes.js"; @@ -71,9 +71,9 @@ export async function generateAction( parallel_tool_calls: true, }); - const [missingMessages, latestTaskPlan] = await Promise.all([ + const [missingMessages, { taskPlan: latestTaskPlan }] = await Promise.all([ getMissingMessages(state, config), - getTaskPlanFromIssue(state, config), + getPlansFromIssue(state, config), ]); const response = await modelWithTools .withConfig({ tags: ["nostream"] }) diff --git a/apps/open-swe/src/graphs/planner/nodes/generate-plan/index.ts b/apps/open-swe/src/graphs/planner/nodes/generate-plan/index.ts index 90ee4fe1..3a2172c1 100644 --- a/apps/open-swe/src/graphs/planner/nodes/generate-plan/index.ts +++ b/apps/open-swe/src/graphs/planner/nodes/generate-plan/index.ts @@ -1,3 +1,4 @@ +import { v4 as uuidv4 } from "uuid"; import { isAIMessage, ToolMessage } from "@langchain/core/messages"; import { createSessionPlanToolFields } from "../../../../tools/index.js"; import { GraphConfig } from "@open-swe/shared/open-swe/types"; @@ -17,6 +18,7 @@ import { z } from "zod"; import { formatCustomRulesPrompt } from "../../../../utils/custom-rules.js"; import { getPlannerNotes } from "../../utils/get-notes.js"; import { PLANNER_NOTES_PROMPT, SYSTEM_PROMPT } from "./prompt.js"; +import { DO_NOT_RENDER_ID_PREFIX } from "@open-swe/shared/constants"; function formatSystemPrompt(state: PlannerGraphState): string { // It's a followup if there's more than one human message. @@ -76,7 +78,8 @@ export async function generatePlan( ...(optionalToolMessage ? [optionalToolMessage] : []), ]); - if (!response.tool_calls?.length) { + const toolCall = response.tool_calls?.[0]; + if (!toolCall) { throw new Error("Failed to generate plan"); } @@ -86,12 +89,19 @@ export async function generatePlan( newSessionId = await stopSandbox(state.sandboxSessionId); } - const proposedPlanArgs = response.tool_calls[0].args as z.infer< + const proposedPlanArgs = toolCall.args as z.infer< typeof sessionPlanTool.schema >; + const toolResponse = new ToolMessage({ + id: `${DO_NOT_RENDER_ID_PREFIX}${uuidv4()}`, + tool_call_id: toolCall.id ?? "", + content: "Successfully saved plan.", + name: sessionPlanTool.name, + }); + return { - messages: [response], + messages: [response, toolResponse], proposedPlanTitle: proposedPlanArgs.title, proposedPlan: proposedPlanArgs.plan, ...(newSessionId && { sandboxSessionId: newSessionId }), diff --git a/apps/open-swe/src/graphs/planner/nodes/index.ts b/apps/open-swe/src/graphs/planner/nodes/index.ts index 6fdfba0d..dd4de483 100644 --- a/apps/open-swe/src/graphs/planner/nodes/index.ts +++ b/apps/open-swe/src/graphs/planner/nodes/index.ts @@ -4,3 +4,4 @@ export * from "./generate-plan/index.js"; export * from "./notetaker.js"; export * from "./proposed-plan.js"; export * from "./prepare-state.js"; +export * from "./determine-needs-context.js"; diff --git a/apps/open-swe/src/graphs/planner/nodes/notetaker.ts b/apps/open-swe/src/graphs/planner/nodes/notetaker.ts index d57b2751..6c8665b3 100644 --- a/apps/open-swe/src/graphs/planner/nodes/notetaker.ts +++ b/apps/open-swe/src/graphs/planner/nodes/notetaker.ts @@ -1,3 +1,4 @@ +import { v4 as uuidv4 } from "uuid"; import { z } from "zod"; import { GraphConfig } from "@open-swe/shared/open-swe/types"; import { @@ -9,6 +10,8 @@ import { getMessageString } from "../../../utils/message/content.js"; import { getUserRequest } from "../../../utils/user-request.js"; import { formatCustomRulesPrompt } from "../../../utils/custom-rules.js"; import { getPlannerNotes } from "../utils/get-notes.js"; +import { ToolMessage } from "@langchain/core/messages"; +import { DO_NOT_RENDER_ID_PREFIX } from "@open-swe/shared/constants"; const PLANNER_NOTES_PROMPT = `You've also taken technical notes throughout the context gathering process. Ensure you include/incorporate these notes, or the highest quality parts of these notes in your conclusion notes. @@ -131,9 +134,15 @@ ${state.messages.map(getMessageString).join("\n")}`; if (!toolCall) { throw new Error("Failed to generate plan"); } + const toolResponse = new ToolMessage({ + id: `${DO_NOT_RENDER_ID_PREFIX}${uuidv4()}`, + tool_call_id: toolCall.id ?? "", + content: "Successfully saved notes.", + name: condenseContextTool.name, + }); return { - messages: [response], + messages: [response, toolResponse], contextGatheringNotes: ( toolCall.args as z.infer ).notes, diff --git a/apps/open-swe/src/graphs/planner/nodes/proposed-plan.ts b/apps/open-swe/src/graphs/planner/nodes/proposed-plan.ts index 341172a8..291b41e8 100644 --- a/apps/open-swe/src/graphs/planner/nodes/proposed-plan.ts +++ b/apps/open-swe/src/graphs/planner/nodes/proposed-plan.ts @@ -21,12 +21,12 @@ import { DO_NOT_RENDER_ID_PREFIX, PROGRAMMER_GRAPH_ID, } from "@open-swe/shared/constants"; -import { - PlannerGraphState, - PlannerGraphUpdate, -} from "@open-swe/shared/open-swe/planner/types"; +import { PlannerGraphState } from "@open-swe/shared/open-swe/planner/types"; import { createLangGraphClient } from "../../../utils/langgraph-client.js"; -import { addTaskPlanToIssue } from "../../../utils/github/issue-task.js"; +import { + addProposedPlanToIssue, + addTaskPlanToIssue, +} from "../../../utils/github/issue-task.js"; import { createLogger, LogLevel } from "../../../utils/logger.js"; import { ACCEPTED_PLAN_NODE_ID, @@ -115,21 +115,25 @@ async function startProgrammerRun(input: { config, runInput.taskPlan, ); - return { - programmerSession: { - threadId: programmerThreadId, - runId: run.run_id, + + return new Command({ + goto: END, + update: { + programmerSession: { + threadId: programmerThreadId, + runId: run.run_id, + }, + sandboxSessionId: runInput.sandboxSessionId, + taskPlan: runInput.taskPlan, + messages: newMessages, }, - sandboxSessionId: runInput.sandboxSessionId, - taskPlan: runInput.taskPlan, - messages: newMessages, - }; + }); } export async function interruptProposedPlan( state: PlannerGraphState, config: GraphConfig, -): Promise { +): Promise { const { proposedPlan } = state; if (!proposedPlan.length) { throw new Error("No proposed plan found."); @@ -174,7 +178,19 @@ export async function interruptProposedPlan( }); } - const interruptRes = interrupt({ + await addProposedPlanToIssue( + { + githubIssueId: state.githubIssueId, + targetRepository: state.targetRepository, + }, + config, + proposedPlan, + ); + + const interruptResponse = interrupt< + HumanInterrupt, + HumanResponse[] | HumanResponse + >({ action_request: { action: PLAN_INTERRUPT_ACTION_TITLE, args: { @@ -189,21 +205,28 @@ export async function interruptProposedPlan( }, description: `A new plan has been generated for your request. Please review it and either approve it, edit it, respond to it, or ignore it. Responses will be passed to an LLM where it will rewrite then plan. If editing the plan, ensure each step in the plan is separated by "${PLAN_INTERRUPT_DELIMITER}".`, - })[0]; + }); - if (interruptRes.type === "response") { - // Plan was responded to, route to the rewrite plan node. - throw new Error("RESPONDING TO PLAN NOT IMPLEMENTED."); + const humanResponse: HumanResponse = Array.isArray(interruptResponse) + ? interruptResponse[0] + : interruptResponse; + + if (humanResponse.type === "response") { + // Plan was responded to, route to the needs-context node which will determine + // if we need more context, or can go right to the planning step. + return new Command({ + goto: "determine-needs-context", + }); } - if (interruptRes.type === "ignore") { + if (humanResponse.type === "ignore") { // Plan was ignored, end the process. return new Command({ goto: END, }); } - if (interruptRes.type === "accept") { + if (humanResponse.type === "accept") { planItems = proposedPlan.map((p, index) => ({ index, plan: p, @@ -216,8 +239,8 @@ export async function interruptProposedPlan( planItems, { existingTaskPlan: state.taskPlan }, ); - } else if (interruptRes.type === "edit") { - const editedPlan = (interruptRes.args as ActionRequest).args.plan + } else if (humanResponse.type === "edit") { + const editedPlan = (humanResponse.args as ActionRequest).args.plan .split(PLAN_INTERRUPT_DELIMITER) .map((step: string) => step.trim()); @@ -234,7 +257,7 @@ export async function interruptProposedPlan( { existingTaskPlan: state.taskPlan }, ); } else { - throw new Error("Unknown interrupt type." + interruptRes.type); + throw new Error("Unknown interrupt type." + humanResponse.type); } return await startProgrammerRun({ @@ -247,7 +270,7 @@ export async function interruptProposedPlan( createAcceptedPlanMessage({ planTitle: state.proposedPlanTitle, planItems, - interruptType: interruptRes.type, + interruptType: humanResponse.type, }), ], }); diff --git a/apps/open-swe/src/graphs/planner/utils/followup.ts b/apps/open-swe/src/graphs/planner/utils/followup.ts index 213e3085..df0bb59b 100644 --- a/apps/open-swe/src/graphs/planner/utils/followup.ts +++ b/apps/open-swe/src/graphs/planner/utils/followup.ts @@ -11,7 +11,8 @@ const followupMessagePrompt = ` The user is sending a followup request for you to generate a plan for. You are provided with the following context to aid in your new plan context gathering steps: - The previous user requests, along with the tasks, and task summaries you generated for these previous requests. - The summaries of the actions you took, and their results from previous planning sessions. - - You are only provided this information as context to reference when gathering context for the new plan, or for making changes to the previously generated plan. + - You are only provided this information as context to reference when gathering context for the new plan, or for making changes to the proposed plan. + - If the user requests changes/additions to the proposed plan, your goal is to make as few changes/additions as possible, only addressing the specific changes the user requested. {PREVIOUS_PLAN} `; diff --git a/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts b/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts index 853cb06f..4e8e7f21 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts @@ -24,7 +24,7 @@ import { } from "./prompt.js"; import { getRepoAbsolutePath } from "@open-swe/shared/git"; import { getMissingMessages } from "../../../../utils/github/issue-messages.js"; -import { getTaskPlanFromIssue } from "../../../../utils/github/issue-task.js"; +import { getPlansFromIssue } from "../../../../utils/github/issue-task.js"; import { createRgTool } from "../../../../tools/rg.js"; import { createInstallDependenciesTool } from "../../../../tools/install-dependencies.js"; import { formatCustomRulesPrompt } from "../../../../utils/custom-rules.js"; @@ -97,9 +97,9 @@ export async function generateAction( parallel_tool_calls: true, }); - const [missingMessages, latestTaskPlan] = await Promise.all([ + const [missingMessages, { taskPlan: latestTaskPlan }] = await Promise.all([ getMissingMessages(state, config), - getTaskPlanFromIssue(state, config), + getPlansFromIssue(state, config), ]); const response = await modelWithTools.invoke([ diff --git a/apps/open-swe/src/graphs/shared/initialize-sandbox.ts b/apps/open-swe/src/graphs/shared/initialize-sandbox.ts index 4a007adc..b8f27997 100644 --- a/apps/open-swe/src/graphs/shared/initialize-sandbox.ts +++ b/apps/open-swe/src/graphs/shared/initialize-sandbox.ts @@ -243,7 +243,7 @@ export async function initializeSandbox( emitStepEvent(baseCloneRepoAction, "pending"); const cloneRepoRes = await cloneRepo(sandbox, targetRepository, { githubInstallationToken, - stateBranchName: state.branchName, + stateBranchName: branchName, }); if (cloneRepoRes.exitCode !== 0) { emitStepEvent( diff --git a/apps/open-swe/src/utils/github/issue-messages.ts b/apps/open-swe/src/utils/github/issue-messages.ts index 9895999b..ef74faac 100644 --- a/apps/open-swe/src/utils/github/issue-messages.ts +++ b/apps/open-swe/src/utils/github/issue-messages.ts @@ -98,14 +98,14 @@ export async function getMissingMessages( return [...(issueMessage ? [issueMessage] : []), ...untrackedCommentMessages]; } -const DEFAULT_ISSUE_TITLE = "New Open SWE Request"; +export const DEFAULT_ISSUE_TITLE = "New Open SWE Request"; export const ISSUE_TITLE_OPEN_TAG = ""; export const ISSUE_TITLE_CLOSE_TAG = ""; export const ISSUE_CONTENT_OPEN_TAG = ""; export const ISSUE_CONTENT_CLOSE_TAG = ""; export function extractIssueTitleAndContentFromMessage(content: string) { - let messageTitle = DEFAULT_ISSUE_TITLE; + let messageTitle: string | null = null; let messageContent = content; if ( content.includes(ISSUE_TITLE_OPEN_TAG) && diff --git a/apps/open-swe/src/utils/github/issue-task.ts b/apps/open-swe/src/utils/github/issue-task.ts index 59988730..2dc3501a 100644 --- a/apps/open-swe/src/utils/github/issue-task.ts +++ b/apps/open-swe/src/utils/github/issue-task.ts @@ -12,6 +12,9 @@ const logger = createLogger(LogLevel.INFO, "IssueTaskString"); export const TASK_OPEN_TAG = ""; export const TASK_CLOSE_TAG = ""; +export const PROPOSED_PLAN_OPEN_TAG = ""; +export const PROPOSED_PLAN_CLOSE_TAG = ""; + function typeNarrowTaskPlan(taskPlan: unknown): taskPlan is TaskPlan { return !!( typeof taskPlan === "object" && @@ -50,15 +53,44 @@ export function extractTasksFromIssueContent(content: string): TaskPlan | null { } } +function extractProposedPlanFromIssueContent(content: string): string[] | null { + if ( + !content.includes(PROPOSED_PLAN_OPEN_TAG) || + !content.includes(PROPOSED_PLAN_CLOSE_TAG) + ) { + return null; + } + const proposedPlanString = content + .split(PROPOSED_PLAN_OPEN_TAG)?.[1] + ?.split(PROPOSED_PLAN_CLOSE_TAG)?.[0]; + try { + const parsedProposedPlan = JSON.parse(proposedPlanString.trim()); + return parsedProposedPlan; + } catch (e) { + logger.error("Failed to parse proposed plan", { + proposedPlanString, + ...(e instanceof Error && { + name: e.name, + message: e.message, + stack: e.stack, + }), + }); + return null; + } +} + type GetIssueTaskPlanInput = { githubIssueId: number; targetRepository: TargetRepository; }; -export async function getTaskPlanFromIssue( +export async function getPlansFromIssue( input: GetIssueTaskPlanInput, config: GraphConfig, -): Promise { +): Promise<{ + taskPlan: TaskPlan | null; + proposedPlan: string[] | null; +}> { const issue = await getIssue({ owner: input.targetRepository.owner, repo: input.targetRepository.repo, @@ -72,7 +104,97 @@ export async function getTaskPlanFromIssue( ); } - return extractTasksFromIssueContent(issue.body); + const taskPlan = extractTasksFromIssueContent(issue.body); + const proposedPlan = extractProposedPlanFromIssueContent(issue.body); + return { + taskPlan, + proposedPlan, + }; +} + +function insertPlanToIssueBody( + issueBody: string, + planString: string, + planType: "taskPlan" | "proposedPlan", +) { + const openingPlanTag = + planType === "taskPlan" ? TASK_OPEN_TAG : PROPOSED_PLAN_OPEN_TAG; + const closingPlanTag = + planType === "taskPlan" ? TASK_CLOSE_TAG : PROPOSED_PLAN_CLOSE_TAG; + + const wrappedPlan = `${openingPlanTag} +${planString} +${closingPlanTag}`; + + let newBody = ""; + + if ( + !issueBody.includes(openingPlanTag) && + !issueBody.includes(closingPlanTag) + ) { + if ( + !issueBody.includes(DETAILS_OPEN_TAG) && + !issueBody.includes(DETAILS_CLOSE_TAG) + ) { + newBody = `${issueBody} +${DETAILS_OPEN_TAG} +${AGENT_CONTEXT_DETAILS_SUMMARY} +${wrappedPlan} +${DETAILS_CLOSE_TAG}`; + } else { + // No plan present yet, but details already exists. + const contentBeforeDetailsTag = issueBody.split(DETAILS_OPEN_TAG)?.[0]; + const contentAfterDetailsTag = issueBody.split(DETAILS_CLOSE_TAG)?.[0]; + + newBody = `${contentBeforeDetailsTag} +${wrappedPlan} +${contentAfterDetailsTag}`; + } + } else { + const contentBeforeOpenTag = issueBody.split(openingPlanTag)?.[0]; + const contentAfterCloseTag = issueBody.split(closingPlanTag)?.[1]; + + newBody = `${contentBeforeOpenTag} +${wrappedPlan} +${contentAfterCloseTag}`; + } + + return newBody; +} + +export async function addProposedPlanToIssue( + input: GetIssueTaskPlanInput, + config: GraphConfig, + proposedPlan: string[], +) { + const issue = await getIssue({ + owner: input.targetRepository.owner, + repo: input.targetRepository.repo, + issueNumber: input.githubIssueId, + githubInstallationToken: + getGitHubTokensFromConfig(config).githubInstallationToken, + }); + if (!issue || !issue.body) { + throw new Error( + "No issue found when attempting to get task plan from issue", + ); + } + + const proposedPlanString = JSON.stringify(proposedPlan, null, 2); + const newBody = insertPlanToIssueBody( + issue.body, + proposedPlanString, + "proposedPlan", + ); + + await updateIssue({ + owner: input.targetRepository.owner, + repo: input.targetRepository.repo, + issueNumber: input.githubIssueId, + githubInstallationToken: + getGitHubTokensFromConfig(config).githubInstallationToken, + body: newBody, + }); } const DETAILS_OPEN_TAG = "
"; @@ -97,35 +219,7 @@ export async function addTaskPlanToIssue( } const taskPlanString = JSON.stringify(taskPlan, null, 2); - let newBody = ""; - - if ( - !issue.body.includes(TASK_OPEN_TAG) && - !issue.body.includes(TASK_CLOSE_TAG) - ) { - newBody = `${issue.body} - -${DETAILS_OPEN_TAG} -${AGENT_CONTEXT_DETAILS_SUMMARY} - -${TASK_OPEN_TAG} -${taskPlanString} -${TASK_CLOSE_TAG} - -${DETAILS_CLOSE_TAG}`; - } else { - const contentBeforeOpenTag = issue.body.split(TASK_OPEN_TAG)?.[0]; - const contentAfterCloseTag = issue.body.split(TASK_CLOSE_TAG)?.[1]; - const newTaskPlanString = JSON.stringify(taskPlan, null, 2); - - newBody = `${contentBeforeOpenTag} - -${TASK_OPEN_TAG} -${newTaskPlanString} -${TASK_CLOSE_TAG} - -${contentAfterCloseTag}`; - } + const newBody = insertPlanToIssueBody(issue.body, taskPlanString, "taskPlan"); await updateIssue({ owner: input.targetRepository.owner, diff --git a/apps/web/src/components/thread/messages/interrupt.tsx b/apps/web/src/components/thread/messages/interrupt.tsx index 040fed25..9d8c657c 100644 --- a/apps/web/src/components/thread/messages/interrupt.tsx +++ b/apps/web/src/components/thread/messages/interrupt.tsx @@ -6,7 +6,7 @@ import { useStream } from "@langchain/langgraph-sdk/react"; interface InterruptProps { interruptValue?: unknown; isLastMessage: boolean; - hasNoAIOrToolMessages: boolean; + hasNoAIOrToolMessages?: boolean; forceRenderInterrupt?: boolean; thread: ReturnType; } diff --git a/apps/web/src/components/v2/actions-renderer.tsx b/apps/web/src/components/v2/actions-renderer.tsx index 1d26e4d9..46830181 100644 --- a/apps/web/src/components/v2/actions-renderer.tsx +++ b/apps/web/src/components/v2/actions-renderer.tsx @@ -22,6 +22,7 @@ import { PlannerGraphState } from "@open-swe/shared/open-swe/planner/types"; import { GraphState, PlanItem } from "@open-swe/shared/open-swe/types"; import { HumanResponse } from "@langchain/langgraph/prebuilt"; import { LoadingActionsCardContent } from "./thread-view-loading"; +import { Interrupt } from "../thread/messages/interrupt"; interface AcceptedPlanEventData { planTitle: string; @@ -210,6 +211,13 @@ export function ActionsRenderer({ !isHumanMessageSDK(m) && !(m.id && m.id.startsWith(DO_NOT_RENDER_ID_PREFIX)), ); + const isLastMessageHidden = !!( + stream.messages?.length > 0 && + stream.messages[stream.messages.length - 1].id && + stream.messages[stream.messages.length - 1].id?.startsWith( + DO_NOT_RENDER_ID_PREFIX, + ) + ); // TODO: Need a better way to handle this. Not great like this... useEffect(() => { @@ -264,6 +272,14 @@ export function ActionsRenderer({ interruptType={acceptedPlanEvents[0].data.interruptType} /> )} + {/* If the last message is hidden, but there's an interrupt, we must manually render the interrupt */} + {isLastMessageHidden && stream.interrupt ? ( + >} + /> + ) : null} ); } diff --git a/apps/web/src/components/v2/manager-chat.tsx b/apps/web/src/components/v2/manager-chat.tsx index a597dce4..6778f0fe 100644 --- a/apps/web/src/components/v2/manager-chat.tsx +++ b/apps/web/src/components/v2/manager-chat.tsx @@ -11,6 +11,7 @@ import { Button } from "../ui/button"; import { useStream } from "@langchain/langgraph-sdk/react"; import { ManagerGraphState } from "@open-swe/shared/open-swe/manager/types"; import { cn } from "@/lib/utils"; +import { isAIMessageSDK } from "@/lib/langchain-messages"; function MessageCopyButton({ content }: { content: string }) { const [copied, setCopied] = useState(false); @@ -67,6 +68,19 @@ interface ManagerChatProps { cancelRun: () => void; } +function extractResponseFromMessage(message: Message): string { + if (!isAIMessageSDK(message)) { + return getMessageContentString(message.content); + } + const toolCall = message.tool_calls?.[0]; + const response = toolCall?.args?.response; + + if (!toolCall || !response) { + return getMessageContentString(message.content); + } + return response; +} + export function ManagerChat({ messages, chatInput, @@ -87,39 +101,41 @@ export function ManagerChat({ contentClassName="space-y-4 p-4" content={ <> - {messages.map((message) => ( -
-
- {message.type === "human" ? ( -
- -
- ) : ( -
- -
- )} -
-
-
- - {message.type === "human" ? "You" : "Agent"} - + {messages.map((message) => { + const messageContentString = + extractResponseFromMessage(message); + return ( +
+
+ {message.type === "human" ? ( +
+ +
+ ) : ( +
+ +
+ )}
-
- {getMessageContentString(message.content)} -
-
- +
+
+ + {message.type === "human" ? "You" : "Agent"} + +
+
+ {messageContentString} +
+
+ +
-
- ))} + ); + })} } footer={