feat: Implement a new summarization and context management system (#30)

* feat: Implement a new summarization and context managment system

* spelling

* fix

* include task status msg in history

* fix: Diagnose error

* cr

* cr

* cr
This commit is contained in:
Brace Sproul 2025-05-27 17:57:36 -07:00 • committed by GitHub
parent 2e4cf9ed63
commit ea793b9912
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
23 changed files with 707 additions and 254 deletions

View file

@ -19,17 +19,15 @@ async function runFromPlan() {
messages: [
{
role: "user",
content:
"This repo contains the react/next.js code for my persona/portfolio site. It currently has static values set for the number of stars on the repositories I highlight. I want this to be accurate, but I do NOT want it to make requests to GitHub every time a user visits. Instead, please implement a solution which will run once a day, fetch the number of stars from a list of repos, then write them to vercel's KV store. Finally, update the UI to make a request to the KV store when the user visits my page and render the accurate star counts.",
},
{
id: "msg_01UM7mQS4P37haW1LrABGZez",
role: "assistant",
content:
"The project is structured as a monorepo, with some apps located inside the /apps directory. In this directory, there is an /auth directory. This directory only contains the scaffolding for a new app in the monorepo, but is not yet implemented. Please take the following plan/task description and implement it in the /auth directory:\nThis monorepo is for an AI coding agent. The app runs and edits the code in the cloud in a sandboxed environment. Right now, we require users to generate a GitHub PAT, which we store in a .env file and can use to authenticate with GitHub. This is not idea, and instead we want to have a github oauth app which users can authenticate with.\nPlease implement a new auth server inside the /auth directory which can do this.\nYou will not have any access to secrets, so you will not be able to run the server to test it.\nI want the server to be able to authenticate users with GitHub, such that we will be able to take the following actions:\n1. clone repositories they give us access to\n2. checkout existing and create new branches on the repositories they give us access to\n3. make pull requests and push changes to the repositories they give us access to\nOnce you're done, ensure you've documented the development process in the readme of this new app.",
additional_kwargs: {
summary_message: true,
},
content: `The project is structured as a monorepo, with some apps located inside the /apps directory. In this directory, there is an /auth directory. This directory only contains the scaffolding for a new app in the monorepo, but is not yet implemented. Please take the following plan/task description and implement it in the /auth directory:
This monorepo is for an AI coding agent. The app runs and edits the code in the cloud in a sandboxed environment. Right now, we require users to generate a GitHub PAT, which we store in a .env file and can use to authenticate with GitHub. This is not idea, and instead we want to have a github oauth app which users can authenticate with.
Please implement a new auth server inside the /auth directory which can do this.
You will not have any access to secrets, so you will not be able to run the server to test it.
I want the server to be able to authenticate users with GitHub, such that we will be able to take the following actions:
1. clone repositories they give us access to
2. checkout existing and create new branches on the repositories they give us access to
3. make pull requests and push changes to the repositories they give us access to
Once you're done, ensure you've documented the development process in the readme of this new app.`,
},
],
plan: [
@ -37,46 +35,37 @@ async function runFromPlan() {
index: 0,
plan: "Set up the Express.js server with TypeScript configuration and necessary dependencies for GitHub OAuth authentication",
completed: false,
summary: undefined,
},
{
index: 1,
plan: "Implement GitHub OAuth flow endpoints including authorization redirect and callback handling",
completed: false,
summary: undefined,
},
{
index: 2,
plan: "Create middleware for JWT token generation and validation for authenticated sessions",
completed: false,
summary: undefined,
},
{
index: 3,
plan: "Add environment configuration management for OAuth app credentials and server settings",
completed: false,
summary: undefined,
},
{
index: 4,
plan: "Create user session management and token storage mechanisms",
plan: "Add comprehensive error handling throughout the authentication flow",
completed: false,
summary: undefined,
},
{
index: 5,
plan: "Implement API endpoints for checking authentication status and user permissions",
completed: false,
},
{
index: 6,
plan: "Add comprehensive error handling and logging throughout the authentication flow",
completed: false,
},
{
index: 7,
plan: "Create comprehensive README documentation covering setup, configuration, and development process",
completed: false,
},
{
index: 8,
plan: "Add TypeScript type definitions for GitHub OAuth responses and internal data structures",
completed: false,
summary: undefined,
},
],
proposedPlan: [
@ -84,12 +73,39 @@ async function runFromPlan() {
"Implement GitHub OAuth flow endpoints including authorization redirect and callback handling",
"Create middleware for JWT token generation and validation for authenticated sessions",
"Add environment configuration management for OAuth app credentials and server settings",
"Create user session management and token storage mechanisms",
"Implement API endpoints for checking authentication status and user permissions",
"Add comprehensive error handling and logging throughout the authentication flow",
"Add comprehensive error handling throughout the authentication flow",
"Create comprehensive README documentation covering setup, configuration, and development process",
"Add TypeScript type definitions for GitHub OAuth responses and internal data structures",
],
planContextSummary: `## User Request Summary
The user wants to implement a GitHub OAuth authentication server in the \`/apps/auth\` directory of a monorepo for an AI coding agent. The goal is to replace the current GitHub PAT authentication system with OAuth to enable:
1. Cloning repositories users give access to
2. Checking out existing and creating new branches
3. Making pull requests and pushing changes
## Codebase Files and Descriptions
- **Project root**: \`/home/user/open-swe/\` - Main monorepo directory
- **Apps directory**: \`/home/user/open-swe/apps/\` - Contains multiple apps including auth, docs, and open-swe
- **Auth app directory**: \`/home/user/open-swe/apps/auth/\` - Target directory for implementation, currently contains only scaffolding
- **Auth package.json**: \`/home/user/open-swe/apps/auth/package.json\` - Contains basic TypeScript/Node.js setup with name "@open-swe/auth", includes dev dependencies for TypeScript, Jest, ESLint, Prettier
- **Auth src directory**: \`/home/user/open-swe/apps/auth/src/\` - Contains only an empty \`index.ts\` file
- **Auth config files**: Directory includes standard config files (.gitignore, .dockerignore, .prettierrc, eslint.config.js, jest.config.js, tsconfig.json, turbo.json)
## Key Repository Insights and Learnings
- The monorepo uses Yarn as package manager (version 3.5.1)
- TypeScript is used throughout with version ~5.7.2
- The auth app is set up as an ES module (type: "module" in package.json)
- Standard tooling includes ESLint, Prettier, Jest for testing
- The auth directory is completely empty except for scaffolding - no existing implementation
- The project appears to be part of the LangChain AI organization based on repository URL
- No access to secrets/environment variables for testing
- Need to document the development process in a README for the auth app
## Implementation Requirements
- Implement GitHub OAuth flow for authentication
- Ensure the server can handle repository operations (clone, branch management, PR creation)
- Create comprehensive documentation in README
- Follow existing monorepo patterns and tooling setup`,
codebaseContext: "",
planChangeRequest: undefined,
sandboxSessionId: undefined,
branchName: `open-swe/${threadId}`,

View file

@ -9,25 +9,11 @@ import {
progressPlanStep,
summarizeTaskSteps,
generateConclusion,
diagnoseError,
} from "./nodes/index.js";
import { isAIMessage } from "@langchain/core/messages";
import { plannerGraph } from "./subgraphs/index.js";
/**
* @param {GraphState} state - The current graph state.
* @returns {"interrupt-plan" | typeof END} The next node to execute, or END if the process should stop.
*/
function routeAfterPlan(state: GraphState): "interrupt-plan" | typeof END {
const { messages } = state;
const lastMessage = messages[messages.length - 1];
if (isAIMessage(lastMessage) && !lastMessage.tool_calls) {
// The last message is an AI message without tool calls. This indicates the LLM generated followup questions.
return END;
}
return "interrupt-plan";
}
/**
* Routes to the next appropriate node after taking action.
* If the last message is an AI message with tool calls, it routes to "take-action".
@ -55,29 +41,27 @@ const workflow = new StateGraph(GraphAnnotation, GraphConfiguration)
.addNode("generate-plan-subgraph", plannerGraph)
.addNode("rewrite-plan", rewritePlan)
.addNode("interrupt-plan", interruptPlan, {
// TODO: Hookup `Command` in interruptPlan node so this actually works.
ends: [END, "rewrite-plan", "generate-action"],
})
.addNode("generate-action", generateAction)
.addNode("take-action", takeAction)
.addNode("take-action", takeAction, {
ends: ["progress-plan-step", "diagnose-error"],
})
.addNode("progress-plan-step", progressPlanStep, {
ends: ["summarize-task-steps", "generate-action"],
ends: ["summarize-task-steps", "generate-action", "generate-conclusion"],
})
.addNode("summarize-task-steps", summarizeTaskSteps, {
ends: ["generate-action", "generate-conclusion"],
})
.addNode("generate-conclusion", generateConclusion)
.addNode("diagnose-error", diagnoseError)
.addEdge(START, "initialize")
.addEdge("initialize", "generate-plan-subgraph")
// TODO: Update routing to work w/ new interrupt node.
.addConditionalEdges("generate-plan-subgraph", routeAfterPlan, [
"interrupt-plan",
END,
])
.addEdge("generate-plan-subgraph", "interrupt-plan")
// Always interrupt after rewriting the plan.
.addEdge("rewrite-plan", "interrupt-plan")
.addConditionalEdges("generate-action", takeActionOrEnd, ["take-action", END])
.addEdge("take-action", "progress-plan-step")
.addEdge("diagnose-error", "generate-action")
.addEdge("generate-conclusion", END);
// Zod types are messed up

View file

@ -0,0 +1,160 @@
import {
BaseMessage,
isToolMessage,
ToolMessage,
} from "@langchain/core/messages";
import { GraphConfig, GraphState, GraphUpdate, PlanItem } from "../types.js";
import { formatPlanPromptWithSummaries } from "../utils/plan-prompt.js";
import {
getMessageContentString,
getMessageString,
} from "../utils/message/content.js";
import { loadModel, Task } from "../utils/load-model.js";
import { z } from "zod";
import { createLogger, LogLevel } from "../utils/logger.js";
const logger = createLogger(LogLevel.INFO, "DiagnoseError");
/**
* Whether or not enough errored tool calls have occurred to interrupt the graph.
* This will return true if the last tool call was an error, and 7 of the last 10
* tool calls have been errors.
* @param toolMessages
*
* @TODO Implement this. Should interrupt after generating a diagnosis for 7 consecutive errors.
*/
// function shouldInterruptError(toolMessages: ToolMessage[]): boolean {
// if (toolMessages[toolMessages.length - 1].status !== "error") {
// return false;
// }
// return toolMessages.slice(-10).filter((m) => m.status === "error").length >= 7;
// }
const systemPrompt = `You are operating as a terminal-based agentic coding assistant built by LangChain. It wraps LLM models to enable natural language interaction with a local codebase. You are expected to be precise, safe, and helpful.
The last command you tried to execute failed with an error. Please carefully diagnose the error, and provide a helpful explanation of exactly what the issue is, and how you can fix it.
Following these rules when diagnosing the error:
- You should provide a clear, concise, and helpful explanation of exactly what the issue is, and how you can fix it.
- You do not want to be overly verbose in your diagnosis. You should only include information which is directly relevant to diagnosing and fixing the error.
- NEVER make up reasons, or make a guess as to what the issue is. Your reasoning must ALWAYS be grounded in the information provided to you.
- Making up reasons, or making a guess can lead to more problems, so it's best to say you don't know rather than make up a reason.
- Reference specific lines of code, or context from the conversation history to support your diagnosis.
Here is the result of the last two failed commands:
{FAILED_ACTION_OUTPUT}
Here is the current task you're working on:
{CURRENT_TASK}
And here are all of the tasks you've completed so far, along with their summaries:
{PLAN_PROMPT}
Finally, here is a summary of general context about the codebase:
{CODEBASE_CONTEXT}
Please carefully go over all of this information, and provide a helpful explanation of exactly what the issue is, and how you can fix it. When you are ready to provide your diagnosis, call the \`diagnose_error\` tool.
`;
const userPrompt = `Here is the full conversation history from the steps taken to complete the current task, along with the user's initial request:
{CONVERSATION_HISTORY}
Please carefully go over all of this information, and provide a helpful explanation of exactly what the issue is, and how you can fix it. When you are ready to provide your diagnosis, call the \`diagnose_error\` tool.`;
const diagnoseErrorToolSchema = z.object({
diagnosis: z.string().describe("The diagnosis of the error."),
});
const diagnoseErrorTool = {
name: "diagnose_error",
description: "Diagnoses an error given a diagnosis.",
schema: diagnoseErrorToolSchema,
};
const formatSystemPrompt = (
lastFailedActionContent: string,
plan: PlanItem[],
codebaseContext: string,
): string => {
const currentTask = plan.find((p) => !p.completed);
const completedTasks = plan.filter((p) => p.completed);
return systemPrompt
.replace(
"{FAILED_ACTION_OUTPUT}",
`<failed-action-output>${lastFailedActionContent}</failed-action-output>`,
)
.replace(
"{CURRENT_TASK}",
`<current-task index="${currentTask?.index}">${currentTask?.plan}</current-task>`,
)
.replace("{PLAN_PROMPT}", formatPlanPromptWithSummaries(completedTasks))
.replace("{CODEBASE_CONTEXT}", codebaseContext);
};
const formatUserPrompt = (messages: BaseMessage[]): string => {
return userPrompt.replace(
"{CONVERSATION_HISTORY}",
messages.map(getMessageString).join("\n"),
);
};
export async function diagnoseError(
state: GraphState,
config: GraphConfig,
): Promise<GraphUpdate> {
const lastFailedAction = state.messages.findLast(
(m) => isToolMessage(m) && m.status === "error",
);
if (!lastFailedAction?.content) {
throw new Error("No failed action found in messages");
}
logger.info("The last two tool calls resulted in errors. Diagnosing error.");
const model = await loadModel(config, Task.SUMMARIZER);
const modelWithTools = model.bindTools([diagnoseErrorTool], {
tool_choice: diagnoseErrorTool.name,
});
const response = await modelWithTools.invoke([
{
role: "system",
content: formatSystemPrompt(
getMessageContentString(lastFailedAction.content),
state.plan,
state.codebaseContext,
),
},
{
role: "user",
content: formatUserPrompt(state.messages),
},
]);
const toolCall = response.tool_calls?.[0];
if (!toolCall) {
throw new Error("Failed to generate a tool call when diagnosing error.");
}
logger.info("Diagnosed error successfully.", {
diagnosis: (toolCall.args as z.infer<typeof diagnoseErrorToolSchema>)
.diagnosis,
});
const toolMessage = new ToolMessage({
tool_call_id: toolCall.id ?? "",
content: `Successfully diagnosed error. Please use the diagnosis to continue with the next action.`,
name: toolCall.name,
status: "success",
additional_kwargs: {
is_diagnosis: true,
},
});
return {
messages: [response, toolMessage],
};
}

View file

@ -34,7 +34,7 @@ export async function generateConclusion(
const firstUserMessage = state.messages.find(isHumanMessage);
const userMessage = `The user's initial request is as follows:
${getMessageContentString(firstUserMessage?.content ?? "No user message found")}
${getMessageContentString(firstUserMessage?.content || "No user message found")}
The conversation history is as follows:
${state.messages.map(getMessageString).join("\n")}

View file

@ -1,4 +1,4 @@
import { GraphState, GraphConfig, GraphUpdate, PlanItem } from "../types.js";
import { GraphState, GraphConfig, GraphUpdate } from "../types.js";
import { loadModel, Task } from "../utils/load-model.js";
import { shellTool, applyPatchTool } from "../tools/index.js";
import { getRepoAbsolutePath } from "../utils/git/index.js";
@ -16,9 +16,13 @@ You can:
- Apply patches, run commands, and manage user approvals based on policy.
- Work inside a sandboxed, git-backed workspace with rollback support.
You work based on a plan which was generated in a previous step. The plan items are as follows:
You work based on a plan which was generated in a previous step. After each task in a plan is completed, a summary of the task is generated, and included in the plan list below. These messages are then removed from the conversation history, so ensure you always weigh the task summaries highly when making decisions.
{PLAN_PROMPT}
The plan tasks and summaries are as follows:
{PLAN_PROMPT_WITH_SUMMARIES}
When you were generating this plan, you also generated a summary of the actions you took in order to come up with this plan. Ensure you use this as context about the codebase, and plan generation process.
{PLAN_GENERATION_SUMMARY}
You are an agent - please keep going until the user's query is completely resolved, before ending your turn and yielding back to the user.
Only terminate your turn when you are sure that the problem is solved.
@ -38,10 +42,10 @@ You MUST adhere to the following criteria when executing the task:
- Analyzing code for vulnerabilities is allowed.
- Showing user code and tool call details is allowed.
- Remember to always properly format and quote your shell commands.
- Take advantage of the condensed context messages in the conversation history (under the names \`condense_task_context\` and \`condense_planning_context\`). These contain summarized/condensed context from previously completed steps. Ensure you always read these messages to avoid duplicate work (e.g.: searching for file paths).
- These summary messages may include a section called 'Codebase files and descriptions' which contains a list of files, and descriptions of the files' contents. If you need context on a file, or directory, ensure you first check this section of the summary messages to avoid duplicate work.
- Take advantage of the task summaries from completed tasks in the prompt above. Ensure you always read these summaries to avoid duplicate work, and so you always have up to date context on the codebase, and tasks you've completed.
- Each summary message will include a short description of the task it completed, how it did so, and every change it made to the codebase during this task. This section will be titled 'Repository modifications summary'.
- The summary messages may also include a section called 'Key repository insights and learnings'. This contains key insights, learnings, and facts the model discovered while completing a task.
- Each summary message will also include a short description of the task it completed, how it did so, and every change it made to the codebase during this task. This section will be titled 'Repository modifications summary'.
- Additionally, you're also provided with a section titled 'Codebase context' which contains an up to date list of files, and descriptions of the files' contents. If you need context on a file, or directory, ensure you first check to see if you can find it in the codebase context before performing an action to read/find it.
- All changes are automatically committed, so you should not worry about creating backups, or committing changes.
- Use \`apply_patch\` to edit files. This tool accepts diffs and file paths. It will then apply the given diff to the file.
- You should NOT try to create empty files with \`apply_patch\`. If you need to create a file, use the \`shell\` tool, and pass \`touch <file path>\` to create the file.
@ -76,13 +80,31 @@ You MUST adhere to the following criteria when executing the task:
- Always use \`rg\` instead of \`grep/ls -R\` because it is much faster and respects gitignore.
- Always use glob patterns when searching with \`rg\` for specific file types. For example, to search for all TSX files, use \`rg -i star -g **/*.tsx project-directory/\`. This is because \`rg\` does not have built in file types for every language.
- Only make changes to the existing Git repo ({REPO_DIRECTORY}). Any changes outside this repo will not be detected, so do not attempt to create new files or directories outside of this repo.
Below, is a collection of useful context about the codebase. It is updated after each completed task, and is provided to you to help you make decisions, and avoid duplicate work:
{CODEBASE_CONTEXT}
Once again, here are the completed tasks, remaining tasks, and the current task you're working on:
{PLAN_PROMPT}
`;
const formatPrompt = (plan: PlanItem[], config: GraphConfig): string => {
const formatPrompt = (state: GraphState, config: GraphConfig): string => {
const repoDirectory = getRepoAbsolutePath(config);
return systemPrompt
.replace("{PLAN_PROMPT}", formatPlanPrompt(plan))
.replaceAll("{REPO_DIRECTORY}", repoDirectory);
.replaceAll(
"{PLAN_PROMPT_WITH_SUMMARIES}",
formatPlanPrompt(state.plan, { includeSummaries: true }),
)
.replaceAll("{PLAN_PROMPT}", formatPlanPrompt(state.plan))
.replaceAll("{REPO_DIRECTORY}", repoDirectory)
.replaceAll(
"{PLAN_GENERATION_SUMMARY}",
`<plan-generation-summary>\n${state.planContextSummary}\n</plan-generation-summary>`,
)
.replaceAll(
"{CODEBASE_CONTEXT}",
`<codebase-context>\n${state.codebaseContext || "No codebase context generated yet. Please use the conversation below as context."}\n</codebase-context>`,
);
};
export async function generateAction(
@ -96,7 +118,7 @@ export async function generateAction(
const response = await modelWithTools.invoke([
{
role: "system",
content: formatPrompt(state.plan, config),
content: formatPrompt(state, config),
},
...state.messages,
]);

View file

@ -6,3 +6,4 @@ export * from "./interrupt-plan.js";
export * from "./progress-plan-step.js";
export * from "./summarize-task-steps.js";
export * from "./generate-conclusion.js";
export * from "./diagnose-error.js";

View file

@ -113,7 +113,7 @@ export async function initialize(
const checkoutBranchRes = await checkoutBranch(
absoluteRepoDir,
state.branchName ?? getBranchName(config),
state.branchName || getBranchName(config),
sandbox,
);

View file

@ -47,6 +47,7 @@ export async function interruptPlan(state: GraphState): Promise<Command> {
index,
plan: p,
completed: false,
summary: undefined,
})),
sandboxSessionId: newSandboxSessionId,
},
@ -68,6 +69,7 @@ export async function interruptPlan(state: GraphState): Promise<Command> {
index,
plan: p,
completed: false,
summary: undefined,
})),
sandboxSessionId: newSandboxSessionId,
},

View file

@ -4,10 +4,7 @@ import { GraphConfig, GraphState, PlanItem } from "../types.js";
import { loadModel, Task } from "../utils/load-model.js";
import { formatPlanPrompt } from "../utils/plan-prompt.js";
import { Command } from "@langchain/langgraph";
import {
getMessageContentString,
getMessageString,
} from "../utils/message/content.js";
import { getMessageString } from "../utils/message/content.js";
import { isHumanMessage } from "@langchain/core/messages";
import { removeFirstHumanMessage } from "../utils/message/modify-array.js";
@ -17,32 +14,41 @@ const systemPrompt = `You are operating as a terminal-based agentic coding assis
In your workflow, you generate a plan, then act on said plan. It may take many actions to complete a single step, or a single action to complete the step.
Here is the plan:
Here is the plan, along with the summaries of each completed task:
{PLAN_PROMPT}
Analyze the tasks you've completed, the tasks which are remaining, and the current task you just took an action on. In addition to this, you're also provided the full conversation history between you and the user. All of the messages in this conversation are from the previous steps/actions you've taken, and any user input.
Analyze the tasks you've completed, the tasks which are remaining, and the current task you just took an action on.
In addition to this, you're also provided the full conversation history between you and the user. All of the messages in this conversation are from the previous steps/actions you've taken, and any user input.
Take all of this information, and determine whether or not you have completed this task in the plan. Be careful to not mark a task as completed if it is not, this can cause cascading issues in the workflow.
If you determine a task has been completed, you should call the \`confirm_task_completion\` tool. If you do NOT think the current task has been completed, do not call the tool and instead respond with \`not completed.\`.`;
Take all of this information, and determine whether or not you have completed this task in the plan.
Once you've determined the status of the current task, call the \`set_task_status\` tool.
`;
const confirmTaskCompletionToolSchema = z.object({
const setTaskStatusToolSchema = z.object({
reasoning: z
.string()
.describe("Reasoning for whether or not the task has been completed."),
current_task_completed: z
.boolean()
.describe("Whether or not the current task has been completed."),
.describe(
"A concise reasoning summary for the status of the current task, explaining why you think it is completed or not completed.",
),
task_status: z
.enum(["completed", "not_completed"])
.describe(
"The status of the current task, based on the reasoning provided.",
),
});
const confirmTaskCompletionTool = {
name: "confirm_task_completion",
description: "Whether or not the current task has been completed.",
schema: confirmTaskCompletionToolSchema,
const setTaskStatusTool = {
name: "set_task_status",
description:
"The status of the current task, along with a concise reasoning summary to support the status.",
schema: setTaskStatusToolSchema,
};
const formatPrompt = (plan: PlanItem[]): string => {
return systemPrompt.replace("{PLAN_PROMPT}", formatPlanPrompt(plan));
return systemPrompt.replace(
"{PLAN_PROMPT}",
formatPlanPrompt(plan, { includeSummaries: true }),
);
};
export async function progressPlanStep(
@ -50,8 +56,8 @@ export async function progressPlanStep(
config: GraphConfig,
): Promise<Command> {
const model = await loadModel(config, Task.PROGRESS_PLAN_CHECKER);
const modelWithTools = model.bindTools([confirmTaskCompletionTool], {
tool_choice: "auto",
const modelWithTools = model.bindTools([setTaskStatusTool], {
tool_choice: setTaskStatusTool.name,
});
const firstUserMessage = state.messages.find(isHumanMessage);
@ -60,10 +66,8 @@ export async function progressPlanStep(
${removeFirstHumanMessage(state.messages).map(getMessageString).join("\n")}
Take all of this information, and determine whether or not you have completed this task in the plan. Be careful to not mark a task as completed if it is not, this can cause cascading issues in the workflow.
If you determine a task has been completed, you should call the \`confirm_task_completion\` tool. If you do NOT think the current task has been completed, do not call the tool and instead respond with \`not completed.\`.
ENSURE YOU ONLY CALL THE \`confirm_task_completion\` TOOL IF YOU DETERMINE THE CURRENT TASK HAS BEEN COMPLETED, OR RESPOND WITH 'not completed.'. DO NOT TAKE ANY OTHER ACTION.`;
Take all of this information, and determine whether or not you have completed this task in the plan.
Once you've determined the status of the current task, call the \`set_task_status\` tool.`;
const response = await modelWithTools.invoke([
{
@ -79,18 +83,23 @@ ENSURE YOU ONLY CALL THE \`confirm_task_completion\` TOOL IF YOU DETERMINE THE C
const toolCall = response.tool_calls?.[0];
if (!toolCall) {
logger.info(
"Current task has not been completed, as no tool call was generated. Progressing to the next action.",
{
responseContent: getMessageContentString(response.content),
},
throw new Error(
"Failed to generate a tool call when checking task status.",
);
return new Command({ goto: "generate-action" });
}
const isCompleted = (
toolCall.args as z.infer<typeof confirmTaskCompletionToolSchema>
).current_task_completed;
const isCompleted =
(toolCall.args as z.infer<typeof setTaskStatusToolSchema>).task_status ===
"completed";
const currentTask = state.plan.filter((p) => !p.completed)?.[0];
const toolMessage = {
role: "tool",
tool_call_id: toolCall.id,
content: `Saved task status as ${
toolCall.args.task_status
} for task ${currentTask?.plan || "unknown"}`,
name: toolCall.name,
};
if (!isCompleted) {
logger.info(
@ -99,15 +108,22 @@ ENSURE YOU ONLY CALL THE \`confirm_task_completion\` TOOL IF YOU DETERMINE THE C
reasoning: toolCall.args.reasoning,
},
);
return new Command({ goto: "generate-action" });
return new Command({
goto: "generate-action",
update: { messages: [response, toolMessage] },
});
}
// This should in theory never happen, but ensure we route properly if it does.
const remainingTask = state.plan.find((p) => !p.completed);
if (!remainingTask) {
logger.info(
"Found no remaining tasks in the plan during the check plan step. Progressing to the next action.",
"Found no remaining tasks in the plan during the check plan step. Continuing to the conclusion generation step.",
);
return new Command({ goto: "generate-action" });
return new Command({
goto: "generate-conclusion",
update: { messages: [response, toolMessage] },
});
}
logger.info("Task marked as completed. Routing to task summarization step.", {
@ -120,6 +136,7 @@ ENSURE YOU ONLY CALL THE \`confirm_task_completion\` TOOL IF YOU DETERMINE THE C
return new Command({
goto: "summarize-task-steps",
update: {
messages: [response, toolMessage],
plan: state.plan.map((p) => {
if (p.index === remainingTask.index) {
return {

View file

@ -140,7 +140,7 @@ async function identifyTasksToModifyFunc(
role: "user",
content: formatSysPromptIdentifyTasks(
getMessageContentString(
firstUserMessage?.content ?? "No user message found",
firstUserMessage?.content || "No user message found",
),
state.planChangeRequest,
state.proposedPlan,
@ -205,7 +205,7 @@ async function updatePlanTasksFunc(
role: "user",
content: formatSysPromptRewritePlan(
getMessageContentString(
firstUserMessage?.content ?? "No user message found",
firstUserMessage?.content || "No user message found",
),
state.planChangeRequest,
state.proposedPlan,

View file

@ -1,132 +1,259 @@
import { z } from "zod";
import { v4 as uuidv4 } from "uuid";
import { GraphConfig, GraphState, PlanItem } from "../types.js";
import { loadModel, Task } from "../utils/load-model.js";
import { AIMessage, isHumanMessage } from "@langchain/core/messages";
import { AIMessage, BaseMessage } from "@langchain/core/messages";
import { formatPlanPrompt } from "../utils/plan-prompt.js";
import { createLogger, LogLevel } from "../utils/logger.js";
import { getMessageString } from "../utils/message/content.js";
import {
removeFirstHumanMessage,
removeLastTaskMessages,
} from "../utils/message/modify-array.js";
getMessageContentString,
getMessageString,
} from "../utils/message/content.js";
import { removeLastTaskMessages } from "../utils/message/modify-array.js";
import { Command } from "@langchain/langgraph";
import { ConfigurableModel } from "langchain/chat_models/universal";
import { traceable } from "langsmith/traceable";
const logger = createLogger(LogLevel.INFO, "SummarizeTaskSteps");
const taskSummarySysPrompt = `You are operating as a terminal-based agentic coding assistant built by LangChain. It wraps LLM models to enable natural language interaction with a local codebase. You are expected to be precise, safe, and helpful.
const systemPrompt = `You are operating as a terminal-based agentic coding assistant built by LangChain. It wraps LLM models to enable natural language interaction with a local codebase. You are expected to be precise, safe, and helpful.
You've been given a task to summarize the messages in your conversation history. You just completed a task in your plan, and can now summarize/condense all of the messages in your conversation history which were relevant to that task.
You do not want to keep the entire conversation history, but instead you want to keep the most relevant and important snippets for future context.
Your current task is to look at the conversation history, and generate a concise summary of the steps which were taken to complete the task.
Here are all of your tasks you've completed, remaining, and the current task you're working on. The completed tasks will include summaries of the steps taken to complete them:
{PLAN_PROMPT}
You MUST adhere to the following criteria when summarizing the conversation history:
- Retain context such as file paths, versions, and installed software which future iterations will find useful.
- It is very important to include the file paths of files you've already searched for, along with a description of the file's contents, inside a 'Codebase files and descriptions' section, so that future steps can reuse this information, and will not need to search through the codebase for files again.
- Consider including a section titled 'Key repository insights and learnings' which may include information, insights and learnings you've discovered while completing the task.
- This section should be concise, but still including enough information so following steps will not repeat any mistakes or go down rabbit holes which you already know about.
- If changes were made to the repository during this task, ensure you include a section titled 'Repository modifications summary' which should include a short description of the task it completed, how it did so, and every change it made to the codebase during this task.
- Include insights, and learnings you've discovered about the codebase or specific files while completing the task.
- You should NOT document scripts, file structure, or other context which could be categorized as 'general codebase context'. General codebase context (e.g. scripts, file structure, package managers, etc.) will be generated in a separate step. Inspect the codebase context string provided below for this information.
- If files were created or modified, include short summaries of the changes made.
- What file(s) were modified/created.
- What content was added/removed.
- If you had to make a change which required you to undo previous changes, include that information.
- Do not include the actual changes you made, but rather high level bullet points containing context and descriptions on the modifications made.
- Do not retain any full code snippets.
- Do not retain any full file contents.
- Ensure your summary is concise, but useful for future context.
- If the conversation history contains any key insights or learnings, ensure you retain those.
- Do not retain any full code snippets.
- Do not retain any full file contents.
- Ensure you have an understanding of the context and summaries you've already generated (provided by the user below) and do not repeat any information you've already included.
- Do not duplicate ANY information. Ensure you carefully read and understand the task summaries generated above, and do not repeat any information you've already included.
- You do not need to include specific codebase context here, as codebase context will be generated in a separate step. Your sole task is to generate a concise summary of this specific task you just completed.
- Ensure your summary is as concise as possible, but useful for future context.
With all of this in mind, please carefully summarize and condense the following conversation history. Ensure you pass this condensed context to the \`condense_task_context\` tool.
Here is the current state of the codebase context you've accumulated. Remember YOU SHOULD NOT INCLUDE ANY GENERAL CODEBASE CONTEXT IN YOUR TASK SUMMARY.
{CODEBASE_CONTEXT}
Ensure you do NOT include codebase context in your task summary, as we want to avoid including duplicate information.
With all of this in mind, please carefully summarize and condense the conversation history of the task you just completed, provided by the user below. Remember that this summary should ONLY include details about the completed task, and should NOT include any general codebase context.
Respond ONLY with the task summary. Do not include any additional information, or text before or after the task summary.
`;
const formatPrompt = (plan: PlanItem[]): string =>
systemPrompt.replace(
"{PLAN_PROMPT}",
formatPlanPrompt(plan, { useLastCompletedTask: true }),
const userContextMessage = `Here is the task you just completed:
{COMPLETED_TASK}
The first message in the conversation history is the user's request. Messages from previously completed tasks have already been removed, in favor of task summaries.
With this in mind, please use the following conversation history to generate a concise summary of the task you just completed.
Conversation history:
{CONVERSATION_HISTORY}`;
const updateCodebaseContextSysPrompt = `You are operating as a terminal-based agentic coding assistant built by LangChain. It wraps LLM models to enable natural language interaction with a local codebase. You are expected to be precise, safe, and helpful.
Your current task is to update the codebase context, given the recent actions taken by the agent.
The codebase context should contain:
- Up to date information on the codebase file paths, and their contents.
- Do not include entire file contents, but rather high level descriptions of what a file contains, and what it does.
- Information on the software installed, and used in the codebase, including information such as version numbers, and dependencies.
- High level context about the codebase structure, and style.
- Any other relevant codebase information which may be useful for future context.
- There should be NO task specific context here. ONLY include context about the codebase. This context should be generally applicable and not tied to the specifics of the task.
You have the following codebase context:
{CODEBASE_CONTEXT}
Please inspect this context, and given the rules above, please respond with a full, complete codebase context I can use for future context.
When responding, ensure:
- You do not duplicate information.
- You remove old/stale context from the existing codebase context string if recent messages contradict it.
- You do NOT remove any information from the existing codebase context string if recent messages do not contradict it. We want to ensure we always have a complete picture of the codebase.
- You modify/combine information from the existing codebase context string if if new information is provided which warrants a change.
Please be concise, clear and helpful. Omit any extraneous information. Respond ONLY with the codebase context. Do not include any additional information, or text before or after the codebase context.
`;
const updateCodebaseContextUserMessage = `Here is the task you just completed:
{COMPLETED_TASK}
The first message in the conversation history is the user's request. Messages from previously completed tasks have already been removed, in favor of task summaries.
With this in mind, please use the following conversation history to update the codebase context to include new relevant information.
Conversation history:
{CONVERSATION_HISTORY}`;
const logger = createLogger(LogLevel.INFO, "SummarizeTaskSteps");
const formatPrompt = (plan: PlanItem[], codebaseContext: string): string =>
taskSummarySysPrompt
.replace(
"{PLAN_PROMPT}",
formatPlanPrompt(plan, {
useLastCompletedTask: true,
includeSummaries: true,
}),
)
.replace(
"{CODEBASE_CONTEXT}",
`<codebase-context>\n${codebaseContext || "No codebase context generated yet."}\n</codebase-context>`,
);
const formatUserMessage = (
messages: BaseMessage[],
plans: PlanItem[],
): string => {
const completedTask = plans.find((p) => p.completed);
if (!completedTask) {
throw new Error(
"No completed task found when trying to format user message for task summary.",
);
}
return userContextMessage
.replace("{COMPLETED_TASK}", completedTask.plan)
.replace(
"{CONVERSATION_HISTORY}",
messages.map(getMessageString).join("\n"),
);
};
const formatCodebaseContextPrompt = (codebaseContext: string): string =>
updateCodebaseContextSysPrompt.replace(
"{CODEBASE_CONTEXT}",
`<codebase-context>\n${codebaseContext || "No codebase context generated yet."}\n</codebase-context>`,
);
const condenseContextToolSchema = z.object({
context: z
.string()
.describe(
"The condensed context from the conversation history relevant to the recently completed task.",
),
});
const condenseContextTool = {
name: "condense_task_context",
description:
"Condense the conversation history into a concise summary, while still retaining the most relevant and important snippets.",
schema: condenseContextToolSchema,
const formatUserCodebaseContextMessage = (
messages: BaseMessage[],
plans: PlanItem[],
): string => {
const completedTask = plans.find((p) => p.completed);
if (!completedTask) {
throw new Error(
"No completed task found when trying to format user message for task summary.",
);
}
return updateCodebaseContextUserMessage
.replace("{COMPLETED_TASK}", completedTask.plan)
.replace(
"{CONVERSATION_HISTORY}",
messages.map(getMessageString).join("\n"),
);
};
async function generateTaskSummaryFunc(
state: GraphState,
model: ConfigurableModel,
): Promise<PlanItem[]> {
const lastCompletedTask = state.plan.findLast((p) => p.completed);
if (!lastCompletedTask) {
throw new Error("Unable to find last completed task.");
}
logger.info(`Summarizing task steps...`);
const response = await model.withConfig({ tags: ["nostream"] }).invoke([
{
role: "system",
content: formatPrompt(state.plan, state.codebaseContext),
},
{
role: "user",
content: formatUserMessage(state.messages, state.plan),
},
]);
const contentString = getMessageContentString(response.content);
const newPlanWithSummary = state.plan.map((p) => {
if (p.index !== lastCompletedTask.index) {
return p;
}
return {
...p,
summary: contentString,
};
});
return newPlanWithSummary;
}
const generateTaskSummary = traceable(generateTaskSummaryFunc, {
name: "generate_task_summary",
});
async function updateCodebaseContextFunc(
state: GraphState,
model: ConfigurableModel,
): Promise<string> {
logger.info(`Updating codebase context...`);
const response = await model.withConfig({ tags: ["nostream"] }).invoke([
{
role: "system",
content: formatCodebaseContextPrompt(state.codebaseContext),
},
{
role: "user",
content: formatUserCodebaseContextMessage(state.messages, state.plan),
},
]);
const contentString = getMessageContentString(response.content);
return contentString;
}
const updateCodebaseContext = traceable(updateCodebaseContextFunc, {
name: "update_codebase_context",
});
export async function summarizeTaskSteps(
state: GraphState,
config: GraphConfig,
): Promise<Command> {
const model = await loadModel(config, Task.SUMMARIZER);
const modelWithTools = model.bindTools([condenseContextTool], {
tool_choice: condenseContextTool.name,
});
const firstUserMessage = state.messages.find(isHumanMessage);
const conversationHistoryStr = `Here is the full conversation history for the task after the user's request.
This history includes any previous summarization/condensation of the conversation history. Ensure you do NOT summarize those messages, or duplicate any information present in them, but do use them as context so you know what has already been seen and summarized.
${removeFirstHumanMessage(state.messages).map(getMessageString).join("\n")}
Given this full conversation history please generate a concise, and useful summary of the conversation history for this task. Ensure you pass this condensed context to the \`condense_task_context\` tool.`;
logger.info(`Summarizing task steps...`);
const response = await modelWithTools.invoke([
{
role: "system",
content: formatPrompt(state.plan),
},
...(firstUserMessage ? [firstUserMessage] : []),
{
role: "user",
content: conversationHistoryStr,
},
]);
const toolCall = response.tool_calls?.[0];
if (!toolCall) {
throw new Error("Failed to generate plan");
const lastCompletedTask = state.plan.findLast((p) => p.completed);
if (!lastCompletedTask) {
throw new Error("Unable to find last completed task.");
}
const model = await loadModel(config, Task.SUMMARIZER);
const [updatedPlan, updatedCodebaseContext] = await Promise.all([
generateTaskSummary(state, model),
updateCodebaseContext(state, model),
]);
const removedMessages = removeLastTaskMessages(state.messages);
logger.info(`Removing ${removedMessages.length} message(s) from state.`);
const allTasksCompleted = state.plan.every((p) => p.completed);
// Ensure all tool calls are removed from the message.
delete response.tool_call_chunks;
delete response.tool_calls;
delete response.invalid_tool_calls;
const messageWithoutToolCall = new AIMessage({
...response,
content:
"Condensed Task Context:\n\n" +
(toolCall.args as z.infer<typeof condenseContextToolSchema>).context,
const condensedTaskMessage = new AIMessage({
id: uuidv4(),
content: `Successfully condensed task context for task: "${lastCompletedTask.plan}". This task's summary can be found in the system prompt.`,
additional_kwargs: {
...response.additional_kwargs,
summary_message: true,
},
});
const newMessagesStateUpdate = [...removedMessages, condensedTaskMessage];
const newMessagesStateUpdate = [...removedMessages, messageWithoutToolCall];
if (!allTasksCompleted) {
const allTasksCompleted = state.plan.every((p) => p.completed);
if (allTasksCompleted) {
return new Command({
goto: "generate-action",
goto: "generate-conclusion",
update: {
messages: newMessagesStateUpdate,
plan: updatedPlan,
codebaseContext: updatedCodebaseContext,
},
});
}
return new Command({
goto: "generate-conclusion",
goto: "generate-action",
update: {
messages: newMessagesStateUpdate,
plan: updatedPlan,
codebaseContext: updatedCodebaseContext,
},
});
}

View file

@ -1,28 +1,60 @@
import { isAIMessage, ToolMessage } from "@langchain/core/messages";
import {
isAIMessage,
isToolMessage,
ToolMessage,
} from "@langchain/core/messages";
import { createLogger, LogLevel } from "../utils/logger.js";
import { applyPatchTool, shellTool } from "../tools/index.js";
import { GraphState, GraphConfig, GraphUpdate } from "../types.js";
import { GraphState, GraphConfig } from "../types.js";
import {
checkoutBranchAndCommit,
getChangedFilesStatus,
getRepoAbsolutePath,
} from "../utils/git/index.js";
import { Sandbox } from "@e2b/code-interpreter";
import { zodSchemaToString } from "../utils/zod-to-string.js";
import {
getMissingKeysFromObjectSchema,
zodSchemaToString,
} from "../utils/zod-to-string.js";
import { z } from "zod";
import { Command } from "@langchain/langgraph";
const logger = createLogger(LogLevel.INFO, "TakeAction");
function formatBadArgsError(schema: z.ZodTypeAny, args: any) {
const missingKeys = getMissingKeysFromObjectSchema(schema, args);
return `Invalid arguments for tool call. Expected:\n${zodSchemaToString(
schema,
)}.\nGot:\n${JSON.stringify(args)}`;
)}.\nGot:\n${JSON.stringify(args)}\nMissing keys:\n - ${missingKeys.join(
"\n - ",
)}\n`;
}
/**
* Whether or not to route to the diagnose error step. This is true if:
* - the last two tool messages are of an error status
* - two of the last three messages are an error status, including the last tool message
* @param toolMessages The tool messages to check the status of.
*/
function shouldDiagnoseError(toolMessages: ToolMessage[]) {
if (
toolMessages[toolMessages.length - 1].status !== "error" ||
toolMessages.length < 2
) {
// Last message is not an error, then neither of the below two conditions should be true.
return false;
}
return (
// Two of the three last tool calls are errors, return true
// (this is either the last two, or the 3rd, and last since the check above ensures the last is an error)
toolMessages.slice(-3).filter((m) => m.status === "error").length >= 2
);
}
export async function takeAction(
state: GraphState,
config: GraphConfig,
): Promise<GraphUpdate> {
): Promise<Command> {
const lastMessage = state.messages[state.messages.length - 1];
if (!isAIMessage(lastMessage) || !lastMessage.tool_calls?.length) {
@ -52,10 +84,15 @@ export async function takeAction(
}
let result = "";
let toolCallStatus: "success" | "error" = "success";
try {
// @ts-expect-error tool.invoke types are weird here...
result = await tool.invoke(toolCall.args);
const toolResult: { result: string; status: "success" | "error" } =
// @ts-expect-error tool.invoke types are weird here...
await tool.invoke(toolCall.args);
result = toolResult.result;
toolCallStatus = toolResult.status;
} catch (e) {
toolCallStatus = "error";
if (
e instanceof Error &&
e.message === "Received tool input did not match expected schema"
@ -80,6 +117,7 @@ export async function takeAction(
tool_call_id: toolCall.id ?? "",
content: result,
name: toolCall.name,
status: toolCallStatus,
});
// Always check if there are changed files after running a tool.
@ -100,8 +138,17 @@ export async function takeAction(
});
}
return {
messages: [toolMessage],
...(branchName && { branchName }),
};
const shouldRouteDiagnoseNode = shouldDiagnoseError(
[...state.messages, toolMessage].filter(
(m): m is ToolMessage =>
isToolMessage(m) && !m.additional_kwargs?.is_diagnosis,
),
);
return new Command({
goto: shouldRouteDiagnoseNode ? "diagnose-error" : "progress-plan-step",
update: {
messages: [toolMessage],
...(branchName && { branchName }),
},
});
}

View file

@ -29,7 +29,7 @@ export async function generateAction(
const firstUserMessage = state.messages.find(isHumanMessage);
const response = await modelWithTools
.withConfig({ tags: ["langsmith:nostream"] })
.withConfig({ tags: ["nostream"] })
.invoke([
{
role: "system",

View file

@ -51,7 +51,7 @@ export async function generatePlan(
}
const response = await modelWithTools
.withConfig({ tags: ["langsmith:nostream"] })
.withConfig({ tags: ["nostream"] })
.invoke([
{
role: "system",

View file

@ -2,7 +2,7 @@ import { z } from "zod";
import { GraphConfig } from "../../../types.js";
import { PlannerGraphState, PlannerGraphUpdate } from "../types.js";
import { loadModel, Task } from "../../../utils/load-model.js";
import { AIMessage, isHumanMessage } from "@langchain/core/messages";
import { isHumanMessage } from "@langchain/core/messages";
import {
getMessageContentString,
getMessageString,
@ -64,7 +64,7 @@ ${state.plannerMessages.map(getMessageString).join("\n")}`;
role: "system",
content: formatPrompt(
getMessageContentString(
firstUserMessage?.content ?? "No user request provided.",
firstUserMessage?.content || "No user request provided.",
),
),
},
@ -79,22 +79,7 @@ ${state.plannerMessages.map(getMessageString).join("\n")}`;
throw new Error("Failed to generate plan");
}
delete response.tool_call_chunks;
delete response.tool_calls;
delete response.invalid_tool_calls;
const messageWithoutToolCall = new AIMessage({
...response,
content:
"Condensed Planning Context:\n\n" +
(toolCall.args as z.infer<typeof condenseContextToolSchema>).context,
additional_kwargs: {
...response.additional_kwargs,
summary_message: true,
},
});
return {
messages: [messageWithoutToolCall],
planContextSummary: toolCall.args.context,
};
}

View file

@ -11,7 +11,11 @@ import { createLogger, LogLevel } from "../utils/logger.js";
const logger = createLogger(LogLevel.INFO, "ApplyPatchTool");
const applyPatchToolSchema = z.object({
diff: z.string().describe("The diff to apply. Use a standard diff format."),
diff: z
.string()
.describe(
"The diff to apply. Use a standard diff format. Ensure this field is ALWAYS provided.",
),
file_path: z.string().describe("The file path to apply the diff to."),
workdir: z
.string()
@ -22,7 +26,7 @@ const applyPatchToolSchema = z.object({
});
export const applyPatchTool = tool(
async (input) => {
async (input): Promise<{ result: string; status: "success" | "error" }> => {
const state = getCurrentTaskInput<GraphState>();
const { sandboxSessionId } = state;
if (!sandboxSessionId) {
@ -44,8 +48,11 @@ export const applyPatchTool = tool(
},
);
if (!readFileSuccess) {
logger.error("Failed to read file", readFileOutput);
return readFileOutput;
logger.error(readFileOutput);
return {
result: readFileOutput,
status: "error",
};
}
let patchedContent: string | false;
@ -62,14 +69,20 @@ export const applyPatchTool = tool(
: { error: e }),
});
const errMessage = e instanceof Error ? e.message : "Unknown error";
return `FAILED TO APPLY PATCH: The diff could not be applied to file '${file_path}'.\n\nError: ${errMessage}`;
return {
result: `FAILED TO APPLY PATCH: The diff could not be applied to file '${file_path}'.\n\nError: ${errMessage}`,
status: "error",
};
}
if (patchedContent === false) {
logger.error(
`FAILED TO APPLY PATCH: The diff could not be applied to file '${file_path}'. This may be due to an invalid diff format or conflicting changes with the file's current content. Original content length: ${readFileOutput.length}, Diff: ${diff.substring(0, 100)}...`,
);
return `FAILED TO APPLY PATCH: The diff could not be applied to file '${file_path}'. This may be due to an invalid diff format or conflicting changes with the file's current content. Original content length: ${readFileOutput.length}, Diff: ${diff.substring(0, 100)}...`;
return {
result: `FAILED TO APPLY PATCH: The diff could not be applied to file '${file_path}'. This may be due to an invalid diff format or conflicting changes with the file's current content. Original content length: ${readFileOutput.length}, Diff: ${diff.substring(0, 100)}...`,
status: "error",
};
}
const { success: writeFileSuccess, output: writeFileOutput } =
@ -80,17 +93,24 @@ export const applyPatchTool = tool(
logger.error("Failed to write file", {
writeFileOutput,
});
return writeFileOutput;
return {
result: writeFileOutput,
status: "error",
};
}
logger.info(
`Successfully applied diff to \`${file_path}\` and saved changes.`,
);
return `Successfully applied diff to \`${file_path}\` and saved changes.`;
return {
result: `Successfully applied diff to \`${file_path}\` and saved changes.`,
status: "success",
};
},
{
name: "apply_patch",
description: "Applies a diff to a file given a file path and diff content.",
description:
"Applies a diff to a file given a file path and diff content. Ensure you ALWAYS pass a valid file path to this tool. The combination of `workdir` and `file_path` should point to a valid file in the sandbox. Ensure you do not omit parts of the path between `workdir` and `file_path`.",
schema: applyPatchToolSchema,
},
);

View file

@ -29,7 +29,7 @@ const shellToolSchema = z.object({
});
export const shellTool = tool(
async (input) => {
async (input): Promise<{ result: string; status: "success" | "error" }> => {
try {
const state = getCurrentTaskInput<GraphState>();
const { sandboxSessionId } = state;
@ -57,10 +57,16 @@ export const shellTool = tool(
error_result: result,
input,
});
return `Command failed. Exit code: ${result.exitCode}\nError: ${result.error}\nStderr:\n${result.stderr}`;
return {
result: `Command failed. Exit code: ${result.exitCode}\nError: ${result.error}\nStderr:\n${result.stderr}`,
status: "error",
};
}
return result.stdout;
return {
result: result.stdout,
status: "success",
};
} catch (e) {
const errorFields = getSandboxErrorFields(e);
if (errorFields) {
@ -68,7 +74,10 @@ export const shellTool = tool(
input,
error: errorFields,
});
return `Command failed. Exit code: ${errorFields.exitCode}\nError: ${errorFields.error}\nStderr:\n${errorFields.stderr}\nStdout:\n${errorFields.stdout}`;
return {
result: `Command failed. Exit code: ${errorFields.exitCode}\nError: ${errorFields.error}\nStderr:\n${errorFields.stderr}\nStdout:\n${errorFields.stdout}`,
status: "error",
};
}
logger.error(

View file

@ -21,6 +21,10 @@ export type PlanItem = {
* Whether or not the plan item has been completed.
*/
completed: boolean;
/**
* A summary of the completed task.
*/
summary?: string;
};
export type TargetRepository = {
@ -47,6 +51,14 @@ export const GraphAnnotation = z.object({
.nullable()
.default(() => null)
.langgraph.reducer((_state, update) => update),
planContextSummary: z
.string()
.default(() => "")
.langgraph.reducer((_state, update) => update),
codebaseContext: z
.string()
.default(() => "")
.langgraph.reducer((_state, update) => update),
/**
* The session ID of the Sandbox to use.
*/

View file

@ -41,7 +41,9 @@ export function getToolMessageString(message: ToolMessage): string {
const content = getMessageContentString(message.content);
const toolCallId = message.tool_call_id;
const toolCallName = message.name;
return `<tool message-id=${message.id ?? "No ID"}>\nTool Call ID: ${toolCallId}\nTool Call Name: ${toolCallName}\nContent: ${content}\n</tool>`;
const toolStatus = message.status || "success";
return `<tool message-id=${message.id ?? "No ID"} status="${toolStatus}">\nTool Call ID: ${toolCallId}\nTool Call Name: ${toolCallName}\nContent: ${content}\n</tool>`;
}
export function getSystemMessageString(message: SystemMessage): string {

View file

@ -4,6 +4,7 @@ export const PLAN_PROMPT = `## Completed Tasks
{COMPLETED_TASKS}
## Remaining Tasks
(This list does not include the current task)
{REMAINING_TASKS}
## Current Task
@ -14,31 +15,68 @@ export const PLAN_PROMPT = `## Completed Tasks
* @param plan The plan to format
* @param options Options for formatting the plan
* @param options.useLastCompletedTask Whether to use the last completed task as the current task
* @param options.includeSummaries Whether to include summaries of completed tasks
* @returns The formatted plan
*/
export function formatPlanPrompt(
plan: PlanItem[],
options?: {
useLastCompletedTask?: boolean;
includeSummaries?: boolean;
},
): string {
const completedTasks = plan.filter((p) => p.completed);
const remainingTasks = plan.filter((p) => !p.completed);
const currentTask = options?.useLastCompletedTask
? completedTasks.sort((a, b) => a.index - b.index)[0]
: remainingTasks.sort((a, b) => a.index - b.index)[0];
let completedTasks = plan.filter((p) => p.completed);
let remainingTasks = plan.filter((p) => !p.completed);
let currentTask: PlanItem | undefined;
if (options?.useLastCompletedTask) {
currentTask = completedTasks.sort((a, b) => a.index - b.index)[0];
// Remove the current task from the completed tasks list:
completedTasks = completedTasks.filter(
(p) => p.index !== currentTask?.index,
);
} else {
currentTask = remainingTasks.sort((a, b) => a.index - b.index)[0];
// Remove the current task from the remaining tasks list:
remainingTasks = remainingTasks.filter(
(p) => p.index !== currentTask?.index,
);
}
return PLAN_PROMPT.replace(
"{COMPLETED_TASKS}",
completedTasks?.length
? completedTasks.map((task) => `${task.index}. ${task.plan}`).join("\n")
? options?.includeSummaries
? formatPlanPromptWithSummaries(completedTasks)
: completedTasks
.map(
(task) =>
`<completed-task index="${task.index}">${task.plan}</completed-task>`,
)
.join("\n")
: "No completed tasks.",
)
.replace(
"{REMAINING_TASKS}",
remainingTasks?.length
? remainingTasks.map((task) => `${task.index}. ${task.plan}`).join("\n")
? remainingTasks
.map(
(task) =>
`<remaining-task index="${task.index}">${task.plan}</remaining-task>`,
)
.join("\n")
: "No remaining tasks.",
)
.replace("{CURRENT_TASK}", currentTask?.plan || "No current task.");
.replace(
"{CURRENT_TASK}",
`<current-task index="${currentTask?.index}">${currentTask?.plan || "No current task found."}</current-task>`,
);
}
export function formatPlanPromptWithSummaries(plan: PlanItem[]): string {
return plan
.map(
(p) =>
`<task index="${p.index}">\n${p.plan}\n <task-summary>\n${p.summary || "No task summary found"}\n </task-summary>\n</task>`,
)
.join("\n");
}

View file

@ -56,14 +56,10 @@ async function readFileFunc(
});
return {
success: false,
output: `FAILED TO READ FILE from sandbox '${filePath}'. Exit code: ${readOutput.exitCode}.\nStderr: ${readOutput.stderr}.\nStdout: ${readOutput.stdout}`,
output: `FAILED TO READ FILE from sandbox '${filePath}'. Exit code: ${readOutput.exitCode}.\nStderr: ${readOutput.stderr}\nStdout: ${readOutput.stdout}`,
};
}
if (readOutput.stderr) {
logger.warn(
`Stderr while reading file '${filePath}' from sandbox via cat: ${readOutput.stderr}`,
);
}
return {
success: true,
output: readOutput.stdout,
@ -93,11 +89,15 @@ async function readFileFunc(
let outputMessage = `FAILED TO EXECUTE READ COMMAND for sandbox '${filePath}'.`;
const errorFields = getSandboxErrorFields(e);
if (errorFields) {
outputMessage += `\nExit code: ${errorFields.exitCode}.\nStderr: ${errorFields.stderr}.\nStdout: ${errorFields.stdout}`;
outputMessage += `\nExit code: ${errorFields.exitCode}\nStderr: ${errorFields.stderr}\nStdout: ${errorFields.stdout}`;
} else {
outputMessage += ` Error: ${(e as Error).message || String(e)}`;
}
if (outputMessage.includes("No such file or directory")) {
outputMessage += `\nPlease check the file paths you passed to \`workdir\` and \`file_path\` to ensure they are valid, and when combined they point to a valid file in the sandbox.`;
}
return {
success: false,
output: outputMessage,
@ -137,7 +137,7 @@ ${delimiter}`;
});
return {
success: false,
output: `FAILED TO WRITE FILE to sandbox '${filePath}'. Exit code: ${writeOutput.exitCode}. Stderr: ${writeOutput.stderr}. Stdout: ${writeOutput.stdout}`,
output: `FAILED TO WRITE FILE to sandbox '${filePath}'. Exit code: ${writeOutput.exitCode}\nStderr: ${writeOutput.stderr}\nStdout: ${writeOutput.stdout}`,
};
}
if (writeOutput.stderr) {
@ -162,7 +162,7 @@ ${delimiter}`;
let outputMessage = `FAILED TO EXECUTE WRITE COMMAND for sandbox '${filePath}'.`;
const errorFields = getSandboxErrorFields(e);
if (errorFields) {
outputMessage += `\nExit code: ${errorFields.exitCode}.\nStderr: ${errorFields.stderr}.\nStdout: ${errorFields.stdout}`;
outputMessage += `\nExit code: ${errorFields.exitCode}\nStderr: ${errorFields.stderr}\nStdout: ${errorFields.stdout}`;
} else {
outputMessage += ` Error: ${(e as Error).message || String(e)}`;
}

View file

@ -1,5 +1,16 @@
import { z } from "zod";
export function getMissingKeysFromObjectSchema(
schema: z.ZodTypeAny,
obj: Record<string, any>,
): string[] {
if (!(schema instanceof z.ZodObject)) {
throw new Error("Schema must be a ZodObject.");
}
return Object.keys(schema._def.shape()).filter((key) => !(key in obj));
}
export function zodSchemaToString(schema: z.ZodTypeAny, indent = 0): string {
const spaces = " ".repeat(indent);

View file

@ -2,7 +2,7 @@
"extends": "@tsconfig/recommended",
"compilerOptions": {
"target": "ES2021",
"lib": ["ES2021", "ES2022.Object", "DOM"],
"lib": ["ES2023"],
"module": "NodeNext",
"moduleResolution": "nodenext",
"esModuleInterop": true,