fix: Improve planner prompt (#276)

* fix: Improve planner prompt

* add pkgs and drop console log

* cr

* cr
This commit is contained in:
Brace Sproul 2025-06-19 16:04:02 -07:00 • committed by GitHub
parent 4f88413002
commit e8c742dc71
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
13 changed files with 209 additions and 121 deletions

View file

@ -12,7 +12,7 @@ import {
generatePlan,
interruptProposedPlan,
prepareGraphState,
summarizer,
notetaker,
takeAction,
} from "./nodes/index.js";
import { isAIMessage } from "@langchain/core/messages";
@ -24,10 +24,10 @@ function takeActionOrGeneratePlan(
): "take-plan-action" | "generate-plan" {
const { messages } = state;
const lastMessage = messages[messages.length - 1];
// If the last message is a tool call, and we have executed less than 6 actions, take action.
// If the last message is a tool call, and we have executed less than 75 actions, take action.
// Max actions count is calculated as: maxContextActions * 2 + 1
// This is because each action generates 2 messages (AI request + tool result) plus 1 initial human message
const maxContextActions = config.configurable?.maxContextActions ?? 6;
const maxContextActions = config.configurable?.maxContextActions ?? 75;
const maxActionsCount = maxContextActions * 2 + 1;
if (
isAIMessage(lastMessage) &&
@ -49,7 +49,7 @@ const workflow = new StateGraph(PlannerGraphStateObj, GraphConfiguration)
.addNode("generate-plan-context-action", generateAction)
.addNode("take-plan-action", takeAction)
.addNode("generate-plan", generatePlan)
.addNode("summarizer", summarizer)
.addNode("notetaker", notetaker)
.addNode("interrupt-proposed-plan", interruptProposedPlan)
.addEdge(START, "prepare-graph-state")
.addEdge("initialize-sandbox", "generate-plan-context-action")
@ -59,8 +59,8 @@ const workflow = new StateGraph(PlannerGraphStateObj, GraphConfiguration)
["take-plan-action", "generate-plan"],
)
.addEdge("take-plan-action", "generate-plan-context-action")
.addEdge("generate-plan", "summarizer")
.addEdge("summarizer", "interrupt-proposed-plan")
.addEdge("generate-plan", "notetaker")
.addEdge("notetaker", "interrupt-proposed-plan")
.addEdge("interrupt-proposed-plan", END);
export const graph = workflow.compile();

View file

@ -42,13 +42,17 @@ Your sole objective in this phase is to gather comprehensive context about the c
2. **Make high-quality, targeted tool calls**: Each command should have a clear purpose in building your understanding of the codebase. Think strategically about what information you need.
3. **Leverage efficient search tools**: Use \`rg\` (ripgrep) for all file searches because it respects .gitignore patterns and provides significantly faster results than alternatives like grep or ls -R.
3. **Gather all of the context necessary**: Ensure you gather all of the necessary context to generate a plan, and then execute that plan without having to gather additional context.
- You do not want to have to generate tasks such as 'Locate the XYZ file', 'Examine the structure of the codebase', or 'Do X if Y is true, otherwise to Z'.
- To ensure the above does not happen, you should be thorough in your context gathering. Always gather enough context to cover all edge cases, and prevent unclear instructions.
4. **Leverage efficient search tools**: Use \`rg\` (ripgrep) for all file searches because it respects .gitignore patterns and provides significantly faster results than alternatives like grep or ls -R.
- When searching for specific file types, use glob patterns: \`rg -i pattern -g **/*.tsx project-directory/\`
- This explicit pattern matching ensures accurate results across all file extensions
4. **Format shell commands precisely**: Ensure all shell commands include proper quoting and escaping. Well-formatted commands prevent errors and provide reliable results.
5. **Format shell commands precisely**: Ensure all shell commands include proper quoting and escaping. Well-formatted commands prevent errors and provide reliable results.
5. **Signal completion clearly**: When you have gathered sufficient context, respond with exactly 'done' without any tool calls. This indicates readiness to proceed to the planning phase.
6. **Signal completion clearly**: When you have gathered sufficient context, respond with exactly 'done' without any tool calls. This indicates readiness to proceed to the planning phase.
</context_gathering_guidelines>
<workspace_information>

View file

@ -14,28 +14,54 @@ import {
import { stopSandbox } from "../../../utils/sandbox.js";
import { filterHiddenMessages } from "../../../utils/message/filter-hidden.js";
const systemPrompt = `You are operating as a terminal-based agentic coding assistant built by LangChain. It wraps LLM models to enable natural language interaction with a local codebase. You are expected to be precise, safe, and helpful.
{FOLLOWUP_MESSAGE_PROMPT}
const systemPrompt = `You are a terminal-based agentic coding assistant built by LangChain, designed to enable natural language interaction with local codebases through wrapped LLM models.
In this step, you are expected to generate a high-level plan to address the user's request. The plan should be a list of actions to take, in order, to address the user's request. You should not include any code in the plan, only a list of actions to take.
<context>{FOLLOWUP_MESSAGE_PROMPT}
You have already gathered comprehensive context from the repository through the conversation history below. All previous messages will be deleted after this planning step, so your plan must be self-contained and actionable without referring back to this context.
</context>
You MUST adhere to the following criteria when generating the plan:
- You have already gathered context from the repository the user has requested you take actions on, and are now ready to generate a plan based on it.
- This context is provided in the conversation history below.
- Your plan should be high-level in nature, but should still be specific enough to be actionable.
- Ensure your plan is as concise as possible. Omit any unnecessary details or steps the user did not request, or are not required to complete the task.
- Your goal is to complete the task outlined by the user in the least number of steps possible.
- Do not pack multiple complex tasks into a single plan item. Each high level task you'll need to complete should have its own plan item.
- When you are ready to generate the plan, ensure you call the 'session_plan' tool. You are REQUIRED to call this tool.
- Your plan should be as simple as possible, while still containing all the tasks required to complete the user's request.
- If the user did not explicitly request you write tests, do not include a task to write tests.
- If the user did not explicitly request you write documentation, do not include a task to do so.
- You should aim to complete the user's request in the least number of steps possible.
<task>
Generate a high-level execution plan to address the user's request. Your plan will guide the implementation phase, so each action must be specific and actionable.
The user's request is as follows. Ensure you generate your plan in accordance with the user's request.
<user_request>
{USER_REQUEST}
`;
</user_request>
</task>
<instructions>
Create your plan following these guidelines:
1. **Structure each action item to include:**
- The specific task to accomplish
- Key technical details needed for execution
- File paths, function names, or other concrete references from the context you've gathered
2. **Write actionable items that:**
- Focus on implementation steps, not information gathering
- Can be executed independently without additional context discovery
- Build upon each other in logical sequence
- Are not open ended, and require additional context to execute
3. **Optimize for efficiency by:**
- Completing the request in the minimum number of steps
- Reusing existing code and patterns wherever possible
- Writing reusable components when code will be used multiple times
4. **Include only what's requested:**
- Add testing steps only if the user explicitly requested tests
- Add documentation steps only if the user explicitly requested documentation
- Focus solely on fulfilling the stated requirements
</instructions>
<output_format>
When ready, call the 'session_plan' tool with your plan. Each plan item should be a complete, self-contained action that can be executed without referring back to this conversation.
Structure your plan items as clear directives, for example:
- "Implement function X in file Y that performs Z using the existing pattern from file A"
- "Modify the authentication middleware in /src/auth.js to add rate limiting using the Express rate-limit package"
</output_format>
Remember: Your goal is to create a focused, executable plan that efficiently accomplishes the user's request using the context you've already gathered.`;
function formatSystemPrompt(state: PlannerGraphState): string {
// It's a followup if there's more than one human message.
@ -46,7 +72,9 @@ function formatSystemPrompt(state: PlannerGraphState): string {
.replace(
"{FOLLOWUP_MESSAGE_PROMPT}",
isFollowup
? formatFollowupMessagePrompt(state.taskPlan, state.proposedPlan)
? "\n" +
formatFollowupMessagePrompt(state.taskPlan, state.proposedPlan) +
"\n\n"
: "",
)
.replace("{USER_REQUEST}", userRequest);

View file

@ -1,6 +1,6 @@
export * from "./generate-message/index.js";
export * from "./take-action.js";
export * from "./generate-plan.js";
export * from "./summarizer.js";
export * from "./notetaker.js";
export * from "./proposed-plan.js";
export * from "./prepare-state.js";

View file

@ -0,0 +1,107 @@
import { z } from "zod";
import { GraphConfig } from "@open-swe/shared/open-swe/types";
import {
PlannerGraphState,
PlannerGraphUpdate,
} from "@open-swe/shared/open-swe/planner/types";
import { loadModel, Task } from "../../../utils/load-model.js";
import { getMessageString } from "../../../utils/message/content.js";
import { getUserRequest } from "../../../utils/user-request.js";
import { BaseMessage } from "@langchain/core/messages";
const systemPrompt = `You are operating as a terminal-based agentic coding assistant built by LangChain. It wraps LLM models to enable natural language interaction with a local codebase. You are expected to be precise, safe, and helpful.
You've just finished gathering context to aid in generating a development plan to address the user's request. The context you've gathered is provided in the conversation history below.
After this, the conversation history will be deleted, and you'll start executing on the plan.
Your task is to carefully read over the conversation history, and take notes on the most important and useful actions you performed which will be helpful to you when you go and execute on the plan.
The notes you extract should be thoughtful, and should include technical details about the codebase, files, patterns, dependencies and setup instructions you discovered during the context gathering step, which you believe will be helpful when you go to execute on the plan.
These notes should not be overly verbose, as you'll be able to gather additional context when executing.
Your goal is to generate notes on all of the low-hanging fruit from the conversation history, to speed up the execution so that you don't need to duplicate work to gather context.
You MUST adhere to the following criteria when generating your notes:
- Do not retain any full code snippets.
- Do not retain any full file contents.
- Only take notes on the context provided below, and do not make up, or attempt to infer any information/context which is not explicitly provided.
Here is the user's request
## User request:
{USER_REQUEST}
Here is the conversation history:
## Conversation history:
{CONVERSATION_HISTORY}
And here is the plan you just generated:
## Proposed plan:
{PROPOSED_PLAN}
With all of this in mind, please carefully inspect the conversation history, and the plan you generated. Then, determine which actions and context from the conversation history will be most useful to you when you execute the plan. After you're done analyzing, call the \`write_technical_notes\` tool.
`;
const formatPrompt = (
userRequest: string,
conversationHistory: BaseMessage[],
proposedPlan: string[],
): string =>
systemPrompt
.replace("{USER_REQUEST}", userRequest)
.replace(
"{CONVERSATION_HISTORY}",
conversationHistory.map(getMessageString).join("\n"),
)
.replace("{PROPOSED_PLAN}", ` - ${proposedPlan.join("\n - ")}`);
const condenseContextToolSchema = z.object({
notes: z
.string()
.describe("The notes you've generated based on the conversation history."),
});
const condenseContextTool = {
name: "write_technical_notes",
description:
"Write technical notes based on the conversation history provided. Ensure these notes are concise, but still containing enough information to be useful to you when you go to execute the plan.",
schema: condenseContextToolSchema,
};
export async function notetaker(
state: PlannerGraphState,
config: GraphConfig,
): Promise<PlannerGraphUpdate> {
const model = await loadModel(config, Task.SUMMARIZER);
const modelWithTools = model.bindTools([condenseContextTool], {
tool_choice: condenseContextTool.name,
parallel_tool_calls: false,
});
const userRequest = getUserRequest(state.messages);
const conversationHistoryStr = `Here is the full conversation history:
${state.messages.map(getMessageString).join("\n")}`;
const response = await modelWithTools.invoke([
{
role: "system",
content: formatPrompt(
userRequest || "No user request provided.",
state.messages,
state.proposedPlan,
),
},
{
role: "user",
content: conversationHistoryStr,
},
]);
const toolCall = response.tool_calls?.[0];
if (!toolCall) {
throw new Error("Failed to generate plan");
}
return {
messages: [response],
contextGatheringNotes: (
toolCall.args as z.infer<typeof condenseContextToolSchema>
).notes,
};
}

View file

@ -99,7 +99,7 @@ export async function prepareGraphState(
.map((m: BaseMessage) => new RemoveMessage({ id: m.id ?? "" }));
const summaryMessage = new AIMessage({
id: uuidv4(),
content: state.planContextSummary,
content: state.contextGatheringNotes,
additional_kwargs: {
summaryMessage: true,
},
@ -111,7 +111,7 @@ export async function prepareGraphState(
...untrackedComments,
],
// Reset plan context summary as it's now included in the messages array.
planContextSummary: "",
contextGatheringNotes: "",
};
return new Command({

View file

@ -76,7 +76,7 @@ export async function interruptProposedPlan(
const userRequest = getUserRequest(state.messages);
const runInput: GraphUpdate = {
planContextSummary: state.planContextSummary,
contextGatheringNotes: state.contextGatheringNotes,
branchName: state.branchName,
targetRepository: state.targetRepository,
githubIssueId: state.githubIssueId,

View file

@ -1,82 +0,0 @@
import { z } from "zod";
import { GraphConfig } from "@open-swe/shared/open-swe/types";
import {
PlannerGraphState,
PlannerGraphUpdate,
} from "@open-swe/shared/open-swe/planner/types";
import { loadModel, Task } from "../../../utils/load-model.js";
import { getMessageString } from "../../../utils/message/content.js";
import { getUserRequest } from "../../../utils/user-request.js";
const systemPrompt = `You are operating as a terminal-based agentic coding assistant built by LangChain. It wraps LLM models to enable natural language interaction with a local codebase. You are expected to be precise, safe, and helpful.
You've been given a task to summarize the messages in your conversation history. You just finished gathering context to be used when generating an development plan to address the user's request.
You do not want to keep the entire conversation history, but instead you want to keep the most relevant and important snippets for future context
You MUST adhere to the following criteria when summarizing the conversation history:
- Retain context such as file paths, versions, and installed software.
- It is very important to include the file paths of files you've already searched for, along with a description of the file's contents, inside a 'Codebase files and descriptions' section, so that future steps can reuse this information, and will not need to search through the codebase for files again.
- Consider including a section titled 'Key repository insights and learnings' which may include information, insights and learnings you've discovered while gathering context for the user's request.
- This section should be concise, but still including enough information so following steps will not repeat any mistakes or go down rabbit holes which you already know about.
- Do not retain any full code snippets.
- Do not retain any full file contents.
- Ensure your summary is concise, but useful for future context.
Here is the user's request
## User request:
{USER_REQUEST}
With all of this in mind, please carefully summarize and condense the following conversation history. Ensure you pass this condensed context to the \`condense_planning_context\` tool.
`;
const formatPrompt = (userRequest: string): string =>
systemPrompt.replace("{USER_REQUEST}", userRequest);
const condenseContextToolSchema = z.object({
context: z
.string()
.describe("The condensed context to be used when generating a plan."),
});
const condenseContextTool = {
name: "condense_planning_context",
description:
"Condense the conversation history into a concise summary, while still retaining the most relevant and important snippets.",
schema: condenseContextToolSchema,
};
export async function summarizer(
state: PlannerGraphState,
config: GraphConfig,
): Promise<PlannerGraphUpdate> {
const model = await loadModel(config, Task.SUMMARIZER);
const modelWithTools = model.bindTools([condenseContextTool], {
tool_choice: condenseContextTool.name,
parallel_tool_calls: false,
});
const userRequest = getUserRequest(state.messages);
const conversationHistoryStr = `Here is the full conversation history:
${state.messages.map(getMessageString).join("\n")}`;
const response = await modelWithTools.invoke([
{
role: "system",
content: formatPrompt(userRequest || "No user request provided."),
},
{
role: "user",
content: conversationHistoryStr,
},
]);
const toolCall = response.tool_calls?.[0];
if (!toolCall) {
throw new Error("Failed to generate plan");
}
return {
messages: [response],
planContextSummary: toolCall.args.context,
};
}

View file

@ -41,8 +41,8 @@ const formatPrompt = (state: GraphState): string => {
)
.replaceAll("{REPO_DIRECTORY}", repoDirectory)
.replaceAll(
"{PLAN_GENERATION_SUMMARY}",
`<plan-generation-summary>\n${state.planContextSummary}\n</plan-generation-summary>`,
"{PLAN_GENERATION_NOTES}",
`<plan-generation-notes>\n${state.contextGatheringNotes}\n</plan-generation-notes>`,
)
.replaceAll(
"{CODEBASE_TREE}",

View file

@ -161,8 +161,9 @@ When modifying files:
## Generated Plan with Summaries
{PLAN_PROMPT_WITH_SUMMARIES}
## Plan Generation Summary
{PLAN_GENERATION_SUMMARY}
## Plan Generation Notes
These are notes you took while gathering context for the plan:
{PLAN_GENERATION_NOTES}
## Current Task Status
{PLAN_PROMPT}

View file

@ -0,0 +1,30 @@
import {
BaseMessage,
isAIMessage,
isHumanMessage,
isToolMessage,
} from "@langchain/core/messages";
import { getMessageContentString } from "@open-swe/shared/messages";
export function calculateConversationHistoryTokenCount(
messages: BaseMessage[],
) {
let totalChars = 0;
messages.forEach((m) => {
if (isAIMessage(m)) {
const contentString = getMessageContentString(m.content);
totalChars += contentString.length;
m.tool_calls?.forEach((tc) => {
totalChars += tc.name.length;
totalChars += JSON.stringify(tc.args).length;
});
}
if (isHumanMessage(m) || isToolMessage(m)) {
const contentString = getMessageContentString(m.content);
totalChars += contentString.length;
}
});
// Estimate 1 token for every 4 characters.
return Math.ceil(totalChars / 4);
}

View file

@ -42,7 +42,7 @@ export const PlannerGraphStateObj = MessagesZodState.extend({
},
default: (): string[] => [],
}),
planContextSummary: withLangGraph(z.custom<string>(), {
contextGatheringNotes: withLangGraph(z.custom<string>(), {
reducer: {
schema: z.custom<string>(),
fn: (_state, update) => update,

View file

@ -149,9 +149,9 @@ export const GraphAnnotation = MessagesZodState.extend({
},
}),
/**
* The summary of actions taken by the planning agent.
* Notes taken based on the actions preformed by the planning agent.
*/
planContextSummary: withLangGraph(z.custom<string>(), {
contextGatheringNotes: withLangGraph(z.custom<string>(), {
reducer: {
schema: z.custom<string>(),
fn: (_state, update) => update,