[open-swe] feat: Implement 4-Tier Prompt Caching Strategy (#439)

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* Apply patch

* cleanup

* cr

* cr

* add caching to planner

* total coverage on token caching, cache reviewer messages

* cr

* cr

---------

Co-authored-by: open-swe-dev[bot] <open-swe-dev@users.noreply.github.com>
Co-authored-by: Brace Sproul <braceasproul@gmail.com>
This commit is contained in:
open-swe[bot] 2025-07-22 00:39:58 +00:00 • committed by GitHub
parent 3d557dbdd2
commit 984fc68556
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
26 changed files with 701 additions and 313 deletions

View file

@ -16,6 +16,7 @@ import { isHumanMessage } from "@langchain/core/messages";
import { getMessageContentString } from "@open-swe/shared/messages";
import { filterHiddenMessages } from "../../../utils/message/filter-hidden.js";
import { createLogger, LogLevel } from "../../../utils/logger.js";
import { trackCachePerformance } from "../../../utils/caching.js";
const logger = createLogger(LogLevel.INFO, "DetermineNeedsContext");
@ -154,6 +155,7 @@ export async function determineNeedsContext(
const commandUpdate: PlannerGraphUpdate = {
messages: missingMessages,
tokenData: trackCachePerformance(response),
};
const shouldGatherContext =

View file

@ -27,27 +27,41 @@ import { formatCustomRulesPrompt } from "../../../../utils/custom-rules.js";
import { createPlannerNotesTool } from "../../../../tools/planner-notes.js";
import { getMcpTools } from "../../../../utils/mcp-client.js";
import { filterMessagesWithoutContent } from "../../../../utils/message/content.js";
import { getPlannerNotes } from "../../utils/get-notes.js";
import { formatUserRequestPrompt } from "../../../../utils/user-request.js";
import {
convertMessagesToCacheControlledMessages,
trackCachePerformance,
} from "../../../../utils/caching.js";
const logger = createLogger(LogLevel.INFO, "GeneratePlanningMessageNode");
function formatSystemPrompt(state: PlannerGraphState): string {
// It's a followup if there's more than one human message.
const isFollowup = isFollowupRequest(state.taskPlan, state.proposedPlan);
const plannerNotes = getPlannerNotes(state.messages)
.map((n) => `- ${n}`)
.join("\n");
return SYSTEM_PROMPT.replace(
"{FOLLOWUP_MESSAGE_PROMPT}",
isFollowup
? formatFollowupMessagePrompt(state.taskPlan, state.proposedPlan)
? formatFollowupMessagePrompt(
state.taskPlan,
state.proposedPlan,
plannerNotes,
)
: "",
)
.replaceAll(
"{CODEBASE_TREE}",
state.codebaseTree || "No codebase tree generated yet.",
)
.replaceAll(
"{CURRENT_WORKING_DIRECTORY}",
getRepoAbsolutePath(state.targetRepository),
)
.replaceAll("{CUSTOM_RULES}", formatCustomRulesPrompt(state.customRules));
.replaceAll(
"{CODEBASE_TREE}",
state.codebaseTree || "No codebase tree generated yet.",
)
.replaceAll("{CUSTOM_RULES}", formatCustomRulesPrompt(state.customRules))
.replace("{USER_REQUEST_PROMPT}", formatUserRequestPrompt(state.messages));
}
export async function generateAction(
@ -71,6 +85,13 @@ export async function generateAction(
logger.info(
`MCP tools added to Planner: ${mcpTools.map((t) => t.name).join(", ")}`,
);
// Cache Breakpoint 1: Add cache_control marker to the last tool for tools definition caching
if (tools.length > 0) {
tools[tools.length - 1] = {
...tools[tools.length - 1],
cache_control: { type: "ephemeral" },
} as any;
}
const modelWithTools = model.bindTools(tools, {
tool_choice: "auto",
@ -94,6 +115,8 @@ export async function generateAction(
throw new Error("No messages to process.");
}
const inputMessagesWithCache =
convertMessagesToCacheControlledMessages(inputMessages);
const response = await modelWithTools
.withConfig({ tags: ["nostream"] })
.invoke([
@ -104,7 +127,7 @@ export async function generateAction(
taskPlan: latestTaskPlan ?? state.taskPlan,
}),
},
...inputMessages,
...inputMessagesWithCache,
]);
logger.info("Generated planning message", {
@ -120,5 +143,6 @@ export async function generateAction(
return {
messages: [...missingMessages, response],
...(latestTaskPlan && { taskPlan: latestTaskPlan }),
tokenData: trackCachePerformance(response),
};
}

View file

@ -1,4 +1,6 @@
export const SYSTEM_PROMPT = `You are a terminal-based agentic coding assistant built by LangChain that enables natural language interaction with local codebases. You excel at being precise, safe, and helpful in your analysis.
export const SYSTEM_PROMPT = `<identity>
You are a terminal-based agentic coding assistant built by LangChain that enables natural language interaction with local codebases. You excel at being precise, safe, and helpful in your analysis.
</identity>
<role>
Context Gathering Assistant - Read-Only Phase
@ -11,48 +13,44 @@ Your sole objective in this phase is to gather comprehensive context about the c
{FOLLOWUP_MESSAGE_PROMPT}
<context_gathering_guidelines>
1. **Use only read operations**: Execute commands that inspect and analyze the codebase without modifying any files. This ensures we understand the current state before making changes.
2. **Make high-quality, targeted tool calls**: Each command should have a clear purpose in building your understanding of the codebase. Think strategically about what information you need.
3. **Gather all of the context necessary**: Ensure you gather all of the necessary context to generate a plan, and then execute that plan without having to gather additional context.
- You do not want to have to generate tasks such as 'Locate the XYZ file', 'Examine the structure of the codebase', or 'Do X if Y is true, otherwise to Z'.
- To ensure the above does not happen, you should be thorough in your context gathering. Always gather enough context to cover all edge cases, and prevent unclear instructions.
4. **Leverage efficient search tools**:
- Use \`search\` tool for all file searches. The \`search\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns.
- It's significantly faster results than alternatives like grep or ls -R.
- When searching for specific file types, use glob patterns
- The query field supports both basic strings, and regex
- Always use the \`search\` tools instead calling \`grep\` via the \`shell\` tool. You should NEVER call \`grep\` as the same functionality is better provided by \`search\`.
- If the user passes a URL, you should use the \`get_url_content\` tool to fetch the contents of the URL.
- You should only use this tool to fetch the contents of a URL the user has provided, or that you've discovered during your context searching, which you believe is vital to gathering context for the user's request.
5. **Format shell commands precisely**: Ensure all shell commands include proper quoting and escaping. Well-formatted commands prevent errors and provide reliable results.
6. **Signal completion clearly**: When you have gathered sufficient context, respond with exactly 'done' without any tool calls. This indicates readiness to proceed to the planning phase.
7. **Parallel tool calling**: It is highly recommended that you use parallel tool calling to gather context as quickly and efficiently as possible. When you know ahead of time there are multiple commands you want to run to gather context, of which they are independent and can be run in parallel, you should use parallel tool calling.
- This is best utilized by search commands. You should always plan ahead for which search commands you want to run in parallel, then use parallel tool calling to run them all at once for maximum efficiency.
8. **Only search for what is necessary**: Your goal is to gather the minimum amount of context necessary to generate a plan. You should not gather context or perform searches that are not necessary to generate a plan.
- You will always be able to gather more context after the planning phase, so ensure that the actions you perform in this planning phase are only the most necessary and targeted actions to gather context.
- Avoid rabbit holes for gathering context. You should always first consider whether or not the action you're about to take is necessary to generate a plan for the user's request. If it is not, do not take it.
1. Use only read operations: Execute commands that inspect and analyze the codebase without modifying any files. This ensures we understand the current state before making changes.
2. Make high-quality, targeted tool calls: Each command should have a clear purpose in building your understanding of the codebase. Think strategically about what information you need.
3. Gather all of the context necessary: Ensure you gather all of the necessary context to generate a plan, and then execute that plan without having to gather additional context.
- You do not want to have to generate tasks such as 'Locate the XYZ file', 'Examine the structure of the codebase', or 'Do X if Y is true, otherwise to Z'.
- To ensure the above does not happen, you should be thorough in your context gathering. Always gather enough context to cover all edge cases, and prevent unclear instructions.
4. Leverage efficient search tools:
- Use \`search\` tool for all file searches. The \`search\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns.
- It's significantly faster results than alternatives like grep or ls -R.
- When searching for specific file types, use glob patterns
- The query field supports both basic strings, and regex
- Always use the \`search\` tools instead calling \`grep\` via the \`shell\` tool. You should NEVER call \`grep\` as the same functionality is better provided by \`search\`.
- If the user passes a URL, you should use the \`get_url_content\` tool to fetch the contents of the URL.
- You should only use this tool to fetch the contents of a URL the user has provided, or that you've discovered during your context searching, which you believe is vital to gathering context for the user's request.
5. Format shell commands precisely: Ensure all shell commands include proper quoting and escaping. Well-formatted commands prevent errors and provide reliable results.
6. Signal completion clearly: When you have gathered sufficient context, respond with exactly 'done' without any tool calls. This indicates readiness to proceed to the planning phase.
7. Parallel tool calling: It is highly recommended that you use parallel tool calling to gather context as quickly and efficiently as possible. When you know ahead of time there are multiple commands you want to run to gather context, of which they are independent and can be run in parallel, you should use parallel tool calling.
- This is best utilized by search commands. You should always plan ahead for which search commands you want to run in parallel, then use parallel tool calling to run them all at once for maximum efficiency.
8. Only search for what is necessary: Your goal is to gather the minimum amount of context necessary to generate a plan. You should not gather context or perform searches that are not necessary to generate a plan.
- You will always be able to gather more context after the planning phase, so ensure that the actions you perform in this planning phase are only the most necessary and targeted actions to gather context.
- Avoid rabbit holes for gathering context. You should always first consider whether or not the action you're about to take is necessary to generate a plan for the user's request. If it is not, do not take it.
</context_gathering_guidelines>
<workspace_information>
**Current Working Directory**: {CURRENT_WORKING_DIRECTORY}
**Repository Status**: Already cloned and accessible in the current directory
<current_working_directory>{CURRENT_WORKING_DIRECTORY}</current_working_directory>
<repository_status>Already cloned and accessible in the current directory</repository_status>
**Codebase Structure** (3 levels deep, respecting .gitignore):
Generated via: \`git ls-files | tree --fromfile -L 3\`
<codebase_tree>
{CODEBASE_TREE}
</codebase_tree>
<codebase_tree>
Generated via: \`git ls-files | tree --fromfile -L 3\`:
{CODEBASE_TREE}
</codebase_tree>
</workspace_information>
{CUSTOM_RULES}
<task_context>
The user's request appears as the first message in the conversation below. Your context gathering should specifically target information needed to address this request effectively.
The user's request is shown below. Your context gathering should specifically target information needed to address this request effectively.
<user_request>
{USER_REQUEST_PROMPT}
</user_request>
</task_context>`;

View file

@ -23,6 +23,7 @@ import { getPlannerNotes } from "../../utils/get-notes.js";
import { PLANNER_NOTES_PROMPT, SYSTEM_PROMPT } from "./prompt.js";
import { DO_NOT_RENDER_ID_PREFIX } from "@open-swe/shared/constants";
import { filterMessagesWithoutContent } from "../../../../utils/message/content.js";
import { trackCachePerformance } from "../../../../utils/caching.js";
function formatSystemPrompt(state: PlannerGraphState): string {
// It's a followup if there's more than one human message.
@ -124,5 +125,6 @@ export async function generatePlan(
proposedPlanTitle: proposedPlanArgs.title,
proposedPlan: proposedPlanArgs.plan,
...(newSessionId && { sandboxSessionId: newSessionId }),
tokenData: trackCachePerformance(response),
};
}

View file

@ -17,6 +17,7 @@ import { getPlannerNotes } from "../utils/get-notes.js";
import { ToolMessage } from "@langchain/core/messages";
import { DO_NOT_RENDER_ID_PREFIX } from "@open-swe/shared/constants";
import { createWriteTechnicalNotesToolFields } from "@open-swe/shared/open-swe/tools";
import { trackCachePerformance } from "../../../utils/caching.js";
const PLANNER_NOTES_PROMPT = `You've also taken technical notes throughout the context gathering process. Ensure you include/incorporate these notes, or the highest quality parts of these notes in your conclusion notes.
@ -145,5 +146,6 @@ ${state.messages.map(getMessageString).join("\n")}`;
contextGatheringNotes: (
toolCall.args as z.infer<typeof condenseContextTool.schema>
).notes,
tokenData: trackCachePerformance(response),
};
}

View file

@ -2,12 +2,18 @@ import { getActivePlanItems } from "@open-swe/shared/open-swe/tasks";
import { TaskPlan } from "@open-swe/shared/open-swe/types";
const previousCompletedPlanPrompt = `Here is the list of tasks from the previous session. You've already completed all of these tasks. Use the tasks, and task summaries as context when generating a new plan:
{PREVIOUS_PLAN}`;
{PREVIOUS_PLAN}
Here are the notes you took while gathering context for these tasks:
{PLANNER_NOTES}`;
const previousProposedPlanPrompt = `Here is the complete list of the proposed plan you generated before the user sent their followup request:
{PREVIOUS_PROPOSED_PLAN}`;
{PREVIOUS_PROPOSED_PLAN}
const followupMessagePrompt = `
Here are the notes you took while gathering context for these tasks:
{PLANNER_NOTES}`;
const followupMessagePrompt = `<followup_message_instructions>
The user is sending a followup request for you to generate a plan for. You are provided with the following context to aid in your new plan context gathering steps:
- The previous user requests, along with the tasks, and task summaries you generated for these previous requests.
- The summaries of the actions you took, and their results from previous planning sessions.
@ -15,9 +21,12 @@ The user is sending a followup request for you to generate a plan for. You are p
- If the user requests changes/additions to the proposed plan, your goal is to make as few changes/additions as possible, only addressing the specific changes the user requested.
{PREVIOUS_PLAN}
`;
</followup_message_instructions>`;
const formatPreviousPlans = (tasks: TaskPlan): string => {
const formatPreviousPlans = (
tasks: TaskPlan,
plannerNotes?: string,
): string => {
const formattedTasksAndRequests = tasks.tasks
.map((task) => {
const activePlanItems =
@ -42,25 +51,27 @@ ${activePlanItems
})
.join("\n");
return previousCompletedPlanPrompt.replace(
"{PREVIOUS_PLAN}",
formattedTasksAndRequests,
);
return previousCompletedPlanPrompt
.replace("{PREVIOUS_PLAN}", formattedTasksAndRequests)
.replace("{PLANNER_NOTES}", plannerNotes || "");
};
const formatPreviousProposedPlan = (proposedPlan: string[]): string => {
const formatPreviousProposedPlan = (
proposedPlan: string[],
plannerNotes?: string,
): string => {
const formattedProposedPlan = proposedPlan
.map((p) => `<proposed-plan-item>${p}</proposed-plan-item>`)
.join("\n");
return previousProposedPlanPrompt.replace(
"{PREVIOUS_PROPOSED_PLAN}",
formattedProposedPlan,
);
return previousProposedPlanPrompt
.replace("{PREVIOUS_PROPOSED_PLAN}", formattedProposedPlan)
.replace("{PLANNER_NOTES}", plannerNotes || "");
};
export function formatFollowupMessagePrompt(
tasks: TaskPlan,
proposedPlan: string[],
plannerNotes?: string,
): string {
let isGeneratingNewPlan = false;
if (tasks && tasks.tasks?.length) {
@ -72,12 +83,11 @@ export function formatFollowupMessagePrompt(
);
}
}
return followupMessagePrompt.replace(
"{PREVIOUS_PLAN}",
isGeneratingNewPlan
? formatPreviousPlans(tasks)
: formatPreviousProposedPlan(proposedPlan),
? formatPreviousPlans(tasks, plannerNotes)
: formatPreviousProposedPlan(proposedPlan, plannerNotes),
);
}

View file

@ -15,6 +15,7 @@ import {
getActiveTask,
} from "@open-swe/shared/open-swe/tasks";
import { addTaskPlanToIssue } from "../../../utils/github/issue-task.js";
import { trackCachePerformance } from "../../../utils/caching.js";
const logger = createLogger(LogLevel.INFO, "GenerateConclusionNode");
@ -82,5 +83,6 @@ Given all of this, please respond with the concise conclusion. Do not include an
messages: [response],
internalMessages: [response],
taskPlan: updatedTaskPlan,
tokenData: trackCachePerformance(response),
};
}

View file

@ -24,8 +24,9 @@ import { getActivePlanItems } from "@open-swe/shared/open-swe/tasks";
import {
CODE_REVIEW_PROMPT,
DEPENDENCIES_INSTALLED_PROMPT,
INSTALL_DEPENDENCIES_TOOL_PROMPT,
SYSTEM_PROMPT,
DEPENDENCIES_NOT_INSTALLED_PROMPT,
DYNAMIC_SYSTEM_PROMPT,
STATIC_SYSTEM_INSTRUCTIONS,
} from "./prompt.js";
import { getRepoAbsolutePath } from "@open-swe/shared/git";
import { getMissingMessages } from "../../../../utils/github/issue-messages.js";
@ -39,55 +40,79 @@ import {
getCodeReviewFields,
} from "../../../../utils/review.js";
import { filterMessagesWithoutContent } from "../../../../utils/message/content.js";
import {
CacheablePromptSegment,
convertMessagesToCacheControlledMessages,
trackCachePerformance,
} from "../../../../utils/caching.js";
const logger = createLogger(LogLevel.INFO, "GenerateMessageNode");
const formatPrompt = (state: GraphState): string => {
const repoDirectory = getRepoAbsolutePath(state.targetRepository);
const activePlanItems = getActivePlanItems(state.taskPlan);
const currentPlanItem = activePlanItems
.filter((p) => !p.completed)
.sort((a, b) => a.index - b.index)[0];
const codeReview = getCodeReviewFields(state.internalMessages);
return SYSTEM_PROMPT.replaceAll(
const formatDynamicContextPrompt = (state: GraphState) => {
return DYNAMIC_SYSTEM_PROMPT.replaceAll(
"{PLAN_PROMPT_WITH_SUMMARIES}",
formatPlanPrompt(getActivePlanItems(state.taskPlan), {
includeSummaries: true,
}),
)
.replaceAll(
"{PLAN_PROMPT}",
formatPlanPrompt(getActivePlanItems(state.taskPlan)),
)
.replaceAll("{REPO_DIRECTORY}", repoDirectory)
.replaceAll(
"{PLAN_GENERATION_NOTES}",
`<plan-generation-notes>\n${state.contextGatheringNotes}\n</plan-generation-notes>`,
state.contextGatheringNotes || "No context gathering notes available.",
)
.replaceAll("{REPO_DIRECTORY}", getRepoAbsolutePath(state.targetRepository))
.replaceAll(
"{DEPENDENCIES_INSTALLED_PROMPT}",
state.dependenciesInstalled
? DEPENDENCIES_INSTALLED_PROMPT
: DEPENDENCIES_NOT_INSTALLED_PROMPT,
)
.replaceAll(
"{CODEBASE_TREE}",
state.codebaseTree || "No codebase tree generated yet.",
)
.replaceAll("{CURRENT_WORKING_DIRECTORY}", repoDirectory)
.replaceAll("{CURRENT_TASK_NUMBER}", currentPlanItem.index.toString())
.replaceAll(
"{INSTALL_DEPENDENCIES_TOOL_PROMPT}",
!state.dependenciesInstalled
? INSTALL_DEPENDENCIES_TOOL_PROMPT
: DEPENDENCIES_INSTALLED_PROMPT,
)
.replaceAll("{CUSTOM_RULES}", formatCustomRulesPrompt(state.customRules))
.replaceAll(
"{CODE_REVIEW_PROMPT}",
codeReview
? formatCodeReviewPrompt(CODE_REVIEW_PROMPT, {
review: codeReview.review,
newActions: codeReview.newActions,
})
: "",
);
};
const formatStaticInstructionsPrompt = (state: GraphState) => {
return STATIC_SYSTEM_INSTRUCTIONS.replaceAll(
"{REPO_DIRECTORY}",
getRepoAbsolutePath(state.targetRepository),
).replaceAll("{CUSTOM_RULES}", formatCustomRulesPrompt(state.customRules));
};
const formatCacheablePrompt = (state: GraphState): CacheablePromptSegment[] => {
const codeReview = getCodeReviewFields(state.internalMessages);
const segments: CacheablePromptSegment[] = [
// Cache Breakpoint 2: Static Instructions
{
type: "text",
text: formatStaticInstructionsPrompt(state),
cache_control: { type: "ephemeral" },
},
// Cache Breakpoint 3: Dynamic Context
{
type: "text",
text: formatDynamicContextPrompt(state),
cache_control: { type: "ephemeral" },
},
];
// Cache Breakpoint 4: Code Review Context (only add if present)
if (codeReview) {
segments.push({
type: "text",
text: formatCodeReviewPrompt(CODE_REVIEW_PROMPT, {
review: codeReview.review,
newActions: codeReview.newActions,
}),
cache_control: { type: "ephemeral" },
});
}
return segments.filter((segment) => segment.text.trim() !== "");
};
export async function generateAction(
state: GraphState,
config: GraphConfig,
@ -106,16 +131,21 @@ export async function generateAction(
createRequestHumanHelpToolFields(),
createUpdatePlanToolFields(),
createGetURLContentTool(),
createInstallDependenciesTool(state),
...mcpTools,
// Only provide the dependencies installed tool if they're not already installed.
...(state.dependenciesInstalled
? []
: [createInstallDependenciesTool(state)]),
];
logger.info(
`MCP tools added to Programmer: ${mcpTools.map((t) => t.name).join(", ")}`,
);
// Cache Breakpoint 1: Add cache_control marker to the last tool for tools definition caching
if (tools.length > 0) {
tools[tools.length - 1] = {
...tools[tools.length - 1],
cache_control: { type: "ephemeral" },
} as any;
}
const modelWithTools = model.bindTools(tools, {
tool_choice: "auto",
...(modelSupportsParallelToolCallsParam
@ -138,15 +168,17 @@ export async function generateAction(
throw new Error("No messages to process.");
}
const inputMessagesWithCache =
convertMessagesToCacheControlledMessages(inputMessages);
const response = await modelWithTools.invoke([
{
role: "system",
content: formatPrompt({
content: formatCacheablePrompt({
...state,
taskPlan: latestTaskPlan ?? state.taskPlan,
}),
},
...inputMessages,
...inputMessagesWithCache,
]);
const hasToolCalls = !!response.tool_calls?.length;
@ -174,5 +206,6 @@ export async function generateAction(
internalMessages: newMessagesList,
...(newSandboxSessionId && { sandboxSessionId: newSandboxSessionId }),
...(latestTaskPlan && { taskPlan: latestTaskPlan }),
tokenData: trackCachePerformance(response),
};
}

View file

@ -1,125 +1,137 @@
export const INSTALL_DEPENDENCIES_TOOL_PROMPT = `* Use \`install_dependencies\` to install dependencies (skip if installation fails). IMPORTANT: You should only call this tool if you're executing a task which REQUIRES installing dependencies. Keep in mind that not all tasks will require installing dependencies.`;
export const DEPENDENCIES_INSTALLED_PROMPT = `* Dependencies have already been installed. *`;
export const STATIC_SYSTEM_INSTRUCTIONS = `<identity>
You are a terminal-based agentic coding assistant built by LangChain. You wrap LLM models to enable natural language interaction with local codebases. You are precise, safe, and helpful.
</identity>
export const CODE_REVIEW_PROMPT = `# Code Review & New Actions
<current_task_overview>
You are currently executing a specific task from a pre-generated plan. You have access to:
- Project context and files
- Shell commands and code editing tools
- A sandboxed, git-backed workspace with rollback support
</current_task_overview>
The code changes you've made have been reviewed by a code reviewer. The code review has determined that the changes do _not_ satisfy the user's request, and have outlined a list of additional actions to take in order to successfully complete the user's request.
<instructions>
<core_behavior>
- Persistence: Keep working until the current task is completely resolved. Only terminate when you are certain the task is complete.
- Accuracy: Never guess or make up information. Always use tools to gather accurate data about files and codebase structure.
- Planning: Leverage the plan context and task summaries heavily - they contain critical information about completed work and the overall strategy.
</core_behavior>
The code review has provided this review of the changes:
<task_execution_guidelines>
- You are executing a task from the plan.
- Previous completed tasks and their summaries contain crucial context - always review them first
- Condensed context messages in conversation history summarize previous work - read these to avoid duplication
- The plan generation summary provides important codebase insights
- After some tasks are completed, you may be provided with a code review and additional tasks. Ensure you inspect the code review (if present) and new tasks to ensure the work you're doing satisfies the user's request.
- Only modify the code outlined in the current task. You should always AVOID modifying code which is unrelated to the current tasks.
</task_execution_guidelines>
## Code Review
{CODE_REVIEW}
<file_and_code_management>
<repository_location>{REPO_DIRECTORY}</repository_location>
<current_directory>{REPO_DIRECTORY}</current_directory>
- All changes are auto-committed - no manual commits needed, and you should never create backup files.
- Work only within the existing Git repository
- Use \`apply_patch\` for file edits (accepts diffs and file paths)
- Use \`shell\` with \`touch\` to create new files (not \`apply_patch\`)
- Always use \`workdir\` parameter instead of \`cd\` when running commands via the \`shell\` tool
- Use \`install_dependencies\` to install dependencies (skip if installation fails). IMPORTANT: You should only call this tool if you're executing a task which REQUIRES installing dependencies. Keep in mind that not all tasks will require installing dependencies.
</file_and_code_management>
The code review has outlined the following actions to take:
<tool_usage_best_practices>
- Search: Use the \`search\` tool for all file searches. The \`search\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns.
- It's significantly faster results than alternatives like grep or ls -R.
- When searching for specific file types, use glob patterns
- The query field supports both basic strings, and regex
- Dependencies: Use the correct package manager; skip if installation fails
- Use the \`install_dependencies\` tool to install dependencies (skip if installation fails). IMPORTANT: You should only call this tool if you're executing a task which REQUIRES installing dependencies. Keep in mind that not all tasks will require installing dependencies.
- Pre-commit: Run \`pre-commit run --files ...\` if .pre-commit-config.yaml exists
- History: Use \`git log\` and \`git blame\` for additional context when needed
- Parallel Tool Calling: You're allowed, and encouraged to call multiple tools at once, as long as they do not conflict, or depend on each other.
- URL Content: Use the \`get_url_content\` tool to fetch the contents of a URL. You should only use this tool to fetch the contents of a URL the user has provided, or that you've discovered during your context searching, which you believe is vital to gathering context for the user's request.
- File Edits: Use the \`apply_patch\` tool to edit files. You should always read a file, and the specific parts of the file you want to edit before using the \`apply_patch\` tool to edit the file.
- This is important, as you never want to blindly edit a file before reading the part of the file you want to edit.
- Scripts may require dependencies to be installed: Remember that sometimes scripts may require dependencies to be installed before they can be run.
- Always ensure you've installed dependencies before running a script which might require them.
</tool_usage_best_practices>
## Actions to Take
{CODE_REVIEW_ACTIONS}
<coding_standards>
- When modifying files:
- Read files before modifying them
- Fix root causes, not symptoms
- Maintain existing code style
- Update documentation as needed
- Remove unnecessary inline comments after completion
- IMPORTANT: Always us the apply_patch tool to modify files. You should NEVER modify files any other way.
- Comments should only be included if a core maintainer of the codebase would not be able to understand the code without them (this means most of the time, you should not include comments)
- Never add copyright/license headers unless requested
- Ignore unrelated bugs or broken tests
- Write concise and clear code. Do not write overly verbose code
- Any tests written should always be executed after creating them to ensure they pass.
- If you've created a new test, ensure the plan has an explicit step to run this new test. If the plan does not include a step to run the tests, ensure you call the \`update_plan\` tool to add a step to run the tests.
- When running a test, ensure you include the proper flags/environment variables to exclude colors/text formatting. This can cause the output to be unreadable. For example, when running Jest tests you pass the \`--no-colors\` flag. In PyTest you set the \`NO_COLOR\` environment variable (prefix the command with \`export NO_COLOR=1\`)
- Only install trusted, well-maintained packages. If installing a new dependency which is not explicitly requested by the user, ensure it is a well-maintained, and widely used package.
- Ensure package manager files are updated to include the new dependency.
- If a command you run fails (e.g. a test, build, lint, etc.), and you make changes to fix the issue, ensure you always re-run the command after making the changes to ensure the fix was successful.
- IMPORTANT: You are NEVER allowed to create backup files. All changes in the codebase are tracked by git, so never create file copies, or backups.
</coding_standards>
<communication_guidelines>
- For coding tasks: Focus on implementation and provide brief summaries
</communication_guidelines>
<special_tools>
<name>request_human_help</name>
<description>Use only after exhausting all attempts to gather context</description>
<name>update_plan</name>
<description>Use this tool to add or remove tasks from the plan, or to update the plan in any other way</description>
</special_tools>
</instructions>
<custom_rules>
{CUSTOM_RULES}
</custom_rules>
`;
export const SYSTEM_PROMPT = `# Identity
export const DEPENDENCIES_INSTALLED_PROMPT = `Dependencies have already been installed.`;
export const DEPENDENCIES_NOT_INSTALLED_PROMPT = `Dependencies have not been installed.`;
You are a terminal-based agentic coding assistant built by LangChain. You wrap LLM models to enable natural language interaction with local codebases. You are precise, safe, and helpful.
export const CODE_REVIEW_PROMPT = `<code_review>
The code changes you've made have been reviewed by a code reviewer. The code review has determined that the changes do _not_ satisfy the user's request, and have outlined a list of additional actions to take in order to successfully complete the user's request.
You are currently executing a specific task from a pre-generated plan. You have access to:
- Project context and files
- Shell commands and code editing tools
- A sandboxed, git-backed workspace with rollback support
The code review has provided this review of the changes:
<review_feedback>
{CODE_REVIEW}
</review_feedback>
# Instructions
IMPORTANT: The code review has outlined the following actions to take:
<review_actions>
{CODE_REVIEW_ACTIONS}
</review_actions>
</code_review>`;
## Core Behavior
* **Persistence**: Keep working until the current task is completely resolved. Only terminate when you are certain the task is complete.
* **Accuracy**: Never guess or make up information. Always use tools to gather accurate data about files and codebase structure.
* **Planning**: Leverage the plan context and task summaries heavily - they contain critical information about completed work and the overall strategy.
## Task Execution Guidelines
### Working with the Plan
* You are executing task #{CURRENT_TASK_NUMBER} from the plan.
* Previous completed tasks and their summaries contain crucial context - always review them first
* Condensed context messages in conversation history summarize previous work - read these to avoid duplication
* The plan generation summary provides important codebase insights
* After some tasks are completed, you may be provided with a code review and additional tasks. Ensure you inspect the code review (if present) and new tasks to ensure the work you're doing satisfies the user's request.
### File and Code Management
* **Repository location**: {REPO_DIRECTORY}
* **Current directory**: {CURRENT_WORKING_DIRECTORY}
* All changes are auto-committed - no manual commits needed, and you should never create backup files.
* Work only within the existing Git repository
* Use \`apply_patch\` for file edits (accepts diffs and file paths)
* Use \`shell\` with \`touch\` to create new files (not \`apply_patch\`)
* Always use \`workdir\` parameter instead of \`cd\` when running commands via the \`shell\` tool
{INSTALL_DEPENDENCIES_TOOL_PROMPT}
### Tool Usage Best Practices
* **Search**: Use \`search\` tool for all file searches. The \`search\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns.
* It's significantly faster results than alternatives like grep or ls -R.
* When searching for specific file types, use glob patterns
* The query field supports both basic strings, and regex
* **Dependencies**: Use the correct package manager; skip if installation fails
* **Pre-commit**: Run \`pre-commit run --files ...\` if .pre-commit-config.yaml exists
* **History**: Use \`git log\` and \`git blame\` for additional context when needed
* **Parallel Tool Calling**: You're allowed, and encouraged to call multiple tools at once, as long as they do not conflict, or depend on each other.
* **URL Content**: Use the \`get_url_content\` tool to fetch the contents of a URL. You should only use this tool to fetch the contents of a URL the user has provided, or that you've discovered during your context searching, which you believe is vital to gathering context for the user's request.
* **File Edits**: Use the \`apply_patch\` tool to edit files. You should always read a file, and the specific parts of the file you want to edit before using the \`apply_patch\` tool to edit the file.
* This is important, as you never want to blindly edit a file before reading the part of the file you want to edit.
* **Scripts may require dependencies to be installed**: Remember that sometimes scripts may require dependencies to be installed before they can be run.
* Always ensure you've installed dependencies before running a script which might require them.
### Coding Standards
When modifying files:
* Read files before modifying them
* Fix root causes, not symptoms
* Maintain existing code style
* Update documentation as needed
* Remove unnecessary inline comments after completion
* Comments should only be included if a core maintainer of the codebase would not be able to understand the code without them
* Never add copyright/license headers unless requested
* Ignore unrelated bugs or broken tests
* Write concise and clear code. Do not write overly verbose code
* Any tests written should always be executed to ensure they pass.
* If you've created a new test, ensure the plan has an explicit step to run this new test. If the plan does not include a step to run the tests, ensure you call the \`update_plan\` tool to add a step to run the tests.
* When running a test, ensure you include the proper flags/environment variables to exclude colors/text formatting. This can cause the output to be unreadable. For example, when running Jest tests you pass the \`--no-colors\` flag. In PyTest you set the \`NO_COLOR\` environment variable (prefix the command with \`export NO_COLOR=1\`)
* Only install trusted, well-maintained packages. If installing a new dependency which is not explicitly requested by the user, ensure it is a well-maintained, and widely used package.
* Ensure package manager files are updated to include the new dependency.
* If a command you run fails (e.g. a test, build, lint, etc.), and you make changes to fix the issue, ensure you always re-run the command after making the changes to ensure the fix was successful.
### Communication Guidelines
* For coding tasks: Focus on implementation and provide brief summaries
## Special Tools
* **request_human_help**: Use only after exhausting all attempts to gather context
* **update_plan**: Use this tool to add or remove tasks from the plan, or to update the plan in any other way
# Context
export const DYNAMIC_SYSTEM_PROMPT = `<context>
<plan_information>
## Generated Plan with Summaries
- Current plan with summaries
{PLAN_PROMPT_WITH_SUMMARIES}
## Plan Generation Notes
- Plan generation notes
These are notes you took while gathering context for the plan:
{PLAN_GENERATION_NOTES}
## Current Task Statuses
{PLAN_PROMPT}
<plan-generation-notes>
{PLAN_GENERATION_NOTES}
</plan-generation-notes>
</plan_information>
<codebase_structure>
## Codebase Tree (3 levels deep, respecting .gitignore)
Generated via: \`git ls-files | tree --fromfile -L 3\`
Location: {REPO_DIRECTORY}
<repo_directory>{REPO_DIRECTORY}</repo_directory>
<are_dependencies_installed>{DEPENDENCIES_INSTALLED_PROMPT}</are_dependencies_installed>
{CODEBASE_TREE}
<codebase_tree>
Generated via: \`git ls-files | tree --fromfile -L 3\`
{CODEBASE_TREE}
</codebase_tree>
</codebase_structure>
{CODE_REVIEW_PROMPT}
{CUSTOM_RULES}`;
</context>
`;

View file

@ -28,6 +28,7 @@ import { getGitHubTokensFromConfig } from "../../../utils/github-tokens.js";
import { getActivePlanItems } from "@open-swe/shared/open-swe/tasks";
import { getRepoAbsolutePath } from "@open-swe/shared/git";
import { createOpenPrToolFields } from "@open-swe/shared/open-swe/tools";
import { trackCachePerformance } from "../../../utils/caching.js";
const logger = createLogger(LogLevel.INFO, "Open PR");
@ -180,5 +181,6 @@ export async function openPullRequest(
}),
...(codebaseTree && { codebaseTree }),
...(dependenciesInstalled !== null && { dependenciesInstalled }),
tokenData: trackCachePerformance(response),
};
}

View file

@ -36,6 +36,7 @@ import {
MAX_INTERNAL_TOKENS,
} from "../../../utils/tokens.js";
import { z } from "zod";
import { trackCachePerformance } from "../../../utils/caching.js";
const logger = createLogger(LogLevel.INFO, "ProgressPlanStep");
@ -149,6 +150,7 @@ Once you've determined the status of the current task, call either the \`mark_ta
const commandUpdate: GraphUpdate = {
messages: newMessages,
internalMessages: newMessages,
tokenData: trackCachePerformance(response),
};
// Check if we have any messages to summarize, and if we're at or above the max token limit.
@ -204,6 +206,7 @@ Once you've determined the status of the current task, call either the \`mark_ta
internalMessages: newMessages,
// Even though there are no remaining tasks, still mark as completed so the UI reflects that the task is completed.
taskPlan: updatedPlanTasks,
tokenData: trackCachePerformance(response),
};
return new Command({
goto: "route-to-review-or-conclusion",
@ -222,6 +225,7 @@ Once you've determined the status of the current task, call either the \`mark_ta
messages: newMessages,
internalMessages: newMessages,
taskPlan: updatedPlanTasks,
tokenData: trackCachePerformance(response),
};
if (totalInternalTokenCount >= MAX_INTERNAL_TOKENS) {

View file

@ -20,6 +20,7 @@ import { getActivePlanItems } from "@open-swe/shared/open-swe/tasks";
import { createConversationHistorySummaryToolFields } from "@open-swe/shared/open-swe/tools";
import { formatUserRequestPrompt } from "../../../utils/user-request.js";
import { getMessagesSinceLastSummary } from "../../../utils/tokens.js";
import { trackCachePerformance } from "../../../utils/caching.js";
const SINGLE_USER_REQUEST_PROMPT = `Here is the user's request:
<user_request>
@ -189,5 +190,6 @@ export async function summarizeHistory(
return {
messages: summaryMessages,
internalMessages: newInternalMessages,
tokenData: trackCachePerformance(response),
};
}

View file

@ -26,6 +26,7 @@ import { formatPlanPrompt } from "../../../utils/plan-prompt.js";
import { createLogger, LogLevel } from "../../../utils/logger.js";
import { createUpdatePlanToolFields } from "@open-swe/shared/open-swe/tools";
import { formatCustomRulesPrompt } from "../../../utils/custom-rules.js";
import { trackCachePerformance } from "../../../utils/caching.js";
const logger = createLogger(LogLevel.INFO, "UpdatePlanNode");
@ -203,5 +204,6 @@ export async function updatePlan(
messages: [toolMessage],
internalMessages: [toolMessage],
taskPlan: newTaskPlan,
tokenData: trackCachePerformance(response),
};
}

View file

@ -24,6 +24,7 @@ import { z } from "zod";
import { addTaskPlanToIssue } from "../../../utils/github/issue-task.js";
import { getMessageString } from "../../../utils/message/content.js";
import { ToolMessage } from "@langchain/core/messages";
import { trackCachePerformance } from "../../../utils/caching.js";
const SYSTEM_PROMPT = `You are a code reviewer for a software engineer working on a large codebase.
@ -172,5 +173,6 @@ export async function finalReview(
messages: messagesUpdate,
internalMessages: messagesUpdate,
reviewsCount: (state.reviewsCount || 0) + 1,
tokenData: trackCachePerformance(response),
};
}

View file

@ -27,13 +27,17 @@ import {
} from "../../../../utils/review.js";
import { BaseMessage } from "@langchain/core/messages";
import { getMessageString } from "../../../../utils/message/content.js";
import {
CacheablePromptSegment,
convertMessagesToCacheControlledMessages,
trackCachePerformance,
} from "../../../../utils/caching.js";
const logger = createLogger(LogLevel.INFO, "GenerateReviewActionsNode");
function formatSystemPrompt(state: ReviewerGraphState): string {
const activePlan = getActivePlanItems(state.taskPlan);
const tasksString = formatPlanPromptWithSummaries(activePlan);
const codeReview = getCodeReviewFields(state.internalMessages);
return SYSTEM_PROMPT.replaceAll(
"{CODEBASE_TREE}",
@ -54,25 +58,52 @@ function formatSystemPrompt(state: ReviewerGraphState): string {
.replaceAll(
"{USER_REQUEST_PROMPT}",
formatUserRequestPrompt(state.messages),
)
.replaceAll(
"{PREVIOUS_REVIEW_PROMPT}",
codeReview
? formatCodeReviewPrompt(PREVIOUS_REVIEW_PROMPT, {
review: codeReview.review,
newActions: codeReview.newActions,
})
: "",
);
}
function formatUserConversationHistoryMessage(messages: BaseMessage[]): string {
return `Here is the full conversation history of the programmer. This includes all of the actions taken by the programmer, as well as any user input.
const formatCacheablePrompt = (
state: ReviewerGraphState,
): CacheablePromptSegment[] => {
const codeReview = getCodeReviewFields(state.internalMessages);
const segments: CacheablePromptSegment[] = [
{
type: "text",
text: formatSystemPrompt(state),
cache_control: { type: "ephemeral" },
},
];
// Cache Breakpoint 4: Code Review Context (only add if present)
if (codeReview) {
segments.push({
type: "text",
text: formatCodeReviewPrompt(PREVIOUS_REVIEW_PROMPT, {
review: codeReview.review,
newActions: codeReview.newActions,
}),
cache_control: { type: "ephemeral" },
});
}
return segments.filter((segment) => segment.text.trim() !== "");
};
function formatUserConversationHistoryMessage(
messages: BaseMessage[],
): CacheablePromptSegment[] {
return [
{
type: "text",
text: `Here is the full conversation history of the programmer. This includes all of the actions taken by the programmer, as well as any user input.
If the history has been truncated, it is because the conversation was too long. In this case, you should only consider the most recent messages.
<conversation_history>
${messages.map(getMessageString).join("\n")}
</conversation_history>`;
</conversation_history>`,
cache_control: { type: "ephemeral" },
},
];
}
export async function generateReviewActions(
@ -98,16 +129,19 @@ export async function generateReviewActions(
: {}),
});
const reviewerMessagesWithCache = convertMessagesToCacheControlledMessages(
state.reviewerMessages,
);
const response = await modelWithTools.invoke([
{
role: "system",
content: formatSystemPrompt(state),
content: formatCacheablePrompt(state),
},
{
role: "user",
content: formatUserConversationHistoryMessage(state.internalMessages),
},
...state.reviewerMessages,
...reviewerMessagesWithCache,
]);
logger.info("Generated review actions", {
@ -123,5 +157,6 @@ export async function generateReviewActions(
return {
messages: [response],
reviewerMessages: [response],
tokenData: trackCachePerformance(response),
};
}

View file

@ -14,7 +14,9 @@ Given this review and the actions you requested be completed to successfully com
You do not need to provide an extensive review of the entire codebase. You should focus your new review on the actions you outlined above to take, and the changes since the previous review.
</previous_review>`;
export const SYSTEM_PROMPT = `You are a terminal-based agentic coding assistant built by LangChain that enables natural language interaction with local codebases. You excel at being precise, safe, and helpful in your analysis.
export const SYSTEM_PROMPT = `<identity>
You are a terminal-based agentic coding assistant built by LangChain that enables natural language interaction with local codebases. You excel at being precise, safe, and helpful in your analysis.
</identity>
<role>
Reviewer Assistant - Read-Only Phase
@ -26,86 +28,74 @@ By reviewing these actions, and comparing them to the plan and original user req
</primary_objective>
<reviewing_guidelines>
1. **Use only read operations**: Execute commands that inspect and analyze the codebase without modifying any files. This ensures we understand the current state before making changes.
2. **Make high-quality, targeted tool calls**: Each command should have a clear purpose in reviewing the actions taken by the Programmer Assistant.
3. **Use git commands to gather context**: Below you're provided with a section '<changed_files>', which lists all of the files that were modified/created/deleted in the current branch.
- Ensure you use this, paired with commands such as 'git diff {BASE_BRANCH_NAME} <file_path>' to inspect a diff of a file to gather context about the changes made by the Programmer Assistant.
3. **Only search for what is necessary**: Ensure you gather all of the context necessary to provide a review of the changes made by the Programmer Assistant.
- Ensure that the actions you perform in this review phase are only the most necessary and targeted actions to gather context.
- Avoid rabbit holes for gathering context. You should always first consider whether or not the action you're about to take is necessary to generate a review for the user's request. If it is not, do not take it.
4. **Leverage \`search\` tool**: Use \`search\` tool for all file searches. The \`search\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns.
- It's significantly faster results than alternatives like grep or ls -R.
- When searching for specific file types, use glob patterns
- The query field supports both basic strings, and regex
5. **Format shell commands precisely**: Ensure all shell commands include proper quoting and escaping. Well-formatted commands prevent errors and provide reliable results.
6. **Only take necessary actions**: You should only take actions which are absolutely necessary to provide a quality review of ONLY the changes in the current branch & the user's request.
- Think about whether or not the request you're reviewing is a simple one, which would warrant less review actions to take, or a more complex request, which would require a more detailed review.
7. **Parallel tool calling**: It is highly recommended that you use parallel tool calling to gather context as quickly and efficiently as possible.
- When you know ahead of time there are multiple commands you want to run to gather context, of which they are independent and can be run in parallel, you should use parallel tool calling.
8. **Always use the correct package manager**: If taking an action which requires a package manager (e.g. npm/yarn or pip/poetry, etc.), ensure you always search for the package manager used by the codebase, and use that one.
- Using a package manager that is different from the one used by the codebase may result in unexpected behavior, or errors.
9. **Prefer using pre-made scripts**: If taking an action like running tests, formatting, linting, etc., always prefer using pre-made scripts over running commands manually.
- If you want to run a command like this, but are unsure if a pre-made script exists, always search for it first.
10. **Signal completion clearly**: When you have gathered sufficient context, respond with exactly 'done' without any tool calls. This indicates readiness to proceed to the final review phase.
1. Use only read operations: Execute commands that inspect and analyze the codebase without modifying any files. This ensures we understand the current state before making changes.
2. Make high-quality, targeted tool calls: Each command should have a clear purpose in reviewing the actions taken by the Programmer Assistant.
3. Use git commands to gather context: Below you're provided with a section '<changed_files>', which lists all of the files that were modified/created/deleted in the current branch.
- Ensure you use this, paired with commands such as 'git diff {BASE_BRANCH_NAME} <file_path>' to inspect a diff of a file to gather context about the changes made by the Programmer Assistant.
4. Only search for what is necessary: Ensure you gather all of the context necessary to provide a review of the changes made by the Programmer Assistant.
- Ensure that the actions you perform in this review phase are only the most necessary and targeted actions to gather context.
- Avoid rabbit holes for gathering context. You should always first consider whether or not the action you're about to take is necessary to generate a review for the user's request. If it is not, do not take it.
5. Leverage \`search\` tool: Use \`search\` tool for all file searches. The \`search\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns.
- It's significantly faster results than alternatives like grep or ls -R.
- When searching for specific file types, use glob patterns
- The query field supports both basic strings, and regex
6. Format shell commands precisely: Ensure all shell commands include proper quoting and escaping. Well-formatted commands prevent errors and provide reliable results.
7. Only take necessary actions: You should only take actions which are absolutely necessary to provide a quality review of ONLY the changes in the current branch & the user's request.
- Think about whether or not the request you're reviewing is a simple one, which would warrant less review actions to take, or a more complex request, which would require a more detailed review.
8. Parallel tool calling: It is highly recommended that you use parallel tool calling to gather context as quickly and efficiently as possible.
- When you know ahead of time there are multiple commands you want to run to gather context, of which they are independent and can be run in parallel, you should use parallel tool calling.
9. Always use the correct package manager: If taking an action which requires a package manager (e.g. npm/yarn or pip/poetry, etc.), ensure you always search for the package manager used by the codebase, and use that one.
- Using a package manager that is different from the one used by the codebase may result in unexpected behavior, or errors.
10. Prefer using pre-made scripts: If taking an action like running tests, formatting, linting, etc., always prefer using pre-made scripts over running commands manually.
- If you want to run a command like this, but are unsure if a pre-made script exists, always search for it first.
11. Signal completion clearly: When you have gathered sufficient context, respond with exactly 'done' without any tool calls. This indicates readiness to proceed to the final review phase.
</reviewing_guidelines>
<instructions>
You should inspect each of the files modified by the programmer (see the <changed_files> section below), and confirm they properly implement the plan (see the <completed_tasks_and_summaries> section below), and that the user's request has been fully implemented.
You should be reviewing them from the perspective of a quality assurance engineer, ensuring the code written is of the highest quality, fully implements the user's request, and all actions have been taken for the PR to be accepted.
You should inspect each of the files modified by the programmer (see the <changed_files> section below), and confirm they properly implement the plan (see the <completed_tasks_and_summaries> section below), and that the user's request has been fully implemented.
You should be reviewing them from the perspective of a quality assurance engineer, ensuring the code written is of the highest quality, fully implements the user's request, and all actions have been taken for the PR to be accepted.
You're also provided with the conversation history of the actions the programmer has taken, and any user input they've received. The first user message below contains this information.
Ensure you carefully read over all of these messages to ensure you have the proper context and do not duplicate actions the programmer has already taken.
You're also provided with the conversation history of the actions the programmer has taken, and any user input they've received. The first user message below contains this information.
Ensure you carefully read over all of these messages to ensure you have the proper context and do not duplicate actions the programmer has already taken.
Common tasks you should always confirm were executed:
- Linter/formatter scripts were executed
- Unit tests were executed
- If no tests for the code written/updated exists, confirm whether or not tests should be written
- Documentation was updated, if applicable
Common tasks you should always confirm were executed:
- Linter/formatter scripts were executed
- Unit tests were executed
- If no tests for the code written/updated exists, confirm whether or not tests should be written
- Documentation was updated, if applicable
**IMPORTANT**:
Keep in mind that not all requests/changes will need tests to be written, or documentation to be added/updated. Ensure you consider whether or not the standard engineering organization would write tests, or documentation for the changes you're reviewing.
After considering this, you may not need to check if tests should be written, or documentation should be added/updated.
**IMPORTANT**:
Keep in mind that not all requests/changes will need tests to be written, or documentation to be added/updated. Ensure you consider whether or not the standard engineering organization would write tests, or documentation for the changes you're reviewing.
After considering this, you may not need to check if tests should be written, or documentation should be added/updated.
Based on the generated plan, the actions taken and files changed, you should review the modified code and determine if it properly completes the overall task, or if more changes need to be made/existing changes should be modified.
On top of inspecting the changed files, you should also look to see if the programmer missed anything, made changes which do not respect the custom rules, or if the changes are otherwise insufficient to complete the task.
Based on the generated plan, the actions taken and files changed, you should review the modified code and determine if it properly completes the overall task, or if more changes need to be made/existing changes should be modified.
On top of inspecting the changed files, you should also look to see if the programmer missed anything, made changes which do not respect the custom rules, or if the changes are otherwise insufficient to complete the task.
You do not want to do more work than required, but you always should complete tasks which you believe are necessary to complete the user's request, and merge the pull request without further action.
You do not want to do more work than required, but you always should complete tasks which you believe are necessary to complete the user's request, and merge the pull request without further action.
After you're satisfied with the context you've gathered, and are ready to provide a final review, respond with exactly 'done' without any tool calls.
This will redirect you to a final review step where you'll submit your final review, and optionally provide a list of additional actions to take.
After you're satisfied with the context you've gathered, and are ready to provide a final review, respond with exactly 'done' without any tool calls.
This will redirect you to a final review step where you'll submit your final review, and optionally provide a list of additional actions to take.
**REMINDER**:
You are ONLY gathering context. Any non-read actions you believe are necessary to take can be executed after you've provided your final review.
Only gather context right now in order to inform your final review, and to provide any additional steps to take after the review.
**REMINDER**:
You are ONLY gathering context. Any non-read actions you believe are necessary to take can be executed after you've provided your final review.
Only gather context right now in order to inform your final review, and to provide any additional steps to take after the review.
</instructions>
<workspace_information>
**Current Working Directory**: {CURRENT_WORKING_DIRECTORY}
**Repository Status**: Already cloned and accessible in the current directory
**Base Branch Name**: {BASE_BRANCH_NAME}
**Dependencies Installed**: {DEPENDENCIES_INSTALLED}
<current_working_directory>{CURRENT_WORKING_DIRECTORY}</current_working_directory>
<repository_status>Already cloned and accessible in the current directory</repository_status>
<base_branch_name>{BASE_BRANCH_NAME}</base_branch_name>
<dependencies_installed>{DEPENDENCIES_INSTALLED}</dependencies_installed>
**Codebase Structure** (3 levels deep, respecting .gitignore):
Generated via: \`git ls-files | tree --fromfile -L 3\`
<codebase_tree>
{CODEBASE_TREE}
</codebase_tree>
<codebase_tree>
Generated via: \`git ls-files | tree --fromfile -L 3\`:
{CODEBASE_TREE}
</codebase_tree>
**Changed Files**:
Generated via: \`git diff {BASE_BRANCH_NAME} --name-only\`
<changed_files>
{CHANGED_FILES}
</changed_files>
<changed_files>
Generated via: \`git diff {BASE_BRANCH_NAME} --name-only\`:
{CHANGED_FILES}
</changed_files>
</workspace_information>
{CUSTOM_RULES}
@ -114,8 +104,6 @@ Generated via: \`git diff {BASE_BRANCH_NAME} --name-only\`
{COMPLETED_TASKS_AND_SUMMARIES}
</completed_tasks_and_summaries>
{PREVIOUS_REVIEW_PROMPT}
<task_context>
{USER_REQUEST_PROMPT}
</task_context>`;

View file

@ -7,7 +7,7 @@ import {
import { createDiagnoseErrorToolFields } from "@open-swe/shared/open-swe/tools";
import { z } from "zod";
import { GraphConfig } from "@open-swe/shared/open-swe/types";
import { CacheMetrics, GraphConfig } from "@open-swe/shared/open-swe/types";
import { createLogger, LogLevel } from "../../utils/logger.js";
import { getAllLastFailedActions } from "../../utils/tool-message-error.js";
import { getMessageString } from "../../utils/message/content.js";
@ -16,6 +16,7 @@ import {
supportsParallelToolCallsParam,
Task,
} from "../../utils/load-model.js";
import { trackCachePerformance } from "../../utils/caching.js";
const logger = createLogger(LogLevel.INFO, "SharedDiagnoseError");
@ -75,6 +76,7 @@ const formatUserPrompt = (messages: BaseMessage[]): string => {
interface DiagnoseErrorInputs {
messages: BaseMessage[];
codebaseTree: string;
tokenData?: CacheMetrics;
}
type DiagnoseErrorUpdate = Partial<DiagnoseErrorInputs>;
@ -140,5 +142,6 @@ export async function diagnoseError(
return {
messages: [response, toolMessage],
tokenData: trackCachePerformance(response),
};
}

View file

@ -7,9 +7,7 @@ import { createLogger, LogLevel } from "../utils/logger.js";
import { createApplyPatchToolFields } from "@open-swe/shared/open-swe/tools";
import { getRepoAbsolutePath } from "@open-swe/shared/git";
import { getSandboxSessionOrThrow } from "./utils/get-sandbox-id.js";
import * as fs from "fs/promises";
import * as path from "path";
import * as os from "os";
import { Sandbox } from "@daytonaio/sdk";
const logger = createLogger(LogLevel.INFO, "ApplyPatchTool");
@ -21,18 +19,27 @@ const logger = createLogger(LogLevel.INFO, "ApplyPatchTool");
* @returns Object with success status and output or error message
*/
async function applyPatchWithGit(
sandbox: any,
sandbox: Sandbox,
workDir: string,
diffContent: string,
): Promise<{ success: boolean; output: string }> {
let tempDir = "";
try {
// Create a temporary file to store the diff
tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "git-patch-"));
const tempPatchFile = path.join(tempDir, "patch.diff");
const tempPatchFile = `/tmp/patch_${Date.now()}_${Math.random().toString(36).substring(2)}.diff`;
// Write the diff to the temporary file
await fs.writeFile(tempPatchFile, diffContent, "utf8");
try {
// Create the patch file in the sandbox
const createFileResponse = await sandbox.process.executeCommand(
`cat > "${tempPatchFile}" << 'EOF'\n${diffContent}\nEOF`,
workDir,
{},
10, // 10 seconds timeout for file creation
);
if (createFileResponse.exitCode !== 0) {
return {
success: false,
output: `Failed to create patch file: ${createFileResponse.result || "Unknown error"}`,
};
}
// Execute git apply with --verbose for detailed error messages
const response = await sandbox.process.executeCommand(
@ -61,20 +68,6 @@ async function applyPatchWithGit(
? error.message
: "Unknown error applying patch with git",
};
} finally {
// Clean up the temporary file after git apply has completed
if (tempDir) {
try {
await fs.rm(tempDir, { recursive: true, force: true });
} catch (cleanupError) {
logger.warn(`Failed to clean up temporary directory: ${tempDir}`, {
error:
cleanupError instanceof Error
? cleanupError.message
: String(cleanupError),
});
}
}
}
}

View file

@ -0,0 +1,115 @@
import {
AIMessage,
AIMessageChunk,
BaseMessage,
HumanMessage,
isAIMessage,
isHumanMessage,
isToolMessage,
MessageContent,
ToolMessage,
} from "@langchain/core/messages";
import { CacheMetrics } from "@open-swe/shared/open-swe/types";
import { createLogger, LogLevel } from "./logger.js";
import { calculateCostSavings } from "@open-swe/shared/caching";
const logger = createLogger(LogLevel.INFO, "Caching");
export interface CacheablePromptSegment {
type: "text";
text: string;
cache_control?: { type: "ephemeral" };
}
export function trackCachePerformance(response: AIMessageChunk): CacheMetrics {
const metrics: CacheMetrics = {
cacheCreationInputTokens:
response.usage_metadata?.input_token_details?.cache_creation || 0,
cacheReadInputTokens:
response.usage_metadata?.input_token_details?.cache_read || 0,
inputTokens: response.usage_metadata?.input_tokens || 0,
outputTokens: response.usage_metadata?.output_tokens || 0,
};
const totalInputTokens =
metrics.cacheCreationInputTokens +
metrics.cacheReadInputTokens +
metrics.inputTokens;
const cacheHitRate =
totalInputTokens > 0 ? metrics.cacheReadInputTokens / totalInputTokens : 0;
const costSavings = calculateCostSavings(metrics).totalSavings;
logger.info("Cache Performance", {
cacheHitRate: `${(cacheHitRate * 100).toFixed(2)}%`,
costSavings: `$${costSavings.toFixed(4)}`,
...metrics,
});
return metrics;
}
function addCacheControlToMessageContent(
messageContent: MessageContent,
): MessageContent {
if (typeof messageContent === "string") {
return [
{
type: "text",
text: messageContent,
cache_control: { type: "ephemeral" },
},
];
} else if (Array.isArray(messageContent)) {
if ("cache_control" in messageContent[messageContent.length - 1]) {
// Already set, no-op
return messageContent;
}
const newMessageContent = [...messageContent];
newMessageContent[newMessageContent.length - 1] = {
...newMessageContent[newMessageContent.length - 1],
cache_control: { type: "ephemeral" },
};
return newMessageContent;
} else {
logger.warn("Unknown message content type", { messageContent });
return messageContent;
}
}
function convertToCacheControlMessage(message: BaseMessage): BaseMessage {
if (isAIMessage(message)) {
return new AIMessage({
...message,
content: addCacheControlToMessageContent(message.content),
});
} else if (isHumanMessage(message)) {
return new HumanMessage({
...message,
content: addCacheControlToMessageContent(message.content),
});
} else if (isToolMessage(message)) {
return new ToolMessage({
...(message as ToolMessage),
content: addCacheControlToMessageContent(
(message as ToolMessage).content,
),
});
} else {
return message;
}
}
export function convertMessagesToCacheControlledMessages(
messages: BaseMessage[],
) {
if (messages.length === 0) {
return messages;
}
const newMessages = [...messages];
const lastIndex = newMessages.length - 1;
newMessages[lastIndex] = convertToCacheControlMessage(newMessages[lastIndex]);
return newMessages;
}

View file

@ -203,11 +203,12 @@ async function performClone(
await sandbox.git.clone(
cloneUrl,
absoluteRepoDir,
undefined,
targetRepository.branch,
targetRepository.baseCommit,
"git",
githubInstallationToken,
);
logger.info("Successfully cloned repository", {
repoPath: `${targetRepository.owner}/${targetRepository.repo}`,
branch: branchName,

View file

@ -33,6 +33,7 @@ import { Interrupt } from "../thread/messages/interrupt";
import { AlertCircle } from "lucide-react";
import { ErrorState } from "./types";
import { CollapsibleAlert } from "./collapsible-alert";
import { TokenUsage } from "./token-usage";
interface AcceptedPlanEventData {
planTitle: string;
@ -338,6 +339,7 @@ export function ActionsRenderer<State extends PlannerGraphState | GraphState>({
icon={<AlertCircle className="size-4" />}
/>
) : null}
<TokenUsage tokenData={stream.values.tokenData} />
</div>
);
}

View file

@ -0,0 +1,54 @@
import { CacheMetrics } from "@open-swe/shared/open-swe/types";
import { calculateCostSavings } from "@open-swe/shared/caching";
import {
Tooltip,
TooltipContent,
TooltipProvider,
TooltipTrigger,
} from "../ui/tooltip";
import { ChartNoAxesColumnIncreasing } from "lucide-react";
export function TokenUsage({ tokenData }: { tokenData?: CacheMetrics }) {
if (!tokenData) return null;
const metrics = calculateCostSavings(tokenData);
return (
<div className="mt-4 ml-auto flex">
<TooltipProvider>
<Tooltip>
<TooltipTrigger>
<ChartNoAxesColumnIncreasing />
</TooltipTrigger>
<TooltipContent className="flex w-full flex-col gap-1 text-sm">
<p>Token usage data on actions where caching is enabled:</p>
<span className="flex w-full items-center justify-between">
<p>Input Tokens:</p>
<p>{metrics.totalInputTokens.toLocaleString()}</p>
</span>
<span className="flex w-full items-center justify-between">
<p>Output Tokens:</p>
<p>{metrics.totalOutputTokens.toLocaleString()}</p>
</span>
<span className="flex w-full items-center justify-between">
<p>Total Tokens:</p>
<p>{metrics.totalTokens.toLocaleString()}</p>
</span>
<span className="flex w-full items-center justify-between">
<p>Output Tokens Cost:</p>
<p>${metrics.totalOutputTokensCost.toFixed(2)}</p>
</span>
<span className="flex w-full items-center justify-between">
<p>Cache Savings:</p>
<p>${metrics.totalSavings.toFixed(2)}</p>
</span>
<span className="flex w-full items-center justify-between">
<p>Total Cost:</p>
<p>${metrics.totalCost.toFixed(2)}</p>
</span>
</TooltipContent>
</Tooltip>
</TooltipProvider>
</div>
);
}

View file

@ -0,0 +1,65 @@
import { CacheMetrics } from "./open-swe/types.js";
export function calculateCostSavings(metrics: CacheMetrics): {
totalSavings: number;
totalCost: number;
totalTokens: number;
totalInputTokens: number;
totalOutputTokens: number;
totalOutputTokensCost: number;
} {
const SONNET_4_BASE_RATE = 3.0 / 1_000_000; // $3 per MTok
const SONNET_4_OUTPUT_RATE = 15.0 / 1_000_000; // $15 per MTok
const CACHE_WRITE_MULTIPLIER = 1.25;
const CACHE_READ_MULTIPLIER = 0.1;
const cacheWriteCost =
metrics.cacheCreationInputTokens *
SONNET_4_BASE_RATE *
CACHE_WRITE_MULTIPLIER;
const cacheReadCost =
metrics.cacheReadInputTokens * SONNET_4_BASE_RATE * CACHE_READ_MULTIPLIER;
const regularInputCost = metrics.inputTokens * SONNET_4_BASE_RATE;
const totalOutputTokensCost = metrics.outputTokens * SONNET_4_OUTPUT_RATE;
// Cost without caching (all tokens at base rate)
const totalInputTokens =
metrics.cacheCreationInputTokens +
metrics.cacheReadInputTokens +
metrics.inputTokens;
const totalTokens = totalInputTokens + metrics.outputTokens;
const costWithoutCaching = totalInputTokens * SONNET_4_BASE_RATE;
// Actual cost with caching
const actualCost = cacheWriteCost + cacheReadCost + regularInputCost;
return {
totalSavings: costWithoutCaching - actualCost,
totalCost: actualCost,
totalTokens,
totalInputTokens,
totalOutputTokens: metrics.outputTokens,
totalOutputTokensCost,
};
}
export function tokenDataReducer(
state: CacheMetrics | undefined,
update: CacheMetrics,
): CacheMetrics {
if (!state) {
return update;
}
return {
cacheCreationInputTokens:
state.cacheCreationInputTokens + update.cacheCreationInputTokens,
cacheReadInputTokens:
state.cacheReadInputTokens + update.cacheReadInputTokens,
inputTokens: state.inputTokens + update.inputTokens,
outputTokens: state.outputTokens + update.outputTokens,
};
}

View file

@ -3,11 +3,13 @@ import { z } from "zod";
import { MessagesZodState } from "@langchain/langgraph";
import {
AgentSession,
CacheMetrics,
CustomRules,
TargetRepository,
TaskPlan,
} from "../types.js";
import { withLangGraph } from "@langchain/langgraph/zod";
import { tokenDataReducer } from "../../caching.js";
export const PlannerGraphStateObj = MessagesZodState.extend({
sandboxSessionId: withLangGraph(z.string(), {
@ -91,6 +93,12 @@ export const PlannerGraphStateObj = MessagesZodState.extend({
fn: (_state, update) => update,
},
}),
tokenData: withLangGraph(z.custom<CacheMetrics>().optional(), {
reducer: {
schema: z.custom<CacheMetrics>().optional(),
fn: tokenDataReducer,
},
}),
});
export type PlannerGraphState = z.infer<typeof PlannerGraphStateObj>;

View file

@ -5,9 +5,15 @@ import {
messagesStateReducer,
MessagesZodState,
} from "@langchain/langgraph";
import { CustomRules, TargetRepository, TaskPlan } from "../types.js";
import {
CacheMetrics,
CustomRules,
TargetRepository,
TaskPlan,
} from "../types.js";
import { withLangGraph } from "@langchain/langgraph/zod";
import { BaseMessage } from "@langchain/core/messages";
import { tokenDataReducer } from "../../caching.js";
export const ReviewerGraphStateObj = MessagesZodState.extend({
/**
@ -110,6 +116,12 @@ export const ReviewerGraphStateObj = MessagesZodState.extend({
},
default: () => 0,
}),
tokenData: withLangGraph(z.custom<CacheMetrics>().optional(), {
reducer: {
schema: z.custom<CacheMetrics>().optional(),
fn: tokenDataReducer,
},
}),
});
export type ReviewerGraphState = z.infer<typeof ReviewerGraphStateObj>;

View file

@ -24,6 +24,14 @@ import {
} from "../constants.js";
import { withLangGraph } from "@langchain/langgraph/zod";
import { BaseMessage } from "@langchain/core/messages";
import { tokenDataReducer } from "../caching.js";
export interface CacheMetrics {
cacheCreationInputTokens: number;
cacheReadInputTokens: number;
inputTokens: number;
outputTokens: number;
}
export type PlanItem = {
/**
@ -251,6 +259,13 @@ export const GraphAnnotation = MessagesZodState.extend({
default: () => 0,
}),
tokenData: withLangGraph(z.custom<CacheMetrics>().optional(), {
reducer: {
schema: z.custom<CacheMetrics>().optional(),
fn: tokenDataReducer,
},
}),
// ---NOT USED---
ui: z
.custom<UIMessage[]>()