mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 12:43:16 +00:00
feat: New algo for when and how to summarize conversation history (#396)
* feat: New algo for when and how to summarize conversation history * remove summarize task steps node * cr * cr * cr * cr * feat: add convo summary component to ui * cr
This commit is contained in:
parent
d769983483
commit
72a129f691
15 changed files with 549 additions and 303 deletions
|
|
@ -20,9 +20,11 @@ Your sole objective in this phase is to gather comprehensive context about the c
|
|||
- To ensure the above does not happen, you should be thorough in your context gathering. Always gather enough context to cover all edge cases, and prevent unclear instructions.
|
||||
|
||||
4. **Leverage efficient search tools**:
|
||||
- Use \`rg\` (ripgrep) for all file searches because it respects .gitignore patterns and provides significantly faster results than alternatives like grep or ls -R.
|
||||
- Use the \`find_instances_of\` tool when searching for specific keywords/strings in files. This tool utilizes \`rg\` (ripgrep) under the hood, but it is optimized for searching for specific keywords/strings in files.
|
||||
- Use the \`rg\` (ripgrep) tool when performing more complex searches (e.g. regex searches).
|
||||
- When searching for specific file types, use glob patterns: \`rg -i pattern -g **/*.tsx project-directory/\`
|
||||
- This explicit pattern matching ensures accurate results across all file extensions
|
||||
- Always use \`rg\` or \`find_instances_of\` tools instead calling \`grep\` via the \`shell\` tool. You should NEVER call \`grep\` as the same functionality is better provided by \`rg\` or \`find_instances_of\`.
|
||||
- If the user passes a URL, you should use the \`get_url_content\` tool to fetch the contents of the URL.
|
||||
- You should only use this tool to fetch the contents of a URL the user has provided, or that you've discovered during your context searching, which you believe is vital to gathering context for the user's request.
|
||||
|
||||
|
|
@ -31,6 +33,11 @@ Your sole objective in this phase is to gather comprehensive context about the c
|
|||
6. **Signal completion clearly**: When you have gathered sufficient context, respond with exactly 'done' without any tool calls. This indicates readiness to proceed to the planning phase.
|
||||
|
||||
7. **Parallel tool calling**: It is highly recommended that you use parallel tool calling to gather context as quickly and efficiently as possible. When you know ahead of time there are multiple commands you want to run to gather context, of which they are independent and can be run in parallel, you should use parallel tool calling.
|
||||
- This is best utilized by search commands. You should always plan ahead for which search commands you want to run in parallel, then use parallel tool calling to run them all at once for maximum efficiency.
|
||||
|
||||
8. **Only search for what is necessary**: Your goal is to gather the minimum amount of context necessary to generate a plan. You should not gather context or perform searches that are not necessary to generate a plan.
|
||||
- You will always be able to gather more context after the planning phase, so ensure that the actions you perform in this planning phase are only the most necessary and targeted actions to gather context.
|
||||
- Avoid rabbit holes for gathering context. You should always first consider whether or not the action you're about to take is necessary to generate a plan for the user's request. If it is not, do not take it.
|
||||
</context_gathering_guidelines>
|
||||
|
||||
<workspace_information>
|
||||
|
|
|
|||
|
|
@ -147,6 +147,7 @@ export async function interruptProposedPlan(
|
|||
branchName: state.branchName,
|
||||
targetRepository: state.targetRepository,
|
||||
githubIssueId: state.githubIssueId,
|
||||
internalMessages: state.messages,
|
||||
};
|
||||
|
||||
if (state.autoAcceptPlan) {
|
||||
|
|
|
|||
|
|
@ -8,15 +8,17 @@ import {
|
|||
generateAction,
|
||||
takeAction,
|
||||
progressPlanStep,
|
||||
summarizeTaskSteps,
|
||||
generateConclusion,
|
||||
openPullRequest,
|
||||
diagnoseError,
|
||||
requestHelp,
|
||||
updatePlan,
|
||||
summarizeHistory,
|
||||
} from "./nodes/index.js";
|
||||
import { isAIMessage } from "@langchain/core/messages";
|
||||
import { initializeSandbox } from "../shared/initialize-sandbox.js";
|
||||
import { getRemainingPlanItems } from "../../utils/current-task.js";
|
||||
import { getActivePlanItems } from "@open-swe/shared/open-swe/tasks";
|
||||
|
||||
/**
|
||||
* Routes to the next appropriate node after taking action.
|
||||
|
|
@ -24,11 +26,17 @@ import { initializeSandbox } from "../shared/initialize-sandbox.js";
|
|||
* Otherwise, it ends the process.
|
||||
*
|
||||
* @param {GraphState} state - The current graph state.
|
||||
* @returns {"open-pr" | "take-action" | "request-help" | Send} The next node to execute, or END if the process should stop.
|
||||
* @returns {"generate-conclusion" | "take-action" | "request-help" | "generate-action" | Send} The next node to execute, or END if the process should stop.
|
||||
*/
|
||||
async function routeGeneratedAction(
|
||||
state: GraphState,
|
||||
): Promise<"open-pr" | "take-action" | "request-help" | Send> {
|
||||
): Promise<
|
||||
| "generate-conclusion"
|
||||
| "take-action"
|
||||
| "request-help"
|
||||
| "generate-action"
|
||||
| Send
|
||||
> {
|
||||
const { internalMessages } = state;
|
||||
const lastMessage = internalMessages[internalMessages.length - 1];
|
||||
|
||||
|
|
@ -53,8 +61,15 @@ async function routeGeneratedAction(
|
|||
return "take-action";
|
||||
}
|
||||
|
||||
// No tool calls, create PR then end.
|
||||
return "open-pr";
|
||||
const activePlanItems = getActivePlanItems(state.taskPlan);
|
||||
const hasRemainingTasks = getRemainingPlanItems(activePlanItems).length > 0;
|
||||
// If the model did not generate a tool call, but there are remaining tasks, we should route back to the generate action step.
|
||||
if (hasRemainingTasks) {
|
||||
return "generate-action";
|
||||
}
|
||||
|
||||
// No tool calls, generate a conclusion.
|
||||
return "generate-conclusion";
|
||||
}
|
||||
|
||||
const workflow = new StateGraph(GraphAnnotation, GraphConfiguration)
|
||||
|
|
@ -65,10 +80,7 @@ const workflow = new StateGraph(GraphAnnotation, GraphConfiguration)
|
|||
})
|
||||
.addNode("update-plan", updatePlan)
|
||||
.addNode("progress-plan-step", progressPlanStep, {
|
||||
ends: ["summarize-task-steps", "generate-action", "generate-conclusion"],
|
||||
})
|
||||
.addNode("summarize-task-steps", summarizeTaskSteps, {
|
||||
ends: ["generate-action", "generate-conclusion"],
|
||||
ends: ["summarize-history", "generate-action", "generate-conclusion"],
|
||||
})
|
||||
.addNode("generate-conclusion", generateConclusion)
|
||||
.addNode("request-help", requestHelp, {
|
||||
|
|
@ -76,17 +88,20 @@ const workflow = new StateGraph(GraphAnnotation, GraphConfiguration)
|
|||
})
|
||||
.addNode("open-pr", openPullRequest)
|
||||
.addNode("diagnose-error", diagnoseError)
|
||||
.addNode("summarize-history", summarizeHistory)
|
||||
.addEdge(START, "initialize")
|
||||
.addEdge("initialize", "generate-action")
|
||||
.addConditionalEdges("generate-action", routeGeneratedAction, [
|
||||
"take-action",
|
||||
"request-help",
|
||||
"open-pr",
|
||||
"generate-conclusion",
|
||||
"update-plan",
|
||||
"generate-action",
|
||||
])
|
||||
.addEdge("update-plan", "generate-action")
|
||||
.addEdge("generate-conclusion", "open-pr")
|
||||
.addEdge("diagnose-error", "generate-action")
|
||||
.addEdge("summarize-history", "generate-action")
|
||||
.addEdge("open-pr", END);
|
||||
|
||||
// Zod types are messed up
|
||||
|
|
|
|||
|
|
@ -22,10 +22,10 @@ You are currently executing a specific task from a pre-generated plan. You have
|
|||
|
||||
### Working with the Plan
|
||||
|
||||
* You are executing task #{CURRENT_TASK_NUMBER} from the following plan:
|
||||
- Previous completed tasks and their summaries contain crucial context - always review them first
|
||||
- Condensed context messages in conversation history summarize previous work - read these to avoid duplication
|
||||
- The plan generation summary provides important codebase insights
|
||||
* You are executing task #{CURRENT_TASK_NUMBER} from the plan.
|
||||
* Previous completed tasks and their summaries contain crucial context - always review them first
|
||||
* Condensed context messages in conversation history summarize previous work - read these to avoid duplication
|
||||
* The plan generation summary provides important codebase insights
|
||||
|
||||
### File and Code Management
|
||||
|
||||
|
|
@ -40,12 +40,18 @@ You are currently executing a specific task from a pre-generated plan. You have
|
|||
|
||||
### Tool Usage Best Practices
|
||||
|
||||
* **Search**: Use the \`rg\` tool (ripgrep) (not grep/ls -R) with glob patterns (e.g., \`rg -i pattern -g **/*.tsx\`)
|
||||
* **Search**: Use the \`find_instances_of\` tool when searching for specific keywords/strings in files. When performing more complex searches (e.g. regex searches), use the \`rg\` tool (ripgrep) (not grep/ls -R) with glob patterns (e.g., \`rg -i pattern -g **/*.tsx\`).
|
||||
* Both \`find_instances_of\` and \`rg\` are optimized for searching files, and should be used in place of \`grep\` or \`ls -R\`.
|
||||
* **Dependencies**: Use the correct package manager; skip if installation fails
|
||||
* **Pre-commit**: Run \`pre-commit run --files ...\` if .pre-commit-config.yaml exists
|
||||
* **History**: Use \`git log\` and \`git blame\` for additional context when needed
|
||||
* **Parallel Tool Calling**: You're allowed, and encouraged to call multiple tools at once, as long as they do not conflict, or depend on each other.
|
||||
* **URL Content**: Use the \`get_url_content\` tool to fetch the contents of a URL. You should only use this tool to fetch the contents of a URL the user has provided, or that you've discovered during your context searching, which you believe is vital to gathering context for the user's request.
|
||||
* **File Edits**: Use the \`apply_patch\` tool to edit files. You should always read a file, and the specific parts of the file you want to edit before using the \`apply_patch\` tool to edit the file.
|
||||
* This is important, as you never want to blindly edit a file before reading the part of the file you want to edit.
|
||||
* **Scripts may require dependencies to be installed**: Remember that sometimes scripts may require dependencies to be installed before they can be run.
|
||||
* Always ensure you've installed dependencies before running a script which might require them.
|
||||
|
||||
|
||||
### Coding Standards
|
||||
|
||||
|
|
@ -55,9 +61,15 @@ When modifying files:
|
|||
* Maintain existing code style
|
||||
* Update documentation as needed
|
||||
* Remove unnecessary inline comments after completion
|
||||
* Comments should only be included if a core maintainer of the codebase would not be able to understand the code without them
|
||||
* Never add copyright/license headers unless requested
|
||||
* Ignore unrelated bugs or broken tests
|
||||
* Write concise and clear code. Do not write overly verbose code.
|
||||
* Write concise and clear code. Do not write overly verbose code
|
||||
* Any tests written should always be executed to ensure they pass.
|
||||
* If you've created a new test, ensure the plan has an explicit step to run this new test. If the plan does not include a step to run the tests, ensure you call the \`update_plan\` tool to add a step to run the tests.
|
||||
* When running a test, ensure you include the proper flags/environment variables to exclude colors/text formatting. This can cause the output to be unreadable. For example, when running Jest tests you pass the \`--no-colors\` flag. In PyTest you set the \`NO_COLOR\` environment variable (prefix the command with \`export NO_COLOR=1\`)
|
||||
* Only install trusted, well-maintained packages. If installing a new dependency which is not explicitly requested by the user, ensure it is a well-maintained, and widely used package.
|
||||
* Ensure package manager files are updated to include the new dependency.
|
||||
|
||||
### Communication Guidelines
|
||||
|
||||
|
|
@ -66,7 +78,7 @@ When modifying files:
|
|||
## Special Tools
|
||||
|
||||
* **request_human_help**: Use only after exhausting all attempts to gather context
|
||||
* **update_plan**: Use for major plan changes (adding/removing tasks)
|
||||
* **update_plan**: Use this tool to add or remove tasks from the plan, or to update the plan in any other way
|
||||
|
||||
# Context
|
||||
|
||||
|
|
@ -78,7 +90,7 @@ When modifying files:
|
|||
These are notes you took while gathering context for the plan:
|
||||
{PLAN_GENERATION_NOTES}
|
||||
|
||||
## Current Task Status
|
||||
## Current Task Statuses
|
||||
{PLAN_PROMPT}
|
||||
</plan_information>
|
||||
|
||||
|
|
|
|||
|
|
@ -1,9 +1,9 @@
|
|||
export * from "./generate-message/index.js";
|
||||
export * from "./take-action.js";
|
||||
export * from "./progress-plan-step.js";
|
||||
export * from "./summarize-task-steps.js";
|
||||
export * from "./generate-conclusion.js";
|
||||
export * from "./open-pr.js";
|
||||
export * from "./diagnose-error.js";
|
||||
export * from "./request-help.js";
|
||||
export * from "./update-plan.js";
|
||||
export * from "./summarize-history.js";
|
||||
|
|
|
|||
|
|
@ -1,4 +1,3 @@
|
|||
import { z } from "zod";
|
||||
import { createLogger, LogLevel } from "../../../utils/logger.js";
|
||||
import {
|
||||
GraphConfig,
|
||||
|
|
@ -23,7 +22,15 @@ import {
|
|||
} from "../../../utils/current-task.js";
|
||||
import { ToolMessage } from "@langchain/core/messages";
|
||||
import { addTaskPlanToIssue } from "../../../utils/github/issue-task.js";
|
||||
import { createSetTaskStatusToolFields } from "@open-swe/shared/open-swe/tools";
|
||||
import {
|
||||
createMarkTaskNotCompletedToolFields,
|
||||
createMarkTaskCompletedToolFields,
|
||||
} from "@open-swe/shared/open-swe/tools";
|
||||
import {
|
||||
calculateConversationHistoryTokenCount,
|
||||
MAX_INTERNAL_TOKENS,
|
||||
} from "../../../utils/tokens.js";
|
||||
import { z } from "zod";
|
||||
|
||||
const logger = createLogger(LogLevel.INFO, "ProgressPlanStep");
|
||||
|
||||
|
|
@ -37,8 +44,10 @@ Here is the plan, along with the summaries of each completed task:
|
|||
Analyze the tasks you've completed, the tasks which are remaining, and the current task you just took an action on.
|
||||
In addition to this, you're also provided the full conversation history between you and the user. All of the messages in this conversation are from the previous steps/actions you've taken, and any user input.
|
||||
|
||||
Take all of this information, and determine whether or not you have completed this task in the plan.
|
||||
Once you've determined the status of the current task, call the \`set_task_status\` tool.
|
||||
Take all of this information, and determine if the current task is complete, or if you still have work left to do.
|
||||
Once you've determined the status of the current task, call either:
|
||||
- \`mark_task_completed\` if the task is complete.
|
||||
- \`mark_task_not_completed\` if the task is not complete.
|
||||
`;
|
||||
|
||||
const formatPrompt = (taskPlan: PlanItem[]): string => {
|
||||
|
|
@ -52,12 +61,16 @@ export async function progressPlanStep(
|
|||
state: GraphState,
|
||||
config: GraphConfig,
|
||||
): Promise<Command> {
|
||||
const setTaskStatusTool = createSetTaskStatusToolFields();
|
||||
const markNotCompletedTool = createMarkTaskNotCompletedToolFields();
|
||||
const markCompletedTool = createMarkTaskCompletedToolFields();
|
||||
const model = await loadModel(config, Task.PROGRESS_PLAN_CHECKER);
|
||||
const modelWithTools = model.bindTools([setTaskStatusTool], {
|
||||
tool_choice: setTaskStatusTool.name,
|
||||
parallel_tool_calls: false,
|
||||
});
|
||||
const modelWithTools = model.bindTools(
|
||||
[markNotCompletedTool, markCompletedTool],
|
||||
{
|
||||
tool_choice: "any",
|
||||
parallel_tool_calls: false,
|
||||
},
|
||||
);
|
||||
|
||||
const userRequest = getUserRequest(state.internalMessages, {
|
||||
returnFullMessage: true,
|
||||
|
|
@ -67,7 +80,7 @@ export async function progressPlanStep(
|
|||
${removeFirstHumanMessage(state.internalMessages).map(getMessageString).join("\n")}
|
||||
|
||||
Take all of this information, and determine whether or not you have completed this task in the plan.
|
||||
Once you've determined the status of the current task, call the \`set_task_status\` tool.`;
|
||||
Once you've determined the status of the current task, call either the \`mark_task_completed\` or \`mark_task_not_completed\` tool.`;
|
||||
|
||||
const activePlanItems = getActivePlanItems(state.taskPlan);
|
||||
|
||||
|
|
@ -90,42 +103,57 @@ Once you've determined the status of the current task, call the \`set_task_statu
|
|||
);
|
||||
}
|
||||
|
||||
const isCompleted =
|
||||
(toolCall.args as z.infer<typeof setTaskStatusTool.schema>).task_status ===
|
||||
"completed";
|
||||
const isCompleted = toolCall.name === markCompletedTool.name;
|
||||
const currentTask = getCurrentPlanItem(activePlanItems);
|
||||
const toolMessage = new ToolMessage({
|
||||
tool_call_id: toolCall.id ?? "",
|
||||
content: `Saved task status as ${
|
||||
toolCall.args.task_status
|
||||
} for task ${currentTask?.plan || "unknown"}`,
|
||||
content: `Saved task status as ${isCompleted ? "completed" : "not completed"} for task ${currentTask?.plan || "unknown"}`,
|
||||
name: toolCall.name,
|
||||
});
|
||||
|
||||
const newMessages = [response, toolMessage];
|
||||
|
||||
const totalInternalTokenCount = calculateConversationHistoryTokenCount(
|
||||
state.internalMessages,
|
||||
);
|
||||
|
||||
if (!isCompleted) {
|
||||
logger.info(
|
||||
"Current task has not been completed. Progressing to the next action.",
|
||||
{
|
||||
reasoning: toolCall.args.reasoning,
|
||||
},
|
||||
);
|
||||
logger.info("Current task has not been completed.", {
|
||||
reasoning: toolCall.args.reasoning,
|
||||
});
|
||||
const commandUpdate: GraphUpdate = {
|
||||
messages: newMessages,
|
||||
internalMessages: newMessages,
|
||||
};
|
||||
|
||||
if (totalInternalTokenCount >= MAX_INTERNAL_TOKENS) {
|
||||
logger.info(
|
||||
"Internal messages list is at or above the max token limit. Routing to summarize history step.",
|
||||
{
|
||||
totalInternalTokenCount,
|
||||
maxInternalTokenCount: MAX_INTERNAL_TOKENS,
|
||||
},
|
||||
);
|
||||
return new Command({
|
||||
goto: "summarize-history",
|
||||
update: commandUpdate,
|
||||
});
|
||||
}
|
||||
|
||||
return new Command({
|
||||
goto: "generate-action",
|
||||
update: commandUpdate,
|
||||
});
|
||||
}
|
||||
const summary = (toolCall.args as z.infer<typeof markCompletedTool.schema>)
|
||||
.completed_task_summary;
|
||||
|
||||
// LLM marked as completed, so we need to update the plan to reflect that.
|
||||
const updatedPlanTasks = completePlanItem(
|
||||
state.taskPlan,
|
||||
getActiveTask(state.taskPlan).id,
|
||||
currentTask.index,
|
||||
summary,
|
||||
);
|
||||
// Update the github issue to reflect this task as completed.
|
||||
await addTaskPlanToIssue(
|
||||
|
|
@ -168,8 +196,22 @@ Once you've determined the status of the current task, call the \`set_task_statu
|
|||
taskPlan: updatedPlanTasks,
|
||||
};
|
||||
|
||||
if (totalInternalTokenCount >= MAX_INTERNAL_TOKENS) {
|
||||
logger.info(
|
||||
"Internal messages list is at or above the max token limit. Routing to summarize history step.",
|
||||
{
|
||||
totalInternalTokenCount,
|
||||
maxInternalTokenCount: MAX_INTERNAL_TOKENS,
|
||||
},
|
||||
);
|
||||
return new Command({
|
||||
goto: "summarize-history",
|
||||
update: commandUpdate,
|
||||
});
|
||||
}
|
||||
|
||||
return new Command({
|
||||
goto: "summarize-task-steps",
|
||||
goto: "generate-action",
|
||||
update: commandUpdate,
|
||||
});
|
||||
}
|
||||
|
|
|
|||
217
apps/open-swe/src/graphs/programmer/nodes/summarize-history.ts
Normal file
217
apps/open-swe/src/graphs/programmer/nodes/summarize-history.ts
Normal file
|
|
@ -0,0 +1,217 @@
|
|||
import { v4 as uuidv4 } from "uuid";
|
||||
import {
|
||||
GraphConfig,
|
||||
GraphState,
|
||||
GraphUpdate,
|
||||
PlanItem,
|
||||
} from "@open-swe/shared/open-swe/types";
|
||||
import { loadModel, Task } from "../../../utils/load-model.js";
|
||||
import {
|
||||
AIMessage,
|
||||
BaseMessage,
|
||||
RemoveMessage,
|
||||
ToolMessage,
|
||||
} from "@langchain/core/messages";
|
||||
import { formatPlanPrompt } from "../../../utils/plan-prompt.js";
|
||||
import { createLogger, LogLevel } from "../../../utils/logger.js";
|
||||
import { getMessageContentString } from "@open-swe/shared/messages";
|
||||
import { getMessageString } from "../../../utils/message/content.js";
|
||||
import { getActivePlanItems } from "@open-swe/shared/open-swe/tasks";
|
||||
import { getCompletedPlanItems } from "../../../utils/current-task.js";
|
||||
import { createConversationHistorySummaryToolFields } from "@open-swe/shared/open-swe/tools";
|
||||
import { z } from "zod";
|
||||
import { DO_NOT_RENDER_ID_PREFIX } from "@open-swe/shared/constants";
|
||||
import { getUserRequest } from "../../../utils/user-request.js";
|
||||
import { getMessagesSinceLastSummary } from "../../../utils/tokens.js";
|
||||
|
||||
const taskSummarySysPrompt = `You are operating as a terminal-based agentic coding assistant built by LangChain. It wraps LLM models to enable natural language interaction with a local codebase. You are expected to be precise, safe, and helpful.
|
||||
|
||||
<role>
|
||||
Context Extraction Assistant
|
||||
</role>
|
||||
|
||||
<primary_objective>
|
||||
Your sole objective in this task is to extract the highest quality/most relevant context from the conversation history below.
|
||||
</primary_objective>
|
||||
|
||||
<objective_information>
|
||||
You're nearing the total number of input tokens you can accept, so you must extract the highest quality/most relevant pieces of information from your conversation history.
|
||||
This context will then overwrite the conversation history presented below. Because of this, ensure the context you extract is only the most important information to your overall goal.
|
||||
To aid with this, you'll be provided with the user's request, as well as all of the tasks in the plan you generated to fulfil the user's request. Additionally, if a task has already been completed you'll be provided with the summary of the steps taken to complete it.
|
||||
</objective_information>
|
||||
|
||||
Here is the user's request:
|
||||
<user_request>
|
||||
{USER_REQUEST}
|
||||
</user_request>
|
||||
|
||||
Here is the full list of tasks in the plan you're in the middle of, as well as the summary of the completed tasks:
|
||||
<tasks_and_summaries>
|
||||
{PLAN_PROMPT}
|
||||
</tasks_and_summaries>
|
||||
|
||||
<instructions>
|
||||
The conversation history below will be replaced with the context you extract in this step. Because of this, you must do your very best to extract and record all of the most important context from the conversation history.
|
||||
You want to ensure that you don't repeat any actions you've already completed (e.g. file search operations, checking codebase information, etc.), so the context you extract from the conversation history should be focused on the most important information to your overall goal.
|
||||
|
||||
You MUST adhere to the following criteria when extracting the most important context from the conversation history:
|
||||
- Include full file paths for all relevant files to the users request & tasks.
|
||||
- Include file summaries/snippets from the relevant files. Avoid including entire files as you're trying to condense the conversation history.
|
||||
- Include insights, and learnings you've discovered about the codebase or specific files while completing the task.
|
||||
- Only record information once, and avoid duplications. Duplicate information or actions in the conversation history should be merged into a single entry.
|
||||
</instructions>
|
||||
|
||||
Here is the full conversation history you'll be extracting context from, to then replace. Carefully read over it all, and think deeply about what information is most important to your overall goal that should be saved:
|
||||
<conversation_history>
|
||||
{CONVERSATION_HISTORY}
|
||||
</conversation_history>
|
||||
|
||||
With all of this in mind, please carefully read over the entire conversation history, and extract the most important and relevant context to replace it so that you can free up space in the conversation history.
|
||||
Respond ONLY with the extracted context. Do not include any additional information, or text before or after the extracted context.
|
||||
`;
|
||||
|
||||
const logger = createLogger(LogLevel.INFO, "SummarizeConversationHistory");
|
||||
|
||||
const formatPrompt = (inputs: {
|
||||
userRequest: string;
|
||||
plan: PlanItem[];
|
||||
conversationHistoryToSummarize: BaseMessage[];
|
||||
}): string => {
|
||||
return taskSummarySysPrompt
|
||||
.replace(
|
||||
"{PLAN_PROMPT}",
|
||||
formatPlanPrompt(inputs.plan, {
|
||||
useLastCompletedTask: true,
|
||||
includeSummaries: true,
|
||||
}),
|
||||
)
|
||||
.replace("{USER_REQUEST}", inputs.userRequest)
|
||||
.replace(
|
||||
"{CONVERSATION_HISTORY}",
|
||||
inputs.conversationHistoryToSummarize.map(getMessageString).join("\n"),
|
||||
);
|
||||
};
|
||||
|
||||
/**
|
||||
* Create an AI & tool message pair for the generated task summary.
|
||||
* This is not included in the internal message state, but is exposed to
|
||||
* users so they can see the summary of the actions that were taken.
|
||||
*/
|
||||
function createUserFacingConversationSummaryMessages(
|
||||
summary: string,
|
||||
): BaseMessage[] {
|
||||
const conversationSummaryTool = createConversationHistorySummaryToolFields();
|
||||
const conversationSummaryToolCallArgs: z.infer<
|
||||
typeof conversationSummaryTool.schema
|
||||
> = {
|
||||
conversation_history_summary: summary,
|
||||
};
|
||||
const conversationSummaryToolCallId = uuidv4();
|
||||
const conversationSummaryPublicMessages = [
|
||||
new AIMessage({
|
||||
id: uuidv4(),
|
||||
content: "",
|
||||
tool_calls: [
|
||||
{
|
||||
id: conversationSummaryToolCallId,
|
||||
name: conversationSummaryTool.name,
|
||||
args: conversationSummaryToolCallArgs,
|
||||
},
|
||||
],
|
||||
}),
|
||||
new ToolMessage({
|
||||
id: `${DO_NOT_RENDER_ID_PREFIX}${uuidv4()}`,
|
||||
tool_call_id: conversationSummaryToolCallId,
|
||||
content: "",
|
||||
}),
|
||||
];
|
||||
|
||||
return conversationSummaryPublicMessages;
|
||||
}
|
||||
|
||||
function createInternalSummaryMessages(summary: string): BaseMessage[] {
|
||||
const dummySummarizeHistoryToolName = "summarize_conversation_history";
|
||||
const dummySummarizeHistoryToolCallId = uuidv4();
|
||||
return [
|
||||
new AIMessage({
|
||||
id: uuidv4(),
|
||||
content:
|
||||
"Looks like I'm running out of tokens. I'm going to summarize the conversation history to free up space.",
|
||||
tool_calls: [
|
||||
{
|
||||
id: dummySummarizeHistoryToolCallId,
|
||||
name: dummySummarizeHistoryToolName,
|
||||
args: {
|
||||
reasoning:
|
||||
"I'm running out of tokens. I'm going to summarize all of the messages since my last summary message to free up space.",
|
||||
},
|
||||
},
|
||||
],
|
||||
additional_kwargs: {
|
||||
summary_message: true,
|
||||
},
|
||||
}),
|
||||
new ToolMessage({
|
||||
id: uuidv4(),
|
||||
tool_call_id: dummySummarizeHistoryToolCallId,
|
||||
content: summary,
|
||||
additional_kwargs: {
|
||||
summary_message: true,
|
||||
},
|
||||
}),
|
||||
];
|
||||
}
|
||||
|
||||
export async function summarizeHistory(
|
||||
state: GraphState,
|
||||
config: GraphConfig,
|
||||
): Promise<GraphUpdate> {
|
||||
const activePlanItems = getActivePlanItems(state.taskPlan);
|
||||
const lastCompletedTask = getCompletedPlanItems(activePlanItems).pop();
|
||||
if (!lastCompletedTask) {
|
||||
throw new Error("Unable to find last completed task.");
|
||||
}
|
||||
|
||||
const model = await loadModel(config, Task.SUMMARIZER);
|
||||
|
||||
const userRequest = getUserRequest(state.messages);
|
||||
const plan = getActivePlanItems(state.taskPlan);
|
||||
const conversationHistoryToSummarize = getMessagesSinceLastSummary(
|
||||
state.internalMessages,
|
||||
);
|
||||
|
||||
logger.info(
|
||||
`Summarizing ${conversationHistoryToSummarize.length} messages in the conversation history...`,
|
||||
);
|
||||
|
||||
const response = await model.invoke([
|
||||
{
|
||||
role: "user",
|
||||
content: formatPrompt({
|
||||
userRequest,
|
||||
plan,
|
||||
conversationHistoryToSummarize,
|
||||
}),
|
||||
},
|
||||
]);
|
||||
|
||||
const summaryString = getMessageContentString(response.content);
|
||||
const taskSummaryMessages =
|
||||
createUserFacingConversationSummaryMessages(summaryString);
|
||||
|
||||
const newInternalMessages = [
|
||||
...conversationHistoryToSummarize.map(
|
||||
(m) => new RemoveMessage({ id: m.id ?? "" }),
|
||||
),
|
||||
...createInternalSummaryMessages(summaryString),
|
||||
];
|
||||
|
||||
logger.info(
|
||||
`Summarized ${conversationHistoryToSummarize.length} messages in the conversation history. Removing and replacing with a summary message.`,
|
||||
);
|
||||
|
||||
return {
|
||||
messages: taskSummaryMessages,
|
||||
internalMessages: newInternalMessages,
|
||||
};
|
||||
}
|
||||
|
|
@ -1,182 +0,0 @@
|
|||
import { v4 as uuidv4 } from "uuid";
|
||||
import {
|
||||
GraphConfig,
|
||||
GraphState,
|
||||
GraphUpdate,
|
||||
PlanItem,
|
||||
} from "@open-swe/shared/open-swe/types";
|
||||
import { loadModel, Task } from "../../../utils/load-model.js";
|
||||
import { AIMessage, BaseMessage } from "@langchain/core/messages";
|
||||
import { formatPlanPrompt } from "../../../utils/plan-prompt.js";
|
||||
import { createLogger, LogLevel } from "../../../utils/logger.js";
|
||||
import { getMessageContentString } from "@open-swe/shared/messages";
|
||||
import { getMessageString } from "../../../utils/message/content.js";
|
||||
import { removeLastTaskMessages } from "../../../utils/message/modify-array.js";
|
||||
import { Command } from "@langchain/langgraph";
|
||||
import { ConfigurableModel } from "langchain/chat_models/universal";
|
||||
import {
|
||||
completePlanItem,
|
||||
getActivePlanItems,
|
||||
getActiveTask,
|
||||
} from "@open-swe/shared/open-swe/tasks";
|
||||
import { getCompletedPlanItems } from "../../../utils/current-task.js";
|
||||
import { addTaskPlanToIssue } from "../../../utils/github/issue-task.js";
|
||||
|
||||
const taskSummarySysPrompt = `You are operating as a terminal-based agentic coding assistant built by LangChain. It wraps LLM models to enable natural language interaction with a local codebase. You are expected to be precise, safe, and helpful.
|
||||
|
||||
Your current task is to look at the conversation history, and generate a concise summary of the steps which were taken to complete the task.
|
||||
|
||||
Here are all of your tasks you've completed, remaining, and the current task you're working on. The completed tasks will include summaries of the steps taken to complete them:
|
||||
{PLAN_PROMPT}
|
||||
|
||||
You MUST adhere to the following criteria when summarizing the conversation history:
|
||||
- Include insights, and learnings you've discovered about the codebase or specific files while completing the task.
|
||||
- You should NOT document scripts, file structure, or other context which could be categorized as 'general codebase context'. General codebase context is automatically included via the \`tree\` command.
|
||||
- If files were created or modified, include short summaries of the changes made.
|
||||
- What file(s) were modified/created.
|
||||
- What content was added/removed.
|
||||
- If you had to make a change which required you to undo previous changes, include that information.
|
||||
- Do not include the actual changes you made, but rather high level bullet points containing context and descriptions on the modifications made.
|
||||
- Do not retain any full code snippets.
|
||||
- Do not retain any full file contents.
|
||||
- Ensure you have an understanding of the context and summaries you've already generated (provided by the user below) and do not repeat any information you've already included.
|
||||
- Do not duplicate ANY information. Ensure you carefully read and understand the task summaries generated above, and do not repeat any information you've already included.
|
||||
- You do not need to include specific codebase context here, as codebase context will be generated in a separate step. Your sole task is to generate a concise summary of this specific task you just completed.
|
||||
- Ensure your summary is as concise as possible, but useful for future context.
|
||||
|
||||
Ensure you do NOT include codebase context in your task summary, as we want to avoid including duplicate information.
|
||||
|
||||
With all of this in mind, please carefully summarize and condense the conversation history of the task you just completed, provided by the user below. Remember that this summary should ONLY include details about the completed task, and should NOT include any general codebase context.
|
||||
Respond ONLY with the task summary. Do not include any additional information, or text before or after the task summary.
|
||||
`;
|
||||
|
||||
const userContextMessage = `Here is the task you just completed:
|
||||
{COMPLETED_TASK}
|
||||
|
||||
The first message in the conversation history is the user's request. Messages from previously completed tasks have already been removed, in favor of task summaries.
|
||||
With this in mind, please use the following conversation history to generate a concise summary of the task you just completed.
|
||||
|
||||
Conversation history:
|
||||
{CONVERSATION_HISTORY}`;
|
||||
|
||||
const logger = createLogger(LogLevel.INFO, "SummarizeTaskSteps");
|
||||
|
||||
const formatPrompt = (plan: PlanItem[]): string =>
|
||||
taskSummarySysPrompt.replace(
|
||||
"{PLAN_PROMPT}",
|
||||
formatPlanPrompt(plan, {
|
||||
useLastCompletedTask: true,
|
||||
includeSummaries: true,
|
||||
}),
|
||||
);
|
||||
|
||||
const formatUserMessage = (
|
||||
messages: BaseMessage[],
|
||||
plans: PlanItem[],
|
||||
): string => {
|
||||
const completedTask = plans.find((p) => p.completed);
|
||||
if (!completedTask) {
|
||||
throw new Error(
|
||||
"No completed task found when trying to format user message for task summary.",
|
||||
);
|
||||
}
|
||||
|
||||
return userContextMessage
|
||||
.replace("{COMPLETED_TASK}", completedTask.plan)
|
||||
.replace(
|
||||
"{CONVERSATION_HISTORY}",
|
||||
messages.map(getMessageString).join("\n"),
|
||||
);
|
||||
};
|
||||
|
||||
async function generateTaskSummary(
|
||||
state: GraphState,
|
||||
model: ConfigurableModel,
|
||||
): Promise<{ planItemIndex: number; summary: string }> {
|
||||
const activePlanItems = getActivePlanItems(state.taskPlan);
|
||||
const lastCompletedTask = getCompletedPlanItems(activePlanItems).pop();
|
||||
if (!lastCompletedTask) {
|
||||
throw new Error("Unable to find last completed task.");
|
||||
}
|
||||
|
||||
logger.info(`Summarizing task steps...`);
|
||||
const response = await model
|
||||
.withConfig({ tags: ["nostream"], runName: "generate-task-summary" })
|
||||
.invoke([
|
||||
{
|
||||
role: "system",
|
||||
content: formatPrompt(activePlanItems),
|
||||
},
|
||||
{
|
||||
role: "user",
|
||||
content: formatUserMessage(state.internalMessages, activePlanItems),
|
||||
},
|
||||
]);
|
||||
|
||||
return {
|
||||
planItemIndex: lastCompletedTask.index,
|
||||
summary: getMessageContentString(response.content),
|
||||
};
|
||||
}
|
||||
|
||||
export async function summarizeTaskSteps(
|
||||
state: GraphState,
|
||||
config: GraphConfig,
|
||||
): Promise<Command> {
|
||||
const activePlanItems = getActivePlanItems(state.taskPlan);
|
||||
const lastCompletedTask = getCompletedPlanItems(activePlanItems).pop();
|
||||
if (!lastCompletedTask) {
|
||||
throw new Error("Unable to find last completed task.");
|
||||
}
|
||||
|
||||
const model = await loadModel(config, Task.SUMMARIZER);
|
||||
const taskSummary = await generateTaskSummary(state, model);
|
||||
const updatedTaskPlan = completePlanItem(
|
||||
state.taskPlan,
|
||||
getActiveTask(state.taskPlan).id,
|
||||
taskSummary.planItemIndex,
|
||||
taskSummary.summary,
|
||||
);
|
||||
// Update the github issue to include the new task summary.
|
||||
await addTaskPlanToIssue(
|
||||
{
|
||||
githubIssueId: state.githubIssueId,
|
||||
targetRepository: state.targetRepository,
|
||||
},
|
||||
config,
|
||||
updatedTaskPlan,
|
||||
);
|
||||
|
||||
const removedMessages = removeLastTaskMessages(state.internalMessages);
|
||||
logger.info(`Removing ${removedMessages.length} message(s) from state.`);
|
||||
|
||||
const condensedTaskMessage = new AIMessage({
|
||||
id: uuidv4(),
|
||||
content: `Successfully condensed task context for task: "${lastCompletedTask.plan}". This task's summary can be found in the system prompt.`,
|
||||
additional_kwargs: {
|
||||
summary_message: true,
|
||||
},
|
||||
});
|
||||
const newMessagesStateUpdate = [...removedMessages, condensedTaskMessage];
|
||||
|
||||
const allTasksCompleted = activePlanItems.every((p) => p.completed);
|
||||
if (allTasksCompleted) {
|
||||
const commandUpdate: GraphUpdate = {
|
||||
internalMessages: newMessagesStateUpdate,
|
||||
taskPlan: updatedTaskPlan,
|
||||
};
|
||||
return new Command({
|
||||
goto: "generate-conclusion",
|
||||
update: commandUpdate,
|
||||
});
|
||||
}
|
||||
|
||||
const commandUpdate: GraphUpdate = {
|
||||
internalMessages: newMessagesStateUpdate,
|
||||
taskPlan: updatedTaskPlan,
|
||||
};
|
||||
return new Command({
|
||||
goto: "generate-action",
|
||||
update: commandUpdate,
|
||||
});
|
||||
}
|
||||
|
|
@ -1,25 +1,4 @@
|
|||
import {
|
||||
BaseMessage,
|
||||
isAIMessage,
|
||||
isHumanMessage,
|
||||
isToolMessage,
|
||||
RemoveMessage,
|
||||
} from "@langchain/core/messages";
|
||||
|
||||
export function removeLastTaskMessages(messages: BaseMessage[]): BaseMessage[] {
|
||||
return messages
|
||||
.filter((m) => {
|
||||
if (
|
||||
m.additional_kwargs?.summary_message ||
|
||||
(!isAIMessage(m) && !isToolMessage(m)) ||
|
||||
!m.id
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
})
|
||||
.map((m) => new RemoveMessage({ id: m.id ?? "" }));
|
||||
}
|
||||
import { BaseMessage, isHumanMessage } from "@langchain/core/messages";
|
||||
|
||||
export function removeFirstHumanMessage(
|
||||
messages: BaseMessage[],
|
||||
|
|
|
|||
|
|
@ -6,6 +6,9 @@ import {
|
|||
} from "@langchain/core/messages";
|
||||
import { getMessageContentString } from "@open-swe/shared/messages";
|
||||
|
||||
// After 100k tokens, summarize the conversation history.
|
||||
export const MAX_INTERNAL_TOKENS = 100_000;
|
||||
|
||||
export function calculateConversationHistoryTokenCount(
|
||||
messages: BaseMessage[],
|
||||
) {
|
||||
|
|
@ -28,3 +31,12 @@ export function calculateConversationHistoryTokenCount(
|
|||
// Estimate 1 token for every 4 characters.
|
||||
return Math.ceil(totalChars / 4);
|
||||
}
|
||||
|
||||
export function getMessagesSinceLastSummary(
|
||||
messages: BaseMessage[],
|
||||
): BaseMessage[] {
|
||||
const allMessagesAfterLastSummary = messages.slice(
|
||||
messages.findIndex((m) => m.additional_kwargs?.summary_message),
|
||||
);
|
||||
return allMessagesAfterLastSummary;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -3,19 +3,24 @@
|
|||
import { JSX, useState } from "react";
|
||||
import {
|
||||
Terminal,
|
||||
FileCode,
|
||||
Loader2,
|
||||
CheckCircle,
|
||||
XCircle,
|
||||
FileText,
|
||||
ChevronDown,
|
||||
ChevronRight,
|
||||
ChevronUp,
|
||||
MessageSquare,
|
||||
FileText,
|
||||
CloudDownload,
|
||||
Search,
|
||||
AlertCircle,
|
||||
CheckCircle,
|
||||
XCircle,
|
||||
Loader2,
|
||||
Globe,
|
||||
Pencil,
|
||||
Package,
|
||||
FileCode,
|
||||
CloudDownload,
|
||||
Hash,
|
||||
} from "lucide-react";
|
||||
import { MarkdownText } from "../thread/markdown-text";
|
||||
import {
|
||||
createApplyPatchToolFields,
|
||||
createShellToolFields,
|
||||
|
|
@ -34,6 +39,7 @@ import {
|
|||
TooltipTrigger,
|
||||
} from "../ui/tooltip";
|
||||
import { cn } from "@/lib/utils";
|
||||
import { ToolIconWithTooltip } from "./tool-icon-tooltip";
|
||||
|
||||
// Used only for Zod type inference.
|
||||
const dummyRepo = { owner: "dummy", repo: "dummy" };
|
||||
|
|
@ -135,23 +141,6 @@ const ACTION_GENERATING_TEXT_MAP = {
|
|||
[findInstancesOfTool.name]: "Finding instances...",
|
||||
};
|
||||
|
||||
function ToolIconWithTooltip({
|
||||
toolNamePretty,
|
||||
icon,
|
||||
}: {
|
||||
toolNamePretty: string;
|
||||
icon: JSX.Element;
|
||||
}) {
|
||||
return (
|
||||
<TooltipProvider>
|
||||
<Tooltip>
|
||||
<TooltipTrigger asChild>{icon}</TooltipTrigger>
|
||||
<TooltipContent>{toolNamePretty}</TooltipContent>
|
||||
</Tooltip>
|
||||
</TooltipProvider>
|
||||
);
|
||||
}
|
||||
|
||||
function MatchCaseIcon({ matchCase }: { matchCase: boolean }) {
|
||||
return (
|
||||
<TooltipProvider>
|
||||
|
|
@ -443,11 +432,14 @@ function ActionItem(props: ActionItemProps) {
|
|||
let formattedRgCommand = "";
|
||||
try {
|
||||
formattedRgCommand =
|
||||
formatRgCommand({
|
||||
pattern: props.pattern,
|
||||
paths: props.paths,
|
||||
flags: props.flags,
|
||||
})?.join(" ") ?? "";
|
||||
formatRgCommand(
|
||||
{
|
||||
pattern: props.pattern,
|
||||
paths: props.paths,
|
||||
flags: props.flags,
|
||||
},
|
||||
{ excludeRequiredFlags: true },
|
||||
)?.join(" ") ?? "";
|
||||
} catch {
|
||||
// no-op
|
||||
}
|
||||
|
|
|
|||
47
apps/web/src/components/gen-ui/conversation-summary.tsx
Normal file
47
apps/web/src/components/gen-ui/conversation-summary.tsx
Normal file
|
|
@ -0,0 +1,47 @@
|
|||
import { FileText, ChevronDown, ChevronRight } from "lucide-react";
|
||||
import { useState } from "react";
|
||||
import { MarkdownText } from "../thread/markdown-text";
|
||||
import { ToolIconWithTooltip } from "./tool-icon-tooltip";
|
||||
|
||||
/**
|
||||
* ConversationHistorySummary component for rendering conversation history summaries
|
||||
* Styled similarly to ActionItem for UI consistency
|
||||
*/
|
||||
export function ConversationHistorySummary({ summary }: { summary: string }) {
|
||||
const [expanded, setExpanded] = useState(true);
|
||||
|
||||
return (
|
||||
<div className="border-border overflow-hidden rounded-md border">
|
||||
<div className="border-b border-blue-300 bg-blue-100/50 p-2 dark:border-blue-800 dark:bg-blue-900/50">
|
||||
<div className="flex items-center justify-between">
|
||||
<div className="flex items-center gap-2">
|
||||
<ToolIconWithTooltip
|
||||
toolNamePretty="Conversation Summary"
|
||||
icon={<FileText className="h-4 w-4" />}
|
||||
/>
|
||||
<span className="text-xs font-medium">Conversation Summary</span>
|
||||
</div>
|
||||
<button
|
||||
onClick={() => setExpanded(!expanded)}
|
||||
className="flex items-center gap-1 text-xs font-normal text-blue-600 hover:text-blue-700 dark:text-blue-400 dark:hover:text-blue-300"
|
||||
>
|
||||
{expanded ? (
|
||||
<ChevronDown className="h-3 w-3" />
|
||||
) : (
|
||||
<ChevronRight className="h-3 w-3" />
|
||||
)}
|
||||
{expanded ? "Collapse" : "Expand"}
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{expanded && (
|
||||
<div className="p-3">
|
||||
<div className="text-sm">
|
||||
<MarkdownText>{summary}</MarkdownText>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
24
apps/web/src/components/gen-ui/tool-icon-tooltip.tsx
Normal file
24
apps/web/src/components/gen-ui/tool-icon-tooltip.tsx
Normal file
|
|
@ -0,0 +1,24 @@
|
|||
import { JSX } from "react";
|
||||
import {
|
||||
Tooltip,
|
||||
TooltipContent,
|
||||
TooltipProvider,
|
||||
TooltipTrigger,
|
||||
} from "../ui/tooltip";
|
||||
|
||||
export function ToolIconWithTooltip({
|
||||
toolNamePretty,
|
||||
icon,
|
||||
}: {
|
||||
toolNamePretty: string;
|
||||
icon: JSX.Element;
|
||||
}) {
|
||||
return (
|
||||
<TooltipProvider>
|
||||
<Tooltip>
|
||||
<TooltipTrigger asChild>{icon}</TooltipTrigger>
|
||||
<TooltipContent>{toolNamePretty}</TooltipContent>
|
||||
</Tooltip>
|
||||
</TooltipProvider>
|
||||
);
|
||||
}
|
||||
|
|
@ -27,7 +27,8 @@ import { ToolCall } from "@langchain/core/messages/tool";
|
|||
import {
|
||||
createApplyPatchToolFields,
|
||||
createShellToolFields,
|
||||
createSetTaskStatusToolFields,
|
||||
createMarkTaskCompletedToolFields,
|
||||
createMarkTaskNotCompletedToolFields,
|
||||
createRgToolFields,
|
||||
createOpenPrToolFields,
|
||||
createInstallDependenciesToolFields,
|
||||
|
|
@ -36,10 +37,12 @@ import {
|
|||
createGetURLContentToolFields,
|
||||
createFindInstancesOfToolFields,
|
||||
createWriteTechnicalNotesToolFields,
|
||||
createConversationHistorySummaryToolFields,
|
||||
} from "@open-swe/shared/open-swe/tools";
|
||||
import { z } from "zod";
|
||||
import { isAIMessageSDK, isToolMessageSDK } from "@/lib/langchain-messages";
|
||||
import { useStream } from "@langchain/langgraph-sdk/react";
|
||||
import { ConversationHistorySummary } from "@/components/gen-ui/conversation-summary";
|
||||
|
||||
// Used only for Zod type inference.
|
||||
const dummyRepo = { owner: "dummy", repo: "dummy" };
|
||||
|
|
@ -47,8 +50,12 @@ const shellTool = createShellToolFields(dummyRepo);
|
|||
type ShellToolArgs = z.infer<typeof shellTool.schema>;
|
||||
const applyPatchTool = createApplyPatchToolFields(dummyRepo);
|
||||
type ApplyPatchToolArgs = z.infer<typeof applyPatchTool.schema>;
|
||||
const setTaskStatusTool = createSetTaskStatusToolFields();
|
||||
type SetTaskStatusToolArgs = z.infer<typeof setTaskStatusTool.schema>;
|
||||
const markTaskCompletedTool = createMarkTaskCompletedToolFields();
|
||||
type MarkTaskCompletedToolArgs = z.infer<typeof markTaskCompletedTool.schema>;
|
||||
const markTaskNotCompletedTool = createMarkTaskNotCompletedToolFields();
|
||||
type MarkTaskNotCompletedToolArgs = z.infer<
|
||||
typeof markTaskNotCompletedTool.schema
|
||||
>;
|
||||
const rgTool = createRgToolFields(dummyRepo);
|
||||
type RgToolArgs = z.infer<typeof rgTool.schema>;
|
||||
const openPrTool = createOpenPrToolFields();
|
||||
|
|
@ -74,6 +81,12 @@ type WriteTechnicalNotesToolArgs = z.infer<
|
|||
typeof writeTechnicalNotesTool.schema
|
||||
>;
|
||||
|
||||
const conversationHistorySummaryTool =
|
||||
createConversationHistorySummaryToolFields();
|
||||
type ConversationHistorySummaryToolArgs = z.infer<
|
||||
typeof conversationHistorySummaryTool.schema
|
||||
>;
|
||||
|
||||
function CustomComponent({
|
||||
message,
|
||||
thread,
|
||||
|
|
@ -300,8 +313,12 @@ export function AssistantMessage({
|
|||
)
|
||||
: [];
|
||||
|
||||
const taskStatusToolCall = message
|
||||
? aiToolCalls.find((tc) => tc.name === setTaskStatusTool.name)
|
||||
const markTaskCompletedToolCall = message
|
||||
? aiToolCalls.find((tc) => tc.name === markTaskCompletedTool.name)
|
||||
: undefined;
|
||||
|
||||
const markTaskNotCompletedToolCall = message
|
||||
? aiToolCalls.find((tc) => tc.name === markTaskNotCompletedTool.name)
|
||||
: undefined;
|
||||
|
||||
const openPrToolCall = message
|
||||
|
|
@ -316,23 +333,49 @@ export function AssistantMessage({
|
|||
? aiToolCalls.find((tc) => tc.name === writeTechnicalNotesTool.name)
|
||||
: undefined;
|
||||
|
||||
// We can be sure that if the task status tool call is present, it will be the
|
||||
const conversationHistorySummaryToolCall = message
|
||||
? aiToolCalls.find((tc) => tc.name === conversationHistorySummaryTool.name)
|
||||
: undefined;
|
||||
|
||||
// Check if this is a conversation history summary message
|
||||
if (conversationHistorySummaryToolCall && aiToolCalls.length === 1) {
|
||||
const args =
|
||||
conversationHistorySummaryToolCall.args as ConversationHistorySummaryToolArgs;
|
||||
|
||||
return (
|
||||
<div className="flex flex-col gap-4">
|
||||
<ConversationHistorySummary
|
||||
summary={args.conversation_history_summary}
|
||||
/>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
// We can be sure that if either task status tool call is present, it will be the
|
||||
// only tool call/result we need to render for this message.
|
||||
if (taskStatusToolCall) {
|
||||
const args = taskStatusToolCall.args as SetTaskStatusToolArgs;
|
||||
if (markTaskCompletedToolCall || markTaskNotCompletedToolCall) {
|
||||
const toolCall = markTaskCompletedToolCall || markTaskNotCompletedToolCall;
|
||||
const completed = !!markTaskCompletedToolCall;
|
||||
|
||||
const correspondingToolResult = toolResults.find(
|
||||
(tr) => tr && tr.tool_call_id === taskStatusToolCall.id,
|
||||
(tr) => tr && tr.tool_call_id === toolCall!.id,
|
||||
);
|
||||
|
||||
const status = correspondingToolResult ? "done" : "generating";
|
||||
const completed = args.task_status === "completed";
|
||||
|
||||
// Get the appropriate summary text based on which tool was called
|
||||
const summaryText = markTaskCompletedToolCall
|
||||
? (markTaskCompletedToolCall.args as MarkTaskCompletedToolArgs)
|
||||
.completed_task_summary
|
||||
: (markTaskNotCompletedToolCall!.args as MarkTaskNotCompletedToolArgs)
|
||||
.reasoning;
|
||||
|
||||
return (
|
||||
<div className="flex flex-col gap-4">
|
||||
<TaskSummary
|
||||
status={status}
|
||||
completed={completed}
|
||||
summaryText={args.reasoning}
|
||||
summaryText={summaryText}
|
||||
/>
|
||||
</div>
|
||||
);
|
||||
|
|
|
|||
|
|
@ -76,7 +76,7 @@ export function createShellToolFields(targetRepository: TargetRepository) {
|
|||
.optional()
|
||||
.default(TIMEOUT_SEC)
|
||||
.describe(
|
||||
"The maximum time to wait for the command to complete in seconds.",
|
||||
"The maximum time to wait for the command to complete in seconds. For commands which may require a long time to complete, such as running tests, you should increase this value.",
|
||||
),
|
||||
});
|
||||
return {
|
||||
|
|
@ -143,7 +143,12 @@ export function createRgToolFields(targetRepository: TargetRepository) {
|
|||
const _tmpRgToolSchema = createRgToolFields({ owner: "x", repo: "x" }).schema;
|
||||
export type RipgrepCommand = z.infer<typeof _tmpRgToolSchema>;
|
||||
|
||||
export function formatRgCommand(cmd: RipgrepCommand): string[] {
|
||||
export function formatRgCommand(
|
||||
cmd: RipgrepCommand,
|
||||
options?: {
|
||||
excludeRequiredFlags?: boolean;
|
||||
},
|
||||
): string[] {
|
||||
const args = ["rg"];
|
||||
|
||||
// Always include these flags
|
||||
|
|
@ -166,8 +171,10 @@ export function formatRgCommand(cmd: RipgrepCommand): string[] {
|
|||
args.push(...filteredFlags);
|
||||
}
|
||||
|
||||
// Add the required flags
|
||||
args.push(...requiredFlags);
|
||||
if (!options?.excludeRequiredFlags) {
|
||||
// Add the required flags
|
||||
args.push(...requiredFlags);
|
||||
}
|
||||
|
||||
if (cmd.pattern) {
|
||||
args.push(cmd.pattern);
|
||||
|
|
@ -225,28 +232,45 @@ export function createFindInstancesOfToolFields(
|
|||
};
|
||||
}
|
||||
|
||||
export function createSetTaskStatusToolFields() {
|
||||
const setTaskStatusToolSchema = z.object({
|
||||
export function createMarkTaskNotCompletedToolFields() {
|
||||
const markTaskNotCompletedToolSchema = z.object({
|
||||
reasoning: z
|
||||
.string()
|
||||
.describe(
|
||||
"A concise reasoning summary for the status of the current task, explaining why you think it is completed or not completed.",
|
||||
),
|
||||
task_status: z
|
||||
.enum(["completed", "not_completed"])
|
||||
.describe(
|
||||
"The status of the current task, based on the reasoning provided.",
|
||||
"A concise reasoning summary for the status of the current task, explaining why you think it is not completed.",
|
||||
),
|
||||
});
|
||||
|
||||
const setTaskStatusTool = {
|
||||
name: "set_task_status",
|
||||
const markTaskNotCompletedTool = {
|
||||
name: "mark_task_not_completed",
|
||||
description:
|
||||
"The status of the current task, along with a concise reasoning summary to support the status.",
|
||||
schema: setTaskStatusToolSchema,
|
||||
"Mark the current task as not completed, along with a concise reasoning summary to support the status.",
|
||||
schema: markTaskNotCompletedToolSchema,
|
||||
};
|
||||
|
||||
return setTaskStatusTool;
|
||||
return markTaskNotCompletedTool;
|
||||
}
|
||||
|
||||
export function createMarkTaskCompletedToolFields() {
|
||||
const markTaskCompletedToolSchema = z.object({
|
||||
completed_task_summary: z
|
||||
.string()
|
||||
.describe(
|
||||
"A detailed summary of the actions you took to complete the current task. " +
|
||||
"Include specifics into the actions you took, insights you learned about the codebase while completing the task, and any other context which would be useful to another developer reviewing the actions you took. " +
|
||||
"You may include file paths and lists of the changes you made, but do not include full file contents or full code changes. " +
|
||||
"Ensure your summary is concise, thoughtful and helpful.",
|
||||
),
|
||||
});
|
||||
|
||||
const markTaskCompletedTool = {
|
||||
name: "mark_task_completed",
|
||||
description:
|
||||
"Mark the current task as completed, and provide a concise reasoning summary on the actions you took to complete the task.",
|
||||
schema: markTaskCompletedToolSchema,
|
||||
};
|
||||
|
||||
return markTaskCompletedTool;
|
||||
}
|
||||
|
||||
export function createInstallDependenciesToolFields(
|
||||
|
|
@ -359,3 +383,16 @@ export function createWriteTechnicalNotesToolFields() {
|
|||
schema: writeTechnicalNotesSchema,
|
||||
};
|
||||
}
|
||||
|
||||
export function createConversationHistorySummaryToolFields() {
|
||||
const conversationHistorySummarySchema = z.object({
|
||||
conversation_history_summary: z.string(),
|
||||
});
|
||||
|
||||
return {
|
||||
name: "conversation_history_summary",
|
||||
description:
|
||||
"<not used as an actual tool call. only used as shared types between the client and agent>",
|
||||
schema: conversationHistorySummarySchema,
|
||||
};
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue