diff --git a/scripts/run-e2e.ts b/scripts/run-e2e.ts index 7e4f7d28..c273d7fb 100644 --- a/scripts/run-e2e.ts +++ b/scripts/run-e2e.ts @@ -41,7 +41,8 @@ async function runE2E() { console.log(`\nRun started with thread ID: "${threadId}"\n`); for await (const chunk of stream) { - console.dir(chunk.data, { depth: null }); + const node = Object.keys(chunk.data)[0]; + console.log(`${node} completed.\n`); } } @@ -58,17 +59,31 @@ async function resumeGraph(threadId: string) { args: null, }, ]; + const configurable: Omit< + GraphConfig["configurable"], + "thread_id" | "assistant_id" + > = { + target_repository: { + owner: "bracesproul", + repo: "personal-site", + }, + }; const stream = client.runs.stream(threadId, "open-swe", { command: { resume: resumeValue, }, + config: { + configurable, + recursion_limit: 200, + }, streamSubgraphs: true, streamMode: "updates", }); for await (const chunk of stream) { - console.dir(chunk.data, { depth: null }); + const node = Object.keys(chunk.data)[0]; + console.log(`${node} completed.\n`); } } diff --git a/scripts/run-from-plan.ts b/scripts/run-from-plan.ts index 22468181..2d0b0ad9 100644 --- a/scripts/run-from-plan.ts +++ b/scripts/run-from-plan.ts @@ -20,16 +20,16 @@ async function runFromPlan() { "This repo contains the react/next.js code for my persona/portfolio site. It currently has static values set for the number of stars on the repositories I highlight. I want this to be accurate, but I do NOT want it to make requests to GitHub every time a user visits. Instead, please implement a solution which will run once a day, fetch the number of stars from a list of repos, then write them to vercel's KV store. Finally, update the UI to make a request to the KV store when the user visits my page and render the accurate star counts.", }, { - id: "toolu_01BDCDYr7oGxJnfgY6MjXGnx", + id: "toolu_01RNEjViZFBXgCSNMwqKvkVL", role: "assistant", content: [ { index: 0, type: "tool_use", - id: "toolu_01BDCDYr7oGxJnfgY6MjXGnx", + id: "toolu_01RNEjViZFBXgCSNMwqKvkVL", name: "condense_planning_context", input: - '{"context": "The user has a React/Next.js portfolio site with static star counts for GitHub repositories. The project structure shows:\\n\\n1. **Current Implementation**: \\n - Main page at `./personal-site/src/app/page.tsx` contains hardcoded star values for 6 repositories:\\n - langchain-ai/langchainjs (14.3k stars)\\n - langchain-ai/langgraphjs (1.3k stars) \\n - langchain-ai/open-canvas (4.4k stars)\\n - langchain-ai/agent-inbox (496 stars)\\n - langchain-ai/agent-chat-ui (583 stars)\\n - langchain-ai/social-media-agent (1.2k stars)\\n - Stars are displayed in ProjectCard components with a stars prop\\n\\n2. **Dependencies**: \\n - Project already has `@vercel/kv` dependency installed\\n - Uses Next.js 15.2.3 with TypeScript\\n - Has existing API routes in `./personal-site/src/app/api/` including gh-viewer route that already uses Vercel KV\\n\\n3. **Requirements**:\\n - Implement daily GitHub API fetching for star counts\\n - Store results in Vercel KV store \\n - Update UI to fetch from KV store instead of static values\\n - Avoid GitHub API calls on every user visit\\n\\n4. **Repository URLs to track**:\\n - https://github.com/langchain-ai/langchainjs\\n - https://github.com/langchain-ai/langgraphjs \\n - https://github.com/langchain-ai/open-canvas\\n - https://github.com/langchain-ai/agent-inbox\\n - https://github.com/langchain-ai/agent-chat-ui\\n - https://github.com/langchain-ai/social-media-agent"}', + '{"context": "The user has a React/Next.js portfolio site with static star counts for GitHub repositories. The project structure is:\\n\\n- Located in ./personal-site/ directory\\n- Uses Next.js 15.2.3 with TypeScript\\n- Already has @vercel/kv dependency installed\\n- Main page at src/app/page.tsx contains hardcoded star values for 6 repositories:\\n - langchain-ai/langchainjs (14.3k stars)\\n - langchain-ai/langgraphjs (1.3k stars) \\n - langchain-ai/open-canvas (4.4k stars)\\n - langchain-ai/agent-inbox (496 stars)\\n - langchain-ai/agent-chat-ui (583 stars)\\n - langchain-ai/social-media-agent (1.2k stars)\\n\\n- Stars are displayed in ProjectCard components with a stars prop\\n- Existing API routes in src/app/api/ (gh-viewer and gh-viewer-v2)\\n- The gh-viewer route already uses Vercel KV for storing view counts\\n\\nThe user wants to:\\n1. Create a daily cron job to fetch real GitHub star counts\\n2. Store the data in Vercel KV\\n3. Update the UI to fetch from KV instead of using static values\\n4. Avoid making GitHub API calls on every page visit\\n\\nThe project already has the necessary infrastructure (Vercel KV, API routes) and just needs the GitHub star fetching logic and UI updates."}', }, ], tool_calls: [ @@ -37,18 +37,24 @@ async function runFromPlan() { name: "condense_planning_context", args: { context: - "The user has a React/Next.js portfolio site with static star counts for GitHub repositories. The project structure shows:\n\n1. **Current Implementation**: \n - Main page at `./personal-site/src/app/page.tsx` contains hardcoded star values for 6 repositories:\n - langchain-ai/langchainjs (14.3k stars)\n - langchain-ai/langgraphjs (1.3k stars) \n - langchain-ai/open-canvas (4.4k stars)\n - langchain-ai/agent-inbox (496 stars)\n - langchain-ai/agent-chat-ui (583 stars)\n - langchain-ai/social-media-agent (1.2k stars)\n - Stars are displayed in ProjectCard components with a stars prop\n\n2. **Dependencies**: \n - Project already has `@vercel/kv` dependency installed\n - Uses Next.js 15.2.3 with TypeScript\n - Has existing API routes in `./personal-site/src/app/api/` including gh-viewer route that already uses Vercel KV\n\n3. **Requirements**:\n - Implement daily GitHub API fetching for star counts\n - Store results in Vercel KV store \n - Update UI to fetch from KV store instead of static values\n - Avoid GitHub API calls on every user visit\n\n4. **Repository URLs to track**:\n - https://github.com/langchain-ai/langchainjs\n - https://github.com/langchain-ai/langgraphjs \n - https://github.com/langchain-ai/open-canvas\n - https://github.com/langchain-ai/agent-inbox\n - https://github.com/langchain-ai/agent-chat-ui\n - https://github.com/langchain-ai/social-media-agent", + "The user has a React/Next.js portfolio site with static star counts for GitHub repositories. The project structure is:\n\n- Located in ./personal-site/ directory\n- Uses Next.js 15.2.3 with TypeScript\n- Already has @vercel/kv dependency installed\n- Main page at src/app/page.tsx contains hardcoded star values for 6 repositories:\n - langchain-ai/langchainjs (14.3k stars)\n - langchain-ai/langgraphjs (1.3k stars) \n - langchain-ai/open-canvas (4.4k stars)\n - langchain-ai/agent-inbox (496 stars)\n - langchain-ai/agent-chat-ui (583 stars)\n - langchain-ai/social-media-agent (1.2k stars)\n\n- Stars are displayed in ProjectCard components with a stars prop\n- Existing API routes in src/app/api/ (gh-viewer and gh-viewer-v2)\n- The gh-viewer route already uses Vercel KV for storing view counts\n\nThe user wants to:\n1. Create a daily cron job to fetch real GitHub star counts\n2. Store the data in Vercel KV\n3. Update the UI to fetch from KV instead of using static values\n4. Avoid making GitHub API calls on every page visit\n\nThe project already has the necessary infrastructure (Vercel KV, API routes) and just needs the GitHub star fetching logic and UI updates.", }, - id: "toolu_01BDCDYr7oGxJnfgY6MjXGnx", + id: "toolu_01RNEjViZFBXgCSNMwqKvkVL", type: "tool_call", }, ], + additional_kwargs: { + summary_message: true, + }, }, { role: "tool", - tool_call_id: "toolu_01BDCDYr7oGxJnfgY6MjXGnx", + tool_call_id: "toolu_01RNEjViZFBXgCSNMwqKvkVL", name: "condense_planning_context", content: "Successfully summarized planning context.", + additional_kwargs: { + summary_message: true, + }, }, ], plan: [ @@ -143,7 +149,8 @@ async function runFromPlan() { }); for await (const chunk of stream) { - console.dir(chunk.data, { depth: null }); + const node = Object.keys(chunk.data)[0]; + console.log(`${node} completed.\n`); } } diff --git a/src/index.ts b/src/index.ts index d270350c..47454e44 100644 --- a/src/index.ts +++ b/src/index.ts @@ -7,6 +7,7 @@ import { rewritePlan, interruptPlan, progressPlanStep, + summarizeTaskSteps, } from "./nodes/index.js"; import { isAIMessage } from "@langchain/core/messages"; import { plannerGraph } from "./subgraphs/index.js"; @@ -58,7 +59,10 @@ const workflow = new StateGraph(GraphAnnotation, GraphConfiguration) }) .addNode("generate-action", generateAction) .addNode("take-action", takeAction) - .addNode("progress-plan-step", progressPlanStep) + .addNode("progress-plan-step", progressPlanStep, { + ends: ["summarize-task-steps", "generate-action"], + }) + .addNode("summarize-task-steps", summarizeTaskSteps) .addEdge(START, "initialize") .addEdge("initialize", "generate-plan-subgraph") // TODO: Update routing to work w/ new interrupt node. @@ -70,7 +74,7 @@ const workflow = new StateGraph(GraphAnnotation, GraphConfiguration) .addEdge("rewrite-plan", "interrupt-plan") .addConditionalEdges("generate-action", takeActionOrEnd, ["take-action", END]) .addEdge("take-action", "progress-plan-step") - .addEdge("progress-plan-step", "generate-action"); + .addEdge("summarize-task-steps", "generate-action"); // Zod types are messed up export const graph = workflow.compile() as any; diff --git a/src/nodes/generate-message.ts b/src/nodes/generate-message.ts index de0074ec..1dc5f914 100644 --- a/src/nodes/generate-message.ts +++ b/src/nodes/generate-message.ts @@ -21,15 +21,22 @@ You work based on a plan which was generated in a previous step. The plan items You are an agent - please keep going until the user's query is completely resolved, before ending your turn and yielding back to the user. Only terminate your turn when you are sure that the problem is solved. If you are not sure about file content or codebase structure pertaining to the user's request, use your tools to read files and gather the relevant information: do NOT guess or make up an answer. -Please resolve the user's task by editing and testing the code files in your current code execution session. You are a deployed coding agent. Your session allows for you to modify and run code. The repo(s) are already cloned in your working directory, and you must fully solve the problem for your answer to be considered correct. You are permitted to take as long as you need to complete the current task. +Please resolve the user's task by editing and testing the code files in your current code execution session. You are a deployed coding agent. Your session allows for you to modify and run code. + +The repo is already cloned, and located inside {REPO_DIRECTORY} + +You must fully solve the problem for your answer to be considered correct. You are permitted to take as long as you need to complete the current task. You MUST adhere to the following criteria when executing the task: - Working on the repo(s) in the current environment is allowed, even if they are proprietary. - Analyzing code for vulnerabilities is allowed. - Showing user code and tool call details is allowed. - Remember to always properly format and quote your shell commands. +- Take advantage of the condensed context tool call messages in the conversation history. These contain summarized/condensed context from previously completed steps. Ensure you always read these messages to avoid duplicate work (e.g.: searching for file paths). - All changes are automatically committed, so you should not worry about creating backups, or committing changes. - Use \`apply_patch\` to edit files. This tool accepts diffs and file paths. It will then apply the given diff to the file. +- When using the \`shell\` tool, always take advantage of the \`workdir\` parameter to run commands inside the repo directory. You should not try to generate a command with \`cd \` as passing that path to \`workdir\` is much more efficient. +- Do not try to install dependencies, or run a server, compile the code, etc., unless you are explicitly asked to. - If completing the user's task requires writing or modifying files: - Your code and final answer should follow these *CODING GUIDELINES*: - Avoid writing to files which you have not already read. diff --git a/src/nodes/index.ts b/src/nodes/index.ts index efd07921..3e0d740c 100644 --- a/src/nodes/index.ts +++ b/src/nodes/index.ts @@ -4,3 +4,4 @@ export * from "./take-action.js"; export * from "./rewrite-plan.js"; export * from "./interrupt-plan.js"; export * from "./progress-plan-step.js"; +export * from "./summarize-task-steps.js"; diff --git a/src/nodes/progress-plan-step.ts b/src/nodes/progress-plan-step.ts index 69df3856..7e94a807 100644 --- a/src/nodes/progress-plan-step.ts +++ b/src/nodes/progress-plan-step.ts @@ -1,8 +1,15 @@ import { z } from "zod"; import { createLogger, LogLevel } from "../utils/logger.js"; -import { GraphConfig, GraphState, GraphUpdate, PlanItem } from "../types.js"; +import { GraphConfig, GraphState, PlanItem } from "../types.js"; import { loadModel, Task } from "../utils/load-model.js"; import { formatPlanPrompt } from "../utils/plan-prompt.js"; +import { Command } from "@langchain/langgraph"; +import { + getMessageContentString, + getMessageString, +} from "../utils/message/content.js"; +import { isHumanMessage } from "@langchain/core/messages"; +import { removeFirstHumanMessage } from "../utils/message/modify-array.js"; const logger = createLogger(LogLevel.INFO, "ProgressPlanStep"); @@ -14,11 +21,15 @@ Here is the plan: {PLAN_PROMPT} -In this task, you will analyze the plan, the tasks you've completed, the tasks which are left, and the current task you just took an action on. In addition to this, you're also provided the full conversation history between you and the user. All of the messages in this conversation are from the previous steps/actions you've taken, and any user input. +Analyze the tasks you've completed, the tasks which are remaining, and the current task you just took an action on. In addition to this, you're also provided the full conversation history between you and the user. All of the messages in this conversation are from the previous steps/actions you've taken, and any user input. -Take all of this information, and determine whether or not you have completed this task in the plan. To do this, you will call the \`confirm_task_completion\` tool.`; +Take all of this information, and determine whether or not you have completed this task in the plan. Be careful to not mark a task as completed if it is not, this can cause cascading issues in the workflow. +If you determine a task has been completed, you should call the \`confirm_task_completion\` tool. If you do NOT think the current task has been completed, do not call the tool and instead respond with \`not completed.\`.`; const confirmTaskCompletionToolSchema = z.object({ + reasoning: z + .string() + .describe("Reasoning for whether or not the task has been completed."), current_task_completed: z .boolean() .describe("Whether or not the current task has been completed."), @@ -37,23 +48,44 @@ const formatPrompt = (plan: PlanItem[]): string => { export async function progressPlanStep( state: GraphState, config: GraphConfig, -): Promise { +): Promise { const model = await loadModel(config, Task.PROGRESS_PLAN_CHECKER); const modelWithTools = model.bindTools([confirmTaskCompletionTool], { - tool_choice: confirmTaskCompletionTool.name, + tool_choice: "auto", }); + const firstUserMessage = state.messages.find(isHumanMessage); + + const conversationHistoryStr = `Here is the full conversation history after the user's request: + +${removeFirstHumanMessage(state.messages).map(getMessageString).join("\n")} + +Take all of this information, and determine whether or not you have completed this task in the plan. Be careful to not mark a task as completed if it is not, this can cause cascading issues in the workflow. +If you determine a task has been completed, you should call the \`confirm_task_completion\` tool. If you do NOT think the current task has been completed, do not call the tool and instead respond with \`not completed.\`. + +ENSURE YOU ONLY CALL THE \`confirm_task_completion\` TOOL IF YOU DETERMINE THE CURRENT TASK HAS BEEN COMPLETED, OR RESPOND WITH 'not completed.'. DO NOT TAKE ANY OTHER ACTION.`; + const response = await modelWithTools.invoke([ { role: "system", content: formatPrompt(state.plan), }, - ...state.messages, + ...(firstUserMessage ? [firstUserMessage] : []), + { + role: "user", + content: conversationHistoryStr, + }, ]); const toolCall = response.tool_calls?.[0]; if (!toolCall) { - throw new Error("Failed to check plan."); + logger.info( + "Current task has not been completed, as no tool call was generated. Progressing to the next action.", + { + responseContent: getMessageContentString(response.content), + }, + ); + return new Command({ goto: "generate-action" }); } const isCompleted = ( @@ -61,28 +93,42 @@ export async function progressPlanStep( ).current_task_completed; if (!isCompleted) { - // Not completed, no changes need to be made - return {}; + logger.info( + "Current task has not been completed. Progressing to the next action.", + { + reasoning: toolCall.args.reasoning, + }, + ); + return new Command({ goto: "generate-action" }); } const remainingTask = state.plan.find((p) => !p.completed); if (!remainingTask) { - // No remaining tasks, end the process logger.info( - "Found no remaining tasks in the plan during the check plan step.", + "Found no remaining tasks in the plan during the check plan step. Progressing to the next action.", ); - return {}; + return new Command({ goto: "generate-action" }); } - return { - plan: state.plan.map((p) => { - if (p.index === remainingTask.index) { - return { - ...p, - completed: true, - }; - } - return p; - }), - }; + logger.info("Task marked as completed. Routing to task summarization step.", { + remainingTask: { + ...remainingTask, + completed: true, + }, + }); + + return new Command({ + goto: "summarize-task-steps", + update: { + plan: state.plan.map((p) => { + if (p.index === remainingTask.index) { + return { + ...p, + completed: true, + }; + } + return p; + }), + }, + }); } diff --git a/src/nodes/summarize-task-steps.ts b/src/nodes/summarize-task-steps.ts new file mode 100644 index 00000000..9e5e8477 --- /dev/null +++ b/src/nodes/summarize-task-steps.ts @@ -0,0 +1,116 @@ +import { z } from "zod"; +import { GraphConfig, GraphState, GraphUpdate, PlanItem } from "../types.js"; +import { loadModel, Task } from "../utils/load-model.js"; +import { + AIMessage, + isHumanMessage, + ToolMessage, +} from "@langchain/core/messages"; +import { formatPlanPrompt } from "../utils/plan-prompt.js"; +import { createLogger, LogLevel } from "../utils/logger.js"; +import { getMessageString } from "../utils/message/content.js"; +import { + removeFirstHumanMessage, + removeLastTaskMessages, +} from "../utils/message/modify-array.js"; + +const logger = createLogger(LogLevel.INFO, "SummarizeTaskSteps"); + +const systemPrompt = `You are operating as a terminal-based agentic coding assistant built by LangChain. It wraps LLM models to enable natural language interaction with a local codebase. You are expected to be precise, safe, and helpful. + +You've been given a task to summarize the messages in your conversation history. You just completed a task in your plan, and can now summarize/condense all of the messages in your conversation history which were relevant to that task. +You do not want to keep the entire conversation history, but instead you want to keep the most relevant and important snippets for future context. + +{PLAN_PROMPT} + +You MUST adhere to the following criteria when summarizing the conversation history: +- Retain context such as file paths, versions, and installed software. +- Do not retain any full code snippets. +- Do not retain any full file contents. +- Ensure your summary is concise, but useful for future context. +- If the conversation history contains any key insights or learnings, ensure you retain those. + +With all of this in mind, please carefully summarize and condense the following conversation history. Ensure you pass this condensed context to the \`condense_task_context\` tool. +`; + +const formatPrompt = (plan: PlanItem[]): string => + systemPrompt.replace( + "{PLAN_PROMPT}", + formatPlanPrompt(plan, { useLastCompletedTask: true }), + ); + +const condenseContextToolSchema = z.object({ + context: z + .string() + .describe( + "The condensed context from the conversation history relevant to the recently completed task.", + ), +}); +const condenseContextTool = { + name: "condense_task_context", + description: + "Condense the conversation history into a concise summary, while still retaining the most relevant and important snippets.", + schema: condenseContextToolSchema, +}; + +export async function summarizeTaskSteps( + state: GraphState, + config: GraphConfig, +): Promise { + const model = await loadModel(config, Task.SUMMARIZER); + const modelWithTools = model.bindTools([condenseContextTool], { + tool_choice: condenseContextTool.name, + }); + + const firstUserMessage = state.messages.find(isHumanMessage); + + const conversationHistoryStr = `Here is the full conversation history for the task after the user's request. +This history includes any previous summarization/condensation of the conversation history. Ensure you do NOT summarize those messages, or duplicate any information present in them, but do use them as context so you know what has already been seen and summarized. + +${removeFirstHumanMessage(state.messages).map(getMessageString).join("\n")} + +Given this full conversation history please generate a concise, and useful summary of the conversation history for this task. Ensure you pass this condensed context to the \`condense_task_context\` tool.`; + + logger.info(`Summarizing task steps...`); + const response = await modelWithTools.invoke([ + { + role: "system", + content: formatPrompt(state.plan), + }, + ...(firstUserMessage ? [firstUserMessage] : []), + { + role: "user", + content: conversationHistoryStr, + }, + ]); + + const toolCall = response.tool_calls?.[0]; + if (!toolCall) { + throw new Error("Failed to generate plan"); + } + + const toolMessage = new ToolMessage({ + tool_call_id: toolCall.id ?? "", + name: toolCall.name, + content: `Successfully summarized planning context.`, + additional_kwargs: { + summary_message: true, + }, + }); + + const removedMessages = removeLastTaskMessages(state.messages); + logger.info(`Removing ${removedMessages.length} message(s) from state.`); + return { + messages: [ + ...removedMessages, + new AIMessage({ + ...response, + additional_kwargs: { + ...response.additional_kwargs, + summary_message: true, + }, + }), + toolMessage, + ], + }; +} diff --git a/src/subgraphs/planner/nodes/summarizer.ts b/src/subgraphs/planner/nodes/summarizer.ts index 3efa2bde..014c33ff 100644 --- a/src/subgraphs/planner/nodes/summarizer.ts +++ b/src/subgraphs/planner/nodes/summarizer.ts @@ -2,8 +2,15 @@ import { z } from "zod"; import { GraphConfig } from "../../../types.js"; import { PlannerGraphState, PlannerGraphUpdate } from "../types.js"; import { loadModel, Task } from "../../../utils/load-model.js"; -import { isHumanMessage, ToolMessage } from "@langchain/core/messages"; -import { getMessageContentString } from "../../../utils/message-content.js"; +import { + AIMessage, + isHumanMessage, + ToolMessage, +} from "@langchain/core/messages"; +import { + getMessageContentString, + getMessageString, +} from "../../../utils/message/content.js"; const systemPrompt = `You are operating as a terminal-based agentic coding assistant built by LangChain. It wraps LLM models to enable natural language interaction with a local codebase. You are expected to be precise, safe, and helpful. @@ -42,13 +49,17 @@ export async function summarizer( state: PlannerGraphState, config: GraphConfig, ): Promise { - const model = await loadModel(config, Task.PLANNER); + const model = await loadModel(config, Task.SUMMARIZER); const modelWithTools = model.bindTools([condenseContextTool], { tool_choice: condenseContextTool.name, }); const firstUserMessage = state.messages.find(isHumanMessage); + const conversationHistoryStr = `Here is the full conversation history: + +${state.plannerMessages.map(getMessageString).join("\n")}`; + const response = await modelWithTools.invoke([ { role: "system", @@ -58,7 +69,10 @@ export async function summarizer( ), ), }, - ...state.plannerMessages, + { + role: "user", + content: conversationHistoryStr, + }, ]); const toolCall = response.tool_calls?.[0]; @@ -70,9 +84,21 @@ export async function summarizer( tool_call_id: toolCall.id ?? "", name: toolCall.name, content: `Successfully summarized planning context.`, + additional_kwargs: { + summary_message: true, + }, }); return { - messages: [response, toolMessage], + messages: [ + new AIMessage({ + ...response, + additional_kwargs: { + ...response.additional_kwargs, + summary_message: true, + }, + }), + toolMessage, + ], }; } diff --git a/src/types.ts b/src/types.ts index bb20b61c..6caa9bad 100644 --- a/src/types.ts +++ b/src/types.ts @@ -289,6 +289,41 @@ export const GraphConfiguration = z.object({ description: "Controls randomness (0 = deterministic, 2 = creative)", }, }), + + /** + * The model ID to use for summarizing the conversation history. + * @default "anthropic:claude-sonnet-4-0" + */ + summarizerModelName: z + .string() + .optional() + .langgraph.metadata({ + x_oap_ui_config: { + type: "select", + default: "anthropic:claude-sonnet-4-0", + description: + "The model to use for summarizing the conversation history", + options: MODEL_OPTIONS_NO_THINKING, + }, + }), + /** + * The temperature to use for summarizing the conversation history. + * If selecting a reasoning model, this will be ignored. + * @default 0 + */ + summarizerTemperature: z + .number() + .optional() + .langgraph.metadata({ + x_oap_ui_config: { + type: "slider", + default: 0, + min: 0, + max: 2, + step: 0.1, + description: "Controls randomness (0 = deterministic, 2 = creative)", + }, + }), }); export type GraphConfig = LangGraphRunnableConfig< diff --git a/src/utils/load-model.ts b/src/utils/load-model.ts index 0d1a6719..aa853648 100644 --- a/src/utils/load-model.ts +++ b/src/utils/load-model.ts @@ -6,6 +6,7 @@ export enum Task { PLANNER_CONTEXT = "plannerContext", ACTION_GENERATOR = "actionGenerator", PROGRESS_PLAN_CHECKER = "progressPlanChecker", + SUMMARIZER = "summarizer", } const TASK_TO_CONFIG_DEFAULTS_MAP = { @@ -25,6 +26,10 @@ const TASK_TO_CONFIG_DEFAULTS_MAP = { modelName: "anthropic:claude-sonnet-4-0", temperature: 0, }, + [Task.SUMMARIZER]: { + modelName: "anthropic:claude-sonnet-4-0", + temperature: 0, + }, }; export async function loadModel(config: GraphConfig, task: Task) { diff --git a/src/utils/message-content.ts b/src/utils/message-content.ts deleted file mode 100644 index 566ab0af..00000000 --- a/src/utils/message-content.ts +++ /dev/null @@ -1,10 +0,0 @@ -import { MessageContent } from "@langchain/core/messages"; - -export function getMessageContentString(content: MessageContent): string { - if (typeof content === "string") return content; - - return content - .filter((c): c is { type: "text"; text: string } => c.type === "text") - .map((c) => c.text) - .join(" "); -} diff --git a/src/utils/message/content.ts b/src/utils/message/content.ts new file mode 100644 index 00000000..88305327 --- /dev/null +++ b/src/utils/message/content.ts @@ -0,0 +1,68 @@ +import { + AIMessage, + BaseMessage, + HumanMessage, + isAIMessage, + isHumanMessage, + isSystemMessage, + isToolMessage, + MessageContent, + SystemMessage, + ToolMessage, +} from "@langchain/core/messages"; +import { ToolCall } from "@langchain/core/messages/tool"; + +export function getMessageContentString(content: MessageContent): string { + if (typeof content === "string") return content; + + return content + .filter((c): c is { type: "text"; text: string } => c.type === "text") + .map((c) => c.text) + .join(" "); +} + +export function getToolCallsString(toolCalls: ToolCall[] | undefined): string { + if (!toolCalls?.length) return ""; + return toolCalls.map((c) => JSON.stringify(c, null, 2)).join("\n"); +} + +export function getAIMessageString(message: AIMessage): string { + const content = getMessageContentString(message.content); + const toolCalls = getToolCallsString(message.tool_calls); + return `\nContent: ${content}\nTool calls: ${toolCalls}\n`; +} + +export function getHumanMessageString(message: HumanMessage): string { + const content = getMessageContentString(message.content); + return `\nContent: ${content}\n`; +} + +export function getToolMessageString(message: ToolMessage): string { + const content = getMessageContentString(message.content); + const toolCallId = message.tool_call_id; + const toolCallName = message.name; + return `\nTool Call ID: ${toolCallId}\nTool Call Name: ${toolCallName}\nContent: ${content}\n`; +} + +export function getSystemMessageString(message: SystemMessage): string { + const content = getMessageContentString(message.content); + return `\nContent: ${content}\n`; +} + +export function getUnknownMessageString(message: BaseMessage): string { + return `\n${JSON.stringify(message, null, 2)}\n`; +} + +export function getMessageString(message: BaseMessage): string { + if (isAIMessage(message)) { + return getAIMessageString(message); + } else if (isHumanMessage(message)) { + return getHumanMessageString(message); + } else if (isToolMessage(message)) { + return getToolMessageString(message); + } else if (isSystemMessage(message)) { + return getSystemMessageString(message); + } + + return getUnknownMessageString(message); +} diff --git a/src/utils/message/modify-array.ts b/src/utils/message/modify-array.ts new file mode 100644 index 00000000..14434d18 --- /dev/null +++ b/src/utils/message/modify-array.ts @@ -0,0 +1,36 @@ +import { + BaseMessage, + isAIMessage, + isHumanMessage, + isToolMessage, + RemoveMessage, +} from "@langchain/core/messages"; + +export function removeLastTaskMessages(messages: BaseMessage[]): BaseMessage[] { + return messages + .filter((m) => { + if ( + m.additional_kwargs?.summary_message || + (!isAIMessage(m) && !isToolMessage(m)) || + !m.id + ) { + return false; + } + return true; + }) + .map((m) => new RemoveMessage({ id: m.id ?? "" })); +} + +export function removeFirstHumanMessage( + messages: BaseMessage[], +): BaseMessage[] { + let humanMsgFound = false; + return messages.filter((m) => { + if (isHumanMessage(m) && !humanMsgFound) { + humanMsgFound = true; + return false; + } + + return true; + }); +} diff --git a/src/utils/plan-prompt.ts b/src/utils/plan-prompt.ts index 2724694f..b1c57b5b 100644 --- a/src/utils/plan-prompt.ts +++ b/src/utils/plan-prompt.ts @@ -9,10 +9,24 @@ export const PLAN_PROMPT = `## Completed Tasks ## Current Task {CURRENT_TASK}`; -export function formatPlanPrompt(plan: PlanItem[]): string { +/** + * Formats a plan for use in a prompt. + * @param plan The plan to format + * @param options Options for formatting the plan + * @param options.useLastCompletedTask Whether to use the last completed task as the current task + * @returns The formatted plan + */ +export function formatPlanPrompt( + plan: PlanItem[], + options?: { + useLastCompletedTask?: boolean; + }, +): string { const completedTasks = plan.filter((p) => p.completed); const remainingTasks = plan.filter((p) => !p.completed); - const currentTask = remainingTasks.sort((a, b) => a.index - b.index)[0]; + const currentTask = options?.useLastCompletedTask + ? completedTasks.sort((a, b) => a.index - b.index)[0] + : remainingTasks.sort((a, b) => a.index - b.index)[0]; return PLAN_PROMPT.replace( "{COMPLETED_TASKS}",