diff --git a/apps/open-swe/src/graphs/programmer/index.ts b/apps/open-swe/src/graphs/programmer/index.ts index 197fe2e3..b44a33e2 100644 --- a/apps/open-swe/src/graphs/programmer/index.ts +++ b/apps/open-swe/src/graphs/programmer/index.ts @@ -8,19 +8,20 @@ import { import { generateAction, takeAction, - progressPlanStep, generateConclusion, openPullRequest, diagnoseError, requestHelp, updatePlan, summarizeHistory, + handleCompletedTask, } from "./nodes/index.js"; import { BaseMessage, isAIMessage } from "@langchain/core/messages"; import { initializeSandbox } from "../shared/initialize-sandbox.js"; import { graph as reviewerGraph } from "../reviewer/index.js"; import { getRemainingPlanItems } from "../../utils/current-task.js"; import { getActivePlanItems } from "@open-swe/shared/open-swe/tasks"; +import { createMarkTaskCompletedToolFields } from "@open-swe/shared/open-swe/tools"; function lastMessagesMissingToolCalls( messages: BaseMessage[], @@ -40,7 +41,7 @@ function lastMessagesMissingToolCalls( * Otherwise, it ends the process. * * @param {GraphState} state - The current graph state. - * @returns {"route-to-review-or-conclusion" | "take-action" | "request-help" | "generate-action" | Send} The next node to execute, or END if the process should stop. + * @returns {"route-to-review-or-conclusion" | "take-action" | "request-help" | "generate-action" | "handle-completed-task" | Send} The next node to execute, or END if the process should stop. */ function routeGeneratedAction( state: GraphState, @@ -49,6 +50,7 @@ function routeGeneratedAction( | "take-action" | "request-help" | "generate-action" + | "handle-completed-task" | Send { const { internalMessages } = state; const lastMessage = internalMessages[internalMessages.length - 1]; @@ -71,6 +73,12 @@ function routeGeneratedAction( }); } + const taskMarkedCompleted = + toolCall.name === createMarkTaskCompletedToolFields().name; + if (taskMarkedCompleted) { + return "handle-completed-task"; + } + return "take-action"; } @@ -122,10 +130,10 @@ const workflow = new StateGraph(GraphAnnotation, GraphConfiguration) .addNode("initialize", initializeSandbox) .addNode("generate-action", generateAction) .addNode("take-action", takeAction, { - ends: ["progress-plan-step", "diagnose-error"], + ends: ["generate-action", "diagnose-error"], }) .addNode("update-plan", updatePlan) - .addNode("progress-plan-step", progressPlanStep, { + .addNode("handle-completed-task", handleCompletedTask, { ends: [ "summarize-history", "generate-action", @@ -151,6 +159,7 @@ const workflow = new StateGraph(GraphAnnotation, GraphConfiguration) "route-to-review-or-conclusion", "update-plan", "generate-action", + "handle-completed-task", ]) .addEdge("update-plan", "generate-action") .addEdge("diagnose-error", "generate-action") diff --git a/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts b/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts index 7ac49617..98df8f57 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/generate-message/index.ts @@ -1,3 +1,4 @@ +import { v4 as uuidv4 } from "uuid"; import { GraphState, GraphConfig, @@ -45,16 +46,16 @@ import { convertMessagesToCacheControlledMessages, trackCachePerformance, } from "../../../../utils/caching.js"; +import { createMarkTaskCompletedToolFields } from "@open-swe/shared/open-swe/tools"; +import { HumanMessage } from "@langchain/core/messages"; const logger = createLogger(LogLevel.INFO, "GenerateMessageNode"); const formatDynamicContextPrompt = (state: GraphState) => { - return DYNAMIC_SYSTEM_PROMPT.replaceAll( - "{PLAN_PROMPT_WITH_SUMMARIES}", - formatPlanPrompt(getActivePlanItems(state.taskPlan), { - includeSummaries: true, - }), - ) + const planString = getActivePlanItems(state.taskPlan) + .map((i) => `\n${i.plan}\n`) + .join("\n"); + return DYNAMIC_SYSTEM_PROMPT.replaceAll("{PLAN_PROMPT}", planString) .replaceAll( "{PLAN_GENERATION_NOTES}", state.contextGatheringNotes || "No context gathering notes available.", @@ -113,6 +114,27 @@ const formatCacheablePrompt = (state: GraphState): CacheablePromptSegment[] => { return segments.filter((segment) => segment.text.trim() !== ""); }; +const planSpecificPrompt = ` +Here is the task execution plan for the request you're working on. +Ensure you carefully read through all of the instructions, messages, and context provided above. +Once you have a clear understanding of the current state of the task, analyze the plan provided below, and take an action based on it. +You're provided with the full list of tasks, including the completed, current and remaining tasks. + +You are in the process of executing the current task: + +{PLAN_PROMPT} +`; + +const formatSpecificPlanPrompt = (state: GraphState): HumanMessage => { + return new HumanMessage({ + id: uuidv4(), + content: planSpecificPrompt.replace( + "{PLAN_PROMPT}", + formatPlanPrompt(getActivePlanItems(state.taskPlan)), + ), + }); +}; + export async function generateAction( state: GraphState, config: GraphConfig, @@ -123,6 +145,7 @@ export async function generateAction( Task.PROGRAMMER, ); const mcpTools = await getMcpTools(config); + const markTaskCompletedTool = createMarkTaskCompletedToolFields(); const tools = [ createSearchTool(state), @@ -132,6 +155,7 @@ export async function generateAction( createUpdatePlanToolFields(), createGetURLContentTool(), createInstallDependenciesTool(state), + markTaskCompletedTool, ...mcpTools, ]; logger.info( @@ -176,6 +200,7 @@ export async function generateAction( }), }, ...inputMessagesWithCache, + formatSpecificPlanPrompt(state), ]); const hasToolCalls = !!response.tool_calls?.length; @@ -186,6 +211,22 @@ export async function generateAction( newSandboxSessionId = await stopSandbox(state.sandboxSessionId); } + if ( + response.tool_calls?.length && + response.tool_calls?.length > 1 && + response.tool_calls.some((t) => t.name === markTaskCompletedTool.name) + ) { + logger.error( + "Multiple tool calls found, including mark_task_completed. Removing the mark_task_completed call.", + { + toolCalls: JSON.stringify(response.tool_calls, null, 2), + }, + ); + response.tool_calls = response.tool_calls.filter( + (t) => t.name !== markTaskCompletedTool.name, + ); + } + logger.info("Generated action", { currentTask: getCurrentPlanItem(getActivePlanItems(state.taskPlan)).plan, ...(getMessageContentString(response.content) && { diff --git a/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts b/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts index d4bb862c..fd4cd81f 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts @@ -86,6 +86,15 @@ You are a terminal-based agentic coding assistant built by LangChain. You wrap L Use this tool to add or remove tasks from the plan, or to update the plan in any other way + + - When you believe you've completed a task, you may call the \`mark_task_completed\` tool to mark the task as complete. + - The \`mark_task_completed\` tool should NEVER be called in parallel with any other tool calls. Ensure it's the only tool you're calling in this message, if you do determine the task is completed. + - Carefully read over the actions you've taken, and the current task (listed below) to ensure the task is complete. You want to avoid prematurely marking a task as complete. + - If the current task involves fixing an issue, such as a failing test, a broken build, etc., you must validate the issue is ACTUALLY fixed before marking it as complete. + - To verify a fix, ensure you run the test, build, or other command first to validate the fix. + - If you do not believe the task is complete, you do not need to call the \`mark_task_completed\` tool. You can continue working on the task, until you determine it is complete. + + @@ -113,8 +122,10 @@ export const CODE_REVIEW_PROMPT = ` export const DYNAMIC_SYSTEM_PROMPT = ` -- Current plan with summaries -{PLAN_PROMPT_WITH_SUMMARIES} +- Task execution plan + + {PLAN_PROMPT} + - Plan generation notes These are notes you took while gathering context for the plan: diff --git a/apps/open-swe/src/graphs/programmer/nodes/handle-completed-task.ts b/apps/open-swe/src/graphs/programmer/nodes/handle-completed-task.ts new file mode 100644 index 00000000..c2cfcc2f --- /dev/null +++ b/apps/open-swe/src/graphs/programmer/nodes/handle-completed-task.ts @@ -0,0 +1,138 @@ +import { v4 as uuidv4 } from "uuid"; +import { createLogger, LogLevel } from "../../../utils/logger.js"; +import { + GraphConfig, + GraphState, + GraphUpdate, +} from "@open-swe/shared/open-swe/types"; +import { Command } from "@langchain/langgraph"; +import { + completePlanItem, + getActivePlanItems, + getActiveTask, +} from "@open-swe/shared/open-swe/tasks"; +import { + getCurrentPlanItem, + getRemainingPlanItems, +} from "../../../utils/current-task.js"; +import { isAIMessage, ToolMessage } from "@langchain/core/messages"; +import { addTaskPlanToIssue } from "../../../utils/github/issue-task.js"; +import { createMarkTaskCompletedToolFields } from "@open-swe/shared/open-swe/tools"; +import { + calculateConversationHistoryTokenCount, + getMessagesSinceLastSummary, + MAX_INTERNAL_TOKENS, +} from "../../../utils/tokens.js"; +import { z } from "zod"; + +const logger = createLogger(LogLevel.INFO, "HandleCompletedTask"); + +export async function handleCompletedTask( + state: GraphState, + config: GraphConfig, +): Promise { + const markCompletedTool = createMarkTaskCompletedToolFields(); + const markCompletedMessage = + state.internalMessages[state.internalMessages.length - 1]; + if ( + !isAIMessage(markCompletedMessage) || + !markCompletedMessage.tool_calls?.length || + !markCompletedMessage.tool_calls.some( + (tc) => tc.name === markCompletedTool.name, + ) + ) { + throw new Error("Failed to find a tool call when checking task status."); + } + const toolCall = markCompletedMessage.tool_calls?.[0]; + if (!toolCall) { + throw new Error( + "Failed to generate a tool call when checking task status.", + ); + } + + const activePlanItems = getActivePlanItems(state.taskPlan); + const currentTask = getCurrentPlanItem(activePlanItems); + const toolMessage = new ToolMessage({ + id: uuidv4(), + tool_call_id: toolCall.id ?? "", + content: `Saved task status as completed for task ${currentTask?.plan || "unknown"}`, + name: toolCall.name, + }); + + const newMessages = [toolMessage]; + + const newMessageList = [...state.internalMessages, ...newMessages]; + const wouldBeConversationHistoryToSummarize = + await getMessagesSinceLastSummary(newMessageList, { + excludeHiddenMessages: true, + excludeCountFromEnd: 20, + }); + const totalInternalTokenCount = calculateConversationHistoryTokenCount( + wouldBeConversationHistoryToSummarize, + { + // Retain the last 20 messages from state + excludeHiddenMessages: true, + excludeCountFromEnd: 20, + }, + ); + + const summary = (toolCall.args as z.infer) + .completed_task_summary; + + // LLM marked as completed, so we need to update the plan to reflect that. + const updatedPlanTasks = completePlanItem( + state.taskPlan, + getActiveTask(state.taskPlan).id, + currentTask.index, + summary, + ); + // Update the github issue to reflect this task as completed. + await addTaskPlanToIssue( + { + githubIssueId: state.githubIssueId, + targetRepository: state.targetRepository, + }, + config, + updatedPlanTasks, + ); + + const commandUpdate: GraphUpdate = { + messages: newMessages, + internalMessages: newMessages, + // Even though there are no remaining tasks, still mark as completed so the UI reflects that the task is completed. + taskPlan: updatedPlanTasks, + }; + + // This should in theory never happen, but ensure we route properly if it does. + const remainingTask = getRemainingPlanItems(activePlanItems)?.[0]; + if (!remainingTask) { + logger.info( + "Found no remaining tasks in the plan during the check plan step. Continuing to the conclusion generation step.", + ); + + return new Command({ + goto: "route-to-review-or-conclusion", + update: commandUpdate, + }); + } + + if (totalInternalTokenCount >= MAX_INTERNAL_TOKENS) { + logger.info( + "Internal messages list is at or above the max token limit. Routing to summarize history step.", + { + totalInternalTokenCount, + maxInternalTokenCount: MAX_INTERNAL_TOKENS, + }, + ); + + return new Command({ + goto: "summarize-history", + update: commandUpdate, + }); + } + + return new Command({ + goto: "generate-action", + update: commandUpdate, + }); +} diff --git a/apps/open-swe/src/graphs/programmer/nodes/index.ts b/apps/open-swe/src/graphs/programmer/nodes/index.ts index 126f7cff..85a89ed9 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/index.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/index.ts @@ -1,6 +1,6 @@ export * from "./generate-message/index.js"; export * from "./take-action.js"; -export * from "./progress-plan-step.js"; +export * from "./handle-completed-task.js"; export * from "./generate-conclusion.js"; export * from "./open-pr.js"; export * from "./diagnose-error.js"; diff --git a/apps/open-swe/src/graphs/programmer/nodes/progress-plan-step.ts b/apps/open-swe/src/graphs/programmer/nodes/progress-plan-step.ts deleted file mode 100644 index 27fb10bd..00000000 --- a/apps/open-swe/src/graphs/programmer/nodes/progress-plan-step.ts +++ /dev/null @@ -1,249 +0,0 @@ -import { v4 as uuidv4 } from "uuid"; -import { createLogger, LogLevel } from "../../../utils/logger.js"; -import { - GraphConfig, - GraphState, - GraphUpdate, - PlanItem, -} from "@open-swe/shared/open-swe/types"; -import { - loadModel, - supportsParallelToolCallsParam, - Task, -} from "../../../utils/load-model.js"; -import { formatPlanPrompt } from "../../../utils/plan-prompt.js"; -import { Command } from "@langchain/langgraph"; -import { getMessageString } from "../../../utils/message/content.js"; -import { formatUserRequestPrompt } from "../../../utils/user-request.js"; -import { - completePlanItem, - getActivePlanItems, - getActiveTask, -} from "@open-swe/shared/open-swe/tasks"; -import { - getCurrentPlanItem, - getRemainingPlanItems, -} from "../../../utils/current-task.js"; -import { ToolMessage } from "@langchain/core/messages"; -import { addTaskPlanToIssue } from "../../../utils/github/issue-task.js"; -import { - createMarkTaskNotCompletedToolFields, - createMarkTaskCompletedToolFields, -} from "@open-swe/shared/open-swe/tools"; -import { - calculateConversationHistoryTokenCount, - getMessagesSinceLastSummary, - MAX_INTERNAL_TOKENS, -} from "../../../utils/tokens.js"; -import { z } from "zod"; -import { trackCachePerformance } from "../../../utils/caching.js"; - -const logger = createLogger(LogLevel.INFO, "ProgressPlanStep"); - -const systemPrompt = `You are operating as a terminal-based agentic coding assistant built by LangChain. It wraps LLM models to enable natural language interaction with a local codebase. You are expected to be precise, safe, and helpful. - -In your workflow, you generate a plan, then act on said plan. It may take many actions to complete a single step, or a single action to complete the step. - -Here is the plan, along with the summaries of each completed task: -{PLAN_PROMPT} - -Analyze the tasks you've completed, the tasks which are remaining, and the current task you just took an action on. -In addition to this, you're also provided the full conversation history between you and the user. All of the messages in this conversation are from the previous steps/actions you've taken, and any user input. -If the task you're working on is to fix a failing command (e.g. a test, build, lint, etc.), and you've made changes to fix the issue, you must re-run the command to ensure the fix was successful before you can mark the task as complete. - For example: If you have a failing test, and you've applied an update to the file to fix the test, you MUST re-run the test before you can mark the task as complete. - -Take all of this information, and determine if the current task is complete, or if you still have work left to do. -Once you've determined the status of the current task, call either: -- \`mark_task_completed\` if the task is complete. -- \`mark_task_not_completed\` if the task is not complete. -`; - -const formatPrompt = (taskPlan: PlanItem[]): string => { - return systemPrompt.replace( - "{PLAN_PROMPT}", - formatPlanPrompt(taskPlan, { includeSummaries: true }), - ); -}; - -export async function progressPlanStep( - state: GraphState, - config: GraphConfig, -): Promise { - const markNotCompletedTool = createMarkTaskNotCompletedToolFields(); - const markCompletedTool = createMarkTaskCompletedToolFields(); - const model = await loadModel(config, Task.SUMMARIZER); - const modelSupportsParallelToolCallsParam = supportsParallelToolCallsParam( - config, - Task.SUMMARIZER, - ); - const modelWithTools = model.bindTools( - [markNotCompletedTool, markCompletedTool], - { - tool_choice: "any", - ...(modelSupportsParallelToolCallsParam - ? { - parallel_tool_calls: false, - } - : {}), - }, - ); - - const conversationHistoryStr = `Here is the full conversation history including the user's request(s): - -${state.internalMessages.map(getMessageString).join("\n")} - -${formatUserRequestPrompt(state.internalMessages)} - -Take all of this information, and determine whether or not you have completed this task in the plan. -Once you've determined the status of the current task, call either the \`mark_task_completed\` or \`mark_task_not_completed\` tool.`; - - const activePlanItems = getActivePlanItems(state.taskPlan); - - const response = await modelWithTools.invoke([ - { - role: "system", - content: formatPrompt(activePlanItems), - }, - { - role: "user", - content: conversationHistoryStr, - }, - ]); - const toolCall = response.tool_calls?.[0]; - - if (!toolCall) { - throw new Error( - "Failed to generate a tool call when checking task status.", - ); - } - - const isCompleted = toolCall.name === markCompletedTool.name; - const currentTask = getCurrentPlanItem(activePlanItems); - const toolMessage = new ToolMessage({ - id: uuidv4(), - tool_call_id: toolCall.id ?? "", - content: `Saved task status as ${isCompleted ? "completed" : "not completed"} for task ${currentTask?.plan || "unknown"}`, - name: toolCall.name, - }); - - const newMessages = [response, toolMessage]; - - const newMessageList = [...state.internalMessages, ...newMessages]; - const wouldBeConversationHistoryToSummarize = - await getMessagesSinceLastSummary(newMessageList, { - excludeHiddenMessages: true, - excludeCountFromEnd: 20, - }); - const totalInternalTokenCount = calculateConversationHistoryTokenCount( - wouldBeConversationHistoryToSummarize, - { - // Retain the last 20 messages from state - excludeHiddenMessages: true, - excludeCountFromEnd: 20, - }, - ); - - if (!isCompleted) { - logger.info("Current task has not been completed.", { - reasoning: toolCall.args.reasoning, - }); - const commandUpdate: GraphUpdate = { - messages: newMessages, - internalMessages: newMessages, - tokenData: trackCachePerformance(response), - }; - - // Check if we have any messages to summarize, and if we're at or above the max token limit. - if (totalInternalTokenCount >= MAX_INTERNAL_TOKENS) { - logger.info( - "Internal messages list is at or above the max token limit. Routing to summarize history step.", - { - totalInternalTokenCount, - maxInternalTokenCount: MAX_INTERNAL_TOKENS, - wouldBeConversationHistoryToSummarizeLength: - wouldBeConversationHistoryToSummarize.length, - }, - ); - return new Command({ - goto: "summarize-history", - update: commandUpdate, - }); - } - - return new Command({ - goto: "generate-action", - update: commandUpdate, - }); - } - const summary = (toolCall.args as z.infer) - .completed_task_summary; - - // LLM marked as completed, so we need to update the plan to reflect that. - const updatedPlanTasks = completePlanItem( - state.taskPlan, - getActiveTask(state.taskPlan).id, - currentTask.index, - summary, - ); - // Update the github issue to reflect this task as completed. - await addTaskPlanToIssue( - { - githubIssueId: state.githubIssueId, - targetRepository: state.targetRepository, - }, - config, - updatedPlanTasks, - ); - - // This should in theory never happen, but ensure we route properly if it does. - const remainingTask = getRemainingPlanItems(activePlanItems)?.[0]; - if (!remainingTask) { - logger.info( - "Found no remaining tasks in the plan during the check plan step. Continuing to the conclusion generation step.", - ); - const commandUpdate: GraphUpdate = { - messages: newMessages, - internalMessages: newMessages, - // Even though there are no remaining tasks, still mark as completed so the UI reflects that the task is completed. - taskPlan: updatedPlanTasks, - tokenData: trackCachePerformance(response), - }; - return new Command({ - goto: "route-to-review-or-conclusion", - update: commandUpdate, - }); - } - - logger.info("Task marked as completed. Routing to task summarization step.", { - remainingTask: { - ...remainingTask, - completed: true, - }, - }); - - const commandUpdate: GraphUpdate = { - messages: newMessages, - internalMessages: newMessages, - taskPlan: updatedPlanTasks, - tokenData: trackCachePerformance(response), - }; - - if (totalInternalTokenCount >= MAX_INTERNAL_TOKENS) { - logger.info( - "Internal messages list is at or above the max token limit. Routing to summarize history step.", - { - totalInternalTokenCount, - maxInternalTokenCount: MAX_INTERNAL_TOKENS, - }, - ); - return new Command({ - goto: "summarize-history", - update: commandUpdate, - }); - } - - return new Command({ - goto: "generate-action", - update: commandUpdate, - }); -} diff --git a/apps/open-swe/src/graphs/programmer/nodes/take-action.ts b/apps/open-swe/src/graphs/programmer/nodes/take-action.ts index b63ca4f3..556a6f9e 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/take-action.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/take-action.ts @@ -211,7 +211,7 @@ export async function takeAction( }), }; return new Command({ - goto: shouldRouteDiagnoseNode ? "diagnose-error" : "progress-plan-step", + goto: shouldRouteDiagnoseNode ? "diagnose-error" : "generate-action", update: commandUpdate, }); } diff --git a/apps/open-swe/src/utils/plan-prompt.ts b/apps/open-swe/src/utils/plan-prompt.ts index 0b1e70db..12dfba3c 100644 --- a/apps/open-swe/src/utils/plan-prompt.ts +++ b/apps/open-swe/src/utils/plan-prompt.ts @@ -1,14 +1,16 @@ import { PlanItem } from "@open-swe/shared/open-swe/types"; -export const PLAN_PROMPT = `## Completed Tasks -{COMPLETED_TASKS} +export const PLAN_PROMPT = ` + {COMPLETED_TASKS} + -## Remaining Tasks -(This list does not include the current task) -{REMAINING_TASKS} + + (This list does not include the current task) + {REMAINING_TASKS} + -## Current Task -{CURRENT_TASK}`; + {CURRENT_TASK} +`; /** * Formats a plan for use in a prompt. @@ -50,7 +52,7 @@ export function formatPlanPrompt( : completedTasks .map( (task) => - `${task.plan}`, + `\n${task.plan}\n`, ) .join("\n") : "No completed tasks.", @@ -61,14 +63,14 @@ export function formatPlanPrompt( ? remainingTasks .map( (task) => - `${task.plan}`, + `\n${task.plan}\n`, ) .join("\n") : "No remaining tasks.", ) .replace( "{CURRENT_TASK}", - `${currentTask?.plan || "No current task found."}`, + `\n${currentTask?.plan || "No current task found."}\n`, ); } @@ -76,7 +78,7 @@ export function formatPlanPromptWithSummaries(taskPlan: PlanItem[]): string { return taskPlan .map( (p) => - `<${p.completed ? "completed-" : ""}task index="${p.index}">\n${p.plan}\n \n${p.summary || "No task summary found"}\n \n`, + `<${p.completed ? "completed_" : ""}task index="${p.index}">\n${p.plan}\n \n${p.summary || "No task summary found"}\n \n`, ) .join("\n"); } diff --git a/apps/web/src/components/gen-ui/action-step.tsx b/apps/web/src/components/gen-ui/action-step.tsx index 54354f76..4d5c700f 100644 --- a/apps/web/src/components/gen-ui/action-step.tsx +++ b/apps/web/src/components/gen-ui/action-step.tsx @@ -622,9 +622,9 @@ export function ActionStep(props: ActionStepProps) { {showReasoning ? "Hide reasoning" : "Show reasoning"} {showReasoning && ( -

+ {reasoningText || "No reasoning provided."} -

+ )} diff --git a/apps/web/src/components/gen-ui/pull-request-opened.tsx b/apps/web/src/components/gen-ui/pull-request-opened.tsx index 971a4e8e..46981967 100644 --- a/apps/web/src/components/gen-ui/pull-request-opened.tsx +++ b/apps/web/src/components/gen-ui/pull-request-opened.tsx @@ -120,9 +120,9 @@ export function PullRequestOpened({

Description

-
+
             {description}
-          
+ )} diff --git a/apps/web/src/components/gen-ui/push-changes.tsx b/apps/web/src/components/gen-ui/push-changes.tsx index 165a496c..7c7d7fc6 100644 --- a/apps/web/src/components/gen-ui/push-changes.tsx +++ b/apps/web/src/components/gen-ui/push-changes.tsx @@ -81,9 +81,9 @@ export function PushChanges({ {showReasoning ? "Hide reasoning" : "Show reasoning"} {showReasoning && ( -

+ {reasoningText} -

+ )} )} diff --git a/apps/web/src/components/gen-ui/replanning-step.tsx b/apps/web/src/components/gen-ui/replanning-step.tsx index 982ddd69..0af524ec 100644 --- a/apps/web/src/components/gen-ui/replanning-step.tsx +++ b/apps/web/src/components/gen-ui/replanning-step.tsx @@ -61,9 +61,9 @@ export function ReplanningStep({ {showReasoning ? "Hide reasoning" : "Show reasoning"} {showReasoning && ( -

+ {reasoningText} -

+ )} )} diff --git a/apps/web/src/components/gen-ui/task-review.tsx b/apps/web/src/components/gen-ui/task-review.tsx index 84df9505..4be30231 100644 --- a/apps/web/src/components/gen-ui/task-review.tsx +++ b/apps/web/src/components/gen-ui/task-review.tsx @@ -70,9 +70,9 @@ export function MarkTaskCompleted({ {showReasoning ? "Hide reasoning" : "Show reasoning"} {showReasoning && ( -

+ {reasoningText} -

+ )} )} @@ -194,9 +194,9 @@ export function MarkTaskIncomplete({ {showReasoning ? "Hide reasoning" : "Show reasoning"} {showReasoning && ( -

+ {reasoningText} -

+ )} )} diff --git a/apps/web/src/components/tasks/index.tsx b/apps/web/src/components/tasks/index.tsx index b693700b..6758e061 100644 --- a/apps/web/src/components/tasks/index.tsx +++ b/apps/web/src/components/tasks/index.tsx @@ -23,6 +23,7 @@ import { import { cn } from "@/lib/utils"; import { ProgressBar } from "./progress-bar"; import { PlanItem, TaskPlan } from "@open-swe/shared/open-swe/types"; +import { BasicMarkdownText } from "../thread/markdown-text"; interface TasksSidebarProps { isOpen: boolean; @@ -327,7 +328,7 @@ export function TasksSidebar({ <>

- {item.plan} + {item.plan}

@@ -384,7 +385,9 @@ export function TasksSidebar({
- {item.summary} + + {item.summary} +
diff --git a/apps/web/src/components/thread/markdown-text.tsx b/apps/web/src/components/thread/markdown-text.tsx index 44d97c1e..69692cd6 100644 --- a/apps/web/src/components/thread/markdown-text.tsx +++ b/apps/web/src/components/thread/markdown-text.tsx @@ -289,3 +289,109 @@ const BasicMarkdownTextImpl: FC<{ children: string; className?: string }> = ({ }; export const BasicMarkdownText = memo(BasicMarkdownTextImpl); + +const InlineMarkdownTextImpl: FC<{ children: string; className?: string }> = ({ + children, + className, +}) => { + const inlineMarkdownComponents: any = { + // Only include inline elements + strong: ({ className, ...props }: { className?: string }) => ( + + ), + em: ({ className, ...props }: { className?: string }) => ( + + ), + code: ({ + className, + children, + ...props + }: { + className?: string; + children?: React.ReactNode; + }) => { + // Only render inline code, not code blocks + const match = /language-(\w+)/.exec(className || ""); + if (match) { + // If it's a code block, render as plain text to keep it inline + return ( + + {String(children)} + + ); + } + return ( + + {children} + + ); + }, + a: ({ className, ...props }: { className?: string }) => ( + + ), + del: ({ className, ...props }: { className?: string }) => ( + + ), + // Remove all block-level elements by not including them + // This will cause them to render as plain text + }; + + return ( + + + {children} + + + ); +}; + +export const InlineMarkdownText = memo(InlineMarkdownTextImpl); diff --git a/apps/web/src/components/ui/tabs.tsx b/apps/web/src/components/ui/tabs.tsx index 469a958d..3cb77034 100644 --- a/apps/web/src/components/ui/tabs.tsx +++ b/apps/web/src/components/ui/tabs.tsx @@ -42,7 +42,7 @@ function TabsTrigger({ {showReasoning && ( -

+ {reasoningText} -

+ )} )} diff --git a/apps/web/src/components/v2/thread-card.tsx b/apps/web/src/components/v2/thread-card.tsx index 6ab98f1f..d162f9c6 100644 --- a/apps/web/src/components/v2/thread-card.tsx +++ b/apps/web/src/components/v2/thread-card.tsx @@ -19,6 +19,7 @@ import { ThreadUIStatus } from "@/lib/schemas/thread-status"; import { cn } from "@/lib/utils"; import { TaskPlan } from "@open-swe/shared/open-swe/types"; import { getActivePlanItems } from "@open-swe/shared/open-swe/tasks"; +import { InlineMarkdownText } from "../thread/markdown-text"; interface ThreadCardProps { thread: ThreadMetadata; @@ -133,10 +134,10 @@ export function ThreadCard({ }} > -
+
- - {thread.title} + + {thread.title}
diff --git a/apps/web/src/components/v2/thread-view.tsx b/apps/web/src/components/v2/thread-view.tsx index abd84858..d55353b2 100644 --- a/apps/web/src/components/v2/thread-view.tsx +++ b/apps/web/src/components/v2/thread-view.tsx @@ -124,7 +124,7 @@ export function ThreadView({ programmerSession.runId !== joinedProgrammerRunId.current ) { joinedProgrammerRunId.current = programmerSession.runId; - plannerStream.joinStream(programmerSession.runId).catch(console.error); + programmerStream.joinStream(programmerSession.runId).catch(console.error); } else if (!programmerSession?.runId) { joinedProgrammerRunId.current = undefined; } @@ -311,19 +311,9 @@ export function ThreadView({ } >
- - - Planner - - - Programmer - + + Planner + Programmer {programmerTaskPlan && ( @@ -333,7 +323,7 @@ export function ThreadView({ /> )} -
+
{selectedTab === "planner" && plannerStream.isLoading && ( @@ -410,7 +400,7 @@ export function ThreadView({ diff --git a/apps/web/src/components/v2/token-usage.tsx b/apps/web/src/components/v2/token-usage.tsx index 407ca974..935d15ca 100644 --- a/apps/web/src/components/v2/token-usage.tsx +++ b/apps/web/src/components/v2/token-usage.tsx @@ -41,6 +41,15 @@ export function TokenUsage({ tokenData }: TokenUsageProps) { if (!tokenData || tokenData.length === 0) return null; const mergedTokenData = mergeTokenData(tokenData); + const totalCachedInputTokens = + mergedTokenData.cacheCreationInputTokens + + mergedTokenData.cacheReadInputTokens; + const totalUncachedInputTokens = mergedTokenData.inputTokens; + const cachePercentage = ( + (totalCachedInputTokens / + (totalCachedInputTokens + totalUncachedInputTokens)) * + 100 + ).toFixed(2); const metrics = calculateCostSavings(mergedTokenData); return ( @@ -66,7 +75,7 @@ export function TokenUsage({ tokenData }: TokenUsageProps) {
- + Input @@ -77,7 +86,7 @@ export function TokenUsage({ tokenData }: TokenUsageProps) {
- + Output @@ -103,7 +112,7 @@ export function TokenUsage({ tokenData }: TokenUsageProps) {
- + Cost @@ -114,17 +123,30 @@ export function TokenUsage({ tokenData }: TokenUsageProps) {
{metrics.totalSavings > 0 && ( -
- - Cache Savings - - - -${metrics.totalSavings.toFixed(2)} - -
+ <> +
+ + Cache Percentage + + + {cachePercentage}% + +
+
+ + Cache Savings + + + -${metrics.totalSavings.toFixed(2)} + +
+ )}