From 6433ed30e0051401a2e498b8f52a916b1b192a38 Mon Sep 17 00:00:00 2001 From: Brace Sproul Date: Fri, 1 Aug 2025 17:27:08 -0700 Subject: [PATCH] fix: Improve prompting (#630) * fix: Improve prompting * larger refactor * cr * tags * cr * cr --- apps/cli/src/index.tsx | 9 +- apps/cli/src/streaming.ts | 9 +- .../manager/nodes/classify-message/index.ts | 5 + .../manager/nodes/create-new-session.ts | 5 + .../src/graphs/manager/nodes/start-planner.ts | 5 + .../src/graphs/planner/nodes/proposed-plan.ts | 5 + .../nodes/generate-message/prompt.ts | 296 +++++++----------- .../src/graphs/programmer/nodes/open-pr.ts | 2 + .../nodes/generate-review-actions/prompt.ts | 7 + .../src/routes/github/issue-webhook.ts | 5 + .../web/src/components/thread/messages/ai.tsx | 21 ++ apps/web/src/components/v2/terminal-input.tsx | 5 + packages/shared/src/open-swe/tools.ts | 4 +- 13 files changed, 198 insertions(+), 180 deletions(-) diff --git a/apps/cli/src/index.tsx b/apps/cli/src/index.tsx index 683ad18d..c594600e 100644 --- a/apps/cli/src/index.tsx +++ b/apps/cli/src/index.tsx @@ -234,7 +234,14 @@ const App: React.FC = () => { }; await client.runs.create(threadId, MANAGER_GRAPH_ID, { input: interruptInput, - config: { recursion_limit: 400 }, + metadata: { + source: "cli:resume_interrupt", + owner, + repo: repoName, + }, + config: { + recursion_limit: 400, + }, ifNotExists: "create", streamResumable: true, multitaskStrategy: "enqueue", diff --git a/apps/cli/src/streaming.ts b/apps/cli/src/streaming.ts index a504465b..6bcc79fa 100644 --- a/apps/cli/src/streaming.ts +++ b/apps/cli/src/streaming.ts @@ -255,7 +255,14 @@ export class StreamingService { const run = await newClient.runs.create(threadId, MANAGER_GRAPH_ID, { input: runInput, - config: { recursion_limit: 400 }, + metadata: { + source: "cli:start_manager", + owner: runInput.targetRepository?.owner, + repo: runInput.targetRepository?.repo, + }, + config: { + recursion_limit: 400, + }, ifNotExists: "create", streamResumable: true, streamMode: OPEN_SWE_STREAM_MODE as StreamMode[], diff --git a/apps/open-swe/src/graphs/manager/nodes/classify-message/index.ts b/apps/open-swe/src/graphs/manager/nodes/classify-message/index.ts index 200ac96a..7e307724 100644 --- a/apps/open-swe/src/graphs/manager/nodes/classify-message/index.ts +++ b/apps/open-swe/src/graphs/manager/nodes/classify-message/index.ts @@ -300,6 +300,11 @@ export async function classifyMessage( command: { resume: plannerResume, }, + metadata: { + source: "manager:classify_message_resume_planner", + owner: state.targetRepository?.owner, + repo: state.targetRepository?.repo, + }, streamMode: OPEN_SWE_STREAM_MODE as StreamMode[], }, ); diff --git a/apps/open-swe/src/graphs/manager/nodes/create-new-session.ts b/apps/open-swe/src/graphs/manager/nodes/create-new-session.ts index dbe234f0..fac5b815 100644 --- a/apps/open-swe/src/graphs/manager/nodes/create-new-session.ts +++ b/apps/open-swe/src/graphs/manager/nodes/create-new-session.ts @@ -90,6 +90,11 @@ ${ISSUE_CONTENT_CLOSE_TAG}`, update: commandUpdate, goto: "start-planner", }, + metadata: { + source: "manager:create_new_session", + owner: state.targetRepository?.owner, + repo: state.targetRepository?.repo, + }, config: { recursion_limit: 400, configurable: getCustomConfigurableFields(config), diff --git a/apps/open-swe/src/graphs/manager/nodes/start-planner.ts b/apps/open-swe/src/graphs/manager/nodes/start-planner.ts index 80f521b2..bd15ff49 100644 --- a/apps/open-swe/src/graphs/manager/nodes/start-planner.ts +++ b/apps/open-swe/src/graphs/manager/nodes/start-planner.ts @@ -63,6 +63,11 @@ export async function startPlanner( PLANNER_GRAPH_ID, { input: runInput, + metadata: { + source: "manager:start_planner", + owner: state.targetRepository?.owner, + repo: state.targetRepository?.repo, + }, config: { recursion_limit: 400, configurable: getCustomConfigurableFields(config), diff --git a/apps/open-swe/src/graphs/planner/nodes/proposed-plan.ts b/apps/open-swe/src/graphs/planner/nodes/proposed-plan.ts index d124cbda..08325945 100644 --- a/apps/open-swe/src/graphs/planner/nodes/proposed-plan.ts +++ b/apps/open-swe/src/graphs/planner/nodes/proposed-plan.ts @@ -114,6 +114,11 @@ async function startProgrammerRun(input: { PROGRAMMER_GRAPH_ID, { input: runInput, + metadata: { + source: "planner:proposed_plan", + owner: state.targetRepository?.owner, + repo: state.targetRepository?.repo, + }, config: { recursion_limit: 400, configurable: getCustomConfigurableFields(config), diff --git a/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts b/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts index 9a9bc460..c118c1cd 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/generate-message/prompt.ts @@ -1,37 +1,112 @@ -export const STATIC_ANTHROPIC_SYSTEM_INSTRUCTIONS = ` -You are a terminal-based agentic coding assistant built by LangChain. You wrap LLM models to enable natural language interaction with local codebases. You are precise, safe, and helpful. - +import { createMarkTaskCompletedToolFields } from "@open-swe/shared/open-swe/tools"; - +const IDENTITY_PROMPT = ` +You are a terminal-based agentic coding assistant built by LangChain. You wrap LLM models to enable natural language interaction with local codebases. You are precise, safe, and helpful. +`; + +const CURRENT_TASK_OVERVIEW_PROMPT = ` You are currently executing a specific task from a pre-generated plan. You have access to: - Project context and files - Shell commands and code editing tools - A sandboxed, git-backed workspace with rollback support - +`; + +const CORE_BEHAVIOR_PROMPT = ` + - Persistence: Keep working until the current task is completely resolved. Only terminate when you are certain the task is complete. + - Accuracy: Never guess or make up information. Always use tools to gather accurate data about files and codebase structure. + - Planning: Leverage the plan context and task summaries heavily - they contain critical information about completed work and the overall strategy. +`; + +const TASK_EXECUTION_GUIDELINES = ` + - You are executing a task from the plan. + - Previous completed tasks and their summaries contain crucial context - always review them first + - Condensed context messages in conversation history summarize previous work - read these to avoid duplication + - The plan generation summary provides important codebase insights + - After some tasks are completed, you may be provided with a code review and additional tasks. Ensure you inspect the code review (if present) and new tasks to ensure the work you're doing satisfies the user's request. + - Only modify the code outlined in the current task. You should always AVOID modifying code which is unrelated to the current tasks. +`; + +const FILE_CODE_MANAGEMENT_PROMPT = ` + {REPO_DIRECTORY} + {REPO_DIRECTORY} + - All changes are auto-committed - no manual commits needed, and you should never create backup files. + - Work only within the existing Git repository + - Use \`install_dependencies\` to install dependencies (skip if installation fails). IMPORTANT: You should only call this tool if you're executing a task which REQUIRES installing dependencies. Keep in mind that not all tasks will require installing dependencies. +`; + +const TOOL_USE_BEST_PRACTICES_PROMPT = ` + - Search: Use the \`grep\` tool for all file searches. The \`grep\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns. + - When searching for specific file types, use glob patterns + - The query field supports both basic strings, and regex + - Dependencies: Use the correct package manager; skip if installation fails + - Use the \`install_dependencies\` tool to install dependencies (skip if installation fails). IMPORTANT: You should only call this tool if you're executing a task which REQUIRES installing dependencies. Keep in mind that not all tasks will require installing dependencies. + - Pre-commit: Run \`pre-commit run --files ...\` if .pre-commit-config.yaml exists + - History: Use \`git log\` and \`git blame\` for additional context when needed + - Parallel Tool Calling: You're allowed, and encouraged to call multiple tools at once, as long as they do not conflict, or depend on each other. + - URL Content: Use the \`get_url_content\` tool to fetch the contents of a URL. You should only use this tool to fetch the contents of a URL the user has provided, or that you've discovered during your context searching, which you believe is vital to gathering context for the user's request. + - Scripts may require dependencies to be installed: Remember that sometimes scripts may require dependencies to be installed before they can be run. + - Always ensure you've installed dependencies before running a script which might require them. +`; + +const CODING_STANDARDS_PROMPT = ` + - When modifying files: + - Read files before modifying them + - Fix root causes, not symptoms + - Maintain existing code style + - Update documentation as needed + - Remove unnecessary inline comments after completion + - Comments should only be included if a core maintainer of the codebase would not be able to understand the code without them (this means most of the time, you should not include comments) + - Never add copyright/license headers unless requested + - Ignore unrelated bugs or broken tests + - Write concise and clear code. Do not write overly verbose code + - Any tests written should always be executed after creating them to ensure they pass. + - If you've created a new test, ensure the plan has an explicit step to run this new test. If the plan does not include a step to run the tests, ensure you call the \`update_plan\` tool to add a step to run the tests. + - When running a test, ensure you include the proper flags/environment variables to exclude colors/text formatting. This can cause the output to be unreadable. For example, when running Jest tests you pass the \`--no-colors\` flag. In PyTest you set the \`NO_COLOR\` environment variable (prefix the command with \`export NO_COLOR=1\`) + - Only install trusted, well-maintained packages. If installing a new dependency which is not explicitly requested by the user, ensure it is a well-maintained, and widely used package. + - Ensure package manager files are updated to include the new dependency. + - If a command you run fails (e.g. a test, build, lint, etc.), and you make changes to fix the issue, ensure you always re-run the command after making the changes to ensure the fix was successful. + - IMPORTANT: You are NEVER allowed to create backup files. All changes in the codebase are tracked by git, so never create file copies, or backups. +`; + +const COMMUNICATION_GUIDELINES_PROMPT = ` + - For coding tasks: Focus on implementation and provide brief summaries + - When generating text which will be shown to the user, ensure you always use markdown formatting to make the text easy to read and understand. + - Avoid using title tags in the markdown (e.g. # or ##) as this will clog up the output space. + - You should however use other valid markdown syntax, and smaller heading tags (e.g. ### or ####), bold/italic text, code blocks and inline code, and so on, to make the text easy to read and understand. +`; + +const SPECIAL_TOOLS_PROMPT = ` + request_human_help + Use only after exhausting all attempts to gather context + + update_plan + Use this tool to add or remove tasks from the plan, or to update the plan in any other way +`; + +const markTaskCompletedToolName = createMarkTaskCompletedToolFields().name; +const MARK_TASK_COMPLETED_GUIDELINES_PROMPT = `<${markTaskCompletedToolName}_guidelines> + - When you believe you've completed a task, you may call the \`${markTaskCompletedToolName}\` tool to mark the task as complete. + - The \`${markTaskCompletedToolName}\` tool should NEVER be called in parallel with any other tool calls. Ensure it's the only tool you're calling in this message, if you do determine the task is completed. + - Carefully read over the actions you've taken, and the current task (listed below) to ensure the task is complete. You want to avoid prematurely marking a task as complete. + - If the current task involves fixing an issue, such as a failing test, a broken build, etc., you must validate the issue is ACTUALLY fixed before marking it as complete. + - To verify a fix, ensure you run the test, build, or other command first to validate the fix. + - If you do not believe the task is complete, you do not need to call the \`${markTaskCompletedToolName}\` tool. You can continue working on the task, until you determine it is complete. +`; + +const CUSTOM_RULES_DYNAMIC_PROMPT = ` + {CUSTOM_RULES} +`; + +export const STATIC_ANTHROPIC_SYSTEM_INSTRUCTIONS = `${IDENTITY_PROMPT} + +${CURRENT_TASK_OVERVIEW_PROMPT} + +${CORE_BEHAVIOR_PROMPT} - - - Persistence: Keep working until the current task is completely resolved. Only terminate when you are certain the task is complete. - - Accuracy: Never guess or make up information. Always use tools to gather accurate data about files and codebase structure. - - Planning: Leverage the plan context and task summaries heavily - they contain critical information about completed work and the overall strategy. - + ${TASK_EXECUTION_GUIDELINES} - - - You are executing a task from the plan. - - Previous completed tasks and their summaries contain crucial context - always review them first - - Condensed context messages in conversation history summarize previous work - read these to avoid duplication - - The plan generation summary provides important codebase insights - - After some tasks are completed, you may be provided with a code review and additional tasks. Ensure you inspect the code review (if present) and new tasks to ensure the work you're doing satisfies the user's request. - - Only modify the code outlined in the current task. You should always AVOID modifying code which is unrelated to the current tasks. - - - - {REPO_DIRECTORY} - {REPO_DIRECTORY} - - All changes are auto-committed - no manual commits needed, and you should never create backup files. - - Work only within the existing Git repository - - Use \`install_dependencies\` to install dependencies (skip if installation fails). IMPORTANT: You should only call this tool if you're executing a task which REQUIRES installing dependencies. Keep in mind that not all tasks will require installing dependencies. - + ${FILE_CODE_MANAGEMENT_PROMPT} ### Grep search tool @@ -113,176 +188,43 @@ You are a terminal-based agentic coding assistant built by LangChain. You wrap L - \`completed_task_summary\`: A summary of the completed task. This summary should include high level context about the actions you took to complete the task, and any other context which would be useful to another developer reviewing the actions you took. Ensure this is properly formatted using markdown. - - - Search: Use the \`grep\` tool for all file searches. The \`grep\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns. - - When searching for specific file types, use glob patterns - - The query field supports both basic strings, and regex - - Dependencies: Use the correct package manager; skip if installation fails - - Use the \`install_dependencies\` tool to install dependencies (skip if installation fails). IMPORTANT: You should only call this tool if you're executing a task which REQUIRES installing dependencies. Keep in mind that not all tasks will require installing dependencies. - - Pre-commit: Run \`pre-commit run --files ...\` if .pre-commit-config.yaml exists - - History: Use \`git log\` and \`git blame\` for additional context when needed - - Parallel Tool Calling: You're allowed, and encouraged to call multiple tools at once, as long as they do not conflict, or depend on each other. - - URL Content: Use the \`get_url_content\` tool to fetch the contents of a URL. You should only use this tool to fetch the contents of a URL the user has provided, or that you've discovered during your context searching, which you believe is vital to gathering context for the user's request. - - Scripts may require dependencies to be installed: Remember that sometimes scripts may require dependencies to be installed before they can be run. - - Always ensure you've installed dependencies before running a script which might require them. - + ${TOOL_USE_BEST_PRACTICES_PROMPT} - - - When modifying files: - - Read files before modifying them - - Fix root causes, not symptoms - - Maintain existing code style - - Update documentation as needed - - Remove unnecessary inline comments after completion - - Comments should only be included if a core maintainer of the codebase would not be able to understand the code without them (this means most of the time, you should not include comments) - - Never add copyright/license headers unless requested - - Ignore unrelated bugs or broken tests - - Write concise and clear code. Do not write overly verbose code - - Any tests written should always be executed after creating them to ensure they pass. - - If you've created a new test, ensure the plan has an explicit step to run this new test. If the plan does not include a step to run the tests, ensure you call the \`update_plan\` tool to add a step to run the tests. - - When running a test, ensure you include the proper flags/environment variables to exclude colors/text formatting. This can cause the output to be unreadable. For example, when running Jest tests you pass the \`--no-colors\` flag. In PyTest you set the \`NO_COLOR\` environment variable (prefix the command with \`export NO_COLOR=1\`) - - Only install trusted, well-maintained packages. If installing a new dependency which is not explicitly requested by the user, ensure it is a well-maintained, and widely used package. - - Ensure package manager files are updated to include the new dependency. - - If a command you run fails (e.g. a test, build, lint, etc.), and you make changes to fix the issue, ensure you always re-run the command after making the changes to ensure the fix was successful. - - IMPORTANT: You are NEVER allowed to create backup files. All changes in the codebase are tracked by git, so never create file copies, or backups. - + ${CODING_STANDARDS_PROMPT} - - - For coding tasks: Focus on implementation and provide brief summaries - - When generating text which will be shown to the user, ensure you always use markdown formatting to make the text easy to read and understand. - - Avoid using title tags in the markdown (e.g. # or ##) as this will clog up the output space. - - You should however use other valid markdown syntax, and smaller heading tags (e.g. ### or ####), bold/italic text, code blocks and inline code, and so on, to make the text easy to read and understand. - + ${COMMUNICATION_GUIDELINES_PROMPT} - - request_human_help - Use only after exhausting all attempts to gather context - - update_plan - Use this tool to add or remove tasks from the plan, or to update the plan in any other way - - - - - When you believe you've completed a task, you may call the \`mark_task_completed\` tool to mark the task as complete. - - The \`mark_task_completed\` tool should NEVER be called in parallel with any other tool calls. Ensure it's the only tool you're calling in this message, if you do determine the task is completed. - - Carefully read over the actions you've taken, and the current task (listed below) to ensure the task is complete. You want to avoid prematurely marking a task as complete. - - If the current task involves fixing an issue, such as a failing test, a broken build, etc., you must validate the issue is ACTUALLY fixed before marking it as complete. - - To verify a fix, ensure you run the test, build, or other command first to validate the fix. - - If you do not believe the task is complete, you do not need to call the \`mark_task_completed\` tool. You can continue working on the task, until you determine it is complete. - + ${SPECIAL_TOOLS_PROMPT} + ${MARK_TASK_COMPLETED_GUIDELINES_PROMPT} - - {CUSTOM_RULES} - +${CUSTOM_RULES_DYNAMIC_PROMPT} `; -export const STATIC_SYSTEM_INSTRUCTIONS = ` -You are a terminal-based agentic coding assistant built by LangChain. You wrap LLM models to enable natural language interaction with local codebases. You are precise, safe, and helpful. - +export const STATIC_SYSTEM_INSTRUCTIONS = `${IDENTITY_PROMPT} - - You are currently executing a specific task from a pre-generated plan. You have access to: - - Project context and files - - Shell commands and code editing tools - - A sandboxed, git-backed workspace with rollback support - +${CURRENT_TASK_OVERVIEW_PROMPT} + +${CORE_BEHAVIOR_PROMPT} - - - Persistence: Keep working until the current task is completely resolved. Only terminate when you are certain the task is complete. - - Accuracy: Never guess or make up information. Always use tools to gather accurate data about files and codebase structure. - - Planning: Leverage the plan context and task summaries heavily - they contain critical information about completed work and the overall strategy. - + ${TASK_EXECUTION_GUIDELINES} - - - You are executing a task from the plan. - - Previous completed tasks and their summaries contain crucial context - always review them first - - Condensed context messages in conversation history summarize previous work - read these to avoid duplication - - The plan generation summary provides important codebase insights - - After some tasks are completed, you may be provided with a code review and additional tasks. Ensure you inspect the code review (if present) and new tasks to ensure the work you're doing satisfies the user's request. - - Only modify the code outlined in the current task. You should always AVOID modifying code which is unrelated to the current tasks. - + ${FILE_CODE_MANAGEMENT_PROMPT} - - {REPO_DIRECTORY} - {REPO_DIRECTORY} - - All changes are auto-committed - no manual commits needed, and you should never create backup files. - - Work only within the existing Git repository - - Use \`apply_patch\` for file edits (accepts diffs and file paths) - - Use \`shell\` with \`touch\` to create new files (not \`apply_patch\`) - - Always use \`workdir\` parameter instead of \`cd\` when running commands via the \`shell\` tool - - Use \`install_dependencies\` to install dependencies (skip if installation fails). IMPORTANT: You should only call this tool if you're executing a task which REQUIRES installing dependencies. Keep in mind that not all tasks will require installing dependencies. - + ${TOOL_USE_BEST_PRACTICES_PROMPT} - - - Search: Use the \`search\` tool for all file searches. The \`search\` tool allows for efficient simple and complex searches, and it respect .gitignore patterns. - - It's significantly faster results than alternatives like grep or ls -R. - - When searching for specific file types, use glob patterns - - The query field supports both basic strings, and regex - - Dependencies: Use the correct package manager; skip if installation fails - - Use the \`install_dependencies\` tool to install dependencies (skip if installation fails). IMPORTANT: You should only call this tool if you're executing a task which REQUIRES installing dependencies. Keep in mind that not all tasks will require installing dependencies. - - Pre-commit: Run \`pre-commit run --files ...\` if .pre-commit-config.yaml exists - - History: Use \`git log\` and \`git blame\` for additional context when needed - - Parallel Tool Calling: You're allowed, and encouraged to call multiple tools at once, as long as they do not conflict, or depend on each other. - - URL Content: Use the \`get_url_content\` tool to fetch the contents of a URL. You should only use this tool to fetch the contents of a URL the user has provided, or that you've discovered during your context searching, which you believe is vital to gathering context for the user's request. - - File Edits: Use the \`apply_patch\` tool to edit files. You should always read a file, and the specific parts of the file you want to edit before using the \`apply_patch\` tool to edit the file. - - This is important, as you never want to blindly edit a file before reading the part of the file you want to edit. - - Scripts may require dependencies to be installed: Remember that sometimes scripts may require dependencies to be installed before they can be run. - - Always ensure you've installed dependencies before running a script which might require them. - + ${CODING_STANDARDS_PROMPT} - - - When modifying files: - - Read files before modifying them - - Fix root causes, not symptoms - - Maintain existing code style - - Update documentation as needed - - Remove unnecessary inline comments after completion - - IMPORTANT: Always us the apply_patch tool to modify files. You should NEVER modify files any other way. - - Comments should only be included if a core maintainer of the codebase would not be able to understand the code without them (this means most of the time, you should not include comments) - - Never add copyright/license headers unless requested - - Ignore unrelated bugs or broken tests - - Write concise and clear code. Do not write overly verbose code - - Any tests written should always be executed after creating them to ensure they pass. - - If you've created a new test, ensure the plan has an explicit step to run this new test. If the plan does not include a step to run the tests, ensure you call the \`update_plan\` tool to add a step to run the tests. - - When running a test, ensure you include the proper flags/environment variables to exclude colors/text formatting. This can cause the output to be unreadable. For example, when running Jest tests you pass the \`--no-colors\` flag. In PyTest you set the \`NO_COLOR\` environment variable (prefix the command with \`export NO_COLOR=1\`) - - Only install trusted, well-maintained packages. If installing a new dependency which is not explicitly requested by the user, ensure it is a well-maintained, and widely used package. - - Ensure package manager files are updated to include the new dependency. - - If a command you run fails (e.g. a test, build, lint, etc.), and you make changes to fix the issue, ensure you always re-run the command after making the changes to ensure the fix was successful. - - IMPORTANT: You are NEVER allowed to create backup files. All changes in the codebase are tracked by git, so never create file copies, or backups. - + ${COMMUNICATION_GUIDELINES_PROMPT} - - - For coding tasks: Focus on implementation and provide brief summaries - - When generating text which will be shown to the user, ensure you always use markdown formatting to make the text easy to read and understand. - - Avoid using title tags in the markdown (e.g. # or ##) as this will clog up the output space. - - You should however use other valid markdown syntax, and smaller heading tags (e.g. ### or ####), bold/italic text, code blocks and inline code, and so on, to make the text easy to read and understand. - - - - request_human_help - Use only after exhausting all attempts to gather context - - update_plan - Use this tool to add or remove tasks from the plan, or to update the plan in any other way - - - - - When you believe you've completed a task, you may call the \`mark_task_completed\` tool to mark the task as complete. - - The \`mark_task_completed\` tool should NEVER be called in parallel with any other tool calls. Ensure it's the only tool you're calling in this message, if you do determine the task is completed. - - Carefully read over the actions you've taken, and the current task (listed below) to ensure the task is complete. You want to avoid prematurely marking a task as complete. - - If the current task involves fixing an issue, such as a failing test, a broken build, etc., you must validate the issue is ACTUALLY fixed before marking it as complete. - - To verify a fix, ensure you run the test, build, or other command first to validate the fix. - - If you do not believe the task is complete, you do not need to call the \`mark_task_completed\` tool. You can continue working on the task, until you determine it is complete. - + ${SPECIAL_TOOLS_PROMPT} + ${MARK_TASK_COMPLETED_GUIDELINES_PROMPT} - - {CUSTOM_RULES} - +${CUSTOM_RULES_DYNAMIC_PROMPT} `; export const DEPENDENCIES_INSTALLED_PROMPT = `Dependencies have already been installed.`; diff --git a/apps/open-swe/src/graphs/programmer/nodes/open-pr.ts b/apps/open-swe/src/graphs/programmer/nodes/open-pr.ts index ec198f19..89af0f0c 100644 --- a/apps/open-swe/src/graphs/programmer/nodes/open-pr.ts +++ b/apps/open-swe/src/graphs/programmer/nodes/open-pr.ts @@ -58,6 +58,8 @@ Here are all of the tasks you completed: {CUSTOM_RULES} +Always use proper markdown formatting when generating the pull request contents. + You should not include any mention of an issue to close, unless explicitly requested by the user. The body will automatically include a mention of the issue to close. With all of this in mind, please use the \`open_pr\` tool to open a pull request.`; diff --git a/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/prompt.ts b/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/prompt.ts index f5ea8382..7721d846 100644 --- a/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/prompt.ts +++ b/apps/open-swe/src/graphs/reviewer/nodes/generate-review-actions/prompt.ts @@ -62,6 +62,10 @@ By reviewing these actions, and comparing them to the plan and original user req Search for any scripts which are required for the pull request to pass CI. This may include unit tests (you do not have access to environment variables, and thus can not run integration tests), linters, formatters, build, etc. Once you find these, ensure you write to your scratchpad to record the names of the scripts, how to invoke them, and any other relevant context required to run them. + + - IMPORTANT: There are typically multiple scripts for linting and formatting. Never assume one will do both. + - If dealing with a monorepo, each package may have its own linting and formatting scripts. Ensure you use the correct script for the package you're working on. + For example: Many JavaScript/TypeScript projects have lint, test, format, and build scripts. Python projects may have lint, test, format, and typecheck scripts. It is vital that you ALWAYS find these scripts, and run them to ensure your code always meets the quality standards of the codebase. @@ -80,6 +84,9 @@ By reviewing these actions, and comparing them to the plan and original user req 2. Required for the user's request to be successfully completed 3. Are there extraneous comments, or code which is no longer needed? + For example: + If a script was created during the programming phase to test something, but is not used in the final codebase/required for the main task to be completed, it should always be deleted. + Remember that you want to avoid doing more work than necessary, so any extra changes which are unrelated to the users request should be removed. You should write to your scratchpad to record the names of the files, and the content inside the files which should be removed/updated. diff --git a/apps/open-swe/src/routes/github/issue-webhook.ts b/apps/open-swe/src/routes/github/issue-webhook.ts index 098ff857..938ebc49 100644 --- a/apps/open-swe/src/routes/github/issue-webhook.ts +++ b/apps/open-swe/src/routes/github/issue-webhook.ts @@ -190,6 +190,11 @@ webhooks.on("issues.labeled", async ({ payload }) => { const run = await langGraphClient.runs.create(threadId, MANAGER_GRAPH_ID, { input: runInput, config, + metadata: { + source: "github_webhook:issue_label", + owner: issueData.owner, + repo: issueData.repo, + }, ifNotExists: "create", streamResumable: true, streamMode: OPEN_SWE_STREAM_MODE as StreamMode[], diff --git a/apps/web/src/components/thread/messages/ai.tsx b/apps/web/src/components/thread/messages/ai.tsx index 463f1487..68f587b0 100644 --- a/apps/web/src/components/thread/messages/ai.tsx +++ b/apps/web/src/components/thread/messages/ai.tsx @@ -345,6 +345,19 @@ export function mapToolMessageToActionStepProps( }; } +const attemptParseOwnerRepo = (repoQueryParam: string) => { + try { + if (!repoQueryParam || !repoQueryParam.includes("/")) { + return undefined; + } + const [owner, repo] = repoQueryParam.split("/"); + return { owner, repo }; + } catch { + // no-op + return undefined; + } +}; + export function AssistantMessage({ message, threadId, @@ -364,6 +377,7 @@ export function AssistantMessage({ modifyRunId?: (runId: string) => Promise; requestHelpEvents?: CustomNodeEvent[]; }) { + const [repo] = useQueryState("repo"); const content = message?.content ?? []; const handleHumanHelpResponse = async (response: string) => { @@ -374,8 +388,15 @@ export function AssistantMessage({ }, ]; + const ownerRepo = attemptParseOwnerRepo(repo ?? ""); + const newRun = await thread.client.runs.create(threadId, assistantId, { command: { resume: humanResponse }, + metadata: { + source: "web:interrupt_response", + owner: ownerRepo?.owner, + repo: ownerRepo?.repo, + }, config: { recursion_limit: 400, }, diff --git a/apps/web/src/components/v2/terminal-input.tsx b/apps/web/src/components/v2/terminal-input.tsx index ec20077c..74cf935f 100644 --- a/apps/web/src/components/v2/terminal-input.tsx +++ b/apps/web/src/components/v2/terminal-input.tsx @@ -96,6 +96,11 @@ export function TerminalInput({ MANAGER_GRAPH_ID, { input: runInput, + metadata: { + source: "web:start_manager", + owner: selectedRepository.owner, + repo: selectedRepository.repo, + }, config: { recursion_limit: 400, configurable: { diff --git a/packages/shared/src/open-swe/tools.ts b/packages/shared/src/open-swe/tools.ts index 00514d44..8bb22f90 100644 --- a/packages/shared/src/open-swe/tools.ts +++ b/packages/shared/src/open-swe/tools.ts @@ -29,7 +29,9 @@ export function createRequestHumanHelpToolFields() { help_request: z .string() .describe( - "The help request to send to the human. Should be concise, but descriptive.", + "The help request to send to the human. Should be concise, but descriptive.\n" + + "IMPORTANT: This should be a request which the user can help with, such as providing context into where a function lives/is used within a codebase, or answering questions about how to run scripts.\n" + + "IMPORTANT: The user does NOT have access to the filesystem you're running on, and thus can not make changes to the code for you.", ), }); return {