From a7f726df00bb0793f1764bc0023224d0372f1060 Mon Sep 17 00:00:00 2001 From: Brace Sproul Date: Tue, 5 Aug 2025 10:34:26 -0700 Subject: [PATCH] feat: Opus 4.1 support, switch default open-swe-max to Opus 4.1 (#678) --- README.md | 2 +- apps/docs/faq.md | 2 +- apps/docs/usage/best-practices.mdx | 6 +++--- apps/docs/usage/github.mdx | 2 +- .../src/graphs/manager/nodes/classify-message/prompts.ts | 4 ++-- apps/open-swe/src/routes/github/issue-webhook.ts | 6 +++--- packages/shared/src/open-swe/models.ts | 4 ++++ 7 files changed, 15 insertions(+), 11 deletions(-) diff --git a/README.md b/README.md index 0b235c62..0533dfa8 100644 --- a/README.md +++ b/README.md @@ -35,7 +35,7 @@ Open SWE is an open-source cloud-based asynchronous coding agent built with [Lan Open SWE can be used in multiple ways: - 🖥️ **From the UI**. You can create, manage and execute Open SWE tasks from the [web application](https://swe.langchain.com). See the ['From the UI' page](https://docs.langchain.com/labs/swe/usage/ui) in the docs for more information. -- 📝 **From GitHub**. You can start Open SWE tasks directly from GitHub issues simply by adding a label `open-swe`, or `open-swe-auto` (adding `-auto` will cause Open SWE to automatically accept the plan, requiring no intervention from you). For enhanced performance on complex tasks, use `open-swe-max` or `open-swe-max-auto` labels which utilize Claude Opus 4 for both planning and programming. See the ['From GitHub' page](https://docs.langchain.com/labs/swe/usage/github) in the docs for more information. +- 📝 **From GitHub**. You can start Open SWE tasks directly from GitHub issues simply by adding a label `open-swe`, or `open-swe-auto` (adding `-auto` will cause Open SWE to automatically accept the plan, requiring no intervention from you). For enhanced performance on complex tasks, use `open-swe-max` or `open-swe-max-auto` labels which utilize Claude Opus 4.1 for both planning and programming. See the ['From GitHub' page](https://docs.langchain.com/labs/swe/usage/github) in the docs for more information. # Documentation diff --git a/apps/docs/faq.md b/apps/docs/faq.md index a8a06cec..90052071 100644 --- a/apps/docs/faq.md +++ b/apps/docs/faq.md @@ -7,7 +7,7 @@ description: "Frequently Asked Questions" The cost per run varies greatly based on the complexity of the task, the size of the repository, and the number of files that need to be changed. For most tasks, you can expect to pay between `$0.50` -> `$3.00` when using Claude Sonnet 4. - For the same tasks running on Claude Opus 4, you can expect to pay between `$1.50` -> `$9.00`. + For the same tasks running on Claude Opus 4/4.1, you can expect to pay between `$1.50` -> `$9.00`. Always remember to monitor your runs if you're cost conscious. The most expensive run I've seen Open SWE complete was ~50M Opus 4 tokens, costing `$25.00`. diff --git a/apps/docs/usage/best-practices.mdx b/apps/docs/usage/best-practices.mdx index dca487ca..5fbe3503 100644 --- a/apps/docs/usage/best-practices.mdx +++ b/apps/docs/usage/best-practices.mdx @@ -40,7 +40,7 @@ Submit separate requests for different features or fixes. This allows Open SWE t ## Model Selection - **Claude Sonnet 4 (Default)**: The default model for planning, writing code, and reviewing changes. This model offers the best balance of performance, speed and cost. -- **Claude Opus 4**: A larger, more powerful model for difficult, or open-ended tasks. Opus 4 is more expensive and slower, but will provide better results for complex tasks. +- **Claude Opus 4.1**: A larger, more powerful model for difficult, or open-ended tasks. Opus 4.1 is more expensive and slower, but will provide better results for complex tasks. ### Avoid Other Models @@ -74,8 +74,8 @@ If you're running Open SWE against an open-ended or very complex task, you may w - `open-swe`: Manual mode with Sonnet 4 - `open-swe-auto`: Auto mode with Sonnet 4 -- `open-swe-max`: Manual mode with Opus 4 -- `open-swe-max-auto`: Auto mode with Opus 4 +- `open-swe-max`: Manual mode with Opus 4.1 +- `open-swe-max-auto`: Auto mode with Opus 4.1 In development environments, append `-dev` to all labels (e.g., diff --git a/apps/docs/usage/github.mdx b/apps/docs/usage/github.mdx index 85ae2283..d132ebeb 100644 --- a/apps/docs/usage/github.mdx +++ b/apps/docs/usage/github.mdx @@ -33,7 +33,7 @@ Open SWE supports three types of labels that control how the agent operates: **Max Mode (`open-swe-max` and `open-swe-max-auto`)** -- Uses Claude Opus 4 for both planning and programming tasks +- Uses Claude Opus 4.1 for both planning and programming tasks - Provides enhanced performance and reasoning capabilities for complex problems - `open-swe-max`: Requires manual plan approval with premium model performance - `open-swe-max-auto`: Combines automatic execution with premium model capabilities diff --git a/apps/open-swe/src/graphs/manager/nodes/classify-message/prompts.ts b/apps/open-swe/src/graphs/manager/nodes/classify-message/prompts.ts index b8be2ef7..3371584d 100644 --- a/apps/open-swe/src/graphs/manager/nodes/classify-message/prompts.ts +++ b/apps/open-swe/src/graphs/manager/nodes/classify-message/prompts.ts @@ -73,8 +73,8 @@ Your documentation is available at: https://docs.langchain.com/labs/swe You can be invoked by both the web app, or by adding a label to a GitHub issue. These label options are: - \`open-swe\` - trigger a standard Open SWE task. It will interrupt after generating a plan, and the user must approve it before it can continue. Uses Claude Sonnet 4 for all LLM requests. - \`open-swe-auto\` - trigger an 'auto' Open SWE task. It will not interrupt after generating a plan, and instead it will auto-approve the plan, and continue to the programming step without user approval. Uses Claude Sonnet 4 for all LLM requests. -- \`open-swe-max\` - this label acts the same as \`open-swe\`, except it uses a larger, more powerful model for the planning and programming steps: Claude Opus 4. It still uses Claude Sonnet 4 for the reviewer step. -- \`open-swe-max-auto\` - this label acts the same as \`open-swe-auto\`, except it uses a larger, more powerful model for the planning and programming steps: Claude Opus 4. It still uses Claude Sonnet 4 for the reviewer step. +- \`open-swe-max\` - this label acts the same as \`open-swe\`, except it uses a larger, more powerful model for the planning and programming steps: Claude Opus 4.1. It still uses Claude Sonnet 4 for the reviewer step. +- \`open-swe-max-auto\` - this label acts the same as \`open-swe-auto\`, except it uses a larger, more powerful model for the planning and programming steps: Claude Opus 4.1. It still uses Claude Sonnet 4 for the reviewer step. Only provide this information if requested by the user. For example, if the user asks what you can do, you should provide the above information in your response. diff --git a/apps/open-swe/src/routes/github/issue-webhook.ts b/apps/open-swe/src/routes/github/issue-webhook.ts index 098ff857..1941151c 100644 --- a/apps/open-swe/src/routes/github/issue-webhook.ts +++ b/apps/open-swe/src/routes/github/issue-webhook.ts @@ -175,15 +175,15 @@ webhooks.on("issues.labeled", async ({ payload }) => { }, autoAcceptPlan: isAutoAcceptLabel, }; - // Create config object with Claude Opus 4 model configuration for max labels + // Create config object with Claude Opus 4.1 model configuration for max labels const config: Record = { recursion_limit: 400, }; if (isMaxLabel) { config.configurable = { - plannerModelName: "anthropic:claude-opus-4-0", - programmerModelName: "anthropic:claude-opus-4-0", + plannerModelName: "anthropic:claude-opus-4-1", + programmerModelName: "anthropic:claude-opus-4-1", }; } diff --git a/packages/shared/src/open-swe/models.ts b/packages/shared/src/open-swe/models.ts index 039763e8..2656f6e7 100644 --- a/packages/shared/src/open-swe/models.ts +++ b/packages/shared/src/open-swe/models.ts @@ -11,6 +11,10 @@ export const MODEL_OPTIONS = [ label: "Claude Sonnet 4", value: "anthropic:claude-sonnet-4-0", }, + { + label: "Claude Opus 4.1", + value: "anthropic:claude-opus-4-1", + }, { label: "Claude Opus 4", value: "anthropic:claude-opus-4-0",