mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-10-05 07:12:12 +00:00
feat(open-swe): Default to GPT-5.5 medium reasoning (#1224)
* feat: default to GPT-5.5 medium reasoning Use OpenAI GPT-5.5 with medium reasoning as the default model and document the completion-token budget semantics for reasoning models. * fix: use Responses API reasoning config Pass GPT-5.5 reasoning settings through LangChain's Responses API parameter instead of the Chat Completions-only reasoning_effort field. * feat: raise GPT-5.5 output budget Set the default GPT-5.5 output token budget to the model maximum so long-running coding tasks have more room for reasoning and final responses. * feat: align recursion limit with Deep Agents Use Deep Agents' default recursion limit so longer coding runs have room to complete without Open SWE imposing a lower cap. * chore: remove minimal effort level * chore: reduce max tokens to 64_000 --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
This commit is contained in:
parent
59bfc65bb7
commit
448be4a466
5 changed files with 48 additions and 15 deletions
7
.gitignore
vendored
7
.gitignore
vendored
|
|
@ -24,6 +24,11 @@
|
||||||
# misc
|
# misc
|
||||||
.DS_Store
|
.DS_Store
|
||||||
*.pem
|
*.pem
|
||||||
|
*.key
|
||||||
|
*.crt
|
||||||
|
*.p12
|
||||||
|
*.pfx
|
||||||
|
*.jks
|
||||||
|
|
||||||
# debug
|
# debug
|
||||||
npm-debug.log*
|
npm-debug.log*
|
||||||
|
|
@ -33,6 +38,7 @@ yarn-error.log*
|
||||||
# local env files
|
# local env files
|
||||||
.env*.local
|
.env*.local
|
||||||
.env
|
.env
|
||||||
|
.env.*
|
||||||
|
|
||||||
# vercel
|
# vercel
|
||||||
.vercel
|
.vercel
|
||||||
|
|
@ -52,6 +58,7 @@ credentials.json
|
||||||
apps/cli/test_traces/
|
apps/cli/test_traces/
|
||||||
|
|
||||||
# Python
|
# Python
|
||||||
|
.venv/
|
||||||
__pycache__/
|
__pycache__/
|
||||||
**/__pycache__/
|
**/__pycache__/
|
||||||
*.py[cod]
|
*.py[cod]
|
||||||
|
|
|
||||||
|
|
@ -4,8 +4,13 @@ Open SWE is designed to be forked and customized for your org. The core agent is
|
||||||
|
|
||||||
```python
|
```python
|
||||||
# agent/server.py — the key lines
|
# agent/server.py — the key lines
|
||||||
|
model_id = os.environ.get("LLM_MODEL_ID", DEFAULT_LLM_MODEL_ID)
|
||||||
|
model_kwargs = {"max_tokens": DEFAULT_LLM_MAX_TOKENS}
|
||||||
|
if model_id == DEFAULT_LLM_MODEL_ID:
|
||||||
|
model_kwargs["reasoning"] = DEFAULT_LLM_REASONING
|
||||||
|
|
||||||
return create_deep_agent(
|
return create_deep_agent(
|
||||||
model=make_model(os.environ.get("LLM_MODEL_ID", DEFAULT_LLM_MODEL_ID), temperature=0, max_tokens=20_000),
|
model=make_model(model_id, **model_kwargs),
|
||||||
system_prompt=construct_system_prompt(...),
|
system_prompt=construct_system_prompt(...),
|
||||||
tools=[http_request, fetch_url, list_repos, get_branch_name, commit_and_open_pr, linear_comment, slack_thread_reply],
|
tools=[http_request, fetch_url, list_repos, get_branch_name, commit_and_open_pr, linear_comment, slack_thread_reply],
|
||||||
backend=sandbox_backend,
|
backend=sandbox_backend,
|
||||||
|
|
@ -119,14 +124,16 @@ See `deepagents.backends.LangSmithSandbox` and `agent/integrations/langsmith.py`
|
||||||
|
|
||||||
## 2. Model
|
## 2. Model
|
||||||
|
|
||||||
The model is configured in the `get_agent()` function in `agent/server.py`. By default it uses `anthropic:claude-opus-4-6`, but you can override it with the `LLM_MODEL_ID` environment variable:
|
The model is configured in the `get_agent()` function in `agent/server.py`. By default it uses `openai:gpt-5.5` with medium reasoning effort, but you can override the model with the `LLM_MODEL_ID` environment variable:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
# Set the model via environment variable (uses provider:model format)
|
# Set the model via environment variable (uses provider:model format)
|
||||||
LLM_MODEL_ID="anthropic:claude-sonnet-4-6"
|
LLM_MODEL_ID="anthropic:claude-sonnet-4-6"
|
||||||
```
|
```
|
||||||
|
|
||||||
If `LLM_MODEL_ID` is not set, the default model (`anthropic:claude-opus-4-6`) is used.
|
If `LLM_MODEL_ID` is not set, the default model (`openai:gpt-5.5`) is used.
|
||||||
|
|
||||||
|
`max_tokens` is a maximum completion/output token budget, not the model's total context window. For OpenAI reasoning models, this budget can include both internal reasoning tokens and final response tokens.
|
||||||
|
|
||||||
### Switching models
|
### Switching models
|
||||||
|
|
||||||
|
|
@ -137,7 +144,7 @@ Use the `provider:model` format:
|
||||||
model=make_model("anthropic:claude-sonnet-4-6", temperature=0, max_tokens=16_000)
|
model=make_model("anthropic:claude-sonnet-4-6", temperature=0, max_tokens=16_000)
|
||||||
|
|
||||||
# OpenAI (uses Responses API by default)
|
# OpenAI (uses Responses API by default)
|
||||||
model=make_model("openai:gpt-4o", temperature=0, max_tokens=16_000)
|
model=make_model("openai:gpt-5.5", max_tokens=128_000, reasoning={"effort": "medium"})
|
||||||
|
|
||||||
# Google
|
# Google
|
||||||
model=make_model("google_genai:gemini-2.5-pro", temperature=0, max_tokens=16_000)
|
model=make_model("google_genai:gemini-2.5-pro", temperature=0, max_tokens=16_000)
|
||||||
|
|
@ -169,7 +176,7 @@ async def get_agent(config: RunnableConfig) -> Pregel:
|
||||||
model = make_model("anthropic:claude-sonnet-4-6", temperature=0, max_tokens=16_000)
|
model = make_model("anthropic:claude-sonnet-4-6", temperature=0, max_tokens=16_000)
|
||||||
else:
|
else:
|
||||||
# Full model for code changes from Linear
|
# Full model for code changes from Linear
|
||||||
model = make_model("anthropic:claude-opus-4-6", temperature=0, max_tokens=20_000)
|
model = make_model("openai:gpt-5.5", max_tokens=128_000, reasoning={"effort": "medium"})
|
||||||
|
|
||||||
return create_deep_agent(model=model, ...)
|
return create_deep_agent(model=model, ...)
|
||||||
```
|
```
|
||||||
|
|
|
||||||
|
|
@ -41,7 +41,7 @@ Rather than forking an existing agent or building from scratch, Open SWE **compo
|
||||||
|
|
||||||
```python
|
```python
|
||||||
create_deep_agent(
|
create_deep_agent(
|
||||||
model="anthropic:claude-opus-4-6",
|
model="openai:gpt-5.5",
|
||||||
system_prompt=construct_system_prompt(...),
|
system_prompt=construct_system_prompt(...),
|
||||||
tools=[http_request, fetch_url, list_repos, get_branch_name, commit_and_open_pr, linear_comment, slack_thread_reply],
|
tools=[http_request, fetch_url, list_repos, get_branch_name, commit_and_open_pr, linear_comment, slack_thread_reply],
|
||||||
backend=sandbox_backend,
|
backend=sandbox_backend,
|
||||||
|
|
|
||||||
|
|
@ -59,7 +59,7 @@ from .tools import (
|
||||||
)
|
)
|
||||||
from .utils.auth import resolve_github_token
|
from .utils.auth import resolve_github_token
|
||||||
from .utils.github_app import get_github_app_installation_token
|
from .utils.github_app import get_github_app_installation_token
|
||||||
from .utils.model import make_model
|
from .utils.model import ModelKwargs, OpenAIReasoning, make_model
|
||||||
from .utils.sandbox import create_sandbox
|
from .utils.sandbox import create_sandbox
|
||||||
from .utils.sandbox_paths import aresolve_sandbox_work_dir
|
from .utils.sandbox_paths import aresolve_sandbox_work_dir
|
||||||
|
|
||||||
|
|
@ -183,8 +183,10 @@ def graph_loaded_for_execution(config: RunnableConfig) -> bool:
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
DEFAULT_LLM_MODEL_ID = "anthropic:claude-opus-4-6"
|
DEFAULT_LLM_MODEL_ID = "openai:gpt-5.5"
|
||||||
DEFAULT_RECURSION_LIMIT = 1_000
|
DEFAULT_LLM_REASONING: OpenAIReasoning = {"effort": "medium"}
|
||||||
|
DEFAULT_LLM_MAX_TOKENS = 64_000
|
||||||
|
DEFAULT_RECURSION_LIMIT = 9_999
|
||||||
|
|
||||||
|
|
||||||
async def get_agent(config: RunnableConfig) -> Pregel:
|
async def get_agent(config: RunnableConfig) -> Pregel:
|
||||||
|
|
@ -273,12 +275,14 @@ async def get_agent(config: RunnableConfig) -> Pregel:
|
||||||
|
|
||||||
work_dir = await aresolve_sandbox_work_dir(sandbox_backend)
|
work_dir = await aresolve_sandbox_work_dir(sandbox_backend)
|
||||||
|
|
||||||
|
model_id = os.environ.get("LLM_MODEL_ID", DEFAULT_LLM_MODEL_ID)
|
||||||
|
model_kwargs: ModelKwargs = {"max_tokens": DEFAULT_LLM_MAX_TOKENS}
|
||||||
|
if model_id == DEFAULT_LLM_MODEL_ID:
|
||||||
|
model_kwargs["reasoning"] = DEFAULT_LLM_REASONING
|
||||||
|
|
||||||
logger.info("Returning agent with sandbox for thread %s", thread_id)
|
logger.info("Returning agent with sandbox for thread %s", thread_id)
|
||||||
return create_deep_agent(
|
return create_deep_agent(
|
||||||
model=make_model(
|
model=make_model(model_id, **model_kwargs),
|
||||||
os.environ.get("LLM_MODEL_ID", DEFAULT_LLM_MODEL_ID),
|
|
||||||
max_tokens=20_000,
|
|
||||||
),
|
|
||||||
system_prompt=construct_system_prompt(
|
system_prompt=construct_system_prompt(
|
||||||
working_dir=work_dir,
|
working_dir=work_dir,
|
||||||
linear_project_id=linear_project_id,
|
linear_project_id=linear_project_id,
|
||||||
|
|
|
||||||
|
|
@ -1,10 +1,25 @@
|
||||||
|
from typing import Literal, TypedDict, Unpack
|
||||||
|
|
||||||
from langchain.chat_models import init_chat_model
|
from langchain.chat_models import init_chat_model
|
||||||
|
|
||||||
OPENAI_RESPONSES_WS_BASE_URL = "wss://api.openai.com/v1"
|
OPENAI_RESPONSES_WS_BASE_URL = "wss://api.openai.com/v1"
|
||||||
|
|
||||||
|
|
||||||
def make_model(model_id: str, **kwargs: dict):
|
OpenAIReasoningEffort = Literal["none", "low", "medium", "high", "xhigh"]
|
||||||
model_kwargs = kwargs.copy()
|
|
||||||
|
|
||||||
|
class OpenAIReasoning(TypedDict, total=False):
|
||||||
|
effort: OpenAIReasoningEffort
|
||||||
|
|
||||||
|
|
||||||
|
class ModelKwargs(TypedDict, total=False):
|
||||||
|
max_tokens: int | None
|
||||||
|
reasoning: OpenAIReasoning | None
|
||||||
|
temperature: float | None
|
||||||
|
|
||||||
|
|
||||||
|
def make_model(model_id: str, **kwargs: Unpack[ModelKwargs]):
|
||||||
|
model_kwargs: dict[str, object] = kwargs.copy()
|
||||||
|
|
||||||
if model_id.startswith("openai:"):
|
if model_id.startswith("openai:"):
|
||||||
model_kwargs["base_url"] = OPENAI_RESPONSES_WS_BASE_URL
|
model_kwargs["base_url"] = OPENAI_RESPONSES_WS_BASE_URL
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue