mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 16:13:15 +00:00
* feat: upgrade default agent + reviewer model to Opus 4.8 Replace Opus 4.7 with Opus 4.8 (claude-opus-4-8) as the supported Anthropic model surfaced in the profile editor and used by the main agent and reviewer graphs. Effort levels (low/medium/high/xhigh/max) and the high default are unchanged, matching the official Opus 4.8 docs. Updates eval config comment and tests accordingly. * fix: provider-aware fallback for stale stored model ids Dropping claude-opus-4-7 from the supported set meant persisted profile/team-settings still holding it failed the SUPPORTED_MODEL_IDS check and fell through to default_model_pair() — a cross-provider jump to the OpenAI global default. Add provider_fallback_pair: when a stored id is no longer supported but its provider still has a supported model, resolve to that provider's newest supported model (anthropic:claude-opus-4-7 -> 4.8), preserving effort when valid. Resolution order is now: valid stored pair -> same-provider fallback -> global default_model_pair(). Profile overrides keep deferring to the team default when no model is set or the provider is unknown.
159 lines
5.6 KiB
Python
159 lines
5.6 KiB
Python
from unittest.mock import AsyncMock, MagicMock, patch
|
|
|
|
import pytest
|
|
from langgraph.graph.state import RunnableConfig
|
|
|
|
from agent.server import get_agent
|
|
|
|
|
|
class _DummyAgent:
|
|
def with_config(self, config: RunnableConfig) -> "_DummyAgent":
|
|
self.config = config
|
|
return self
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_agent_uses_profile_subagent_model_override() -> None:
|
|
config: RunnableConfig = {
|
|
"configurable": {
|
|
"__is_for_execution__": True,
|
|
"thread_id": "thread-123",
|
|
"github_login": "octocat",
|
|
},
|
|
"metadata": {},
|
|
}
|
|
main_model = MagicMock(name="main_model")
|
|
subagent_model = MagicMock(name="subagent_model")
|
|
captured: dict[str, object] = {}
|
|
|
|
def fake_create_deep_agent(**kwargs: object) -> _DummyAgent:
|
|
captured.update(kwargs)
|
|
return _DummyAgent()
|
|
|
|
with (
|
|
patch(
|
|
"agent.server.resolve_github_token",
|
|
new_callable=AsyncMock,
|
|
return_value=("ghp", "enc", None),
|
|
),
|
|
patch("agent.server.resolve_triggering_user_identity", return_value=None),
|
|
patch(
|
|
"agent.server.ensure_sandbox_for_thread",
|
|
new_callable=AsyncMock,
|
|
return_value=MagicMock(),
|
|
),
|
|
patch(
|
|
"agent.server.aresolve_sandbox_work_dir",
|
|
new_callable=AsyncMock,
|
|
return_value="/workspace",
|
|
),
|
|
patch(
|
|
"agent.server.get_team_default_model",
|
|
new_callable=AsyncMock,
|
|
return_value=("openai:gpt-5.5", "medium"),
|
|
),
|
|
patch(
|
|
"agent.server.get_team_default_subagent_model",
|
|
new_callable=AsyncMock,
|
|
return_value=("openai:gpt-5.5", "low"),
|
|
),
|
|
patch(
|
|
"agent.server.load_profile",
|
|
new_callable=AsyncMock,
|
|
return_value={
|
|
"default_model": "anthropic:claude-opus-4-8",
|
|
"reasoning_effort": "high",
|
|
"default_subagent_model": "openai:gpt-5.5",
|
|
"subagent_reasoning_effort": "xhigh",
|
|
},
|
|
),
|
|
patch("agent.server.fallback_model_id_for", return_value=None),
|
|
patch("agent.server.make_model", side_effect=[main_model, subagent_model]) as make_model,
|
|
patch("agent.server.construct_system_prompt", return_value="prompt"),
|
|
patch("agent.server.create_deep_agent", side_effect=fake_create_deep_agent),
|
|
):
|
|
await get_agent(config)
|
|
|
|
assert captured["model"] is main_model
|
|
subagents = captured["subagents"]
|
|
assert isinstance(subagents, list)
|
|
assert subagents[0]["name"] == "general-purpose"
|
|
assert subagents[0]["model"] is subagent_model
|
|
|
|
main_call = make_model.call_args_list[0]
|
|
assert main_call.args == ("anthropic:claude-opus-4-8",)
|
|
assert main_call.kwargs["thinking"] == {"type": "adaptive"}
|
|
assert main_call.kwargs["effort"] == "high"
|
|
|
|
subagent_call = make_model.call_args_list[1]
|
|
assert subagent_call.args == ("openai:gpt-5.5",)
|
|
assert subagent_call.kwargs["reasoning"] == {"effort": "xhigh"}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_agent_subagent_inherits_profile_model_override_without_explicit_pair() -> None:
|
|
config: RunnableConfig = {
|
|
"configurable": {
|
|
"__is_for_execution__": True,
|
|
"thread_id": "thread-123",
|
|
"github_login": "octocat",
|
|
},
|
|
"metadata": {},
|
|
}
|
|
main_model = MagicMock(name="main_model")
|
|
subagent_model = MagicMock(name="subagent_model")
|
|
captured: dict[str, object] = {}
|
|
|
|
def fake_create_deep_agent(**kwargs: object) -> _DummyAgent:
|
|
captured.update(kwargs)
|
|
return _DummyAgent()
|
|
|
|
with (
|
|
patch(
|
|
"agent.server.resolve_github_token",
|
|
new_callable=AsyncMock,
|
|
return_value=("ghp", "enc", None),
|
|
),
|
|
patch("agent.server.resolve_triggering_user_identity", return_value=None),
|
|
patch(
|
|
"agent.server.ensure_sandbox_for_thread",
|
|
new_callable=AsyncMock,
|
|
return_value=MagicMock(),
|
|
),
|
|
patch(
|
|
"agent.server.aresolve_sandbox_work_dir",
|
|
new_callable=AsyncMock,
|
|
return_value="/workspace",
|
|
),
|
|
patch(
|
|
"agent.server.get_team_default_model",
|
|
new_callable=AsyncMock,
|
|
return_value=("openai:gpt-5.5", "medium"),
|
|
),
|
|
patch(
|
|
"agent.server.get_team_default_subagent_model",
|
|
new_callable=AsyncMock,
|
|
return_value=("openai:gpt-5.5", "low"),
|
|
),
|
|
patch(
|
|
"agent.server.load_profile",
|
|
new_callable=AsyncMock,
|
|
return_value={
|
|
"default_model": "anthropic:claude-opus-4-8",
|
|
"reasoning_effort": "high",
|
|
},
|
|
),
|
|
patch("agent.server.fallback_model_id_for", return_value=None),
|
|
patch("agent.server.make_model", side_effect=[main_model, subagent_model]) as make_model,
|
|
patch("agent.server.construct_system_prompt", return_value="prompt"),
|
|
patch("agent.server.create_deep_agent", side_effect=fake_create_deep_agent),
|
|
):
|
|
await get_agent(config)
|
|
|
|
subagents = captured["subagents"]
|
|
assert isinstance(subagents, list)
|
|
assert subagents[0]["model"] is subagent_model
|
|
assert make_model.call_args_list[0].args == ("anthropic:claude-opus-4-8",)
|
|
assert make_model.call_args_list[1].args == ("anthropic:claude-opus-4-8",)
|
|
assert make_model.call_args_list[1].kwargs["thinking"] == {"type": "adaptive"}
|
|
assert make_model.call_args_list[1].kwargs["effort"] == "high"
|