open-swe/tests/test_agent_subagent_models.py
seahaven-openswe[bot] 0546085672
Some checks are pending
CI / Lint (push) Waiting to run
CI / Format check (push) Waiting to run
CI / Unit tests (push) Waiting to run
CI / Playwright E2E (push) Waiting to run
CI / Docker build smoke (push) Waiting to run
CI / Triage ledger up to date (push) Waiting to run
CI / ui bun.lock in sync (push) Waiting to run
feat: port durable dispatch hardening and startup latency improvements (#160)
* feat: port plan-review & workflow-approval UX (#135)

Port six upstream commits onto dev:

- c03a6be7 (already ported): keep plan guidance high-level
- 546042a4: add workflow approval UI with diff preview, approval URLs,
  web review links, and polling for approval status during active runs
- 216cf181: remove workflow token elevation; approved pushes pass
  through directly without proxy token rewriting
- 3dbc0282: preserve plan redirects after login by accepting relative
  same-origin redirect_to values and rejecting blocked paths
- bb104d93: submit plan comments with cmd+enter
- 90cb6caa: terse Slack replies, shared content via save_plan outside
  plan mode (PLAN_STATUS_SHARED), reject shared-content mutations

Refs: #135

* feat: port durable dispatch hardening and startup latency improvements

Port five upstream PRs onto dev:

- #1621 / #1658: durable dispatch with loopback webhook defense,
  create_durable_run helper, _config_with_prepare_run_id, degradation
  to None for relative/loopback completion webhook URLs
- #1696: run-level completion webhook deduplication (replace
  claim-then-post with post-then-flag per run_id), DeferredErrorModel
  for graph-factory resilience, ToolRetryMiddleware for task subagents,
  TimeoutWrapupMiddleware for all three graphs
- #1697: lazy-load __init__.py for agent.middleware, agent.tools,
  agent.dashboard (PEP 562); defer heavy imports (exa_py in web_search,
  agent.webapp in request_pr_review, deepagents in sandbox.py); add
  ttl_cache.py with stale-while-revalidate for tool loaders

Refs: #137

* fix: restore login page render and clear CI lint/format

The plan-review port removed the authRedirectUrl import from login.tsx
but left its call site, crashing the login page at runtime (blank page,
no 'Sign in to open-swe'). Pass the relative path straight to loginUrl,
matching the plan route and the backend relative-redirect handling.

Also drop an unused os import in the guard test and reformat
workflow_push_guard.py to satisfy ruff.

* fix: restore RepairOrphaned middleware export and repoint model fake to deferred_model boundary

* fix: restore RepairOrphanedToolCallsMiddleware, fix E2E model-fake patch, drop dead ttl_cache

- Re-add RepairOrphanedToolCallsMiddleware to the lazy middleware __init__
  (_MIDDLEWARE_MODULES, __all__, TYPE_CHECKING) so agent.reviewer can import it.
- Reroute E2E model patching to deferred_model.make_model so make_model_or_defer
  (used by all three graph factories) returns the scripted fake instead of
  building a real model with fake credentials.
- Drop unused agent/utils/ttl_cache.py — no agent module imports it.
- Fix import ordering in agent/reviewer.py and agent/analyzer.py (ruff I001).
- Format tests/test_dispatch.py.

* fix: claim-then-post run-level failure dedup; stop permanent suppression

---------

Co-authored-by: amoussa1229 <166072409+amoussa1229@users.noreply.github.com>
Co-authored-by: Adam Moussa <adam@seahavenind.com>
2026-07-09 17:11:25 -04:00

163 lines
5.9 KiB
Python

from unittest.mock import AsyncMock, MagicMock, patch
import pytest
from langgraph.graph.state import RunnableConfig
from agent.server import get_agent
class _DummyAgent:
def with_config(self, config: RunnableConfig) -> "_DummyAgent":
self.config = config
return self
@pytest.mark.asyncio
async def test_agent_uses_profile_subagent_model_override() -> None:
config: RunnableConfig = {
"configurable": {
"__is_for_execution__": True,
"thread_id": "thread-123",
"github_login": "octocat",
},
"metadata": {},
}
main_model = MagicMock(name="main_model")
subagent_model = MagicMock(name="subagent_model")
captured: dict[str, object] = {}
def fake_create_deep_agent(**kwargs: object) -> _DummyAgent:
captured.update(kwargs)
return _DummyAgent()
with (
patch(
"agent.server.resolve_github_token",
new_callable=AsyncMock,
return_value=("ghp", None),
),
patch("agent.server.resolve_triggering_user_identity", return_value=None),
patch(
"agent.server.ensure_sandbox_for_thread",
new_callable=AsyncMock,
return_value=MagicMock(),
),
patch(
"agent.server.aresolve_sandbox_work_dir",
new_callable=AsyncMock,
return_value="/workspace",
),
patch(
"agent.server.get_team_default_model_pair",
new_callable=AsyncMock,
return_value=(
("bedrock_converse:us.anthropic.claude-opus-4-8", "medium"),
("fireworks:accounts/fireworks/models/deepseek-v4-pro", "low"),
),
),
patch(
"agent.server.load_profile",
new_callable=AsyncMock,
return_value={
"default_model": "bedrock_converse:us.anthropic.claude-opus-4-8",
"reasoning_effort": "high",
"default_subagent_model": "fireworks:accounts/fireworks/models/deepseek-v4-pro",
"subagent_reasoning_effort": "xhigh",
},
),
patch("agent.server.fallback_model_id_for", return_value=None),
patch(
"agent.utils.deferred_model.make_model", side_effect=[main_model, subagent_model]
) as make_model,
patch("agent.server.construct_system_prompt", return_value="prompt"),
patch("agent.server.create_deep_agent", side_effect=fake_create_deep_agent),
):
await get_agent(config)
assert captured["model"] is main_model
subagents = captured["subagents"]
assert isinstance(subagents, list)
assert subagents[0]["name"] == "general-purpose"
assert subagents[0]["model"] is subagent_model
main_call = make_model.call_args_list[0]
assert main_call.args == ("bedrock_converse:us.anthropic.claude-opus-4-8",)
assert main_call.kwargs["additional_model_request_fields"] == {
"thinking": {"type": "adaptive", "display": "summarized"},
"output_config": {"effort": "high"},
}
subagent_call = make_model.call_args_list[1]
assert subagent_call.args == ("fireworks:accounts/fireworks/models/deepseek-v4-pro",)
assert subagent_call.kwargs["model_kwargs"] == {"reasoning_effort": "xhigh"}
@pytest.mark.asyncio
async def test_agent_subagent_inherits_profile_model_override_without_explicit_pair() -> None:
config: RunnableConfig = {
"configurable": {
"__is_for_execution__": True,
"thread_id": "thread-123",
"github_login": "octocat",
},
"metadata": {},
}
main_model = MagicMock(name="main_model")
subagent_model = MagicMock(name="subagent_model")
captured: dict[str, object] = {}
def fake_create_deep_agent(**kwargs: object) -> _DummyAgent:
captured.update(kwargs)
return _DummyAgent()
with (
patch(
"agent.server.resolve_github_token",
new_callable=AsyncMock,
return_value=("ghp", None),
),
patch("agent.server.resolve_triggering_user_identity", return_value=None),
patch(
"agent.server.ensure_sandbox_for_thread",
new_callable=AsyncMock,
return_value=MagicMock(),
),
patch(
"agent.server.aresolve_sandbox_work_dir",
new_callable=AsyncMock,
return_value="/workspace",
),
patch(
"agent.server.get_team_default_model_pair",
new_callable=AsyncMock,
return_value=(
("bedrock_converse:us.anthropic.claude-opus-4-8", "medium"),
("fireworks:accounts/fireworks/models/deepseek-v4-pro", "low"),
),
),
patch(
"agent.server.load_profile",
new_callable=AsyncMock,
return_value={
"default_model": "bedrock_converse:us.anthropic.claude-opus-4-8",
"reasoning_effort": "high",
},
),
patch("agent.server.fallback_model_id_for", return_value=None),
patch(
"agent.utils.deferred_model.make_model", side_effect=[main_model, subagent_model]
) as make_model,
patch("agent.server.construct_system_prompt", return_value="prompt"),
patch("agent.server.create_deep_agent", side_effect=fake_create_deep_agent),
):
await get_agent(config)
subagents = captured["subagents"]
assert isinstance(subagents, list)
assert subagents[0]["model"] is subagent_model
assert make_model.call_args_list[0].args == ("bedrock_converse:us.anthropic.claude-opus-4-8",)
assert make_model.call_args_list[1].args == ("bedrock_converse:us.anthropic.claude-opus-4-8",)
assert make_model.call_args_list[1].kwargs["additional_model_request_fields"] == {
"thinking": {"type": "adaptive", "display": "summarized"},
"output_config": {"effort": "high"},
}