mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 20:53:15 +00:00
Some checks are pending
CI / Lint (push) Waiting to run
CI / Format check (push) Waiting to run
CI / Unit tests (push) Waiting to run
CI / Playwright E2E (push) Waiting to run
CI / Docker build smoke (push) Waiting to run
CI / Triage ledger up to date (push) Waiting to run
CI / ui bun.lock in sync (push) Waiting to run
* feat: port plan-review & workflow-approval UX (#135)
Port six upstream commits onto dev:
- c03a6be7 (already ported): keep plan guidance high-level
- 546042a4: add workflow approval UI with diff preview, approval URLs,
web review links, and polling for approval status during active runs
- 216cf181: remove workflow token elevation; approved pushes pass
through directly without proxy token rewriting
- 3dbc0282: preserve plan redirects after login by accepting relative
same-origin redirect_to values and rejecting blocked paths
- bb104d93: submit plan comments with cmd+enter
- 90cb6caa: terse Slack replies, shared content via save_plan outside
plan mode (PLAN_STATUS_SHARED), reject shared-content mutations
Refs: #135
* feat: port durable dispatch hardening and startup latency improvements
Port five upstream PRs onto dev:
- #1621 / #1658: durable dispatch with loopback webhook defense,
create_durable_run helper, _config_with_prepare_run_id, degradation
to None for relative/loopback completion webhook URLs
- #1696: run-level completion webhook deduplication (replace
claim-then-post with post-then-flag per run_id), DeferredErrorModel
for graph-factory resilience, ToolRetryMiddleware for task subagents,
TimeoutWrapupMiddleware for all three graphs
- #1697: lazy-load __init__.py for agent.middleware, agent.tools,
agent.dashboard (PEP 562); defer heavy imports (exa_py in web_search,
agent.webapp in request_pr_review, deepagents in sandbox.py); add
ttl_cache.py with stale-while-revalidate for tool loaders
Refs: #137
* fix: restore login page render and clear CI lint/format
The plan-review port removed the authRedirectUrl import from login.tsx
but left its call site, crashing the login page at runtime (blank page,
no 'Sign in to open-swe'). Pass the relative path straight to loginUrl,
matching the plan route and the backend relative-redirect handling.
Also drop an unused os import in the guard test and reformat
workflow_push_guard.py to satisfy ruff.
* fix: restore RepairOrphaned middleware export and repoint model fake to deferred_model boundary
* fix: restore RepairOrphanedToolCallsMiddleware, fix E2E model-fake patch, drop dead ttl_cache
- Re-add RepairOrphanedToolCallsMiddleware to the lazy middleware __init__
(_MIDDLEWARE_MODULES, __all__, TYPE_CHECKING) so agent.reviewer can import it.
- Reroute E2E model patching to deferred_model.make_model so make_model_or_defer
(used by all three graph factories) returns the scripted fake instead of
building a real model with fake credentials.
- Drop unused agent/utils/ttl_cache.py — no agent module imports it.
- Fix import ordering in agent/reviewer.py and agent/analyzer.py (ruff I001).
- Format tests/test_dispatch.py.
* fix: claim-then-post run-level failure dedup; stop permanent suppression
---------
Co-authored-by: amoussa1229 <166072409+amoussa1229@users.noreply.github.com>
Co-authored-by: Adam Moussa <adam@seahavenind.com>
339 lines
13 KiB
Python
339 lines
13 KiB
Python
from __future__ import annotations
|
|
|
|
from typing import Any
|
|
from unittest.mock import AsyncMock
|
|
|
|
import pytest
|
|
|
|
from agent import completion
|
|
|
|
|
|
class _FakeThreads:
|
|
def __init__(self, metadata: dict[str, Any], *, fail_updates: int = 0) -> None:
|
|
self._metadata = metadata
|
|
self.updates: list[dict[str, Any]] = []
|
|
self._fail_updates = fail_updates
|
|
|
|
async def get(self, thread_id: str) -> dict[str, Any]:
|
|
# Reflect prior claims so a duplicate/retried delivery sees them, exactly
|
|
# as the platform persists thread metadata across webhook deliveries.
|
|
return {"thread_id": thread_id, "metadata": dict(self._metadata)}
|
|
|
|
async def update(self, *, thread_id: str, metadata: dict[str, Any]) -> None:
|
|
if self._fail_updates > 0:
|
|
self._fail_updates -= 1
|
|
raise RuntimeError("simulated metadata write failure")
|
|
self.updates.append(metadata)
|
|
self._metadata.update(metadata)
|
|
|
|
|
|
class _FakeClient:
|
|
def __init__(self, metadata: dict[str, Any], *, fail_updates: int = 0) -> None:
|
|
self.threads = _FakeThreads(metadata, fail_updates=fail_updates)
|
|
|
|
|
|
def _slack_metadata() -> dict[str, Any]:
|
|
return {
|
|
"source": "slack",
|
|
"source_context": {"slack_thread": {"channel_id": "C1", "thread_ts": "123.45"}},
|
|
}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_error_status_posts_slack_failure_reply(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
client = _FakeClient(_slack_metadata())
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
reply = AsyncMock(return_value=True)
|
|
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
|
|
monkeypatch.setattr(
|
|
completion, "dashboard_thread_url", lambda thread_id: f"https://ui/{thread_id}"
|
|
)
|
|
|
|
result = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "run_id": "run-1", "status": "error"}
|
|
)
|
|
|
|
assert result["status"] == "ok"
|
|
reply.assert_awaited_once()
|
|
args = reply.await_args.args
|
|
assert args[0] == "C1"
|
|
assert args[1] == "123.45"
|
|
assert "<https://ui/t1|Open SWE Web>" in args[2]
|
|
assert client.threads.updates == [
|
|
{"failure_reply_posted_run_id": "run-1", "failure_reply_posted_run_ids": ["run-1"]}
|
|
]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_success_status_is_ignored(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
client = _FakeClient(_slack_metadata())
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
reply = AsyncMock(return_value=True)
|
|
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
|
|
|
|
result = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "run_id": "run-1", "status": "success"}
|
|
)
|
|
|
|
assert result["status"] == "ignored"
|
|
reply.assert_not_called()
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_idempotent_when_already_replied(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
metadata = _slack_metadata()
|
|
metadata["failure_reply_posted_run_ids"] = ["run-1"]
|
|
client = _FakeClient(metadata)
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
reply = AsyncMock(return_value=True)
|
|
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
|
|
|
|
result = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "run_id": "run-1", "status": "timeout"}
|
|
)
|
|
|
|
assert result["status"] == "ignored"
|
|
reply.assert_not_called()
|
|
assert client.threads.updates == []
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_later_failed_run_posts_even_if_prior_run_replied(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
metadata = _slack_metadata()
|
|
metadata["failure_reply_posted_run_ids"] = ["run-1"]
|
|
client = _FakeClient(metadata)
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
reply = AsyncMock(return_value=True)
|
|
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
|
|
|
|
result = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "run_id": "run-2", "status": "timeout"}
|
|
)
|
|
|
|
assert result["status"] == "ok"
|
|
reply.assert_awaited_once()
|
|
assert client.threads.updates == [
|
|
{
|
|
"failure_reply_posted_run_id": "run-2",
|
|
"failure_reply_posted_run_ids": ["run-1", "run-2"],
|
|
}
|
|
]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_linear_source_comments_on_issue(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
client = _FakeClient({"source": "linear", "source_context": {"linear_issue": {"id": "iss_1"}}})
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
comment = AsyncMock(return_value=True)
|
|
monkeypatch.setattr(completion, "comment_on_linear_issue", comment)
|
|
|
|
result = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "run_id": "run-1", "status": "timeout"}
|
|
)
|
|
|
|
assert result["status"] == "ok"
|
|
comment.assert_awaited_once()
|
|
assert comment.await_args.args[0] == "iss_1"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_missing_thread_id_is_ignored() -> None:
|
|
result = await completion.handle_run_completion({"run_id": "run-1", "status": "error"})
|
|
assert result["status"] == "ignored"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_no_run_id_and_no_marker_posts_without_dedupe(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
# A payload carrying nothing to distinguish one run from another posts
|
|
# un-deduped (no sticky flag written) rather than being suppressed. It
|
|
# prefers a rare duplicate over permanent silence (SR160-03).
|
|
client = _FakeClient(_slack_metadata())
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
reply = AsyncMock(return_value=True)
|
|
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
|
|
|
|
result = await completion.handle_run_completion({"thread_id": "t1", "status": "error"})
|
|
|
|
assert result["status"] == "ok"
|
|
reply.assert_awaited_once()
|
|
assert client.threads.updates == []
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_legacy_sticky_flag_does_not_suppress_new_failure(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
# An old per-thread sticky boolean left by a prior code version must not
|
|
# permanently silence a later failed run (SR160-03).
|
|
metadata = _slack_metadata()
|
|
metadata["failure_reply_posted"] = True
|
|
client = _FakeClient(metadata)
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
reply = AsyncMock(return_value=True)
|
|
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
|
|
|
|
result = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "run_id": "run-9", "status": "error"}
|
|
)
|
|
|
|
assert result["status"] == "ok"
|
|
reply.assert_awaited_once()
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_no_run_id_dedupes_on_updated_at_marker(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
# Without a run_id the same run's retry (identical updated_at) dedupes...
|
|
client = _FakeClient(_slack_metadata())
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
reply = AsyncMock(return_value=True)
|
|
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
|
|
|
|
first = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "status": "error", "updated_at": "2026-07-09T00:00:00Z"}
|
|
)
|
|
retry = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "status": "error", "updated_at": "2026-07-09T00:00:00Z"}
|
|
)
|
|
|
|
assert first["status"] == "ok"
|
|
assert retry["status"] == "ignored"
|
|
reply.assert_awaited_once()
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_no_run_id_different_runs_each_reply(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
# ...while two genuinely different failed runs (distinct updated_at) on one
|
|
# thread each get their reply — the crux of SR160-03.
|
|
client = _FakeClient(_slack_metadata())
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
reply = AsyncMock(return_value=True)
|
|
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
|
|
|
|
first = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "status": "error", "updated_at": "2026-07-09T00:00:00Z"}
|
|
)
|
|
second = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "status": "timeout", "updated_at": "2026-07-09T01:00:00Z"}
|
|
)
|
|
|
|
assert first["status"] == "ok"
|
|
assert second["status"] == "ok"
|
|
assert reply.await_count == 2
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_claim_written_before_post(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
# Claim-then-post: the dedup key is recorded before the network reply fires.
|
|
client = _FakeClient(_slack_metadata())
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
order: list[str] = []
|
|
|
|
async def _update(*, thread_id: str, metadata: dict[str, Any]) -> None:
|
|
order.append("claim")
|
|
client.threads.updates.append(metadata)
|
|
client.threads._metadata.update(metadata)
|
|
|
|
async def _reply(*args: Any, **kwargs: Any) -> bool:
|
|
order.append("post")
|
|
return True
|
|
|
|
monkeypatch.setattr(client.threads, "update", _update)
|
|
monkeypatch.setattr(completion, "post_slack_thread_reply", _reply)
|
|
|
|
result = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "run_id": "run-1", "status": "error"}
|
|
)
|
|
|
|
assert result["status"] == "ok"
|
|
assert order == ["claim", "post"]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_claim_write_failure_does_not_post_and_retry_posts_once(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
# SR160-02: if the claim write throws, we do NOT post and report an error so
|
|
# the retry re-attempts — the retry then claims and posts exactly once,
|
|
# rather than the old flow double-posting on retry.
|
|
client = _FakeClient(_slack_metadata(), fail_updates=1)
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
reply = AsyncMock(return_value=True)
|
|
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
|
|
|
|
first = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "run_id": "run-1", "status": "error"}
|
|
)
|
|
assert first["status"] == "error"
|
|
reply.assert_not_called()
|
|
|
|
retry = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "run_id": "run-1", "status": "error"}
|
|
)
|
|
assert retry["status"] == "ok"
|
|
reply.assert_awaited_once()
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_duplicate_delivery_posts_exactly_once(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
# A retried/duplicate delivery for one run_id posts exactly once because the
|
|
# first delivery's claim is persisted before the second reads metadata.
|
|
client = _FakeClient(_slack_metadata())
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
reply = AsyncMock(return_value=True)
|
|
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
|
|
|
|
payload = {"thread_id": "t1", "run_id": "run-1", "status": "error"}
|
|
first = await completion.handle_run_completion(dict(payload))
|
|
second = await completion.handle_run_completion(dict(payload))
|
|
|
|
assert first["status"] == "ok"
|
|
assert second["status"] == "ignored"
|
|
reply.assert_awaited_once()
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_no_reply_channel_does_not_flag(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
client = _FakeClient({"source": "schedule"})
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
|
|
result = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "run_id": "run-1", "status": "error"}
|
|
)
|
|
|
|
assert result["status"] == "ignored"
|
|
assert client.threads.updates == []
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_interrupted_status_is_ignored(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
# Follow-ups use multitask_strategy="interrupt", so an interrupted run is a
|
|
# healthy hand-off, not a failure to report.
|
|
client = _FakeClient(_slack_metadata())
|
|
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
|
|
reply = AsyncMock(return_value=True)
|
|
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
|
|
|
|
result = await completion.handle_run_completion(
|
|
{"thread_id": "t1", "run_id": "run-1", "status": "interrupted"}
|
|
)
|
|
|
|
assert result["status"] == "ignored"
|
|
reply.assert_not_called()
|
|
assert client.threads.updates == []
|
|
|
|
|
|
def test_verify_run_complete_token(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
# No secret configured: fail closed (reject everything).
|
|
monkeypatch.setattr(completion, "RUN_COMPLETE_WEBHOOK_SECRET", None)
|
|
assert completion.verify_run_complete_token(None) is False
|
|
assert completion.verify_run_complete_token("whatever") is False
|
|
|
|
# Secret configured: require an exact match.
|
|
monkeypatch.setattr(completion, "RUN_COMPLETE_WEBHOOK_SECRET", "s3cret")
|
|
assert completion.verify_run_complete_token("s3cret") is True
|
|
assert completion.verify_run_complete_token("wrong") is False
|
|
assert completion.verify_run_complete_token(None) is False
|