open-swe/tests/webhooks/test_completion_webhook.py
Adam Moussa ae1f883b4c
refactor: move tests into tests/<domain>/ layout
Applies the plan's C5 step: git mv every test per the domain-reorg
move-map (movemap-m50.txt) into tests/{agent,analyzer,auth,dashboard,
github,middleware,models,reviewer,sandbox,slack,tools,webhooks}/, plus
the 13 fork-only placements from the scoping report §2c (Atlassian
webhook tests -> tests/webhooks/, test_atlassian_connect.py and
test_auth_error_leak.py -> tests/auth/, jira/confluence util tests ->
tests/tools/, test_repo_binding_isolation.py -> tests/sandbox/,
bot-identity/autofix tests -> tests/github/).

Path-only move: the only content edits are parents[1] -> parents[2]
fixes in test_e2b_integration.py and test_daytona_integration.py,
required because their __file__-relative ROOT path gained one more
directory level in the move.

Monkeypatch retargets for these files were already completed in C4;
none remained outstanding here.
2026-07-17 14:42:45 -04:00

355 lines
13 KiB
Python

from __future__ import annotations
from typing import Any
from unittest.mock import AsyncMock
import pytest
from agent import completion
class _FakeThreads:
def __init__(self, metadata: dict[str, Any], *, fail_updates: int = 0) -> None:
self._metadata = metadata
self.updates: list[dict[str, Any]] = []
self._fail_updates = fail_updates
async def get(self, thread_id: str) -> dict[str, Any]:
# Reflect prior claims so a duplicate/retried delivery sees them, exactly
# as the platform persists thread metadata across webhook deliveries.
return {"thread_id": thread_id, "metadata": dict(self._metadata)}
async def update(self, *, thread_id: str, metadata: dict[str, Any]) -> None:
if self._fail_updates > 0:
self._fail_updates -= 1
raise RuntimeError("simulated metadata write failure")
self.updates.append(metadata)
self._metadata.update(metadata)
class _FakeClient:
def __init__(self, metadata: dict[str, Any], *, fail_updates: int = 0) -> None:
self.threads = _FakeThreads(metadata, fail_updates=fail_updates)
def _slack_metadata() -> dict[str, Any]:
return {
"source": "slack",
"source_context": {"slack_thread": {"channel_id": "C1", "thread_ts": "123.45"}},
}
@pytest.mark.asyncio
async def test_error_status_posts_slack_failure_reply(monkeypatch: pytest.MonkeyPatch) -> None:
client = _FakeClient(_slack_metadata())
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
reply = AsyncMock(return_value=True)
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
monkeypatch.setattr(
completion, "dashboard_thread_url", lambda thread_id: f"https://ui/{thread_id}"
)
result = await completion.handle_run_completion(
{"thread_id": "t1", "run_id": "run-1", "status": "error"}
)
assert result["status"] == "ok"
reply.assert_awaited_once()
args = reply.await_args.args
assert args[0] == "C1"
assert args[1] == "123.45"
assert "<https://ui/t1|Open SWE Web>" in args[2]
assert client.threads.updates == [
{"failure_reply_posted_run_id": "run-1", "failure_reply_posted_run_ids": ["run-1"]}
]
@pytest.mark.asyncio
async def test_success_status_is_ignored(monkeypatch: pytest.MonkeyPatch) -> None:
client = _FakeClient(_slack_metadata())
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
reply = AsyncMock(return_value=True)
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
result = await completion.handle_run_completion(
{"thread_id": "t1", "run_id": "run-1", "status": "success"}
)
assert result["status"] == "ignored"
reply.assert_not_called()
@pytest.mark.asyncio
async def test_idempotent_when_already_replied(monkeypatch: pytest.MonkeyPatch) -> None:
metadata = _slack_metadata()
metadata["failure_reply_posted_run_ids"] = ["run-1"]
client = _FakeClient(metadata)
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
reply = AsyncMock(return_value=True)
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
result = await completion.handle_run_completion(
{"thread_id": "t1", "run_id": "run-1", "status": "timeout"}
)
assert result["status"] == "ignored"
reply.assert_not_called()
assert client.threads.updates == []
@pytest.mark.asyncio
async def test_later_failed_run_posts_even_if_prior_run_replied(
monkeypatch: pytest.MonkeyPatch,
) -> None:
metadata = _slack_metadata()
metadata["failure_reply_posted_run_ids"] = ["run-1"]
client = _FakeClient(metadata)
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
reply = AsyncMock(return_value=True)
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
result = await completion.handle_run_completion(
{"thread_id": "t1", "run_id": "run-2", "status": "timeout"}
)
assert result["status"] == "ok"
reply.assert_awaited_once()
assert client.threads.updates == [
{
"failure_reply_posted_run_id": "run-2",
"failure_reply_posted_run_ids": ["run-1", "run-2"],
}
]
@pytest.mark.asyncio
async def test_linear_source_comments_on_issue(monkeypatch: pytest.MonkeyPatch) -> None:
client = _FakeClient({"source": "linear", "source_context": {"linear_issue": {"id": "iss_1"}}})
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
comment = AsyncMock(return_value=True)
monkeypatch.setattr(completion, "comment_on_linear_issue", comment)
result = await completion.handle_run_completion(
{"thread_id": "t1", "run_id": "run-1", "status": "timeout"}
)
assert result["status"] == "ok"
comment.assert_awaited_once()
assert comment.await_args.args[0] == "iss_1"
@pytest.mark.asyncio
async def test_jira_source_comments_on_issue(monkeypatch: pytest.MonkeyPatch) -> None:
client = _FakeClient({"source": "jira", "source_context": {"jira_issue": {"key": "PROJ-42"}}})
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
comment = AsyncMock(return_value=True)
monkeypatch.setattr(completion, "comment_on_jira_issue", comment)
result = await completion.handle_run_completion(
{"thread_id": "t1", "run_id": "run-1", "status": "timeout"}
)
assert result["status"] == "ok"
comment.assert_awaited_once()
assert comment.await_args.args[0] == "PROJ-42"
@pytest.mark.asyncio
async def test_missing_thread_id_is_ignored() -> None:
result = await completion.handle_run_completion({"run_id": "run-1", "status": "error"})
assert result["status"] == "ignored"
@pytest.mark.asyncio
async def test_no_run_id_and_no_marker_posts_without_dedupe(
monkeypatch: pytest.MonkeyPatch,
) -> None:
# A payload carrying nothing to distinguish one run from another posts
# un-deduped (no sticky flag written) rather than being suppressed. It
# prefers a rare duplicate over permanent silence (SR160-03).
client = _FakeClient(_slack_metadata())
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
reply = AsyncMock(return_value=True)
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
result = await completion.handle_run_completion({"thread_id": "t1", "status": "error"})
assert result["status"] == "ok"
reply.assert_awaited_once()
assert client.threads.updates == []
@pytest.mark.asyncio
async def test_legacy_sticky_flag_does_not_suppress_new_failure(
monkeypatch: pytest.MonkeyPatch,
) -> None:
# An old per-thread sticky boolean left by a prior code version must not
# permanently silence a later failed run (SR160-03).
metadata = _slack_metadata()
metadata["failure_reply_posted"] = True
client = _FakeClient(metadata)
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
reply = AsyncMock(return_value=True)
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
result = await completion.handle_run_completion(
{"thread_id": "t1", "run_id": "run-9", "status": "error"}
)
assert result["status"] == "ok"
reply.assert_awaited_once()
@pytest.mark.asyncio
async def test_no_run_id_dedupes_on_updated_at_marker(monkeypatch: pytest.MonkeyPatch) -> None:
# Without a run_id the same run's retry (identical updated_at) dedupes...
client = _FakeClient(_slack_metadata())
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
reply = AsyncMock(return_value=True)
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
first = await completion.handle_run_completion(
{"thread_id": "t1", "status": "error", "updated_at": "2026-07-09T00:00:00Z"}
)
retry = await completion.handle_run_completion(
{"thread_id": "t1", "status": "error", "updated_at": "2026-07-09T00:00:00Z"}
)
assert first["status"] == "ok"
assert retry["status"] == "ignored"
reply.assert_awaited_once()
@pytest.mark.asyncio
async def test_no_run_id_different_runs_each_reply(monkeypatch: pytest.MonkeyPatch) -> None:
# ...while two genuinely different failed runs (distinct updated_at) on one
# thread each get their reply — the crux of SR160-03.
client = _FakeClient(_slack_metadata())
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
reply = AsyncMock(return_value=True)
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
first = await completion.handle_run_completion(
{"thread_id": "t1", "status": "error", "updated_at": "2026-07-09T00:00:00Z"}
)
second = await completion.handle_run_completion(
{"thread_id": "t1", "status": "timeout", "updated_at": "2026-07-09T01:00:00Z"}
)
assert first["status"] == "ok"
assert second["status"] == "ok"
assert reply.await_count == 2
@pytest.mark.asyncio
async def test_claim_written_before_post(monkeypatch: pytest.MonkeyPatch) -> None:
# Claim-then-post: the dedup key is recorded before the network reply fires.
client = _FakeClient(_slack_metadata())
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
order: list[str] = []
async def _update(*, thread_id: str, metadata: dict[str, Any]) -> None:
order.append("claim")
client.threads.updates.append(metadata)
client.threads._metadata.update(metadata)
async def _reply(*args: Any, **kwargs: Any) -> bool:
order.append("post")
return True
monkeypatch.setattr(client.threads, "update", _update)
monkeypatch.setattr(completion, "post_slack_thread_reply", _reply)
result = await completion.handle_run_completion(
{"thread_id": "t1", "run_id": "run-1", "status": "error"}
)
assert result["status"] == "ok"
assert order == ["claim", "post"]
@pytest.mark.asyncio
async def test_claim_write_failure_does_not_post_and_retry_posts_once(
monkeypatch: pytest.MonkeyPatch,
) -> None:
# SR160-02: if the claim write throws, we do NOT post and report an error so
# the retry re-attempts — the retry then claims and posts exactly once,
# rather than the old flow double-posting on retry.
client = _FakeClient(_slack_metadata(), fail_updates=1)
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
reply = AsyncMock(return_value=True)
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
first = await completion.handle_run_completion(
{"thread_id": "t1", "run_id": "run-1", "status": "error"}
)
assert first["status"] == "error"
reply.assert_not_called()
retry = await completion.handle_run_completion(
{"thread_id": "t1", "run_id": "run-1", "status": "error"}
)
assert retry["status"] == "ok"
reply.assert_awaited_once()
@pytest.mark.asyncio
async def test_duplicate_delivery_posts_exactly_once(monkeypatch: pytest.MonkeyPatch) -> None:
# A retried/duplicate delivery for one run_id posts exactly once because the
# first delivery's claim is persisted before the second reads metadata.
client = _FakeClient(_slack_metadata())
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
reply = AsyncMock(return_value=True)
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
payload = {"thread_id": "t1", "run_id": "run-1", "status": "error"}
first = await completion.handle_run_completion(dict(payload))
second = await completion.handle_run_completion(dict(payload))
assert first["status"] == "ok"
assert second["status"] == "ignored"
reply.assert_awaited_once()
@pytest.mark.asyncio
async def test_no_reply_channel_does_not_flag(monkeypatch: pytest.MonkeyPatch) -> None:
client = _FakeClient({"source": "schedule"})
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
result = await completion.handle_run_completion(
{"thread_id": "t1", "run_id": "run-1", "status": "error"}
)
assert result["status"] == "ignored"
assert client.threads.updates == []
@pytest.mark.asyncio
async def test_interrupted_status_is_ignored(monkeypatch: pytest.MonkeyPatch) -> None:
# Follow-ups use multitask_strategy="interrupt", so an interrupted run is a
# healthy hand-off, not a failure to report.
client = _FakeClient(_slack_metadata())
monkeypatch.setattr(completion, "langgraph_client", lambda: client)
reply = AsyncMock(return_value=True)
monkeypatch.setattr(completion, "post_slack_thread_reply", reply)
result = await completion.handle_run_completion(
{"thread_id": "t1", "run_id": "run-1", "status": "interrupted"}
)
assert result["status"] == "ignored"
reply.assert_not_called()
assert client.threads.updates == []
def test_verify_run_complete_token(monkeypatch: pytest.MonkeyPatch) -> None:
# No secret configured: fail closed (reject everything).
monkeypatch.setattr(completion, "RUN_COMPLETE_WEBHOOK_SECRET", None)
assert completion.verify_run_complete_token(None) is False
assert completion.verify_run_complete_token("whatever") is False
# Secret configured: require an exact match.
monkeypatch.setattr(completion, "RUN_COMPLETE_WEBHOOK_SECRET", "s3cret")
assert completion.verify_run_complete_token("s3cret") is True
assert completion.verify_run_complete_token("wrong") is False
assert completion.verify_run_complete_token(None) is False