open-swe/tests/test_slack_start_new_thread_tool.py

260 lines
9.2 KiB
Python
Raw Normal View History

from __future__ import annotations
import importlib
from typing import Any
import pytest
from agent.utils.thread_ids import generate_thread_id_from_slack_thread
slack_breakout_tool = importlib.import_module("agent.tools.slack_start_new_thread")
def _config() -> dict[str, Any]:
return {
"configurable": {
"repo": {"owner": "langchain-ai", "name": "open-swe"},
"github_login": "alice",
"user_email": "alice@example.com",
feat: Re-land deferred upstream features on modular webhooks (#80) (#128) * fix(webhooks): fall back to vision model for Slack/Linear image threads Re-land upstream #1626 onto the modular webhook structure. When a Slack mention or Linear issue carries images but the resolved model is text-only, fall back to a vision-capable model instead of dropping the images. Re-points default_vision_model_pair at the fork's image-capable models (Opus 4.8 default, else any supports_images model) rather than upstream's openai:/anthropic: provider filter. Refs #80, upstream #1626 * fix(slack): persist trace_message_ts so web-handoff updates the trace reply Re-land upstream #1630 onto the modular structure. The first-mention store_slack_run_mapping call did not pass trace_message_ts, so it was never persisted (nothing to preserve from on first mention) and _notify_slack_web_handoff always skipped the trace-reply update on web handoff. Pass it through and cover it with a test. Refs #80, upstream #1630 * feat(slack): include channel context in Slack prompts Re-land upstream #1633 onto the modular structure. Fetch cached Slack channel metadata once per event (_get_slack_channel_context) and thread it through the docs-plz gate, repo resolution, and process_slack_mention so prompts carry the channel name and a clearly-marked untrusted channel description. Avoids duplicate conversations.info calls. Refs #80, upstream #1633 * feat(tools): add slack_start_new_thread breakout tool Re-land upstream #1638 onto the modular structure. Adds the slack_start_new_thread tool (posts a top-level Slack message and dispatches a fresh agent run for a broken-out task via the durable dispatch_agent_run contract), wires it into the agent tool list and tools/__init__, adds prompt guidance, and excludes it from plan mode so it can't bypass the approval flow. Tool imports only live modules. Refs #80, upstream #1638 * feat(plan): notify Slack on plan approval Re-land upstream #1632 onto the modular structure. When a plan is approved via the dashboard approve endpoint, post a thread reply to the originating Slack thread noting the comment count and approver, after the follow-up run is dispatched. Slack post failures never break approval. Adapted to the fork's approve_plan (no plan_markdown read). Refs #80, upstream #1632 * feat(plan): publish plans from sandbox files Re-land upstream #1635 onto the modular structure, completing the partially-ported change so dev is internally consistent. save_plan now takes a plan_file_path, reads the agent-authored Markdown file from /workspace/plans/ (validating extension/location/UTF-8/size) and publishes it, instead of taking a plan_markdown string. Removes write_file/edit_file from PLAN_MODE_EXCLUDED_TOOLS so the agent can author the plan file, updates enter_plan_mode/reject_plan guidance and the e2e fake LLM. Skips the #1610-only update_plan hunk (not on dev). Refs #80, upstream #1635 * fix(security): SSRF-harden server-side image fetch + stop logging raw image URLs INJ-01 (high): fetch_image_block used follow_redirects=True with no per-hop revalidation and discarded the resolved-IP pin, so an attacker-authored Slack/ Linear image URL could 302-redirect the fetch to an internal host / cloud metadata endpoint (blind SSRF), and DNS-rebinding could bypass the one-shot is_url_safe check. Route image fetches through the same per-hop resolve+pin+ revalidate loop the http_request tool uses, lifted into url_safety as the shared request_with_safe_redirects. Also strip the per-host Slack/Linear bearer token on redirect so it can't be replayed to a redirect target. SC-1 (low): linear.py logged full image URLs (which can carry signed tokens) at DEBUG; multimodal logged them at INFO on every fetch. Log host-only. Sink lived in multimodal.py (unchanged by the feature work) but PR #128 widened its reach by no longer dropping images for text-only models. Fixing on the base branch so #130/#129 inherit it on rebase. Adds fetch_image_block SSRF regression tests (redirect-to-internal blocked; auth stripped on redirect).
2026-07-08 18:32:43 -04:00
"agent_model_id": "bedrock_converse:us.anthropic.claude-opus-4-8",
"agent_effort": "high",
"slack_thread": {
"channel_id": "C1",
"thread_ts": "1700000000.000001",
"triggering_user_id": "U1",
"triggering_user_name": "Alice",
"triggering_user_email": "alice@example.com",
"triggering_event_ts": "1700000000.000002",
},
}
}
class _FakeThreadsClient:
def __init__(self, captured: dict[str, Any]) -> None:
self.captured = captured
async def create(self, *, thread_id: str, if_exists: str, metadata: dict[str, Any]) -> None:
self.captured["thread_create"] = {
"thread_id": thread_id,
"if_exists": if_exists,
"metadata": metadata,
}
async def update(self, *, thread_id: str, metadata: dict[str, Any]) -> None:
self.captured["thread_update"] = {"thread_id": thread_id, "metadata": metadata}
class _FakeClient:
def __init__(self, captured: dict[str, Any]) -> None:
self.threads = _FakeThreadsClient(captured)
async def test_slack_start_new_thread_success(monkeypatch: pytest.MonkeyPatch) -> None:
captured: dict[str, Any] = {"stored_mappings": []}
new_ts = "1700000000.111111"
trace_ts = "1700000000.222222"
async def fake_post_top_level(
channel_id: str,
text: str,
*,
unfurl_links: bool = True,
unfurl_media: bool = True,
blocks: list[dict[str, Any]] | None = None,
) -> tuple[str | None, str | None]:
captured["top_level_post"] = {
"channel_id": channel_id,
"text": text,
"unfurl_links": unfurl_links,
"unfurl_media": unfurl_media,
"blocks": blocks,
}
return new_ts, None
async def fake_dispatch_agent_run(
thread_id: str,
content: str,
configurable: dict[str, Any],
*,
source: str,
client: Any,
**kwargs: Any,
) -> dict[str, str]:
captured["dispatch"] = {
"thread_id": thread_id,
"content": content,
"configurable": configurable,
"source": source,
"client": client,
"kwargs": kwargs,
}
return {"run_id": "run-123"}
async def fake_post_trace(channel_id: str, thread_ts: str, thread_id: str) -> str:
captured["trace"] = {
"channel_id": channel_id,
"thread_ts": thread_ts,
"thread_id": thread_id,
}
return trace_ts
async def fake_store_mapping(
client: Any,
channel_id: str,
thread_ts: str,
run_id: str,
*,
message_ts: str | None = None,
triggering_user_id: str | None = None,
) -> None:
captured["stored_mappings"].append(
{
"client": client,
"channel_id": channel_id,
"thread_ts": thread_ts,
"run_id": run_id,
"message_ts": message_ts,
"triggering_user_id": triggering_user_id,
}
)
fake_client = _FakeClient(captured)
monkeypatch.setattr(slack_breakout_tool, "get_config", _config)
monkeypatch.setattr(slack_breakout_tool, "get_client", lambda url: fake_client)
monkeypatch.setattr(
slack_breakout_tool, "post_slack_top_level_message_with_ts", fake_post_top_level
)
monkeypatch.setattr(slack_breakout_tool, "dispatch_agent_run", fake_dispatch_agent_run)
monkeypatch.setattr(slack_breakout_tool, "post_slack_trace_reply", fake_post_trace)
monkeypatch.setattr(slack_breakout_tool, "store_slack_run_mapping", fake_store_mapping)
monkeypatch.setattr(
slack_breakout_tool,
"dashboard_thread_url",
lambda thread_id: f"https://dashboard.example/agents/{thread_id}",
)
result = await slack_breakout_tool.slack_start_new_thread(
"Investigate follow-up",
"Use the same repo and investigate the follow-up aspect in detail.",
)
expected_thread_id = generate_thread_id_from_slack_thread("C1", new_ts)
assert result == {
"success": True,
"thread_id": expected_thread_id,
"thread_ts": new_ts,
"dashboard_url": f"https://dashboard.example/agents/{expected_thread_id}",
}
assert captured["top_level_post"]["channel_id"] == "C1"
assert "Investigate follow-up" in captured["top_level_post"]["text"]
assert "langchain-ai/open-swe" in captured["top_level_post"]["text"]
assert captured["top_level_post"]["unfurl_links"] is False
assert captured["thread_create"]["if_exists"] == "do_nothing"
assert captured["thread_create"]["thread_id"] == expected_thread_id
metadata = captured["thread_update"]["metadata"]
assert metadata["source"] == "slack"
assert metadata["repo"] == {"owner": "langchain-ai", "name": "open-swe"}
assert metadata["github_login"] == "alice"
assert metadata["triggering_user_email"] == "alice@example.com"
assert metadata["source_context"]["slack_thread"]["thread_ts"] == new_ts
assert metadata["source_context"]["slack_thread"]["triggering_user_id"] == "U1"
assert metadata["source_context"]["breakout_from"] == {
"channel_id": "C1",
"thread_ts": "1700000000.000001",
"message_ts": "1700000000.000002",
}
dispatch = captured["dispatch"]
assert dispatch["thread_id"] == expected_thread_id
assert dispatch["source"] == "slack"
assert dispatch["configurable"]["slack_thread"]["thread_ts"] == new_ts
assert dispatch["configurable"]["repo"] == {"owner": "langchain-ai", "name": "open-swe"}
assert dispatch["configurable"]["github_login"] == "alice"
feat: Re-land deferred upstream features on modular webhooks (#80) (#128) * fix(webhooks): fall back to vision model for Slack/Linear image threads Re-land upstream #1626 onto the modular webhook structure. When a Slack mention or Linear issue carries images but the resolved model is text-only, fall back to a vision-capable model instead of dropping the images. Re-points default_vision_model_pair at the fork's image-capable models (Opus 4.8 default, else any supports_images model) rather than upstream's openai:/anthropic: provider filter. Refs #80, upstream #1626 * fix(slack): persist trace_message_ts so web-handoff updates the trace reply Re-land upstream #1630 onto the modular structure. The first-mention store_slack_run_mapping call did not pass trace_message_ts, so it was never persisted (nothing to preserve from on first mention) and _notify_slack_web_handoff always skipped the trace-reply update on web handoff. Pass it through and cover it with a test. Refs #80, upstream #1630 * feat(slack): include channel context in Slack prompts Re-land upstream #1633 onto the modular structure. Fetch cached Slack channel metadata once per event (_get_slack_channel_context) and thread it through the docs-plz gate, repo resolution, and process_slack_mention so prompts carry the channel name and a clearly-marked untrusted channel description. Avoids duplicate conversations.info calls. Refs #80, upstream #1633 * feat(tools): add slack_start_new_thread breakout tool Re-land upstream #1638 onto the modular structure. Adds the slack_start_new_thread tool (posts a top-level Slack message and dispatches a fresh agent run for a broken-out task via the durable dispatch_agent_run contract), wires it into the agent tool list and tools/__init__, adds prompt guidance, and excludes it from plan mode so it can't bypass the approval flow. Tool imports only live modules. Refs #80, upstream #1638 * feat(plan): notify Slack on plan approval Re-land upstream #1632 onto the modular structure. When a plan is approved via the dashboard approve endpoint, post a thread reply to the originating Slack thread noting the comment count and approver, after the follow-up run is dispatched. Slack post failures never break approval. Adapted to the fork's approve_plan (no plan_markdown read). Refs #80, upstream #1632 * feat(plan): publish plans from sandbox files Re-land upstream #1635 onto the modular structure, completing the partially-ported change so dev is internally consistent. save_plan now takes a plan_file_path, reads the agent-authored Markdown file from /workspace/plans/ (validating extension/location/UTF-8/size) and publishes it, instead of taking a plan_markdown string. Removes write_file/edit_file from PLAN_MODE_EXCLUDED_TOOLS so the agent can author the plan file, updates enter_plan_mode/reject_plan guidance and the e2e fake LLM. Skips the #1610-only update_plan hunk (not on dev). Refs #80, upstream #1635 * fix(security): SSRF-harden server-side image fetch + stop logging raw image URLs INJ-01 (high): fetch_image_block used follow_redirects=True with no per-hop revalidation and discarded the resolved-IP pin, so an attacker-authored Slack/ Linear image URL could 302-redirect the fetch to an internal host / cloud metadata endpoint (blind SSRF), and DNS-rebinding could bypass the one-shot is_url_safe check. Route image fetches through the same per-hop resolve+pin+ revalidate loop the http_request tool uses, lifted into url_safety as the shared request_with_safe_redirects. Also strip the per-host Slack/Linear bearer token on redirect so it can't be replayed to a redirect target. SC-1 (low): linear.py logged full image URLs (which can carry signed tokens) at DEBUG; multimodal logged them at INFO on every fetch. Log host-only. Sink lived in multimodal.py (unchanged by the feature work) but PR #128 widened its reach by no longer dropping images for text-only models. Fixing on the base branch so #130/#129 inherit it on rebase. Adds fetch_image_block SSRF regression tests (redirect-to-internal blocked; auth stripped on redirect).
2026-07-08 18:32:43 -04:00
assert (
dispatch["configurable"]["agent_model_id"]
== "bedrock_converse:us.anthropic.claude-opus-4-8"
)
assert "Breakout Instructions" in dispatch["content"]
assert captured["trace"] == {
"channel_id": "C1",
"thread_ts": new_ts,
"thread_id": expected_thread_id,
}
assert [item["message_ts"] for item in captured["stored_mappings"]] == [new_ts, trace_ts]
assert all(item["triggering_user_id"] == "U1" for item in captured["stored_mappings"])
async def test_slack_start_new_thread_requires_slack_config(
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setattr(slack_breakout_tool, "get_config", lambda: {"configurable": {}})
result = await slack_breakout_tool.slack_start_new_thread("Title", "Instructions")
assert result == {"success": False, "error": "Missing slack_thread config"}
@pytest.mark.parametrize(
("title", "instructions", "error"),
[
("", "Instructions", "title is required"),
("Title", "", "instructions is required"),
("x" * 161, "Instructions", "title is too long"),
("Title", "x" * 12001, "instructions is too long"),
],
)
async def test_slack_start_new_thread_validates_text(
title: str,
instructions: str,
error: str,
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setattr(slack_breakout_tool, "get_config", _config)
result = await slack_breakout_tool.slack_start_new_thread(title, instructions)
assert result["success"] is False
assert result["error"] == error
async def test_slack_start_new_thread_rejects_invalid_repo_override(
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setattr(slack_breakout_tool, "get_config", _config)
result = await slack_breakout_tool.slack_start_new_thread(
"Title", "Instructions", default_repo="https://github.com/langchain-ai/open-swe"
)
assert result == {
"success": False,
"error": "default_repo must be a simple owner/name repository string",
}
async def test_slack_start_new_thread_returns_slack_failure_without_dispatch(
monkeypatch: pytest.MonkeyPatch,
) -> None:
captured: dict[str, bool] = {"dispatched": False}
async def fake_post_top_level(*args: Any, **kwargs: Any) -> tuple[str | None, str | None]:
return None, "msg_too_long"
async def fake_dispatch_agent_run(*args: Any, **kwargs: Any) -> dict[str, str]:
captured["dispatched"] = True
return {"run_id": "run-123"}
monkeypatch.setattr(slack_breakout_tool, "get_config", _config)
monkeypatch.setattr(
slack_breakout_tool, "post_slack_top_level_message_with_ts", fake_post_top_level
)
monkeypatch.setattr(slack_breakout_tool, "dispatch_agent_run", fake_dispatch_agent_run)
result = await slack_breakout_tool.slack_start_new_thread("Title", "Instructions")
assert result["success"] is False
assert result["error"] == "msg_too_long"
assert result["slack_error"] == "msg_too_long"
assert "shorter" in result["hint"]
assert captured["dispatched"] is False