mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 06:53:14 +00:00
* feat: activate PR babysitting UI toggles for autofix and trigger mode Remove the "coming soon" gating on the Autofix Mode, Autofix Severity Threshold, and Trigger Mode controls in the review settings page so admins can enable CI auto-fix and review-comment resolution on PRs that Open SWE opens. The backend (ci_autofix.py, webapp.py webhook routing) was already fully wired — only the UI was disabled. Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> * feat: simplify autofix to on/off toggle, remove severity threshold Replace the four-level AutofixMode (off/low/medium/high) and the autofix_severity_threshold setting with a single boolean autofix_enabled toggle. The severity threshold was leftover from the reviewer finding-severity model and does not apply to CI autofix; the agent should fix any failing CI and resolve any comments on PRs it opens. Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> * feat: move autofix toggle to per-user profile, remove team-level setting The autofix toggle is now per-user (auto_fix_ci in the user profile) instead of team-level (admin-only). This uses the existing auto_fix_ci field that was already in ProfileUpdate but never wired up. Changes: - ci_autofix.py: check per-user auto_fix_ci profile flag after resolving the agent thread's github_login, instead of checking team-level autofix_enabled before knowing the PR - webapp.py: removed early is_autofix_enabled() webhook gates; the per-user check now happens in ci_autofix.py once the thread is found - team_settings.py: removed autofix_enabled field, is_autofix_enabled() - cloud-agents.tsx: enabled the auto_fix_ci toggle (was comingSoon) - review.tsx: removed the admin-level autofix switch - Updated tests and AGENTS.md The agent graph (not the reviewer) is what gets dispatched - this was already correct in ci_autofix.py line 223: client.runs.create( thread_id, "agent", ...). Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> * feat: batch PR babysitting events Remove the leftover trigger-mode gate from PR babysitting and batch new CI/review events while an agent run is already active so the running agent can handle the latest PR state before finishing. Also moves review-feedback permission checks behind the per-user opt-out and applies the auto-fix profile gate to merge-conflict babysitting. Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> * fix: consume batched babysitting events Teach the agent queue middleware to turn pending PR babysitting metadata into an injected instruction for the active run, so batched CI/review events are not dropped while still avoiding duplicate run creation. Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> * fix: address review findings in PR babysitting batching - Route batched events through the LangGraph store (read in-process by the message-queue middleware) instead of a per-model-call threads.get on every agent thread. - Only record an attempt / mark the head SHA handled on a real dispatch, not on a batch, so an event isn't permanently dropped if the in-flight run ends before consuming it. - Carry the reviewer's comment through batched review feedback instead of replacing it with a generic re-check nudge. --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
117 lines
4 KiB
Python
117 lines
4 KiB
Python
from __future__ import annotations
|
|
|
|
from typing import Any
|
|
from unittest.mock import MagicMock, patch
|
|
|
|
import pytest
|
|
|
|
from agent.middleware.check_message_queue import (
|
|
DASHBOARD_HANDOFF_MARKER,
|
|
_build_blocks_from_payload,
|
|
check_message_queue_before_model,
|
|
)
|
|
|
|
|
|
class _QueuedItem:
|
|
def __init__(self, value: dict[str, Any]) -> None:
|
|
self.value = value
|
|
|
|
|
|
class _FakeStore:
|
|
def __init__(self, items: dict[tuple[tuple[str, ...], str], dict[str, Any]]) -> None:
|
|
self.items = items
|
|
self.deleted: list[tuple[tuple[str, ...], str]] = []
|
|
|
|
async def aget(self, namespace: tuple[str, ...], key: str) -> _QueuedItem | None:
|
|
value = self.items.get((namespace, key))
|
|
return _QueuedItem(value) if value is not None else None
|
|
|
|
async def adelete(self, namespace: tuple[str, ...], key: str) -> None:
|
|
self.deleted.append((namespace, key))
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_check_message_queue_injects_dashboard_handoff_instruction() -> None:
|
|
store = _FakeStore(
|
|
{
|
|
(("queue", "thread-1"), "pending_messages"): {
|
|
"messages": [
|
|
{"content": {"text": "continue in web", "source": "dashboard"}},
|
|
]
|
|
}
|
|
}
|
|
)
|
|
|
|
with (
|
|
patch(
|
|
"agent.middleware.check_message_queue.get_config",
|
|
return_value={"configurable": {"thread_id": "thread-1"}},
|
|
),
|
|
patch("agent.middleware.check_message_queue.get_store", return_value=store),
|
|
):
|
|
result = await check_message_queue_before_model.abefore_model({}, MagicMock())
|
|
|
|
assert result is not None
|
|
message = result["messages"][0]
|
|
assert message["role"] == "user"
|
|
assert DASHBOARD_HANDOFF_MARKER in message["content"][0]["text"]
|
|
assert message["content"][1] == {"type": "text", "text": "continue in web"}
|
|
assert store.deleted == [(("queue", "thread-1"), "pending_messages")]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_check_message_queue_injects_pending_autofix_event() -> None:
|
|
store = _FakeStore(
|
|
{
|
|
(("autofix", "thread-1"), "pending_event"): {
|
|
"reason": "review_feedback",
|
|
"details": ["Reviewer alice commented: rename to userId"],
|
|
}
|
|
}
|
|
)
|
|
|
|
with (
|
|
patch(
|
|
"agent.middleware.check_message_queue.get_config",
|
|
return_value={"configurable": {"thread_id": "thread-1"}},
|
|
),
|
|
patch("agent.middleware.check_message_queue.get_store", return_value=store),
|
|
):
|
|
result = await check_message_queue_before_model.abefore_model({}, MagicMock())
|
|
|
|
assert result is not None
|
|
message = result["messages"][0]
|
|
assert message["role"] == "user"
|
|
text = message["content"][0]["text"]
|
|
assert "PR babysitting event arrived" in text
|
|
# The reviewer's actual comment is carried through, not dropped for a generic nudge.
|
|
assert "rename to userId" in text
|
|
assert (("autofix", "thread-1"), "pending_event") in store.deleted
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_build_blocks_skips_images_for_text_only_model() -> None:
|
|
payload = {
|
|
"text": "see this screenshot",
|
|
"image_urls": ["https://files.slack.com/fake.png"],
|
|
}
|
|
blocks = await _build_blocks_from_payload(
|
|
payload, model_id="fireworks:accounts/fireworks/models/glm-5p2"
|
|
)
|
|
assert len(blocks) == 1
|
|
assert blocks[0]["type"] == "text"
|
|
assert "does not support image input" in blocks[0]["text"]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_build_blocks_includes_images_for_vision_model() -> None:
|
|
payload: dict[str, Any] = {"text": "see this", "image_urls": []}
|
|
blocks = await _build_blocks_from_payload(payload, model_id="openai:gpt-5.5")
|
|
assert blocks == [{"type": "text", "text": "see this"}]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_build_blocks_no_model_check_fetches_images() -> None:
|
|
payload: dict[str, Any] = {"text": "see this", "image_urls": []}
|
|
blocks = await _build_blocks_from_payload(payload)
|
|
assert blocks == [{"type": "text", "text": "see this"}]
|