mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 09:13:14 +00:00
* feat: implement reviewer findings, publish_review, and watch mode
Build out the reviewer agent end-to-end against the design in
REVIEWER_DESIGN.md:
- Findings as first-class state on the reviewer thread metadata
(`agent/reviewer_findings.py`): Finding TypedDict with start_line/end_line
ranges, suggestion text for ```suggestion blocks, github_review_comment_id
for cross-run reconciliation, diff_hunk for UI rendering. Thread-level
metadata gets `kind=reviewer`, `pr`, `last_reviewed_sha`, `watch` so a
future frontend can list reviewer threads via the langgraph SDK.
- Diff utilities (`agent/reviewer_diff.py`): parse_unified_diff,
compute_diff_line_set for in-diff validation, extract_diff_hunk for
caching the hunk on a Finding, compute_diff_in_sandbox for SHA-to-SHA
diffs against the prepped repo.
- Tools: `add_finding` (validates against the diff line set so out-of-diff
ranges fail at creation, not at GitHub-publish), `update_finding`,
`list_findings`, `publish_review`. The reviewer agent's tool list is
swapped from `[]` (direct shell `gh api` calls) to these four.
- Publish path (`agent/reviewer_publish.py` + `agent/tools/publish_review.py`):
one POST /reviews call with body + inline comments + ```suggestion blocks,
per-comment IDs stored back on findings, GraphQL `resolveReviewThread`
fired for findings transitioning open->resolved on a re-review.
- Reviewer graph: deterministic clone-or-fetch + checkout in the factory
before the agent's first model call (warm- and cold-path symmetric);
computed diff and in-diff line set passed via runnable config; system
prompt rewritten for the single-evolving-findings model, severity ladder,
in-diff-only discipline, and watch-mode reconciliation flow.
- Watch mode in webapp.py: `push` event + `pull_request` closed/reopened
added to supported events. New `process_github_push_event` resolves the
open PR for the pushed branch, gates on the reviewer thread's `watch`
flag, builds a re-review configurable, and triggers a run on the same
canonical thread. `process_github_pr_close` toggles watch on
closed/reopened. `set_reviewer_thread_metadata` is called on first
review to install `kind=reviewer` + PR identity + watch=True.
- Eval harness: target.py now extracts `add_finding` calls (mapped to the
legacy {file, line, body, severity} shape the judge expects) and passes
the right configurable so the prep step has base/head SHAs.
- Tests: new unit suites for findings helpers, diff parsing, finding tools,
publish rendering + GraphQL resolve, and watch-mode webhook handlers
(push triggers re-review only when watching, idempotent on unchanged
head SHA, PR close disables watch). Updated existing reviewer-webhook
tests to mock `set_reviewer_thread_metadata`.
- REVIEWER_EVAL_PLAN.md removed per user request; folded relevant context
into REVIEWER_DESIGN.md.
* fix(reviewer): correct git diff flags, scope, dedup, and review-comments URL
Address PR #1253 review findings:
- compute_diff_in_sandbox dropped the invalid `--no-prefix=false` flag
(`option no-prefix takes no value` — every prep run was failing
silently and the agent saw an empty diff).
- compute_diff_in_sandbox grew a `merge_base` flag. First-review path
now uses three-dot `base...head` (the merge-base diff GitHub renders
on Files-changed) so we don't pick up changes that landed on the base
branch after the PR diverged. Re-review delta keeps two-dot
`last_reviewed_sha..head` since that's exactly the new commits.
- publish_review skips findings that already carry
`github_review_comment_id`. Without this, watched re-reviews
re-posted every previously surfaced finding, and only the most-recent
duplicate's id would later resolve when the issue got addressed.
- fetch_review_comments URL now includes `{pull_number}` —
`/repos/{owner}/{repo}/pulls/{pr_number}/reviews/{review_id}/comments`
is the canonical endpoint; the old form 404s, so comment ids were
never stored and watch-mode resolution couldn't run.
Three new tests cover: three-dot vs two-dot wiring, no `--no-prefix`
flag in the executed command, and that publish_review does not re-post
findings whose `github_review_comment_id` is set.
* fix(reviewer): default publish cap from 15 to 4
A clean PR with one critical issue padded out by three lower-severity
findings is fine; fifteen is review spam. The agent can override per
call when a PR genuinely warrants more.
168 lines
5.4 KiB
Python
168 lines
5.4 KiB
Python
"""Unit tests for the publish_review rendering and orchestration helpers."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import Any
|
|
from unittest.mock import AsyncMock, MagicMock, patch
|
|
|
|
import pytest
|
|
|
|
from agent.reviewer_findings import Finding, new_finding
|
|
from agent.reviewer_publish import (
|
|
render_inline_comment_body,
|
|
render_inline_comment_payload,
|
|
render_review_body,
|
|
resolve_review_thread,
|
|
)
|
|
|
|
|
|
def _f(**overrides: Any) -> Finding:
|
|
base = new_finding(
|
|
severity="high",
|
|
category="correctness",
|
|
file="src/foo.py",
|
|
start_line=10,
|
|
end_line=10,
|
|
description="boom",
|
|
sha="abc",
|
|
)
|
|
base.update(overrides) # type: ignore[arg-type]
|
|
return base
|
|
|
|
|
|
def test_render_inline_comment_body_without_suggestion() -> None:
|
|
body = render_inline_comment_body(_f(description="just text"))
|
|
assert body == "just text"
|
|
|
|
|
|
def test_render_inline_comment_body_with_suggestion_appends_block() -> None:
|
|
body = render_inline_comment_body(
|
|
_f(description="needs fix", suggestion="x = 1\nx += 1"),
|
|
)
|
|
assert "needs fix" in body
|
|
assert "```suggestion" in body
|
|
assert "x = 1\nx += 1" in body
|
|
|
|
|
|
def test_render_inline_comment_payload_single_line() -> None:
|
|
payload = render_inline_comment_payload(_f(start_line=10, end_line=10))
|
|
assert payload == {
|
|
"path": "src/foo.py",
|
|
"line": 10,
|
|
"side": "RIGHT",
|
|
"body": "boom",
|
|
}
|
|
|
|
|
|
def test_render_inline_comment_payload_multi_line_uses_start_fields() -> None:
|
|
payload = render_inline_comment_payload(_f(start_line=8, end_line=12))
|
|
assert payload is not None
|
|
assert payload["start_line"] == 8
|
|
assert payload["start_side"] == "RIGHT"
|
|
assert payload["line"] == 12
|
|
|
|
|
|
def test_render_inline_comment_payload_returns_none_for_file_level() -> None:
|
|
payload = render_inline_comment_payload(_f(start_line=None, end_line=None))
|
|
assert payload is None
|
|
|
|
|
|
def test_render_review_body_includes_summary_and_marker() -> None:
|
|
body = render_review_body(
|
|
pr_number=123,
|
|
surfaced_count=2,
|
|
total_open_count=3,
|
|
severity_threshold="medium",
|
|
summary="LGTM with two notes",
|
|
)
|
|
assert "LGTM with two notes" in body
|
|
assert "<!-- open-swe-reviewer pr=123 -->" in body
|
|
assert "1 lower-severity finding hidden" in body
|
|
|
|
|
|
def test_render_review_body_surfaces_no_findings_message() -> None:
|
|
body = render_review_body(
|
|
pr_number=99,
|
|
surfaced_count=0,
|
|
total_open_count=0,
|
|
severity_threshold="medium",
|
|
summary=None,
|
|
)
|
|
assert "No issues at or above" in body
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_resolve_review_thread_returns_true_on_success() -> None:
|
|
response = MagicMock()
|
|
response.json.return_value = {
|
|
"data": {"resolveReviewThread": {"thread": {"id": "T_1", "isResolved": True}}}
|
|
}
|
|
response.raise_for_status.return_value = None
|
|
|
|
client_cm = AsyncMock()
|
|
client_cm.__aenter__.return_value = client_cm
|
|
client_cm.post = AsyncMock(return_value=response)
|
|
|
|
with patch("agent.reviewer_publish.httpx.AsyncClient", return_value=client_cm):
|
|
ok = await resolve_review_thread(thread_node_id="T_1", token="t")
|
|
assert ok is True
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_resolve_review_thread_returns_false_on_graphql_errors() -> None:
|
|
response = MagicMock()
|
|
response.json.return_value = {"errors": [{"message": "no perms"}]}
|
|
response.raise_for_status.return_value = None
|
|
|
|
client_cm = AsyncMock()
|
|
client_cm.__aenter__.return_value = client_cm
|
|
client_cm.post = AsyncMock(return_value=response)
|
|
|
|
with patch("agent.reviewer_publish.httpx.AsyncClient", return_value=client_cm):
|
|
ok = await resolve_review_thread(thread_node_id="T_1", token="t")
|
|
assert ok is False
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_publish_review_skips_findings_already_published() -> None:
|
|
"""Re-runs must not re-post findings that already have a github_review_comment_id."""
|
|
from agent.tools.publish_review import _publish_review_async
|
|
|
|
findings = [
|
|
_f(id="f_old", severity="high", file="a.py", github_review_comment_id=42),
|
|
_f(id="f_new", severity="high", file="b.py"),
|
|
]
|
|
|
|
list_async = AsyncMock(return_value=findings)
|
|
post_review = AsyncMock(return_value={"id": 999})
|
|
fetch_comments = AsyncMock(return_value=[])
|
|
set_metadata = AsyncMock()
|
|
|
|
with (
|
|
patch("agent.tools.publish_review.get_thread_id_from_runtime", return_value="tid"),
|
|
patch("agent.tools.publish_review.list_findings_async", list_async),
|
|
patch("agent.tools.publish_review.post_pull_request_review", post_review),
|
|
patch("agent.tools.publish_review.fetch_review_comments", fetch_comments),
|
|
patch(
|
|
"agent.tools.publish_review._resolve_threads_for_resolved_findings",
|
|
new_callable=AsyncMock,
|
|
return_value=0,
|
|
),
|
|
patch("agent.tools.publish_review.set_reviewer_thread_metadata", set_metadata),
|
|
):
|
|
result = await _publish_review_async(
|
|
owner="o",
|
|
repo="r",
|
|
pr_number=7,
|
|
head_sha="sha",
|
|
token="t",
|
|
summary=None,
|
|
severity_threshold="medium",
|
|
cap=15,
|
|
)
|
|
|
|
assert result["success"] is True
|
|
assert result["surfaced_count"] == 1
|
|
posted = post_review.await_args.kwargs["inline_comments"]
|
|
paths = {c["path"] for c in posted}
|
|
assert paths == {"b.py"}
|