open-swe/tests/test_reviewer_eval_target.py
Adam Moussa 62d9945df4
refactor: consolidate reviewer modules into agent/review/
Part of the domain-reorg adoption (build plan step C2): fork content,
upstream layout. Nine 1:1 module moves (reviewer_diff/eval_store/
findings/groups/publish/reconcile/trace_context + review_style_
collector/guidance) into agent/review/, with internal relative
imports re-wired to the new package depth. agent/review/__init__.py
mirrors upstream's thin re-export shim (one of the 21 verified "A"
structural adds).

Rewrote the 38 grep hits across importer files (agent/{analyzer,
ci_autofix,reviewer,webapp}.py, agent/dashboard/*, agent/middleware/
settle_review_check.py, agent/tools/*, agent/utils/github_feedback.py,
agent/webhooks/github.py, evals/reviewer/*, and the reviewer test
suite) to point at agent.review.*; 4 of the 38 hits were name
collisions (list_reviewer_findings, reviewer_outcomes,
_reviewer_thread_id, reviewer_thread_id — not the moved modules) and
were left untouched. tests/test_github_checks.py's module-alias
import (`from agent import reviewer_publish`) follows upstream's own
`from agent.review import publish as reviewer_publish` pattern so
downstream `reviewer_publish.*` call sites needed no changes.
agent/reviewer.py and agent/webapp.py stay in place per the hard
rule (fork content, import-only rewire) and are not part of this
package.

Gates: ruff check + ruff format --check, pytest --co -q (1637
collected), full unit suite (1637 passed), and the reviewer/findings
suite in isolation (pytest -k "review or finding", 421 passed).
2026-07-17 13:52:03 -04:00

189 lines
5.4 KiB
Python

from __future__ import annotations
from typing import Any
import pytest
from agent.review.findings import new_finding
from evals.reviewer import target
def test_eval_target_marks_runs_as_eval_dry_run(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.delenv("REVIEWER_EVAL_MODEL_ID", raising=False)
monkeypatch.delenv("REVIEWER_EVAL_REASONING_EFFORT", raising=False)
configurable = target._build_configurable(
{
"repo": "acme/repo",
"pr_number": 1,
"pr_url": "https://github.com/acme/repo/pull/1",
"base_sha": "base",
"head_sha": "head",
"head_ref": "branch",
}
)
assert configurable["reviewer_eval"] is True
assert configurable["eval"] is True
assert configurable["__is_for_execution__"] is True
assert configurable["reviewer_eval_cap"] == 6
assert "source" not in configurable
def test_eval_target_passes_configured_cap(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("REVIEWER_EVAL_CAP", "1")
configurable = target._build_configurable(
{
"repo": "acme/repo",
"pr_number": 1,
"pr_url": "https://github.com/acme/repo/pull/1",
"base_sha": "base",
"head_sha": "head",
"head_ref": "branch",
}
)
assert configurable["reviewer_eval_cap"] == 1
def test_eval_target_passes_model_overrides(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("REVIEWER_EVAL_MODEL_ID", "anthropic:claude-opus-4-8")
monkeypatch.setenv("REVIEWER_EVAL_REASONING_EFFORT", "high")
configurable = target._build_configurable(
{
"repo": "acme/repo",
"pr_number": 1,
"pr_url": "https://github.com/acme/repo/pull/1",
"base_sha": "base",
"head_sha": "head",
"head_ref": "branch",
}
)
assert configurable["reviewer_model_id"] == "anthropic:claude-opus-4-8"
assert configurable["reviewer_reasoning_effort"] == "high"
def _result_with_findings(findings: list[dict[str, Any]]) -> dict[str, Any]:
return {"messages": [{"tool_calls": [{"name": "add_finding", "args": f} for f in findings]}]}
def test_extract_comments_includes_all_confidences() -> None:
result = _result_with_findings(
[
{
"file": "a.py",
"severity": "high",
"confidence": "low",
"description": "lo",
"start_line": 1,
"end_line": 1,
},
{
"file": "b.py",
"severity": "high",
"confidence": "high",
"description": "hi",
"start_line": 2,
"end_line": 2,
},
]
)
comments = target._extract_comments(result)
assert {c["file"] for c in comments} == {"a.py", "b.py"}
@pytest.mark.asyncio
async def test_extract_surfaced_comments_uses_publish_filter(
monkeypatch: pytest.MonkeyPatch,
) -> None:
high = new_finding(
severity="high",
confidence="high",
category="correctness",
file="a.py",
start_line=10,
end_line=10,
description="high signal",
sha="head",
finding_id="f_high",
)
low = new_finding(
severity="low",
confidence="high",
category="style",
file="b.py",
start_line=20,
end_line=20,
description="low signal",
sha="head",
finding_id="f_low",
)
class Threads:
async def get(self, _thread_id: str) -> dict[str, Any]:
return {
"metadata": {
"findings": [high, low],
"reviewer_eval_publication": {
"finding_ids": ["f_high"],
"severity_threshold": "medium",
"cap": 6,
},
}
}
class Client:
threads = Threads()
monkeypatch.setenv("REVIEWER_EVAL_SEVERITY_THRESHOLD", "medium")
monkeypatch.setenv("REVIEWER_EVAL_CAP", "4")
comments, publish_completed = await target._extract_surfaced_comments(Client(), "tid")
assert comments == [
{
"file": "a.py",
"line": 10,
"body": "high signal",
"severity": "high",
}
]
assert publish_completed is True
@pytest.mark.asyncio
async def test_extract_surfaced_comments_requires_publication_snapshot() -> None:
class Threads:
async def get(self, _thread_id: str) -> dict[str, Any]:
return {"metadata": {"findings": []}}
class Client:
threads = Threads()
comments, publish_completed = await target._extract_surfaced_comments(Client(), "tid")
assert comments == []
assert publish_completed is False
def test_extract_comments_deduplicates_identical_tool_calls() -> None:
finding = {
"file": "a.py",
"severity": "high",
"description": "Same issue",
"start_line": 1,
"end_line": 1,
}
comments = target._extract_comments(_result_with_findings([finding, finding]))
assert len(comments) == 1
def test_completed_counter_increments() -> None:
start = target.get_completed_count()
target._record_completed()
target._record_completed()
assert target.get_completed_count() == start + 2