mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 09:13:14 +00:00
Applies the plan's C5 step: git mv every test per the domain-reorg
move-map (movemap-m50.txt) into tests/{agent,analyzer,auth,dashboard,
github,middleware,models,reviewer,sandbox,slack,tools,webhooks}/, plus
the 13 fork-only placements from the scoping report §2c (Atlassian
webhook tests -> tests/webhooks/, test_atlassian_connect.py and
test_auth_error_leak.py -> tests/auth/, jira/confluence util tests ->
tests/tools/, test_repo_binding_isolation.py -> tests/sandbox/,
bot-identity/autofix tests -> tests/github/).
Path-only move: the only content edits are parents[1] -> parents[2]
fixes in test_e2b_integration.py and test_daytona_integration.py,
required because their __file__-relative ROOT path gained one more
directory level in the move.
Monkeypatch retargets for these files were already completed in C4;
none remained outstanding here.
114 lines
3.7 KiB
Python
114 lines
3.7 KiB
Python
from __future__ import annotations
|
|
|
|
import json
|
|
from types import SimpleNamespace
|
|
from unittest.mock import MagicMock, patch
|
|
from uuid import uuid4
|
|
|
|
from evals.reviewer import judge
|
|
|
|
|
|
def _result(match: bool, confidence: float) -> judge.PairResult:
|
|
return {"match": match, "confidence": confidence, "reasoning": "reason"}
|
|
|
|
|
|
def test_select_pairs_maximizes_cardinality_before_confidence() -> None:
|
|
matrix = [
|
|
[_result(True, 0.9), _result(True, 0.8)],
|
|
[_result(True, 0.7), _result(False, 0.0)],
|
|
]
|
|
|
|
assert set(judge._select_pairs(matrix)) == {(0, 1), (1, 0)}
|
|
|
|
|
|
def test_select_pairs_uses_confidence_to_break_cardinality_ties() -> None:
|
|
matrix = [
|
|
[_result(True, 0.9), _result(True, 0.1)],
|
|
[_result(True, 0.2), _result(True, 0.8)],
|
|
]
|
|
|
|
assert set(judge._select_pairs(matrix)) == {(0, 0), (1, 1)}
|
|
|
|
|
|
def test_recall_at_cap_never_exceeds_one() -> None:
|
|
recall_at_cap, ceiling = judge._recall_at_cap(tp=7, golden_count=7, cap=6)
|
|
|
|
assert recall_at_cap == 1.0
|
|
assert ceiling == 6 / 7
|
|
|
|
|
|
def test_judge_match_deduplicates_and_persists_full_matrix() -> None:
|
|
run = SimpleNamespace(
|
|
outputs={
|
|
"comments": [
|
|
{"file": "a.py", "line": 1, "body": "same", "severity": "high"},
|
|
{"file": "a.py", "line": 1, "body": " same ", "severity": "high"},
|
|
{"file": "b.py", "line": 2, "body": "other", "severity": "medium"},
|
|
]
|
|
}
|
|
)
|
|
example = SimpleNamespace(
|
|
id=uuid4(),
|
|
inputs={"repo": "acme/repo"},
|
|
outputs={
|
|
"golden_comments": [
|
|
{"comment": "first", "severity": "High"},
|
|
{"comment": "second", "severity": "Medium"},
|
|
]
|
|
},
|
|
)
|
|
calls: list[tuple[str, str]] = []
|
|
|
|
def _pair(golden: judge.ReviewComment, candidate: judge.ReviewComment) -> judge.PairResult:
|
|
calls.append((golden.get("comment", ""), candidate.get("body", "")))
|
|
return _result(golden.get("comment") == "first" and candidate.get("file") == "a.py", 0.8)
|
|
|
|
with patch("evals.reviewer.judge._judge_pair", side_effect=_pair):
|
|
result = judge.judge_match(run, example)
|
|
|
|
by_key = {item["key"]: item for item in result["results"]}
|
|
assert len(calls) == 4
|
|
assert by_key["n_candidates_raw"]["score"] == 3
|
|
assert by_key["n_candidates"]["score"] == 2
|
|
assert by_key["n_duplicates"]["score"] == 1
|
|
matrix = json.loads(by_key["pairwise_match_matrix"]["value"])
|
|
assert len(matrix["cells"]) == 4
|
|
assert matrix["selected_pairs"] == [{"candidate_index": 0, "golden_index": 0}]
|
|
|
|
|
|
def test_judge_pair_normalizes_malformed_response() -> None:
|
|
model = MagicMock()
|
|
model.invoke.return_value.content = '{"match": "yes", "confidence": 4, "reasoning": 7}'
|
|
|
|
with patch("evals.reviewer.judge._get_judge", return_value=model):
|
|
result = judge._judge_pair({"comment": "gold"}, {"body": "candidate"})
|
|
|
|
assert result == {"match": False, "confidence": 1.0, "reasoning": ""}
|
|
|
|
|
|
def test_aggregate_pr_reports_synthetic_and_medium_plus_metrics() -> None:
|
|
judge._drain_counts()
|
|
base = {
|
|
"tp": 1,
|
|
"fp": 1,
|
|
"fn": 0,
|
|
"precision": 0.5,
|
|
"recall": 1.0,
|
|
"f1": 2 / 3,
|
|
"medium_plus_tp": 1,
|
|
"medium_plus_fp": 0,
|
|
"medium_plus_fn": 0,
|
|
"medium_plus_precision": 1.0,
|
|
"medium_plus_recall": 1.0,
|
|
"medium_plus_f1": 1.0,
|
|
}
|
|
judge._record_counts(uuid4(), {**base, "is_synthetic": False})
|
|
judge._record_counts(uuid4(), {**base, "is_synthetic": True})
|
|
|
|
result = judge.aggregate_pr([], [])
|
|
|
|
keys = {item["key"] for item in result["results"]}
|
|
assert "micro_f1" in keys
|
|
assert "medium_plus_micro_f1" in keys
|
|
assert "synthetic_micro_f1" in keys
|
|
assert "upstream_micro_f1" in keys
|