open-swe/tests/test_reviewer_eval_run.py
Johannes du Plessis 834efbc33c
feat: Adds ability to run evals against deployment (#1311)
* feat: tighten reviewer eval workflow

Require the reviewer to verify and dedupe findings before recording them, and make benchmark runs safe to execute against deployed reviewer graphs without posting GitHub reviews.

* chore: move reviewer eval settings to config

Load reviewer benchmark settings from the default eval config file so deployed eval runs do not require a wide CLI surface.

* feat: allow reviewer eval model overrides

Pass reviewer model and reasoning effort from the eval config into reviewer runs so isolated benchmark deployments can test Opus 4.7 high thinking.

* fix: use adaptive thinking for Opus 4.7

Switch Opus 4.7 model overrides to Anthropic adaptive thinking with effort instead of the deprecated budgeted thinking payload rejected by the API.

* refactor: use latest Anthropic effort API

Remove legacy Anthropic budget-token thinking support and route Anthropic efforts through adaptive thinking plus effort.

* revert prompting
2026-05-18 15:47:13 -07:00

60 lines
2.1 KiB
Python

from __future__ import annotations
import os
from unittest.mock import patch
from evals.reviewer.run_eval import _apply_config_to_env, _coerce_config
def test_reviewer_eval_config_coerces_known_values() -> None:
config = _coerce_config(
{
"dataset_name": "dataset",
"experiment_prefix": "experiment",
"max_concurrency": 2,
"langgraph_url": "https://example.test",
"assistant_id": "reviewer",
"model_id": "anthropic:claude-opus-4-7",
"reasoning_effort": "high",
"score_mode": "surfaced_findings",
"severity_threshold": "medium",
"cap": 4,
"unknown": "ignored",
}
)
assert config == {
"dataset_name": "dataset",
"experiment_prefix": "experiment",
"max_concurrency": 2,
"langgraph_url": "https://example.test",
"assistant_id": "reviewer",
"model_id": "anthropic:claude-opus-4-7",
"reasoning_effort": "high",
"score_mode": "surfaced_findings",
"severity_threshold": "medium",
"cap": 4,
}
def test_reviewer_eval_config_sets_target_env() -> None:
with patch.dict(os.environ, {}, clear=True):
_apply_config_to_env(
{
"langgraph_url": "https://example.test",
"assistant_id": "reviewer",
"model_id": "anthropic:claude-opus-4-7",
"reasoning_effort": "high",
"score_mode": "surfaced_findings",
"severity_threshold": "high",
"cap": 3,
}
)
assert os.environ["LANGGRAPH_URL"] == "https://example.test"
assert os.environ["REVIEWER_ASSISTANT_ID"] == "reviewer"
assert os.environ["REVIEWER_EVAL_MODEL_ID"] == "anthropic:claude-opus-4-7"
assert os.environ["REVIEWER_EVAL_REASONING_EFFORT"] == "high"
assert os.environ["REVIEWER_EVAL_SCORE_MODE"] == "surfaced_findings"
assert os.environ["REVIEWER_EVAL_SEVERITY_THRESHOLD"] == "high"
assert os.environ["REVIEWER_EVAL_CAP"] == "3"