feat: Adds ability to run evals against deployment (#1311)

* feat: tighten reviewer eval workflow

Require the reviewer to verify and dedupe findings before recording them, and make benchmark runs safe to execute against deployed reviewer graphs without posting GitHub reviews.

* chore: move reviewer eval settings to config

Load reviewer benchmark settings from the default eval config file so deployed eval runs do not require a wide CLI surface.

* feat: allow reviewer eval model overrides

Pass reviewer model and reasoning effort from the eval config into reviewer runs so isolated benchmark deployments can test Opus 4.7 high thinking.

* fix: use adaptive thinking for Opus 4.7

Switch Opus 4.7 model overrides to Anthropic adaptive thinking with effort instead of the deprecated budgeted thinking payload rejected by the API.

* refactor: use latest Anthropic effort API

Remove legacy Anthropic budget-token thinking support and route Anthropic efforts through adaptive thinking plus effort.

* revert prompting
This commit is contained in:
Johannes du Plessis 2026-05-18 15:47:13 -07:00 • committed by GitHub
parent f662ad6587
commit 834efbc33c
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
13 changed files with 571 additions and 45 deletions

View file

@ -45,6 +45,9 @@ from .server import (
DEFAULT_LLM_REASONING, DEFAULT_LLM_REASONING,
DEFAULT_RECURSION_LIMIT, DEFAULT_RECURSION_LIMIT,
MODEL_CALL_RECURSION_LIMIT, MODEL_CALL_RECURSION_LIMIT,
_anthropic_effort_for,
_anthropic_thinking_for,
_openai_reasoning_for,
ensure_sandbox_for_thread, ensure_sandbox_for_thread,
graph_loaded_for_execution, graph_loaded_for_execution,
) )
@ -329,10 +332,25 @@ async def get_reviewer_agent(config: RunnableConfig) -> Pregel:
head_sha=head_sha, head_sha=head_sha,
) )
model_id = os.environ.get("LLM_MODEL_ID", DEFAULT_LLM_MODEL_ID) configured_model_id = config["configurable"].get("reviewer_model_id")
model_id = (
configured_model_id
if isinstance(configured_model_id, str) and configured_model_id
else os.environ.get("LLM_MODEL_ID", DEFAULT_LLM_MODEL_ID)
)
configured_effort = config["configurable"].get("reviewer_reasoning_effort")
reasoning_effort = configured_effort if isinstance(configured_effort, str) else None
model_kwargs: ModelKwargs = {"max_tokens": DEFAULT_LLM_MAX_TOKENS} model_kwargs: ModelKwargs = {"max_tokens": DEFAULT_LLM_MAX_TOKENS}
if model_id == DEFAULT_LLM_MODEL_ID: if model_id.startswith("openai:"):
model_kwargs["reasoning"] = DEFAULT_LLM_REASONING reasoning = _openai_reasoning_for(reasoning_effort)
model_kwargs["reasoning"] = reasoning if reasoning is not None else DEFAULT_LLM_REASONING
elif model_id.startswith("anthropic:"):
thinking = _anthropic_thinking_for(reasoning_effort)
if thinking is not None:
model_kwargs["thinking"] = thinking
effort = _anthropic_effort_for(reasoning_effort)
if effort is not None:
model_kwargs["effort"] = effort
system_prompt = _reviewer_system_prompt( system_prompt = _reviewer_system_prompt(
f"{work_dir}/{repo_name}" if repo_name else work_dir, f"{work_dir}/{repo_name}" if repo_name else work_dir,

View file

@ -64,6 +64,7 @@ from .utils.auth import resolve_github_token
from .utils.authorship import resolve_triggering_user_identity from .utils.authorship import resolve_triggering_user_identity
from .utils.github_app import get_github_app_installation_token from .utils.github_app import get_github_app_installation_token
from .utils.model import ( from .utils.model import (
AnthropicEffort,
AnthropicThinking, AnthropicThinking,
ModelKwargs, ModelKwargs,
OpenAIReasoning, OpenAIReasoning,
@ -359,30 +360,24 @@ def _openai_reasoning_for(profile_effort: str | None) -> OpenAIReasoning | None:
return None return None
# Mapping mirrors Claude Code's effort levels for Opus 4.7. Numbers are tuned _ANTHROPIC_EFFORTS: set[AnthropicEffort] = {"low", "medium", "high", "xhigh", "max"}
# so each level meaningfully separates from the next while leaving headroom
# under DEFAULT_LLM_MAX_TOKENS (64k) for the model's actual output.
_ANTHROPIC_THINKING_BUDGETS: dict[str, int] = {
"low": 1_024,
"medium": 4_000,
"high": 12_000,
"xhigh": 32_000,
"max": 60_000,
}
def _anthropic_thinking_for(profile_effort: str | None) -> AnthropicThinking | None: def _anthropic_thinking_for(profile_effort: str | None) -> AnthropicThinking | None:
"""Map a profile effort string to an Anthropic thinking kwarg. """Map a profile effort string to an Anthropic thinking kwarg.
Returns ``None`` when the effort doesn't have a known budget so we leave Latest Claude models use adaptive thinking with an effort hint instead of
the model's default thinking behaviour alone. manual thinking token budgets.
""" """
if not profile_effort: if profile_effort in _ANTHROPIC_EFFORTS:
return None return {"type": "adaptive"}
budget = _ANTHROPIC_THINKING_BUDGETS.get(profile_effort) return None
if budget is None:
return None
return {"type": "enabled", "budget_tokens": budget} def _anthropic_effort_for(profile_effort: str | None) -> AnthropicEffort | None:
if profile_effort in _ANTHROPIC_EFFORTS:
return profile_effort
return None
def _get_cached_sandbox_backend(thread_id: str) -> SandboxBackendProtocol: def _get_cached_sandbox_backend(thread_id: str) -> SandboxBackendProtocol:
@ -450,6 +445,9 @@ async def get_agent(config: RunnableConfig) -> Pregel:
thinking = _anthropic_thinking_for(profile_effort) thinking = _anthropic_thinking_for(profile_effort)
if thinking is not None: if thinking is not None:
model_kwargs["thinking"] = thinking model_kwargs["thinking"] = thinking
effort = _anthropic_effort_for(profile_effort)
if effort is not None:
model_kwargs["effort"] = effort
fallback_model_id = os.environ.get("LLM_FALLBACK_MODEL_ID") or fallback_model_id_for(model_id) fallback_model_id = os.environ.get("LLM_FALLBACK_MODEL_ID") or fallback_model_id_for(model_id)
fallback_middleware: list[Any] = [] fallback_middleware: list[Any] = []

View file

@ -90,6 +90,15 @@ def publish_review(
if not isinstance(head_sha, str) or not head_sha: if not isinstance(head_sha, str) or not head_sha:
return {"success": False, "error": "Missing head_sha in run config"} return {"success": False, "error": "Missing head_sha in run config"}
if _is_reviewer_eval_mode(configurable):
return asyncio.run(
_publish_review_eval_dry_run_async(
head_sha=head_sha,
severity_threshold=_cast_severity(severity_threshold),
cap=cap,
)
)
token = get_github_token() token = get_github_token()
if not token: if not token:
return {"success": False, "error": "No GitHub token available"} return {"success": False, "error": "No GitHub token available"}
@ -125,6 +134,46 @@ def _cast_severity(value: str) -> Severity:
return value # type: ignore[return-value] return value # type: ignore[return-value]
def _is_reviewer_eval_mode(configurable: dict[str, Any]) -> bool:
return configurable.get("reviewer_eval") is True or configurable.get("eval") is True
async def _publish_review_eval_dry_run_async(
*,
head_sha: str,
severity_threshold: Severity,
cap: int,
) -> dict[str, Any]:
"""Simulate publish_review for benchmark runs without posting to GitHub."""
thread_id = get_thread_id_from_runtime()
findings = await list_findings_async(thread_id)
unpublished_findings = [
f for f in findings if not isinstance(f.get("github_review_comment_id"), int)
]
open_unpublished = [f for f in unpublished_findings if f.get("status", "open") == "open"]
eligible = filter_findings_for_publish(
unpublished_findings,
severity_threshold=severity_threshold,
cap=cap,
)
inline_comments = [
payload
for finding in eligible
if (payload := render_inline_comment_payload(finding)) is not None
]
await set_reviewer_thread_metadata(thread_id, last_reviewed_sha=head_sha)
return {
"success": True,
"dry_run": True,
"review_id": None,
"surfaced_count": len(inline_comments),
"hidden_count": max(len(open_unpublished) - len(inline_comments), 0),
"resolved_thread_count": 0,
}
async def _publish_review_async( async def _publish_review_async(
*, *,
owner: str, owner: str,

View file

@ -10,7 +10,8 @@ DEFAULT_MAX_RETRIES = 6
OpenAIReasoningEffort = Literal["none", "low", "medium", "high", "xhigh"] OpenAIReasoningEffort = Literal["none", "low", "medium", "high", "xhigh"]
AnthropicThinkingType = Literal["enabled", "disabled"] AnthropicThinkingType = Literal["adaptive"]
AnthropicEffort = Literal["low", "medium", "high", "xhigh", "max"]
class OpenAIReasoning(TypedDict, total=False): class OpenAIReasoning(TypedDict, total=False):
@ -19,13 +20,13 @@ class OpenAIReasoning(TypedDict, total=False):
class AnthropicThinking(TypedDict, total=False): class AnthropicThinking(TypedDict, total=False):
type: AnthropicThinkingType type: AnthropicThinkingType
budget_tokens: int
class ModelKwargs(TypedDict, total=False): class ModelKwargs(TypedDict, total=False):
max_tokens: int | None max_tokens: int | None
reasoning: OpenAIReasoning | None reasoning: OpenAIReasoning | None
thinking: AnthropicThinking | None thinking: AnthropicThinking | None
effort: AnthropicEffort | None
temperature: float | None temperature: float | None
max_retries: int | None max_retries: int | None

View file

@ -10,6 +10,7 @@ root for the full design.
evals/reviewer/ evals/reviewer/
├── golden_comments/ # 50 PRs × golden comments (copied from martian benchmark) ├── golden_comments/ # 50 PRs × golden comments (copied from martian benchmark)
├── build_dataset.py # martian JSON → LangSmith dataset (resolves SHAs via gh) ├── build_dataset.py # martian JSON → LangSmith dataset (resolves SHAs via gh)
├── config.toml # default benchmark run config
├── judge.py # claude-opus-4-5 pairwise match evaluator + aggregate ├── judge.py # claude-opus-4-5 pairwise match evaluator + aggregate
├── target.py # invokes the reviewer graph over langgraph_sdk ├── target.py # invokes the reviewer graph over langgraph_sdk
└── run_eval.py # client.aevaluate entrypoint └── run_eval.py # client.aevaluate entrypoint
@ -45,9 +46,7 @@ example schema, and must emit a `submit_review` tool call (or set
`state["review"]["comments"]`) with `[{file, line, severity, body}, ...]`. `state["review"]["comments"]`) with `[{file, line, severity, body}, ...]`.
```bash ```bash
uv run python -m evals.reviewer.run_eval \ uv run python -m evals.reviewer.run_eval
--experiment-prefix openswe-reviewer-baseline \
--max-concurrency 5
``` ```
Smoke-test with 3 PRs first: Smoke-test with 3 PRs first:
@ -56,6 +55,19 @@ Smoke-test with 3 PRs first:
uv run python -m evals.reviewer.run_eval --limit 3 uv run python -m evals.reviewer.run_eval --limit 3
``` ```
The runner reads benchmark settings from `evals/reviewer/config.toml`. Set the
deployment URL there (or leave it blank to use `LANGGRAPH_URL` / local dev).
The target sets `reviewer_eval` for every run, so `publish_review` does not post
to GitHub.
By default the judge scores final `add_finding` calls. Set
`score_mode = "surfaced_findings"` in the config to score only findings that
would pass the production threshold/cap.
`model_id` and `reasoning_effort` in the config are passed to the reviewer run,
so isolated benchmark deployments can test a specific model/effort without
changing deployment-wide defaults.
## Comparing against Devin Review ## Comparing against Devin Review
Both tools are scored on the same 50 PRs with the same judge model Both tools are scored on the same 50 PRs with the same judge model

View file

@ -0,0 +1,15 @@
dataset_name = "openswe-reviewer-v1"
experiment_prefix = "openswe-reviewer-verified-deployed"
max_concurrency = 2
# Leave blank to use LANGGRAPH_URL or local dev.
langgraph_url = ""
assistant_id = "reviewer"
model_id = "anthropic:claude-opus-4-7"
reasoning_effort = "high"
# Use "surfaced_findings" to score only findings that pass the production
# severity threshold and cap.
score_mode = "all_findings"
severity_threshold = "medium"
cap = 4

View file

@ -1,17 +1,18 @@
"""Run the reviewer eval against the LangSmith dataset. """Run the reviewer eval against the LangSmith dataset.
Usage: Usage:
uv run python -m evals.reviewer.run_eval \\ uv run python -m evals.reviewer.run_eval
--dataset-name openswe-reviewer-v1 \\
--experiment-prefix openswe-reviewer-baseline \\
--max-concurrency 5
""" """
from __future__ import annotations from __future__ import annotations
import argparse import argparse
import logging import logging
import os
import tomllib
from collections.abc import Iterable from collections.abc import Iterable
from pathlib import Path
from typing import Any, Literal, TypedDict
from dotenv import load_dotenv from dotenv import load_dotenv
from langgraph_sdk import get_client from langgraph_sdk import get_client
@ -19,12 +20,97 @@ from langsmith import Client, aevaluate
from langsmith.schemas import Example from langsmith.schemas import Example
from evals.reviewer.judge import aggregate_pr, judge_match from evals.reviewer.judge import aggregate_pr, judge_match
from evals.reviewer.target import LANGGRAPH_URL, drain_thread_ids, review_pr from evals.reviewer.target import drain_thread_ids, get_langgraph_url, review_pr
load_dotenv() load_dotenv()
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
CONFIG_PATH = Path(__file__).with_name("config.toml")
ScoreMode = Literal["all_findings", "surfaced_findings"]
Severity = Literal["informational", "low", "medium", "high", "critical"]
class ReviewerEvalConfig(TypedDict, total=False):
dataset_name: str
experiment_prefix: str
max_concurrency: int
langgraph_url: str
assistant_id: str
model_id: str
reasoning_effort: str
score_mode: ScoreMode
severity_threshold: Severity
cap: int
def _load_config() -> ReviewerEvalConfig:
if not CONFIG_PATH.exists():
return {}
with CONFIG_PATH.open("rb") as f:
raw = tomllib.load(f)
return _coerce_config(raw)
def _coerce_config(raw: dict[str, Any]) -> ReviewerEvalConfig:
config: ReviewerEvalConfig = {}
dataset_name = raw.get("dataset_name")
if isinstance(dataset_name, str) and dataset_name:
config["dataset_name"] = dataset_name
experiment_prefix = raw.get("experiment_prefix")
if isinstance(experiment_prefix, str) and experiment_prefix:
config["experiment_prefix"] = experiment_prefix
langgraph_url = raw.get("langgraph_url")
if isinstance(langgraph_url, str) and langgraph_url:
config["langgraph_url"] = langgraph_url
assistant_id = raw.get("assistant_id")
if isinstance(assistant_id, str) and assistant_id:
config["assistant_id"] = assistant_id
model_id = raw.get("model_id")
if isinstance(model_id, str) and model_id:
config["model_id"] = model_id
reasoning_effort = raw.get("reasoning_effort")
if isinstance(reasoning_effort, str) and reasoning_effort:
config["reasoning_effort"] = reasoning_effort
max_concurrency = raw.get("max_concurrency")
if isinstance(max_concurrency, int) and max_concurrency > 0:
config["max_concurrency"] = max_concurrency
score_mode = raw.get("score_mode")
if score_mode in {"all_findings", "surfaced_findings"}:
config["score_mode"] = score_mode
severity_threshold = raw.get("severity_threshold")
if severity_threshold in {"informational", "low", "medium", "high", "critical"}:
config["severity_threshold"] = severity_threshold
cap = raw.get("cap")
if isinstance(cap, int) and cap >= 0:
config["cap"] = cap
return config
def _apply_config_to_env(config: ReviewerEvalConfig) -> None:
env_mapping = {
"langgraph_url": "LANGGRAPH_URL",
"assistant_id": "REVIEWER_ASSISTANT_ID",
"model_id": "REVIEWER_EVAL_MODEL_ID",
"reasoning_effort": "REVIEWER_EVAL_REASONING_EFFORT",
"score_mode": "REVIEWER_EVAL_SCORE_MODE",
"severity_threshold": "REVIEWER_EVAL_SEVERITY_THRESHOLD",
"cap": "REVIEWER_EVAL_CAP",
}
for config_key, env_key in env_mapping.items():
value = config.get(config_key)
if value is not None:
os.environ[env_key] = str(value)
async def _cleanup_threads(thread_ids: Iterable[str]) -> None: async def _cleanup_threads(thread_ids: Iterable[str]) -> None:
"""Delete LangGraph threads created during the eval. """Delete LangGraph threads created during the eval.
@ -32,7 +118,7 @@ async def _cleanup_threads(thread_ids: Iterable[str]) -> None:
Underlying sandboxes are reclaimed by the provider's TTL — this only Underlying sandboxes are reclaimed by the provider's TTL — this only
drops the LangGraph checkpoint/metadata records. drops the LangGraph checkpoint/metadata records.
""" """
sdk = get_client(url=LANGGRAPH_URL) sdk = get_client(url=get_langgraph_url())
for tid in thread_ids: for tid in thread_ids:
try: try:
await sdk.threads.delete(tid) await sdk.threads.delete(tid)
@ -41,10 +127,10 @@ async def _cleanup_threads(thread_ids: Iterable[str]) -> None:
async def main() -> None: async def main() -> None:
config = _load_config()
_apply_config_to_env(config)
ap = argparse.ArgumentParser() ap = argparse.ArgumentParser()
ap.add_argument("--dataset-name", default="openswe-reviewer-v1")
ap.add_argument("--experiment-prefix", default="openswe-reviewer-baseline")
ap.add_argument("--max-concurrency", type=int, default=5)
ap.add_argument("--limit", type=int, default=None, help="Run only the first N examples.") ap.add_argument("--limit", type=int, default=None, help="Run only the first N examples.")
ap.add_argument( ap.add_argument(
"--no-cleanup", "--no-cleanup",
@ -53,12 +139,16 @@ async def main() -> None:
) )
args = ap.parse_args() args = ap.parse_args()
dataset_name = config.get("dataset_name", "openswe-reviewer-v1")
experiment_prefix = config.get("experiment_prefix", "openswe-reviewer-baseline")
max_concurrency = config.get("max_concurrency", 5)
data: str | list[Example] data: str | list[Example]
if args.limit: if args.limit:
client = Client() client = Client()
data = list(client.list_examples(dataset_name=args.dataset_name, limit=args.limit)) data = list(client.list_examples(dataset_name=dataset_name, limit=args.limit))
else: else:
data = args.dataset_name data = dataset_name
try: try:
await aevaluate( await aevaluate(
@ -66,8 +156,8 @@ async def main() -> None:
data=data, data=data,
evaluators=[judge_match], evaluators=[judge_match],
summary_evaluators=[aggregate_pr], summary_evaluators=[aggregate_pr],
experiment_prefix=args.experiment_prefix, experiment_prefix=experiment_prefix,
max_concurrency=args.max_concurrency, max_concurrency=max_concurrency,
num_repetitions=1, num_repetitions=1,
) )
finally: finally:

View file

@ -11,12 +11,20 @@ from __future__ import annotations
import os import os
import threading import threading
from typing import Any from typing import Any, Literal, cast
from dotenv import load_dotenv
from langgraph_sdk import get_client from langgraph_sdk import get_client
REVIEWER_ASSISTANT_ID = os.getenv("REVIEWER_ASSISTANT_ID", "reviewer") from agent.reviewer_findings import Finding, Severity, filter_findings_for_publish
LANGGRAPH_URL = os.getenv("LANGGRAPH_URL", "http://localhost:2024")
load_dotenv()
DEFAULT_REVIEWER_ASSISTANT_ID = "reviewer"
DEFAULT_LANGGRAPH_URL = "http://localhost:2024"
ScoreMode = Literal["all_findings", "surfaced_findings"]
_VALID_SCORE_MODES: set[ScoreMode] = {"all_findings", "surfaced_findings"}
_VALID_SEVERITIES: set[Severity] = {"informational", "low", "medium", "high", "critical"}
_THREAD_IDS: set[str] = set() _THREAD_IDS: set[str] = set()
_THREAD_IDS_LOCK = threading.Lock() _THREAD_IDS_LOCK = threading.Lock()
@ -40,6 +48,31 @@ def drain_thread_ids() -> set[str]:
return snapshot return snapshot
def get_langgraph_url() -> str:
return os.getenv("LANGGRAPH_URL", DEFAULT_LANGGRAPH_URL)
def get_reviewer_assistant_id() -> str:
return os.getenv("REVIEWER_ASSISTANT_ID", DEFAULT_REVIEWER_ASSISTANT_ID)
def get_score_mode() -> ScoreMode:
value = os.getenv("REVIEWER_EVAL_SCORE_MODE", "all_findings")
if value in _VALID_SCORE_MODES:
return cast(ScoreMode, value)
return "all_findings"
def get_reviewer_model_id() -> str | None:
value = os.getenv("REVIEWER_EVAL_MODEL_ID")
return value if value else None
def get_reviewer_reasoning_effort() -> str | None:
value = os.getenv("REVIEWER_EVAL_REASONING_EFFORT")
return value if value else None
def _build_user_message(inputs: dict[str, Any]) -> str: def _build_user_message(inputs: dict[str, Any]) -> str:
return ( return (
f"Review pull request {inputs['pr_url']}.\n\n" f"Review pull request {inputs['pr_url']}.\n\n"
@ -58,8 +91,10 @@ def _build_user_message(inputs: dict[str, Any]) -> str:
def _build_configurable(inputs: dict[str, Any]) -> dict[str, Any]: def _build_configurable(inputs: dict[str, Any]) -> dict[str, Any]:
repo = inputs.get("repo", "") repo = inputs.get("repo", "")
owner, _, name = repo.partition("/") if isinstance(repo, str) else ("", "", "") owner, _, name = repo.partition("/") if isinstance(repo, str) else ("", "", "")
return { configurable: dict[str, Any] = {
"__is_for_execution__": True, "__is_for_execution__": True,
"reviewer_eval": True,
"eval": True,
"repo": {"owner": owner, "name": name}, "repo": {"owner": owner, "name": name},
"pr_number": inputs.get("pr_number"), "pr_number": inputs.get("pr_number"),
"pr_url": inputs.get("pr_url", ""), "pr_url": inputs.get("pr_url", ""),
@ -67,20 +102,29 @@ def _build_configurable(inputs: dict[str, Any]) -> dict[str, Any]:
"head_sha": inputs.get("head_sha", ""), "head_sha": inputs.get("head_sha", ""),
"branch_name": inputs.get("head_ref", ""), "branch_name": inputs.get("head_ref", ""),
} }
model_id = get_reviewer_model_id()
if model_id:
configurable["reviewer_model_id"] = model_id
reasoning_effort = get_reviewer_reasoning_effort()
if reasoning_effort:
configurable["reviewer_reasoning_effort"] = reasoning_effort
return configurable
async def review_pr(inputs: dict[str, Any]) -> dict[str, Any]: async def review_pr(inputs: dict[str, Any]) -> dict[str, Any]:
"""LangSmith target: run the reviewer agent on one PR.""" """LangSmith target: run the reviewer agent on one PR."""
client = get_client(url=LANGGRAPH_URL) client = get_client(url=get_langgraph_url())
thread = await client.threads.create() thread = await client.threads.create()
thread_id: str = thread["thread_id"] thread_id: str = thread["thread_id"]
_record_thread_id(thread_id) _record_thread_id(thread_id)
result = await client.runs.wait( result = await client.runs.wait(
thread_id, thread_id,
assistant_id=REVIEWER_ASSISTANT_ID, assistant_id=get_reviewer_assistant_id(),
input={"messages": [{"role": "user", "content": _build_user_message(inputs)}]}, input={"messages": [{"role": "user", "content": _build_user_message(inputs)}]},
config={"configurable": _build_configurable(inputs)}, config={"configurable": _build_configurable(inputs)},
) )
if get_score_mode() == "surfaced_findings":
return {"comments": await _extract_surfaced_comments(client, thread_id)}
return {"comments": _extract_comments(result)} return {"comments": _extract_comments(result)}
@ -118,3 +162,57 @@ def _extract_comments(result: Any) -> list[dict[str, Any]]:
} }
) )
return comments return comments
async def _extract_surfaced_comments(client: Any, thread_id: str) -> list[dict[str, Any]]:
thread = await client.threads.get(thread_id)
metadata = thread.get("metadata") if isinstance(thread, dict) else None
findings_value = metadata.get("findings") if isinstance(metadata, dict) else None
findings = _coerce_findings(findings_value)
surfaced = filter_findings_for_publish(
findings,
severity_threshold=_score_severity_threshold(),
cap=_score_cap(),
)
return [_normalize_finding(finding) for finding in surfaced]
def _coerce_findings(value: Any) -> list[Finding]:
if not isinstance(value, list):
return []
findings: list[Finding] = []
for item in value:
if not isinstance(item, dict):
continue
if not isinstance(item.get("id"), str):
continue
findings.append(cast(Finding, item))
return findings
def _normalize_finding(finding: Finding) -> dict[str, Any]:
line = finding.get("end_line")
if line is None:
line = finding.get("start_line")
return {
"file": finding.get("file"),
"line": line,
"body": finding.get("description", ""),
"severity": finding.get("severity"),
}
def _score_severity_threshold() -> Severity:
value = os.getenv("REVIEWER_EVAL_SEVERITY_THRESHOLD", "medium")
if value in _VALID_SEVERITIES:
return cast(Severity, value)
return "medium"
def _score_cap() -> int:
raw = os.getenv("REVIEWER_EVAL_CAP", "4")
try:
cap = int(raw)
except ValueError:
return 4
return max(cap, 0)

View file

@ -0,0 +1,13 @@
from __future__ import annotations
from agent.server import _anthropic_effort_for, _anthropic_thinking_for
def test_anthropic_uses_adaptive_thinking_and_effort() -> None:
assert _anthropic_thinking_for("high") == {"type": "adaptive"}
assert _anthropic_effort_for("high") == "high"
def test_anthropic_ignores_unknown_effort() -> None:
assert _anthropic_thinking_for("unknown") is None
assert _anthropic_effort_for("unknown") is None

View file

@ -54,3 +54,42 @@ async def test_reviewer_uses_cached_thread_token_for_slack_review_request() -> N
assert metadata["github_token_encrypted"] == "encrypted-token" assert metadata["github_token_encrypted"] == "encrypted-token"
mock_get_thread_token.assert_awaited_once_with("reviewer-thread-id") mock_get_thread_token.assert_awaited_once_with("reviewer-thread-id")
mock_resolve_token.assert_not_called() mock_resolve_token.assert_not_called()
@pytest.mark.asyncio
async def test_reviewer_applies_eval_model_and_effort_overrides() -> None:
config: RunnableConfig = {
"configurable": {
"__is_for_execution__": True,
"thread_id": "reviewer-thread-id",
"repo": {"owner": "acme", "name": "repo"},
"pr_number": 1,
"pr_url": "https://github.com/acme/repo/pull/1",
"base_sha": "base",
"head_sha": "head",
"reviewer_model_id": "anthropic:claude-opus-4-7",
"reviewer_reasoning_effort": "high",
},
"metadata": {},
}
dummy_agent = _DummyAgent()
with (
patch(
"agent.reviewer.ensure_sandbox_for_thread",
new_callable=AsyncMock,
return_value=MagicMock(),
),
patch(
"agent.reviewer.aresolve_sandbox_work_dir",
new_callable=AsyncMock,
return_value="/workspace",
),
patch("agent.reviewer.make_model", return_value=MagicMock()) as make_model,
patch("agent.reviewer.create_deep_agent", return_value=dummy_agent),
):
await reviewer.get_reviewer_agent(config)
assert make_model.call_args.args == ("anthropic:claude-opus-4-7",)
assert make_model.call_args.kwargs["thinking"] == {"type": "adaptive"}
assert make_model.call_args.kwargs["effort"] == "high"

View file

@ -0,0 +1,60 @@
from __future__ import annotations
import os
from unittest.mock import patch
from evals.reviewer.run_eval import _apply_config_to_env, _coerce_config
def test_reviewer_eval_config_coerces_known_values() -> None:
config = _coerce_config(
{
"dataset_name": "dataset",
"experiment_prefix": "experiment",
"max_concurrency": 2,
"langgraph_url": "https://example.test",
"assistant_id": "reviewer",
"model_id": "anthropic:claude-opus-4-7",
"reasoning_effort": "high",
"score_mode": "surfaced_findings",
"severity_threshold": "medium",
"cap": 4,
"unknown": "ignored",
}
)
assert config == {
"dataset_name": "dataset",
"experiment_prefix": "experiment",
"max_concurrency": 2,
"langgraph_url": "https://example.test",
"assistant_id": "reviewer",
"model_id": "anthropic:claude-opus-4-7",
"reasoning_effort": "high",
"score_mode": "surfaced_findings",
"severity_threshold": "medium",
"cap": 4,
}
def test_reviewer_eval_config_sets_target_env() -> None:
with patch.dict(os.environ, {}, clear=True):
_apply_config_to_env(
{
"langgraph_url": "https://example.test",
"assistant_id": "reviewer",
"model_id": "anthropic:claude-opus-4-7",
"reasoning_effort": "high",
"score_mode": "surfaced_findings",
"severity_threshold": "high",
"cap": 3,
}
)
assert os.environ["LANGGRAPH_URL"] == "https://example.test"
assert os.environ["REVIEWER_ASSISTANT_ID"] == "reviewer"
assert os.environ["REVIEWER_EVAL_MODEL_ID"] == "anthropic:claude-opus-4-7"
assert os.environ["REVIEWER_EVAL_REASONING_EFFORT"] == "high"
assert os.environ["REVIEWER_EVAL_SCORE_MODE"] == "surfaced_findings"
assert os.environ["REVIEWER_EVAL_SEVERITY_THRESHOLD"] == "high"
assert os.environ["REVIEWER_EVAL_CAP"] == "3"

View file

@ -0,0 +1,94 @@
from __future__ import annotations
from typing import Any
import pytest
from agent.reviewer_findings import new_finding
from evals.reviewer import target
def test_eval_target_marks_runs_as_eval_dry_run(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.delenv("REVIEWER_EVAL_MODEL_ID", raising=False)
monkeypatch.delenv("REVIEWER_EVAL_REASONING_EFFORT", raising=False)
configurable = target._build_configurable(
{
"repo": "acme/repo",
"pr_number": 1,
"pr_url": "https://github.com/acme/repo/pull/1",
"base_sha": "base",
"head_sha": "head",
"head_ref": "branch",
}
)
assert configurable["reviewer_eval"] is True
assert configurable["eval"] is True
assert configurable["__is_for_execution__"] is True
assert "source" not in configurable
def test_eval_target_passes_model_overrides(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("REVIEWER_EVAL_MODEL_ID", "anthropic:claude-opus-4-7")
monkeypatch.setenv("REVIEWER_EVAL_REASONING_EFFORT", "high")
configurable = target._build_configurable(
{
"repo": "acme/repo",
"pr_number": 1,
"pr_url": "https://github.com/acme/repo/pull/1",
"base_sha": "base",
"head_sha": "head",
"head_ref": "branch",
}
)
assert configurable["reviewer_model_id"] == "anthropic:claude-opus-4-7"
assert configurable["reviewer_reasoning_effort"] == "high"
@pytest.mark.asyncio
async def test_extract_surfaced_comments_uses_publish_filter(
monkeypatch: pytest.MonkeyPatch,
) -> None:
high = new_finding(
severity="high",
category="correctness",
file="a.py",
start_line=10,
end_line=10,
description="high signal",
sha="head",
finding_id="f_high",
)
low = new_finding(
severity="low",
category="style",
file="b.py",
start_line=20,
end_line=20,
description="low signal",
sha="head",
finding_id="f_low",
)
class Threads:
async def get(self, _thread_id: str) -> dict[str, Any]:
return {"metadata": {"findings": [high, low]}}
class Client:
threads = Threads()
monkeypatch.setenv("REVIEWER_EVAL_SEVERITY_THRESHOLD", "medium")
monkeypatch.setenv("REVIEWER_EVAL_CAP", "4")
comments = await target._extract_surfaced_comments(Client(), "tid")
assert comments == [
{
"file": "a.py",
"line": 10,
"body": "high signal",
"severity": "high",
}
]

View file

@ -86,6 +86,45 @@ def test_render_review_body_no_findings_message() -> None:
assert "<!-- open-swe-reviewer pr=99 -->" in body assert "<!-- open-swe-reviewer pr=99 -->" in body
def test_publish_review_eval_mode_does_not_call_github() -> None:
from agent.tools.publish_review import publish_review
findings = [
_f(id="f_high", severity="high", file="a.py", start_line=1, end_line=1),
_f(id="f_low", severity="low", file="b.py", start_line=2, end_line=2),
]
with (
patch(
"agent.tools.publish_review.get_config",
return_value={
"configurable": {
"thread_id": "tid",
"repo": {"owner": "o", "name": "r"},
"pr_number": 7,
"head_sha": "sha",
"reviewer_eval": True,
},
"metadata": {},
},
),
patch("agent.tools.publish_review.get_thread_id_from_runtime", return_value="tid"),
patch("agent.tools.publish_review.list_findings_async", AsyncMock(return_value=findings)),
patch("agent.tools.publish_review.set_reviewer_thread_metadata", AsyncMock()) as set_meta,
patch("agent.tools.publish_review.get_github_token") as get_token,
patch("agent.tools.publish_review.post_pull_request_review", AsyncMock()) as post_review,
):
result = publish_review()
assert result["success"] is True
assert result["dry_run"] is True
assert result["surfaced_count"] == 1
assert result["hidden_count"] == 1
get_token.assert_not_called()
post_review.assert_not_called()
set_meta.assert_awaited_once_with("tid", last_reviewed_sha="sha")
@pytest.mark.asyncio @pytest.mark.asyncio
async def test_resolve_review_thread_returns_true_on_success() -> None: async def test_resolve_review_thread_returns_true_on_success() -> None:
response = MagicMock() response = MagicMock()