open-swe/agent/dashboard/eval_jobs.py
Adam Moussa 62d9945df4
refactor: consolidate reviewer modules into agent/review/
Part of the domain-reorg adoption (build plan step C2): fork content,
upstream layout. Nine 1:1 module moves (reviewer_diff/eval_store/
findings/groups/publish/reconcile/trace_context + review_style_
collector/guidance) into agent/review/, with internal relative
imports re-wired to the new package depth. agent/review/__init__.py
mirrors upstream's thin re-export shim (one of the 21 verified "A"
structural adds).

Rewrote the 38 grep hits across importer files (agent/{analyzer,
ci_autofix,reviewer,webapp}.py, agent/dashboard/*, agent/middleware/
settle_review_check.py, agent/tools/*, agent/utils/github_feedback.py,
agent/webhooks/github.py, evals/reviewer/*, and the reviewer test
suite) to point at agent.review.*; 4 of the 38 hits were name
collisions (list_reviewer_findings, reviewer_outcomes,
_reviewer_thread_id, reviewer_thread_id — not the moved modules) and
were left untouched. tests/test_github_checks.py's module-alias
import (`from agent import reviewer_publish`) follows upstream's own
`from agent.review import publish as reviewer_publish` pattern so
downstream `reviewer_publish.*` call sites needed no changes.
agent/reviewer.py and agent/webapp.py stay in place per the hard
rule (fork content, import-only rewire) and are not part of this
package.

Gates: ruff check + ruff format --check, pytest --co -q (1637
collected), full unit suite (1637 passed), and the reviewer/findings
suite in isolation (pytest -k "review or finding", 421 passed).
2026-07-17 13:52:03 -04:00

179 lines
5.4 KiB
Python

"""Track the reviewer eval for the admin dashboard.
The eval itself runs in the ``Reviewer eval`` GitHub Action (durable runner,
isolated from the serving deployment). The Action's harness reports progress
into a LangGraph store record (namespace ``["evals"]``, key ``"reviewer"``) via
``evals.reviewer.store_reporter``; this module reads that record for the
dashboard and reconciles a run whose heartbeat has gone stale (e.g. the Action
was killed) to ``failed``.
"""
from __future__ import annotations
import logging
import os
from datetime import UTC, datetime
from typing import Any, Literal, TypedDict
from langgraph_sdk import get_client
from agent.review.eval_store import (
_HEARTBEAT_STALE_SECONDS,
DEFAULT_EVAL_PROJECT,
EVALS_NAMESPACE,
REVIEWER_EVAL_KEY,
)
from agent.review.findings import REVIEW_FINDING_CAP
logger = logging.getLogger(__name__)
EvalStatus = Literal["idle", "running", "completed", "failed"]
ScoreMode = Literal["all_findings", "surfaced_findings"]
Severity = Literal["low", "medium", "high", "critical"]
class ReviewerEvalConfig(TypedDict):
dataset_name: str
experiment_prefix: str
max_concurrency: int
langsmith_project: str
langgraph_url: str
assistant_id: str
model_id: str
reasoning_effort: str
score_mode: ScoreMode
severity_threshold: Severity
cap: int
DEFAULT_REVIEWER_EVAL_CONFIG: ReviewerEvalConfig = {
"dataset_name": "openswe-reviewer-v1",
"experiment_prefix": "openswe-review-confidence",
"max_concurrency": 5,
"langsmith_project": DEFAULT_EVAL_PROJECT,
"langgraph_url": "",
"assistant_id": "reviewer",
"model_id": "bedrock_converse:us.anthropic.claude-opus-4-8",
"reasoning_effort": "medium",
"score_mode": "surfaced_findings",
"severity_threshold": "low",
"cap": REVIEW_FINDING_CAP,
}
def _client():
return get_client()
def _now_iso() -> str:
return datetime.now(UTC).isoformat()
def _resolve_langgraph_url() -> str | None:
return os.environ.get("LANGGRAPH_URL") or os.environ.get("LANGGRAPH_URL_PROD")
def _eval_project() -> str:
return os.environ.get("EVAL_LANGSMITH_PROJECT") or DEFAULT_EVAL_PROJECT
def _resolve_eval_config(config: ReviewerEvalConfig | None = None) -> ReviewerEvalConfig:
resolved: ReviewerEvalConfig = {
**DEFAULT_REVIEWER_EVAL_CONFIG,
"langsmith_project": _eval_project(),
"langgraph_url": _resolve_langgraph_url() or "",
}
if config is not None:
resolved.update(config)
return resolved
def _idle_record() -> dict[str, Any]:
config = _resolve_eval_config()
return {
"name": REVIEWER_EVAL_KEY,
"status": "idle",
"run_name": config["experiment_prefix"],
"langsmith_project": config["langsmith_project"],
"limit": None,
"config_snapshot": config,
"started_at": None,
"finished_at": None,
"created_by": None,
"pid": None,
"exit_code": None,
"experiment_url": None,
"error": None,
"log_tail": None,
"worker_id": None,
"heartbeat": None,
"progress": None,
"github_run_url": None,
"trigger": None,
"updated_at": _now_iso(),
}
async def _get_record() -> dict[str, Any] | None:
try:
item = await _client().store.get_item(EVALS_NAMESPACE, REVIEWER_EVAL_KEY)
except Exception as e:
logger.debug("store get_item failed for reviewer eval: %s", e)
return None
if item is None:
return None
value = item.get("value") if isinstance(item, dict) else getattr(item, "value", None)
return value if isinstance(value, dict) else None
async def _put_record(record: dict[str, Any]) -> dict[str, Any]:
record = {**record, "updated_at": _now_iso()}
try:
await _client().store.put_item(EVALS_NAMESPACE, REVIEWER_EVAL_KEY, record)
except Exception:
logger.exception("Failed to persist reviewer eval status")
return record
def _heartbeat_age_seconds(record: dict[str, Any]) -> float | None:
"""Seconds since the record's heartbeat, or ``None`` if absent/unparseable."""
hb = record.get("heartbeat")
if not isinstance(hb, str) or not hb:
return None
try:
ts = datetime.fromisoformat(hb)
except ValueError:
return None
if ts.tzinfo is None:
ts = ts.replace(tzinfo=UTC)
return (datetime.now(UTC) - ts).total_seconds()
def _is_heartbeat_fresh(record: dict[str, Any]) -> bool:
age = _heartbeat_age_seconds(record)
return age is not None and age <= _HEARTBEAT_STALE_SECONDS
async def get_reviewer_eval_status() -> dict[str, Any]:
"""Return the latest reviewer-eval status, reconciling a stale ``running``.
The GitHub Action refreshes the record's heartbeat while it runs. A poll
only marks the run failed once the heartbeat is stale, so a healthy run is
left untouched and a killed Action surfaces as ``failed`` within the stale
threshold.
"""
record = await _get_record()
if record is None:
return _idle_record()
if record.get("status") != "running":
return record
if _is_heartbeat_fresh(record):
return record
return await _put_record(
{
**record,
"status": "failed",
"finished_at": record.get("finished_at") or _now_iso(),
"error": "Eval process is no longer tracked (GitHub Action stopped?).",
}
)