mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 20:53:15 +00:00
* Run reviewer eval in a GitHub Action; make dashboard a read-only progress view The dashboard launched the eval as a subprocess inside the serving deployment worker, so a container recycle killed long runs and discarded results that had already completed server-side. Move the harness to a workflow_dispatch Action (run on prod). run_eval now publishes status/progress/log-tail to the LangGraph store record the dashboard reads, so /admin/evals stays a live view; a killed Action surfaces as failed via the stale-heartbeat reconcile. * reviewer_eval workflow: pass inputs via env, no shell interpolation Addresses the reviewer finding: workflow_dispatch string inputs were interpolated into the run: block (limit unquoted), allowing shell injection in a job holding LANGSMITH/ANTHROPIC keys. Pass inputs through env and reference quoted "$VARS"; validate limit is numeric and build its flag in bash. --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
95 lines
3.6 KiB
Python
95 lines
3.6 KiB
Python
from __future__ import annotations
|
|
|
|
from unittest.mock import AsyncMock, MagicMock, patch
|
|
|
|
import pytest
|
|
|
|
from evals.reviewer import store_reporter
|
|
from evals.reviewer.store_reporter import StoreReporter, github_run_url, is_enabled
|
|
|
|
_CONFIG = {
|
|
"experiment_prefix": "openswe-review-confidence",
|
|
"langsmith_project": "open-swe-evals",
|
|
"model_id": "google_genai:gemini-3.5-flash",
|
|
}
|
|
|
|
|
|
def _make_reporter(
|
|
monkeypatch: pytest.MonkeyPatch, completed: int = 0
|
|
) -> tuple[StoreReporter, MagicMock]:
|
|
monkeypatch.setenv("LANGGRAPH_URL", "https://lg.test")
|
|
monkeypatch.setenv("GITHUB_SERVER_URL", "https://github.com")
|
|
monkeypatch.setenv("GITHUB_REPOSITORY", "langchain-ai/open-swe")
|
|
monkeypatch.setenv("GITHUB_RUN_ID", "12345")
|
|
monkeypatch.setenv("GITHUB_ACTOR", "octocat")
|
|
client = MagicMock()
|
|
client.store.put_item = AsyncMock()
|
|
with patch.object(store_reporter, "get_client", return_value=client):
|
|
reporter = StoreReporter(
|
|
config=dict(_CONFIG),
|
|
limit=3,
|
|
total=10,
|
|
created_by=None,
|
|
completed_getter=lambda: completed,
|
|
tail_getter=lambda: "tail",
|
|
experiment_url_getter=lambda: "https://smith.langchain.com/exp",
|
|
)
|
|
return reporter, client
|
|
|
|
|
|
def test_is_enabled(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
monkeypatch.delenv("REVIEWER_EVAL_REPORT_STORE", raising=False)
|
|
monkeypatch.delenv("LANGGRAPH_URL", raising=False)
|
|
assert is_enabled() is False
|
|
monkeypatch.setenv("REVIEWER_EVAL_REPORT_STORE", "1")
|
|
assert is_enabled() is False # still needs LANGGRAPH_URL
|
|
monkeypatch.setenv("LANGGRAPH_URL", "https://lg.test")
|
|
assert is_enabled() is True
|
|
|
|
|
|
def test_github_run_url(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
monkeypatch.setenv("GITHUB_SERVER_URL", "https://github.com")
|
|
monkeypatch.setenv("GITHUB_REPOSITORY", "langchain-ai/open-swe")
|
|
monkeypatch.setenv("GITHUB_RUN_ID", "999")
|
|
assert github_run_url() == "https://github.com/langchain-ai/open-swe/actions/runs/999"
|
|
monkeypatch.delenv("GITHUB_RUN_ID")
|
|
assert github_run_url() is None
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_start_writes_running_record(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
reporter, client = _make_reporter(monkeypatch, completed=2)
|
|
await reporter.start()
|
|
|
|
client.store.put_item.assert_awaited_once()
|
|
namespace, key, record = client.store.put_item.await_args.args
|
|
assert namespace == ["evals"]
|
|
assert key == "reviewer"
|
|
assert record["status"] == "running"
|
|
assert record["trigger"] == "github_action"
|
|
assert record["progress"] == {"completed": 2, "total": 10}
|
|
assert record["github_run_url"] == "https://github.com/langchain-ai/open-swe/actions/runs/12345"
|
|
assert record["created_by"] == "octocat" # falls back to GITHUB_ACTOR
|
|
assert record["worker_id"] == "12345"
|
|
assert record["run_name"] == "openswe-review-confidence"
|
|
assert record["limit"] == 3
|
|
assert record["heartbeat"]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_finish_writes_terminal_record(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
reporter, client = _make_reporter(monkeypatch)
|
|
await reporter.finish(status="failed", error="boom")
|
|
|
|
_, _, record = client.store.put_item.await_args.args
|
|
assert record["status"] == "failed"
|
|
assert record["error"] == "boom"
|
|
assert record["finished_at"]
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_put_swallows_store_errors(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
reporter, client = _make_reporter(monkeypatch)
|
|
client.store.put_item = AsyncMock(side_effect=RuntimeError("store down"))
|
|
# Should not raise — store failures must not crash the eval.
|
|
await reporter.start()
|