mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-10-01 09:43:14 +00:00
* feat: trigger reviewer evals from the admin page Add an admin-only "Reviewer eval" section + endpoints that launch the reviewer benchmark as an isolated subprocess against the running deployment, with live status and the LangSmith experiment link. Route eval traces to a dedicated open-swe-evals project so they stay out of the production tracing project. Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> * fix: reconcile reviewer eval status via heartbeat, not local process The persisted record is shared across workers but _PROCS is process-local. The owning worker now refreshes a heartbeat while the subprocess runs, and status is only reconciled to failed once the heartbeat is stale, so a poll on a worker without the local handle no longer kills a live run (and a duplicate start is rejected across workers). Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
22 lines
755 B
TOML
22 lines
755 B
TOML
dataset_name = "openswe-reviewer-v1"
|
|
experiment_prefix = "openswe-review-confidence"
|
|
max_concurrency = 5
|
|
|
|
# LangSmith tracing project for eval runs. Keeps eval traces out of the
|
|
# deployment's production project.
|
|
langsmith_project = "open-swe-evals"
|
|
|
|
# Leave blank to use LANGGRAPH_URL or local dev.
|
|
langgraph_url = ""
|
|
assistant_id = "reviewer"
|
|
# models: openai:gpt-5.5, anthropic:claude-opus-4-8, google_genai:gemini-3.5-flash
|
|
model_id = "google_genai:gemini-3.5-flash"
|
|
reasoning_effort = "medium"
|
|
|
|
# score_mode:
|
|
# - "all_findings" — score every add_finding the agent emits (no gating).
|
|
# - "surfaced_findings" — only findings that pass the production severity
|
|
# threshold and cap.
|
|
score_mode = "all_findings"
|
|
severity_threshold = "medium"
|
|
cap = 4
|