mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 08:03:15 +00:00
* fix: align reviewer eval with published findings (#1713) * fix: make reviewer eval reflect published findings Serialize and deduplicate finding persistence, align review calibration around the final six-finding publication, and make judge matching order-independent and auditable. * fix: honor reviewer eval limits Forward configured caps into publication snapshots and keep recall-at-cap bounded for diagnostic all-findings runs. (cherry picked from commit 71e3b8183882bcc42e318f3f220c291617ebcb67) Co-authored-by: Johannes du Plessis <johannes@langchain.dev> * Empty commit to trigger CI --------- Co-authored-by: Johannes du Plessis <johannes@langchain.dev>
22 lines
831 B
TOML
22 lines
831 B
TOML
dataset_name = "openswe-reviewer-v1"
|
|
experiment_prefix = "openswe-review-confidence"
|
|
max_concurrency = 5
|
|
|
|
# LangSmith tracing project for eval runs. Keeps eval traces out of the
|
|
# deployment's production project.
|
|
langsmith_project = "open-swe-evals"
|
|
|
|
# Leave blank to use LANGGRAPH_URL or local dev.
|
|
langgraph_url = ""
|
|
assistant_id = "reviewer"
|
|
# models (post Bedrock/Fireworks migration): bedrock_converse:us.anthropic.claude-opus-4-8,
|
|
# or any fireworks:* id in agent/dashboard/options.py SUPPORTED_MODELS.
|
|
model_id = "bedrock_converse:us.anthropic.claude-opus-4-8"
|
|
reasoning_effort = "medium"
|
|
|
|
# score_mode:
|
|
# - "all_findings" — diagnostic mode for deduplicated add_finding calls.
|
|
# - "surfaced_findings" — score the exact final eval publication snapshot.
|
|
score_mode = "surfaced_findings"
|
|
severity_threshold = "low"
|
|
cap = 6
|