open-swe/evals/reviewer/config.toml

16 lines
445 B
TOML
Raw Normal View History

dataset_name = "openswe-reviewer-v1"
experiment_prefix = "openswe-reviewer-verified-deployed"
max_concurrency = 2
# Leave blank to use LANGGRAPH_URL or local dev.
langgraph_url = ""
assistant_id = "reviewer"
model_id = "anthropic:claude-opus-4-7"
reasoning_effort = "high"
# Use "surfaced_findings" to score only findings that pass the production
# severity threshold and cap.
score_mode = "all_findings"
severity_threshold = "medium"
cap = 4