dataset_name = "openswe-reviewer-v1" experiment_prefix = "openswe-review-confidence" max_concurrency = 5 # LangSmith tracing project for eval runs. Keeps eval traces out of the # deployment's production project. langsmith_project = "open-swe-evals" # Leave blank to use LANGGRAPH_URL or local dev. langgraph_url = "" assistant_id = "reviewer" # models (post Bedrock/Fireworks migration): bedrock_converse:us.anthropic.claude-opus-4-8, # or any fireworks:* id in agent/dashboard/options.py SUPPORTED_MODELS. model_id = "bedrock_converse:us.anthropic.claude-opus-4-8" reasoning_effort = "medium" # score_mode: # - "all_findings" — diagnostic mode for deduplicated add_finding calls. # - "surfaced_findings" — score the exact final eval publication snapshot. score_mode = "surfaced_findings" severity_threshold = "low" cap = 6