mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-10-01 13:13:14 +00:00
* feat: upgrade default agent + reviewer model to Opus 4.8 Replace Opus 4.7 with Opus 4.8 (claude-opus-4-8) as the supported Anthropic model surfaced in the profile editor and used by the main agent and reviewer graphs. Effort levels (low/medium/high/xhigh/max) and the high default are unchanged, matching the official Opus 4.8 docs. Updates eval config comment and tests accordingly. * fix: provider-aware fallback for stale stored model ids Dropping claude-opus-4-7 from the supported set meant persisted profile/team-settings still holding it failed the SUPPORTED_MODEL_IDS check and fell through to default_model_pair() — a cross-provider jump to the OpenAI global default. Add provider_fallback_pair: when a stored id is no longer supported but its provider still has a supported model, resolve to that provider's newest supported model (anthropic:claude-opus-4-7 -> 4.8), preserving effort when valid. Resolution order is now: valid stored pair -> same-provider fallback -> global default_model_pair(). Profile overrides keep deferring to the team default when no model is set or the provider is unknown.
18 lines
610 B
TOML
18 lines
610 B
TOML
dataset_name = "openswe-reviewer-v1"
|
|
experiment_prefix = "openswe-review-confidence"
|
|
max_concurrency = 5
|
|
|
|
# Leave blank to use LANGGRAPH_URL or local dev.
|
|
langgraph_url = ""
|
|
assistant_id = "reviewer"
|
|
# models: openai:gpt-5.5, anthropic:claude-opus-4-8, google_genai:gemini-3.5-flash
|
|
model_id = "google_genai:gemini-3.5-flash"
|
|
reasoning_effort = "medium"
|
|
|
|
# score_mode:
|
|
# - "all_findings" — score every add_finding the agent emits (no gating).
|
|
# - "surfaced_findings" — only findings that pass the production severity
|
|
# threshold and cap.
|
|
score_mode = "all_findings"
|
|
severity_threshold = "medium"
|
|
cap = 4
|