mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 12:43:16 +00:00
* feat: upgrade default agent + reviewer model to Opus 4.8 Replace Opus 4.7 with Opus 4.8 (claude-opus-4-8) as the supported Anthropic model surfaced in the profile editor and used by the main agent and reviewer graphs. Effort levels (low/medium/high/xhigh/max) and the high default are unchanged, matching the official Opus 4.8 docs. Updates eval config comment and tests accordingly. * fix: provider-aware fallback for stale stored model ids Dropping claude-opus-4-7 from the supported set meant persisted profile/team-settings still holding it failed the SUPPORTED_MODEL_IDS check and fell through to default_model_pair() — a cross-provider jump to the OpenAI global default. Add provider_fallback_pair: when a stored id is no longer supported but its provider still has a supported model, resolve to that provider's newest supported model (anthropic:claude-opus-4-7 -> 4.8), preserving effort when valid. Resolution order is now: valid stored pair -> same-provider fallback -> global default_model_pair(). Profile overrides keep deferring to the team default when no model is set or the provider is unknown.
60 lines
2.1 KiB
Python
60 lines
2.1 KiB
Python
from __future__ import annotations
|
|
|
|
import os
|
|
from unittest.mock import patch
|
|
|
|
from evals.reviewer.run_eval import _apply_config_to_env, _coerce_config
|
|
|
|
|
|
def test_reviewer_eval_config_coerces_known_values() -> None:
|
|
config = _coerce_config(
|
|
{
|
|
"dataset_name": "dataset",
|
|
"experiment_prefix": "experiment",
|
|
"max_concurrency": 2,
|
|
"langgraph_url": "https://example.test",
|
|
"assistant_id": "reviewer",
|
|
"model_id": "anthropic:claude-opus-4-8",
|
|
"reasoning_effort": "high",
|
|
"score_mode": "surfaced_findings",
|
|
"severity_threshold": "medium",
|
|
"cap": 4,
|
|
"unknown": "ignored",
|
|
}
|
|
)
|
|
|
|
assert config == {
|
|
"dataset_name": "dataset",
|
|
"experiment_prefix": "experiment",
|
|
"max_concurrency": 2,
|
|
"langgraph_url": "https://example.test",
|
|
"assistant_id": "reviewer",
|
|
"model_id": "anthropic:claude-opus-4-8",
|
|
"reasoning_effort": "high",
|
|
"score_mode": "surfaced_findings",
|
|
"severity_threshold": "medium",
|
|
"cap": 4,
|
|
}
|
|
|
|
|
|
def test_reviewer_eval_config_sets_target_env() -> None:
|
|
with patch.dict(os.environ, {}, clear=True):
|
|
_apply_config_to_env(
|
|
{
|
|
"langgraph_url": "https://example.test",
|
|
"assistant_id": "reviewer",
|
|
"model_id": "anthropic:claude-opus-4-8",
|
|
"reasoning_effort": "high",
|
|
"score_mode": "surfaced_findings",
|
|
"severity_threshold": "high",
|
|
"cap": 3,
|
|
}
|
|
)
|
|
|
|
assert os.environ["LANGGRAPH_URL"] == "https://example.test"
|
|
assert os.environ["REVIEWER_ASSISTANT_ID"] == "reviewer"
|
|
assert os.environ["REVIEWER_EVAL_MODEL_ID"] == "anthropic:claude-opus-4-8"
|
|
assert os.environ["REVIEWER_EVAL_REASONING_EFFORT"] == "high"
|
|
assert os.environ["REVIEWER_EVAL_SCORE_MODE"] == "surfaced_findings"
|
|
assert os.environ["REVIEWER_EVAL_SEVERITY_THRESHOLD"] == "high"
|
|
assert os.environ["REVIEWER_EVAL_CAP"] == "3"
|