From 5350b63c85abeefbe2af00fce52341104c5e425c Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Fri, 10 Jul 2026 14:05:20 -0400 Subject: [PATCH] feat(open-swe): re-add Fable 5 behind an admin toggle (Bedrock) (#172) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(models): re-add Fable 5 with admin disable toggle (port of upstream #1677) * refactor(models): convert re-added Fable 5 to Bedrock model IDs * fix(open-swe): correct Fable copy to describe provider data sharing, not ZDR The ported admin toggle description and code comments described Fable 5 as incompatible with Zero Data Retention. That is backwards: Fable 5 requires the account to opt into Bedrock provider_data_share — prompts/completions are retained and shared with Anthropic (up to 30 days, incl. human review). The old UI copy would lead an admin to believe the opposite of what enabling the toggle does. Reword the toggle description and the gate_fable_model / team_settings comments accordingly. Still off by default. Refs #171. --- .github/workflows/ci.yml | 1 + agent/chat.py | 15 ++- agent/dashboard/options.py | 39 ++++++++ agent/dashboard/routes.py | 26 ++++- agent/dashboard/schedules.py | 10 +- agent/dashboard/team_settings.py | 32 +++++++ agent/dashboard/thread_api.py | 6 +- agent/reviewer.py | 12 +++ agent/server.py | 19 +++- docs/upstream-sync/triage.jsonl | 2 +- docs/upstream-sync/triage.md | 2 +- tests/test_agent_schedules.py | 8 +- tests/test_agent_subagent_models.py | 68 +++++++++++++ tests/test_anthropic_model.py | 55 +++++++++++ tests/test_dashboard_thread_api.py | 83 +++++++++++++++- tests/test_model_fallback_resolution.py | 37 +++++++ tests/test_team_settings_fable.py | 122 ++++++++++++++++++++++++ ui/src/lib/api.ts | 1 + ui/src/routes/admin.tsx | 41 ++++++++ 19 files changed, 563 insertions(+), 16 deletions(-) create mode 100644 tests/test_anthropic_model.py create mode 100644 tests/test_team_settings_fable.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 608fef06..a040077c 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -80,6 +80,7 @@ jobs: working-directory: tests/e2e run: | npm ci + sudo rm -f /etc/apt/sources.list.d/azure-cli.list /etc/apt/sources.list.d/microsoft-prod.list npx playwright install --with-deps chromium # Playwright's webServer boots `langgraph dev`; globalSetup builds the real # ui/ SPA. The fake LLM/GitHub/Slack boundaries need no secrets. diff --git a/agent/chat.py b/agent/chat.py index 132f83d8..45dae8a4 100644 --- a/agent/chat.py +++ b/agent/chat.py @@ -28,8 +28,16 @@ warnings.filterwarnings("ignore", message=".*Pydantic V1.*", category=UserWarnin from deepagents import create_deep_agent from langchain.agents.middleware import ModelCallLimitMiddleware -from .dashboard.options import SUPPORTED_MODEL_IDS, model_supports_effort -from .dashboard.team_settings import get_effective_gateway_enabled, get_team_default_model +from .dashboard.options import ( + SUPPORTED_MODEL_IDS, + gate_fable_model, + model_supports_effort, +) +from .dashboard.team_settings import ( + get_effective_gateway_enabled, + get_team_default_model, + get_team_fable_enabled, +) from .middleware import ( ExcludeToolsMiddleware, SanitizeFireworksMessagesMiddleware, @@ -125,6 +133,9 @@ async def get_chat_agent(config: RunnableConfig) -> Pregel: configurable["chat_github_token"] = token model_id, effort = await _resolve_chat_model(configurable) + model_id, effort = gate_fable_model( + model_id, effort, fable_enabled=await get_team_fable_enabled() + ) use_gateway = await get_effective_gateway_enabled() model_kwargs = provider_model_kwargs( model_id, diff --git a/agent/dashboard/options.py b/agent/dashboard/options.py index 401e411a..3e22bd3c 100644 --- a/agent/dashboard/options.py +++ b/agent/dashboard/options.py @@ -28,6 +28,13 @@ SUPPORTED_MODELS: list[ModelOption] = [ "default_effort": "high", "supports_images": True, }, + { + "id": "bedrock_converse:us.anthropic.claude-fable-5", + "label": "Fable 5", + "efforts": ["low", "medium", "high", "xhigh", "max"], + "default_effort": "high", + "supports_images": True, + }, { "id": "fireworks:accounts/fireworks/models/kimi-k2p7-code", "label": "Kimi K2.7", @@ -74,6 +81,38 @@ SUPPORTED_MODELS: list[ModelOption] = [ SUPPORTED_MODEL_IDS: frozenset[str] = frozenset(m["id"] for m in SUPPORTED_MODELS) +_FABLE_ID_PREFIX = "bedrock_converse:us.anthropic.claude-fable" + +FABLE_MODEL_IDS: frozenset[str] = frozenset( + m["id"] for m in SUPPORTED_MODELS if m["id"].startswith(_FABLE_ID_PREFIX) +) + + +def fable_disabled_fallback(effort: object = None) -> tuple[str, str]: + """Newest supported non-Fable Claude model (keeps the Claude family), + else the global default. Substitutes a Fable selection when Fable is + disabled workspace-wide, preserving ``effort`` when the fallback supports it.""" + for m in SUPPORTED_MODELS: + if m["id"].startswith("bedrock_converse:us.anthropic.") and m["id"] not in FABLE_MODEL_IDS: + return m["id"], _fallback_effort_for(m, effort) or m["default_effort"] + return default_model_pair() + + +def gate_fable_model( + model_id: str, effort: str | None, *, fable_enabled: bool +) -> tuple[str, str | None]: + """Provider-data-share gate: if Fable is disabled but a Fable id was + resolved, swap in a safe non-Fable model. Fable requires the account to opt + into Bedrock ``provider_data_share`` (prompts/completions shared with the + provider), so it is gated off by default. Non-Fable selections pass through + unchanged. Applied at every model-construction entrypoint so a disabled + Fable model can never reach ``make_model``, no matter which layer selected + it.""" + if not fable_enabled and isinstance(model_id, str) and model_id in FABLE_MODEL_IDS: + return fable_disabled_fallback(effort) + return model_id, effort + + DEFAULT_MODEL_ID: str = "bedrock_converse:us.anthropic.claude-opus-4-8" DEFAULT_MODEL_EFFORT: str = "medium" diff --git a/agent/dashboard/routes.py b/agent/dashboard/routes.py index 04b8df48..5793cb81 100644 --- a/agent/dashboard/routes.py +++ b/agent/dashboard/routes.py @@ -59,7 +59,7 @@ from .oauth import ( require_session, sanitize_redirect_to, ) -from .options import SUPPORTED_MODELS +from .options import FABLE_MODEL_IDS, SUPPORTED_MODELS, gate_fable_model from .profiles import ( ProfileUpdate, get_profile, @@ -145,6 +145,7 @@ from .team_settings import ( TeamSettingsUpdate, get_team_default_model, get_team_default_subagent_model, + get_team_fable_enabled, get_team_settings, upsert_team_settings, ) @@ -445,8 +446,23 @@ async def me(session: dict[str, Any] = _SESSION_DEP) -> dict[str, Any]: async def options() -> dict[str, Any]: agent_model, agent_effort = await get_team_default_model("agent") subagent_model, subagent_effort = await get_team_default_subagent_model("agent") + fable_enabled = await get_team_fable_enabled() + # Never advertise a default that isn't in the selectable list: when Fable is + # off, gate a stale Fable default down to its non-Fable fallback so the Cloud + # Agents page (and the PUT /profile it drives) don't choke on it. + agent_model, agent_effort = gate_fable_model( + agent_model, agent_effort, fable_enabled=fable_enabled + ) + subagent_model, subagent_effort = gate_fable_model( + subagent_model, subagent_effort, fable_enabled=fable_enabled + ) + models = ( + SUPPORTED_MODELS + if fable_enabled + else [m for m in SUPPORTED_MODELS if m["id"] not in FABLE_MODEL_IDS] + ) return { - "models": SUPPORTED_MODELS, + "models": models, "default_agent_model": agent_model, "default_agent_reasoning_effort": agent_effort, "default_agent_subagent_model": subagent_model, @@ -468,6 +484,12 @@ async def put_my_profile( session: dict[str, Any] = _SESSION_DEP, ) -> dict[str, Any]: update.validate_pairing() + if not await get_team_fable_enabled(): + if ( + update.default_model in FABLE_MODEL_IDS + or update.default_subagent_model in FABLE_MODEL_IDS + ): + raise HTTPException(400, "Fable is disabled for this workspace") return await upsert_profile(session["sub"], session.get("email") or "", update) diff --git a/agent/dashboard/schedules.py b/agent/dashboard/schedules.py index 8356035b..5dff79b2 100644 --- a/agent/dashboard/schedules.py +++ b/agent/dashboard/schedules.py @@ -12,9 +12,10 @@ from fastapi import HTTPException from pydantic import BaseModel, Field, field_validator from ..utils.thread_ops import langgraph_client -from .options import SUPPORTED_MODEL_IDS, model_supports_effort +from .options import SUPPORTED_MODEL_IDS, gate_fable_model, model_supports_effort from .profiles import get_profile, get_valid_access_token from .repo_access import repo_config_for_user, require_repo_access_for_user +from .team_settings import get_team_fable_enabled from .thread_api import _agent_version_metadata, _now_ms, _resolve_run_email logger = logging.getLogger(__name__) @@ -405,7 +406,7 @@ def _agent_run_metadata(record: dict[str, Any], thread_id: str) -> dict[str, Any return metadata -def _agent_run_config(record: dict[str, Any], thread_id: str) -> dict[str, Any]: +async def _agent_run_config(record: dict[str, Any], thread_id: str) -> dict[str, Any]: configurable: dict[str, Any] = { "thread_id": thread_id, "source": "schedule", @@ -418,6 +419,9 @@ def _agent_run_config(record: dict[str, Any], thread_id: str) -> dict[str, Any]: configurable["repo"] = repo model, effort = _normalize_model_choice(record.get("model"), record.get("effort")) if model and effort: + model, effort = gate_fable_model( + model, effort, fable_enabled=await get_team_fable_enabled() + ) configurable["agent_model_id"] = model configurable["agent_effort"] = effort report_channel = record.get("slack_report_channel") @@ -471,7 +475,7 @@ async def launch_scheduled_agent_run(schedule_id: str) -> dict[str, Any]: thread_id, _AGENT_ASSISTANT_ID, input={"messages": [{"role": "user", "content": record["prompt"]}]}, - config=_agent_run_config(record, thread_id), + config=await _agent_run_config(record, thread_id), if_not_exists="create", stream_mode=["values", "updates", "messages-tuple"], stream_resumable=True, diff --git a/agent/dashboard/team_settings.py b/agent/dashboard/team_settings.py index 98299ed5..5c65b971 100644 --- a/agent/dashboard/team_settings.py +++ b/agent/dashboard/team_settings.py @@ -17,8 +17,10 @@ from pydantic import BaseModel, field_validator, model_validator from ..utils.gateway import resolve_gateway_enabled from .options import ( + FABLE_MODEL_IDS, SUPPORTED_MODEL_IDS, default_model_pair, + gate_fable_model, model_supports_effort, provider_fallback_pair, ) @@ -56,6 +58,7 @@ class TeamSettingsUpdate(BaseModel): # Tri-state LLM Gateway toggle: True/False is authoritative, None inherits the # LANGSMITH_GATEWAY_ENABLED deployment default. gateway_enabled: bool | None = None + fable_enabled: bool = False review_tracing_project: str | None = None org_guidelines: str | None = None default_agent_model: str | None = None @@ -131,6 +134,26 @@ class TeamSettingsUpdate(BaseModel): _validate_model_effort_pair( self.default_chat_model, self.default_chat_reasoning_effort, "review chat" ) + if not self.fable_enabled: + # Disabling Fable is the provider-data-share kill switch and must always succeed: rather + # than reject a payload that still carries a Fable default, swap each + # Fable default to its safe non-Fable fallback (mirrors the runtime + # gate_fable_model guard) so the stored record can't advertise Fable. + for model_field, effort_field in ( + ("default_agent_model", "default_agent_reasoning_effort"), + ("default_agent_subagent_model", "default_agent_subagent_reasoning_effort"), + ("default_reviewer_model", "default_reviewer_reasoning_effort"), + ("default_reviewer_subagent_model", "default_reviewer_subagent_reasoning_effort"), + ("default_grouping_model", "default_grouping_reasoning_effort"), + ("default_chat_model", "default_chat_reasoning_effort"), + ): + model = getattr(self, model_field) + if model in FABLE_MODEL_IDS: + new_model, new_effort = gate_fable_model( + model, getattr(self, effort_field), fable_enabled=False + ) + setattr(self, model_field, new_model) + setattr(self, effort_field, new_effort) return self @@ -171,6 +194,7 @@ def _default_settings() -> dict[str, Any]: "pr_summaries": True, "review_trace_links": True, "gateway_enabled": None, + "fable_enabled": False, "review_tracing_project": None, "org_guidelines": DEFAULT_ORG_REVIEW_GUIDELINES, "default_agent_model": fallback_model, @@ -226,6 +250,7 @@ async def upsert_team_settings(update: TeamSettingsUpdate) -> dict[str, Any]: "pr_summaries": update.pr_summaries, "review_trace_links": update.review_trace_links, "gateway_enabled": update.gateway_enabled, + "fable_enabled": update.fable_enabled, "review_tracing_project": update.review_tracing_project, "org_guidelines": update.org_guidelines, "default_agent_model": update.default_agent_model, @@ -353,6 +378,13 @@ async def get_team_gateway_enabled() -> bool | None: return value if isinstance(value, bool) else None +async def get_team_fable_enabled() -> bool: + """Return whether Fable models are enabled for the team.""" + settings = await get_team_settings() + value = settings.get("fable_enabled") + return bool(value) if isinstance(value, bool) else False + + async def get_effective_gateway_enabled() -> bool: """Resolve whether LLM Gateway routing is on: team setting, else env default.""" return resolve_gateway_enabled(await get_team_gateway_enabled()) diff --git a/agent/dashboard/thread_api.py b/agent/dashboard/thread_api.py index 5bb69e42..a6bf0af6 100644 --- a/agent/dashboard/thread_api.py +++ b/agent/dashboard/thread_api.py @@ -31,12 +31,13 @@ from .agent_overrides import normalize_profile_overrides from .options import ( SUPPORTED_MODEL_IDS, default_vision_model_pair, + gate_fable_model, model_supports_effort, model_supports_images, ) from .pr_diff import build_pr_diff_files from .profiles import get_profile, get_valid_access_token -from .team_settings import get_team_default_model +from .team_settings import get_team_default_model, get_team_fable_enabled from .user_mappings import email_for_login logger = logging.getLogger(__name__) @@ -153,6 +154,9 @@ async def _resolve_agent_model_choice( chosen_model, chosen_effort = _normalize_model_choice(model_id, effort) if chosen_model and chosen_effort: resolved_model, resolved_effort = chosen_model, chosen_effort + resolved_model, resolved_effort = gate_fable_model( + resolved_model, resolved_effort, fable_enabled=await get_team_fable_enabled() + ) return resolved_model, resolved_effort diff --git a/agent/reviewer.py b/agent/reviewer.py index 18151c0b..4c27b349 100644 --- a/agent/reviewer.py +++ b/agent/reviewer.py @@ -33,11 +33,13 @@ from deepagents import create_deep_agent from langchain.agents.middleware import ModelCallLimitMiddleware from langchain_core.language_models.chat_models import BaseChatModel +from .dashboard.options import gate_fable_model from .dashboard.team_settings import ( get_effective_gateway_enabled, get_org_review_guidelines, get_team_default_grouping_model, get_team_default_model_pair, + get_team_fable_enabled, ) from .middleware import ( RepairOrphanedToolCallsMiddleware, @@ -822,6 +824,9 @@ async def _resolve_grouping_model( effort = configured_effort if isinstance(configured_effort, str) else None else: model_id, effort = await get_team_default_grouping_model() + model_id, effort = gate_fable_model( + model_id, effort, fable_enabled=await get_team_fable_enabled() + ) model_kwargs = provider_model_kwargs( model_id, effort, @@ -1124,6 +1129,13 @@ async def get_reviewer_agent(config: RunnableConfig) -> Pregel: subagent_effort = ( configured_subagent_effort if isinstance(configured_subagent_effort, str) else None ) + fable_enabled = await get_team_fable_enabled() + model_id, reasoning_effort = gate_fable_model( + model_id, reasoning_effort, fable_enabled=fable_enabled + ) + subagent_model_id, subagent_effort = gate_fable_model( + subagent_model_id, subagent_effort, fable_enabled=fable_enabled + ) model_kwargs = provider_model_kwargs( model_id, reasoning_effort, diff --git a/agent/server.py b/agent/server.py index 619d3cfa..628c1135 100644 --- a/agent/server.py +++ b/agent/server.py @@ -41,12 +41,18 @@ from .dashboard.agent_overrides import ( resolve_github_login, ) from .dashboard.agent_usage import record_agent_thread_usage -from .dashboard.options import DEFAULT_MODEL_ID, SUPPORTED_MODEL_IDS, model_supports_effort +from .dashboard.options import ( + DEFAULT_MODEL_ID, + SUPPORTED_MODEL_IDS, + gate_fable_model, + model_supports_effort, +) from .dashboard.repo_snapshots import resolve_repo_snapshot_id from .dashboard.team_settings import ( get_effective_gateway_enabled, get_team_default_model_pair, get_team_default_repo, + get_team_fable_enabled, ) from .dashboard.user_mappings import email_for_login from .integrations.corridor_mcp import load_corridor_tools @@ -765,6 +771,7 @@ async def get_agent(config: RunnableConfig) -> Pregel: ) team_defaults_task = asyncio.create_task(get_team_default_model_pair("agent")) gateway_task = asyncio.create_task(get_effective_gateway_enabled()) + fable_task = asyncio.create_task(get_team_fable_enabled()) profile_task = asyncio.create_task(load_profile(profile_login)) if profile_login else None try: ( @@ -772,11 +779,13 @@ async def get_agent(config: RunnableConfig) -> Pregel: sandbox_backend, team_defaults, use_gateway, + fable_enabled, ) = await asyncio.gather( triggering_user_identity_task, sandbox_task, team_defaults_task, gateway_task, + fable_task, ) except SandboxRepoMismatchError as exc: # Repo-binding refusal at the run boundary: log for alarming and surface the @@ -787,6 +796,7 @@ async def get_agent(config: RunnableConfig) -> Pregel: triggering_user_identity_task, team_defaults_task, gateway_task, + fable_task, profile_task, ): if pending is not None and not pending.done(): @@ -860,6 +870,13 @@ async def get_agent(config: RunnableConfig) -> Pregel: if always_create_prs: logger.info("Always Create PRs enabled by profile for %s", profile_login) + model_id, profile_effort = gate_fable_model( + model_id, profile_effort, fable_enabled=fable_enabled + ) + subagent_model_id, subagent_effort = gate_fable_model( + subagent_model_id, subagent_effort, fable_enabled=fable_enabled + ) + model_kwargs = provider_model_kwargs( model_id, profile_effort, diff --git a/docs/upstream-sync/triage.jsonl b/docs/upstream-sync/triage.jsonl index 3f7d848e..d854ac89 100644 --- a/docs/upstream-sync/triage.jsonl +++ b/docs/upstream-sync/triage.jsonl @@ -70,7 +70,7 @@ {"sha": "216cf181", "pr": 1699, "subject": "fix: keep workflow HITL without token downscoping (#1699)", "disposition": "landed", "reason": "DIVERGES-FROM-UPSTREAM: fork deliberately does NOT adopt #1699's standing-token workflows:write broadening. Security review (#159) BLOCKed it — the standing ALWAYS-ON proxy token carrying workflows:write turns the HITL guard's git-push-parser gaps (obfuscated-expansion push, `gh api` REST contents PUT, cross-branch refspecs) into live unapproved-workflow-push exploits. Fork keeps BASE without workflows:write and restores the transient per-approval elevation (_run_with_workflow_token mints WORKFLOW_RUNTIME_PROXY_TOKEN_PERMISSIONS around the approved, guard-normalized fixed_command, then downscopes to RUNTIME then BASE): the token scope is the backstop the parser relies on, so a bypass hits GitHub 403. HITL diff-preview/approval-URL/Slack-card additions from #159 retained; token-model divergence only.", "branch": "plan-approval", "local_sha": null, "updated": "2026-07-09T20:00:00Z"} {"sha": "67abf5b0", "pr": 1659, "subject": "fix: surface attributed PR creation failures (#1659)", "disposition": "deferred", "reason": "PR-attribution-failure guard (new mw, safe imports); heavy conflict on diverged open_pull_request.py", "branch": "pr-attribution", "local_sha": null, "updated": "2026-07-08T20:14:42Z"} {"sha": "3dbc0282", "pr": 1676, "subject": "fix: preserve plan redirects after login (#1676)", "disposition": "landed", "reason": "FLAG-HUMAN: follow-on to landed #1668 refining sanitize_redirect_to (open-redirect auth surface); not a dup", "branch": "plan-approval", "local_sha": null, "updated": "2026-07-09T17:10:22Z"} -{"sha": "c75cbb1f", "pr": 1677, "subject": "feat: re-add Fable 5 with an admin toggle to disable it (#1677)", "disposition": "deferred", "reason": "FLAG-HUMAN: re-adds Fable 5 via anthropic: — contradicts dev's deliberate hide (#1483) + Bedrock migration (#62); wont-merge candidate", "branch": "fable-admin-toggle", "local_sha": null, "updated": "2026-07-08T20:14:42Z"} +{"sha": "c75cbb1f", "pr": 1677, "subject": "feat: re-add Fable 5 with an admin toggle to disable it (#1677)", "disposition": "landed", "reason": "Ported + Bedrock-converted onto dev via feat/readd-fable5-bedrock; anthropic: Fable ID mapped to bedrock_converse:us.anthropic.claude-fable-5.", "branch": "fable-admin-toggle", "local_sha": null, "updated": "2026-07-10T17:07:19Z"} {"sha": "bb104d93", "pr": 1679, "subject": "fix: submit plan comments with cmd enter (#1679)", "disposition": "landed", "reason": "applies clean but edits fork-diverged PlanReview.tsx (#130); needs UI/e2e validation — separate PR", "branch": "plan-approval", "local_sha": null, "updated": "2026-07-09T17:10:22Z"} {"sha": "304032fa", "pr": 1680, "subject": "chore: clarify question answering prompt (#1680)", "disposition": "deferred", "reason": "reword Slack info-only answer guidance; conflicts w/ fork's customized Slack prompt", "branch": "prompt-tweaks", "local_sha": null, "updated": "2026-07-08T20:14:42Z"} {"sha": "5003c953", "pr": 1683, "subject": "feat: open Linear-triggered PRs as the triggering user (#1683)", "disposition": "deferred", "reason": "FLAG-HUMAN: adds linear to resolve_github_token per-user OAuth branch (auth surface); depends on #1626 linear.py", "branch": "linear-pr-as-user", "local_sha": null, "updated": "2026-07-08T20:14:42Z"} diff --git a/docs/upstream-sync/triage.md b/docs/upstream-sync/triage.md index bc053790..101342a7 100644 --- a/docs/upstream-sync/triage.md +++ b/docs/upstream-sync/triage.md @@ -54,6 +54,7 @@ Rows key on the **upstream SHA** (stable across local cherry-picks). Deferred ro | `5f7f5fbd` | #1697 | fix: Reduce graph import and loader startup latency (#1697) | Landed | import-hygiene refactor; cross-cutting, references many deferred upstream-only modules | durable-dispatch | | `216cf181` | #1699 | fix: keep workflow HITL without token downscoping (#1699) | Landed | DIVERGES-FROM-UPSTREAM: fork deliberately does NOT adopt #1699's standing-token workflows:write broadening. Security review (#159) BLOCKed it — the standing ALWAYS-ON proxy token carrying workflows:write turns the HITL guard's git-push-parser gaps (obfuscated-expansion push, `gh api` REST contents PUT, cross-branch refspecs) into live unapproved-workflow-push exploits. Fork keeps BASE without workflows:write and restores the transient per-approval elevation (_run_with_workflow_token mints WORKFLOW_RUNTIME_PROXY_TOKEN_PERMISSIONS around the approved, guard-normalized fixed_command, then downscopes to RUNTIME then BASE): the token scope is the backstop the parser relies on, so a bypass hits GitHub 403. HITL diff-preview/approval-URL/Slack-card additions from #159 retained; token-model divergence only. | plan-approval | | `3dbc0282` | #1676 | fix: preserve plan redirects after login (#1676) | Landed | FLAG-HUMAN: follow-on to landed #1668 refining sanitize_redirect_to (open-redirect auth surface); not a dup | plan-approval | +| `c75cbb1f` | #1677 | feat: re-add Fable 5 with an admin toggle to disable it (#1677) | Landed | Ported + Bedrock-converted onto dev via feat/readd-fable5-bedrock; anthropic: Fable ID mapped to bedrock_converse:us.anthropic.claude-fable-5. | fable-admin-toggle | | `bb104d93` | #1679 | fix: submit plan comments with cmd enter (#1679) | Landed | applies clean but edits fork-diverged PlanReview.tsx (#130); needs UI/e2e validation — separate PR | plan-approval | | `feb7ac98` | #1689 | feat(web): surface thread sandbox ID with touch-friendly menu (#1689) | Landed | Ported to dev via feat/thread-sandbox-id-sidebar. | dashboard-ui | | `7f7af715` | #1684 | feat: auto-load scoped AGENTS on reads (#1684) | Landed | ported in #129 (SubdirAgentsReadMiddleware) | subdir-agents | @@ -89,7 +90,6 @@ Rows key on the **upstream SHA** (stable across local cherry-picks). Deferred ro | `4f8bc2dd` | #1692 | refactor: simplify open-swe agent sandbox lifecycle (#1692) | Deferred | FLAG-HUMAN: structural rewrite of ensure_sandbox_for_thread (drops __creating__ 4-case sentinel) | sandbox-refactor | | `48217b68` | #1489 | feat(open-swe): add E2B sandbox provider (#1489) | Deferred | additive E2B provider; separable but ships on the async sandbox.py base | sandbox-refactor | | `67abf5b0` | #1659 | fix: surface attributed PR creation failures (#1659) | Deferred | PR-attribution-failure guard (new mw, safe imports); heavy conflict on diverged open_pull_request.py | pr-attribution | -| `c75cbb1f` | #1677 | feat: re-add Fable 5 with an admin toggle to disable it (#1677) | Deferred | FLAG-HUMAN: re-adds Fable 5 via anthropic: — contradicts dev's deliberate hide (#1483) + Bedrock migration (#62); wont-merge candidate | fable-admin-toggle | | `304032fa` | #1680 | chore: clarify question answering prompt (#1680) | Deferred | reword Slack info-only answer guidance; conflicts w/ fork's customized Slack prompt | prompt-tweaks | | `5003c953` | #1683 | feat: open Linear-triggered PRs as the triggering user (#1683) | Deferred | FLAG-HUMAN: adds linear to resolve_github_token per-user OAuth branch (auth surface); depends on #1626 linear.py | linear-pr-as-user | | `f53caff1` | #1701 | fix: fall back to core GitHub App scope when optional grants missing (#1701) | Deferred | FLAG-HUMAN: GitHub-App permission-ladder degrade (auth surface); heavy conflict on diverged github_app.py/_resolve_proxy_token | github-app-scope | diff --git a/tests/test_agent_schedules.py b/tests/test_agent_schedules.py index df2509b3..dbde7c42 100644 --- a/tests/test_agent_schedules.py +++ b/tests/test_agent_schedules.py @@ -196,7 +196,7 @@ async def test_update_agent_schedule_clears_slack_report_channel(fake_client) -> assert result["slackReportChannel"] is None -def test_agent_run_config_seeds_slack_thread_channel() -> None: +async def test_agent_run_config_seeds_slack_thread_channel() -> None: record = { "id": "sched_1", "model": "Default", @@ -206,12 +206,12 @@ def test_agent_run_config_seeds_slack_thread_channel() -> None: "slack_report_channel": "C0123ABCD", } - config = schedules._agent_run_config(record, "thread_1") + config = await schedules._agent_run_config(record, "thread_1") assert config["configurable"]["slack_thread"] == {"channel_id": "C0123ABCD"} -def test_agent_run_config_omits_slack_thread_without_channel() -> None: +async def test_agent_run_config_omits_slack_thread_without_channel() -> None: record = { "id": "sched_1", "model": "Default", @@ -221,7 +221,7 @@ def test_agent_run_config_omits_slack_thread_without_channel() -> None: "slack_report_channel": None, } - config = schedules._agent_run_config(record, "thread_1") + config = await schedules._agent_run_config(record, "thread_1") assert "slack_thread" not in config["configurable"] diff --git a/tests/test_agent_subagent_models.py b/tests/test_agent_subagent_models.py index a3f5573c..c3f1ab8d 100644 --- a/tests/test_agent_subagent_models.py +++ b/tests/test_agent_subagent_models.py @@ -161,3 +161,71 @@ async def test_agent_subagent_inherits_profile_model_override_without_explicit_p "thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "high"}, } + + +@pytest.mark.asyncio +async def test_agent_gate_swaps_disabled_fable_profile_to_opus() -> None: + config: RunnableConfig = { + "configurable": { + "__is_for_execution__": True, + "thread_id": "thread-123", + "github_login": "octocat", + }, + "metadata": {}, + } + main_model = MagicMock(name="main_model") + subagent_model = MagicMock(name="subagent_model") + captured: dict[str, object] = {} + + def fake_create_deep_agent(**kwargs: object) -> _DummyAgent: + captured.update(kwargs) + return _DummyAgent() + + with ( + patch( + "agent.server.resolve_github_token", new_callable=AsyncMock, return_value=("ghp", None) + ), + patch("agent.server.resolve_triggering_user_identity", return_value=None), + patch( + "agent.server.ensure_sandbox_for_thread", + new_callable=AsyncMock, + return_value=MagicMock(), + ), + patch( + "agent.server.aresolve_sandbox_work_dir", + new_callable=AsyncMock, + return_value="/workspace", + ), + patch( + "agent.server.get_team_default_model_pair", + new_callable=AsyncMock, + return_value=( + ("bedrock_converse:us.anthropic.claude-opus-4-8", "medium"), + ("bedrock_converse:us.anthropic.claude-opus-4-8", "low"), + ), + ), + # Profile selected Fable back when it was allowed; it's now disabled. + patch( + "agent.server.load_profile", + new_callable=AsyncMock, + return_value={ + "default_model": "bedrock_converse:us.anthropic.claude-fable-5", + "reasoning_effort": "high", + }, + ), + patch("agent.server.get_team_fable_enabled", new_callable=AsyncMock, return_value=False), + patch("agent.server.fallback_model_id_for", return_value=None), + patch( + "agent.utils.deferred_model.make_model", side_effect=[main_model, subagent_model] + ) as make_model, + patch("agent.server.construct_system_prompt", return_value="prompt"), + patch("agent.server.create_deep_agent", side_effect=fake_create_deep_agent), + ): + await get_agent(config) + + # Fable was scrubbed to Opus for both main and subagent; effort preserved. + assert make_model.call_args_list[0].args == ("bedrock_converse:us.anthropic.claude-opus-4-8",) + assert make_model.call_args_list[0].kwargs["additional_model_request_fields"][ + "output_config" + ] == {"effort": "high"} + assert make_model.call_args_list[1].args == ("bedrock_converse:us.anthropic.claude-opus-4-8",) diff --git a/tests/test_anthropic_model.py b/tests/test_anthropic_model.py new file mode 100644 index 00000000..aec1777e --- /dev/null +++ b/tests/test_anthropic_model.py @@ -0,0 +1,55 @@ +import pytest + +from agent.dashboard.options import SUPPORTED_MODELS, provider_fallback_pair +from agent.utils.model import provider_model_kwargs + +SONNET_5_ID = "bedrock_converse:us.anthropic.claude-sonnet-5" +FABLE_5_ID = "bedrock_converse:us.anthropic.claude-fable-5" + + +def test_sonnet_5_is_supported_with_documented_efforts() -> None: + sonnet = next(m for m in SUPPORTED_MODELS if m["id"] == SONNET_5_ID) + assert sonnet["label"] == "Sonnet 5 (Bedrock)" + assert sonnet["efforts"] == ["low", "medium", "high", "xhigh", "max"] + assert sonnet["default_effort"] == "high" + assert sonnet["supports_images"] is True + + +@pytest.mark.parametrize("effort", ["low", "medium", "high", "xhigh", "max"]) +def test_sonnet_5_efforts_map_to_bedrock_kwargs(effort: str) -> None: + kwargs = provider_model_kwargs(SONNET_5_ID, effort, max_tokens=16_000) + assert kwargs["max_tokens"] == 16_000 + fields = kwargs["additional_model_request_fields"] + assert fields["output_config"] == {"effort": effort} + assert fields["thinking"] == {"type": "adaptive", "display": "summarized"} + + +def test_sonnet_46_fallback_uses_sonnet_5() -> None: + assert provider_fallback_pair("bedrock_converse:us.anthropic.claude-sonnet-4-6", "xhigh") == ( + SONNET_5_ID, + "xhigh", + ) + + +def test_opus_fallback_stays_on_opus_family() -> None: + assert provider_fallback_pair("bedrock_converse:us.anthropic.claude-opus-4-7", "xhigh") == ( + "bedrock_converse:us.anthropic.claude-opus-4-8", + "xhigh", + ) + + +def test_fable_5_is_supported_with_documented_efforts() -> None: + fable = next(m for m in SUPPORTED_MODELS if m["id"] == FABLE_5_ID) + assert fable["label"] == "Fable 5" + assert fable["efforts"] == ["low", "medium", "high", "xhigh", "max"] + assert fable["default_effort"] == "high" + assert fable["supports_images"] is True + + +@pytest.mark.parametrize("effort", ["low", "medium", "high", "xhigh", "max"]) +def test_fable_5_efforts_map_to_bedrock_kwargs(effort: str) -> None: + kwargs = provider_model_kwargs(FABLE_5_ID, effort, max_tokens=16_000) + assert kwargs["max_tokens"] == 16_000 + fields = kwargs["additional_model_request_fields"] + assert fields["output_config"] == {"effort": effort} + assert fields["thinking"] == {"type": "adaptive", "display": "summarized"} diff --git a/tests/test_dashboard_thread_api.py b/tests/test_dashboard_thread_api.py index 856856f8..067391e7 100644 --- a/tests/test_dashboard_thread_api.py +++ b/tests/test_dashboard_thread_api.py @@ -1,16 +1,19 @@ import base64 import json from types import SimpleNamespace +from unittest.mock import AsyncMock, patch import pytest from fastapi import HTTPException -from agent.dashboard import thread_api +from agent.dashboard import routes, thread_api from agent.dashboard.agent_overrides import resolve_agent_model_id from agent.dashboard.options import model_supports_images _TEXT_ONLY_MODEL = "fireworks:accounts/fireworks/models/deepseek-v4-pro" _VISION_MODEL = "bedrock_converse:us.anthropic.claude-opus-4-8" +_FABLE = "bedrock_converse:us.anthropic.claude-fable-5" +_PAIR = ("bedrock_converse:us.anthropic.claude-opus-4-8", "medium") def _image() -> thread_api.DashboardImageBody: @@ -1427,3 +1430,81 @@ async def test_status_filter_refreshes_threads_missing_run_status(monkeypatch) - assert {item["id"] for item in result["items"]} == {"t0"} assert result["items"][0]["status"] == "finished" assert set(run_list_thread_ids) == {"t0", "t1"} + + +@pytest.mark.asyncio +async def test_options_omits_fable_when_disabled() -> None: + with ( + patch( + "agent.dashboard.routes.get_team_fable_enabled", + new_callable=AsyncMock, + return_value=False, + ), + patch( + "agent.dashboard.routes.get_team_default_model", + new_callable=AsyncMock, + return_value=_PAIR, + ), + patch( + "agent.dashboard.routes.get_team_default_subagent_model", + new_callable=AsyncMock, + return_value=_PAIR, + ), + ): + payload = await routes.options() + assert _FABLE not in [m["id"] for m in payload["models"]] + + +@pytest.mark.asyncio +async def test_options_includes_fable_when_enabled() -> None: + with ( + patch( + "agent.dashboard.routes.get_team_fable_enabled", + new_callable=AsyncMock, + return_value=True, + ), + patch( + "agent.dashboard.routes.get_team_default_model", + new_callable=AsyncMock, + return_value=_PAIR, + ), + patch( + "agent.dashboard.routes.get_team_default_subagent_model", + new_callable=AsyncMock, + return_value=_PAIR, + ), + ): + payload = await routes.options() + assert _FABLE in [m["id"] for m in payload["models"]] + + +@pytest.mark.asyncio +async def test_options_gates_stale_fable_default_when_disabled() -> None: + # A stale Fable team default must not be advertised as the default while Fable + # is omitted from the selectable list, or the Cloud Agents page would offer a + # default that PUT /profile then rejects. + fable_pair = (_FABLE, "high") + with ( + patch( + "agent.dashboard.routes.get_team_fable_enabled", + new_callable=AsyncMock, + return_value=False, + ), + patch( + "agent.dashboard.routes.get_team_default_model", + new_callable=AsyncMock, + return_value=fable_pair, + ), + patch( + "agent.dashboard.routes.get_team_default_subagent_model", + new_callable=AsyncMock, + return_value=fable_pair, + ), + ): + payload = await routes.options() + model_ids = [m["id"] for m in payload["models"]] + assert _FABLE not in model_ids + assert payload["default_agent_model"] != _FABLE + assert payload["default_agent_subagent_model"] != _FABLE + assert payload["default_agent_model"] in model_ids + assert payload["default_agent_subagent_model"] in model_ids diff --git a/tests/test_model_fallback_resolution.py b/tests/test_model_fallback_resolution.py index dc11f9df..8f0305f3 100644 --- a/tests/test_model_fallback_resolution.py +++ b/tests/test_model_fallback_resolution.py @@ -5,7 +5,10 @@ import pytest from agent.dashboard.agent_overrides import normalize_profile_overrides from agent.dashboard.options import ( DEFAULT_MODEL_ID, + FABLE_MODEL_IDS, default_model_pair, + fable_disabled_fallback, + gate_fable_model, provider_fallback_pair, ) from agent.dashboard.team_settings import get_team_default_model @@ -88,3 +91,37 @@ def test_profile_unknown_provider_defers_to_team_default() -> None: def test_global_default_is_supported() -> None: model, _ = default_model_pair() assert model == DEFAULT_MODEL_ID + + +def test_gate_fable_passthrough_when_enabled() -> None: + assert gate_fable_model( + "bedrock_converse:us.anthropic.claude-fable-5", "high", fable_enabled=True + ) == ( + "bedrock_converse:us.anthropic.claude-fable-5", + "high", + ) + + +def test_gate_fable_swaps_to_opus_when_disabled() -> None: + assert gate_fable_model( + "bedrock_converse:us.anthropic.claude-fable-5", "high", fable_enabled=False + ) == ( + "bedrock_converse:us.anthropic.claude-opus-4-8", + "high", + ) + + +def test_gate_fable_leaves_non_fable_ids_alone() -> None: + assert gate_fable_model( + "fireworks:accounts/fireworks/models/deepseek-v4-pro", "high", fable_enabled=False + ) == ( + "fireworks:accounts/fireworks/models/deepseek-v4-pro", + "high", + ) + + +def test_fable_disabled_fallback_is_non_fable_claude() -> None: + model, effort = fable_disabled_fallback("high") + assert model == "bedrock_converse:us.anthropic.claude-opus-4-8" + assert model not in FABLE_MODEL_IDS + assert effort == "high" diff --git a/tests/test_team_settings_fable.py b/tests/test_team_settings_fable.py new file mode 100644 index 00000000..6e4ccefa --- /dev/null +++ b/tests/test_team_settings_fable.py @@ -0,0 +1,122 @@ +from __future__ import annotations + +from unittest.mock import AsyncMock, patch + +import pytest + +from agent.dashboard.options import FABLE_MODEL_IDS +from agent.dashboard.team_settings import TeamSettingsUpdate, get_team_fable_enabled + +_FABLE = "bedrock_converse:us.anthropic.claude-fable-5" + +# --- accessor: get_team_fable_enabled (async, patched store) --- + + +@pytest.mark.asyncio +async def test_fable_enabled_defaults_false_when_absent() -> None: + # Legacy record with no fable_enabled key -> off. + with patch( + "agent.dashboard.team_settings.get_team_settings", + new_callable=AsyncMock, + return_value={}, + ): + assert await get_team_fable_enabled() is False + + +@pytest.mark.asyncio +async def test_fable_enabled_true_when_set() -> None: + with patch( + "agent.dashboard.team_settings.get_team_settings", + new_callable=AsyncMock, + return_value={"fable_enabled": True}, + ): + assert await get_team_fable_enabled() is True + + +@pytest.mark.asyncio +async def test_fable_enabled_false_for_non_bool_value() -> None: + # Fail-closed: any non-bool (e.g. a stray string) resolves to False. + with patch( + "agent.dashboard.team_settings.get_team_settings", + new_callable=AsyncMock, + return_value={"fable_enabled": "true"}, + ): + assert await get_team_fable_enabled() is False + + +# --- validation: TeamSettingsUpdate (sync) --- + + +def test_update_defaults_fable_disabled() -> None: + assert TeamSettingsUpdate().fable_enabled is False + + +def test_update_coerces_fable_model_when_disabled() -> None: + # Disabling Fable must never fail: a lingering Fable default is swapped for a + # safe non-Fable fallback (effort preserved) rather than rejected. + update = TeamSettingsUpdate( + default_agent_model=_FABLE, + default_agent_reasoning_effort="high", + ) + assert update.default_agent_model not in FABLE_MODEL_IDS + assert update.default_agent_reasoning_effort == "high" + + +def test_update_coerces_fable_in_any_role_when_disabled() -> None: + # Same coercion applies to every model field, e.g. review chat. + update = TeamSettingsUpdate( + default_chat_model=_FABLE, + default_chat_reasoning_effort="high", + ) + assert update.default_chat_model not in FABLE_MODEL_IDS + assert update.default_chat_reasoning_effort == "high" + + +def test_disable_transition_coerces_all_fable_defaults() -> None: + # The kill-switch flow: an admin who had picked Fable everywhere flips the + # toggle off, and the UI re-sends the whole settings blob with the Fable + # defaults still attached. The update must succeed and strip every Fable id. + update = TeamSettingsUpdate( + fable_enabled=False, + default_agent_model=_FABLE, + default_agent_reasoning_effort="high", + default_agent_subagent_model=_FABLE, + default_agent_subagent_reasoning_effort="high", + default_reviewer_model=_FABLE, + default_reviewer_reasoning_effort="high", + default_reviewer_subagent_model=_FABLE, + default_reviewer_subagent_reasoning_effort="high", + default_grouping_model=_FABLE, + default_grouping_reasoning_effort="high", + default_chat_model=_FABLE, + default_chat_reasoning_effort="high", + ) + for field in ( + "default_agent_model", + "default_agent_subagent_model", + "default_reviewer_model", + "default_reviewer_subagent_model", + "default_grouping_model", + "default_chat_model", + ): + assert getattr(update, field) not in FABLE_MODEL_IDS, field + + +def test_update_leaves_non_fable_defaults_untouched_when_disabled() -> None: + # Coercion only rewrites Fable ids; other selections pass through unchanged. + update = TeamSettingsUpdate( + default_agent_model="bedrock_converse:us.anthropic.claude-sonnet-5", + default_agent_reasoning_effort="medium", + ) + assert update.default_agent_model == "bedrock_converse:us.anthropic.claude-sonnet-5" + assert update.default_agent_reasoning_effort == "medium" + + +def test_update_accepts_fable_model_when_enabled() -> None: + update = TeamSettingsUpdate( + fable_enabled=True, + default_agent_model=_FABLE, + default_agent_reasoning_effort="high", + ) + assert update.fable_enabled is True + assert update.default_agent_model == _FABLE diff --git a/ui/src/lib/api.ts b/ui/src/lib/api.ts index d3321176..fe0652db 100644 --- a/ui/src/lib/api.ts +++ b/ui/src/lib/api.ts @@ -166,6 +166,7 @@ export interface TeamSettings { review_trace_links: boolean /** Tri-state LLM Gateway toggle; null inherits the LANGSMITH_GATEWAY_ENABLED default. */ gateway_enabled?: boolean | null + fable_enabled?: boolean review_tracing_project?: string | null org_guidelines?: string | null default_agent_model?: string | null diff --git a/ui/src/routes/admin.tsx b/ui/src/routes/admin.tsx index 7ec9b2b4..15b4ece6 100644 --- a/ui/src/routes/admin.tsx +++ b/ui/src/routes/admin.tsx @@ -22,6 +22,7 @@ import { SelectValue, } from "@/components/ui/select" import { Skeleton } from "@/components/ui/skeleton" +import { Switch } from "@/components/ui/switch" import { api } from "@/lib/api" import { RequireLogin } from "@/lib/auth-redirect" import { useSession } from "@/lib/session" @@ -57,6 +58,8 @@ function AdminPage() { + + @@ -694,6 +697,44 @@ function PRTraceResolutionSection() { ) } +function FableSection() { + const qc = useQueryClient() + const settings = useQuery({ queryKey: ["teamSettings"], queryFn: api.getTeamSettings }) + const [error, setError] = useState(null) + const save = useMutation({ + mutationFn: (body: TeamSettings) => api.saveTeamSettings(body), + onSuccess: (saved) => { + qc.setQueryData(["teamSettings"], saved) + qc.invalidateQueries({ queryKey: ["options"] }) // refresh pickers so Fable appears/disappears + setError(null) + }, + onError: (e: Error) => setError(e.message), + }) + return ( + + + + settings.data && save.mutate({ ...settings.data, fable_enabled: next }) + } + disabled={!settings.data || save.isPending} + /> + } + /> + + {error && {error}} + + ) +} + function GlobalDefaultsSection({ models }: { models: Array }) { const qc = useQueryClient() const settings = useQuery({
{error}