diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 608fef06..a040077c 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -80,6 +80,7 @@ jobs: working-directory: tests/e2e run: | npm ci + sudo rm -f /etc/apt/sources.list.d/azure-cli.list /etc/apt/sources.list.d/microsoft-prod.list npx playwright install --with-deps chromium # Playwright's webServer boots `langgraph dev`; globalSetup builds the real # ui/ SPA. The fake LLM/GitHub/Slack boundaries need no secrets. diff --git a/agent/chat.py b/agent/chat.py index 132f83d8..45dae8a4 100644 --- a/agent/chat.py +++ b/agent/chat.py @@ -28,8 +28,16 @@ warnings.filterwarnings("ignore", message=".*Pydantic V1.*", category=UserWarnin from deepagents import create_deep_agent from langchain.agents.middleware import ModelCallLimitMiddleware -from .dashboard.options import SUPPORTED_MODEL_IDS, model_supports_effort -from .dashboard.team_settings import get_effective_gateway_enabled, get_team_default_model +from .dashboard.options import ( + SUPPORTED_MODEL_IDS, + gate_fable_model, + model_supports_effort, +) +from .dashboard.team_settings import ( + get_effective_gateway_enabled, + get_team_default_model, + get_team_fable_enabled, +) from .middleware import ( ExcludeToolsMiddleware, SanitizeFireworksMessagesMiddleware, @@ -125,6 +133,9 @@ async def get_chat_agent(config: RunnableConfig) -> Pregel: configurable["chat_github_token"] = token model_id, effort = await _resolve_chat_model(configurable) + model_id, effort = gate_fable_model( + model_id, effort, fable_enabled=await get_team_fable_enabled() + ) use_gateway = await get_effective_gateway_enabled() model_kwargs = provider_model_kwargs( model_id, diff --git a/agent/dashboard/options.py b/agent/dashboard/options.py index 401e411a..3e22bd3c 100644 --- a/agent/dashboard/options.py +++ b/agent/dashboard/options.py @@ -28,6 +28,13 @@ SUPPORTED_MODELS: list[ModelOption] = [ "default_effort": "high", "supports_images": True, }, + { + "id": "bedrock_converse:us.anthropic.claude-fable-5", + "label": "Fable 5", + "efforts": ["low", "medium", "high", "xhigh", "max"], + "default_effort": "high", + "supports_images": True, + }, { "id": "fireworks:accounts/fireworks/models/kimi-k2p7-code", "label": "Kimi K2.7", @@ -74,6 +81,38 @@ SUPPORTED_MODELS: list[ModelOption] = [ SUPPORTED_MODEL_IDS: frozenset[str] = frozenset(m["id"] for m in SUPPORTED_MODELS) +_FABLE_ID_PREFIX = "bedrock_converse:us.anthropic.claude-fable" + +FABLE_MODEL_IDS: frozenset[str] = frozenset( + m["id"] for m in SUPPORTED_MODELS if m["id"].startswith(_FABLE_ID_PREFIX) +) + + +def fable_disabled_fallback(effort: object = None) -> tuple[str, str]: + """Newest supported non-Fable Claude model (keeps the Claude family), + else the global default. Substitutes a Fable selection when Fable is + disabled workspace-wide, preserving ``effort`` when the fallback supports it.""" + for m in SUPPORTED_MODELS: + if m["id"].startswith("bedrock_converse:us.anthropic.") and m["id"] not in FABLE_MODEL_IDS: + return m["id"], _fallback_effort_for(m, effort) or m["default_effort"] + return default_model_pair() + + +def gate_fable_model( + model_id: str, effort: str | None, *, fable_enabled: bool +) -> tuple[str, str | None]: + """Provider-data-share gate: if Fable is disabled but a Fable id was + resolved, swap in a safe non-Fable model. Fable requires the account to opt + into Bedrock ``provider_data_share`` (prompts/completions shared with the + provider), so it is gated off by default. Non-Fable selections pass through + unchanged. Applied at every model-construction entrypoint so a disabled + Fable model can never reach ``make_model``, no matter which layer selected + it.""" + if not fable_enabled and isinstance(model_id, str) and model_id in FABLE_MODEL_IDS: + return fable_disabled_fallback(effort) + return model_id, effort + + DEFAULT_MODEL_ID: str = "bedrock_converse:us.anthropic.claude-opus-4-8" DEFAULT_MODEL_EFFORT: str = "medium" diff --git a/agent/dashboard/routes.py b/agent/dashboard/routes.py index 04b8df48..5793cb81 100644 --- a/agent/dashboard/routes.py +++ b/agent/dashboard/routes.py @@ -59,7 +59,7 @@ from .oauth import ( require_session, sanitize_redirect_to, ) -from .options import SUPPORTED_MODELS +from .options import FABLE_MODEL_IDS, SUPPORTED_MODELS, gate_fable_model from .profiles import ( ProfileUpdate, get_profile, @@ -145,6 +145,7 @@ from .team_settings import ( TeamSettingsUpdate, get_team_default_model, get_team_default_subagent_model, + get_team_fable_enabled, get_team_settings, upsert_team_settings, ) @@ -445,8 +446,23 @@ async def me(session: dict[str, Any] = _SESSION_DEP) -> dict[str, Any]: async def options() -> dict[str, Any]: agent_model, agent_effort = await get_team_default_model("agent") subagent_model, subagent_effort = await get_team_default_subagent_model("agent") + fable_enabled = await get_team_fable_enabled() + # Never advertise a default that isn't in the selectable list: when Fable is + # off, gate a stale Fable default down to its non-Fable fallback so the Cloud + # Agents page (and the PUT /profile it drives) don't choke on it. + agent_model, agent_effort = gate_fable_model( + agent_model, agent_effort, fable_enabled=fable_enabled + ) + subagent_model, subagent_effort = gate_fable_model( + subagent_model, subagent_effort, fable_enabled=fable_enabled + ) + models = ( + SUPPORTED_MODELS + if fable_enabled + else [m for m in SUPPORTED_MODELS if m["id"] not in FABLE_MODEL_IDS] + ) return { - "models": SUPPORTED_MODELS, + "models": models, "default_agent_model": agent_model, "default_agent_reasoning_effort": agent_effort, "default_agent_subagent_model": subagent_model, @@ -468,6 +484,12 @@ async def put_my_profile( session: dict[str, Any] = _SESSION_DEP, ) -> dict[str, Any]: update.validate_pairing() + if not await get_team_fable_enabled(): + if ( + update.default_model in FABLE_MODEL_IDS + or update.default_subagent_model in FABLE_MODEL_IDS + ): + raise HTTPException(400, "Fable is disabled for this workspace") return await upsert_profile(session["sub"], session.get("email") or "", update) diff --git a/agent/dashboard/schedules.py b/agent/dashboard/schedules.py index 8356035b..5dff79b2 100644 --- a/agent/dashboard/schedules.py +++ b/agent/dashboard/schedules.py @@ -12,9 +12,10 @@ from fastapi import HTTPException from pydantic import BaseModel, Field, field_validator from ..utils.thread_ops import langgraph_client -from .options import SUPPORTED_MODEL_IDS, model_supports_effort +from .options import SUPPORTED_MODEL_IDS, gate_fable_model, model_supports_effort from .profiles import get_profile, get_valid_access_token from .repo_access import repo_config_for_user, require_repo_access_for_user +from .team_settings import get_team_fable_enabled from .thread_api import _agent_version_metadata, _now_ms, _resolve_run_email logger = logging.getLogger(__name__) @@ -405,7 +406,7 @@ def _agent_run_metadata(record: dict[str, Any], thread_id: str) -> dict[str, Any return metadata -def _agent_run_config(record: dict[str, Any], thread_id: str) -> dict[str, Any]: +async def _agent_run_config(record: dict[str, Any], thread_id: str) -> dict[str, Any]: configurable: dict[str, Any] = { "thread_id": thread_id, "source": "schedule", @@ -418,6 +419,9 @@ def _agent_run_config(record: dict[str, Any], thread_id: str) -> dict[str, Any]: configurable["repo"] = repo model, effort = _normalize_model_choice(record.get("model"), record.get("effort")) if model and effort: + model, effort = gate_fable_model( + model, effort, fable_enabled=await get_team_fable_enabled() + ) configurable["agent_model_id"] = model configurable["agent_effort"] = effort report_channel = record.get("slack_report_channel") @@ -471,7 +475,7 @@ async def launch_scheduled_agent_run(schedule_id: str) -> dict[str, Any]: thread_id, _AGENT_ASSISTANT_ID, input={"messages": [{"role": "user", "content": record["prompt"]}]}, - config=_agent_run_config(record, thread_id), + config=await _agent_run_config(record, thread_id), if_not_exists="create", stream_mode=["values", "updates", "messages-tuple"], stream_resumable=True, diff --git a/agent/dashboard/team_settings.py b/agent/dashboard/team_settings.py index 98299ed5..5c65b971 100644 --- a/agent/dashboard/team_settings.py +++ b/agent/dashboard/team_settings.py @@ -17,8 +17,10 @@ from pydantic import BaseModel, field_validator, model_validator from ..utils.gateway import resolve_gateway_enabled from .options import ( + FABLE_MODEL_IDS, SUPPORTED_MODEL_IDS, default_model_pair, + gate_fable_model, model_supports_effort, provider_fallback_pair, ) @@ -56,6 +58,7 @@ class TeamSettingsUpdate(BaseModel): # Tri-state LLM Gateway toggle: True/False is authoritative, None inherits the # LANGSMITH_GATEWAY_ENABLED deployment default. gateway_enabled: bool | None = None + fable_enabled: bool = False review_tracing_project: str | None = None org_guidelines: str | None = None default_agent_model: str | None = None @@ -131,6 +134,26 @@ class TeamSettingsUpdate(BaseModel): _validate_model_effort_pair( self.default_chat_model, self.default_chat_reasoning_effort, "review chat" ) + if not self.fable_enabled: + # Disabling Fable is the provider-data-share kill switch and must always succeed: rather + # than reject a payload that still carries a Fable default, swap each + # Fable default to its safe non-Fable fallback (mirrors the runtime + # gate_fable_model guard) so the stored record can't advertise Fable. + for model_field, effort_field in ( + ("default_agent_model", "default_agent_reasoning_effort"), + ("default_agent_subagent_model", "default_agent_subagent_reasoning_effort"), + ("default_reviewer_model", "default_reviewer_reasoning_effort"), + ("default_reviewer_subagent_model", "default_reviewer_subagent_reasoning_effort"), + ("default_grouping_model", "default_grouping_reasoning_effort"), + ("default_chat_model", "default_chat_reasoning_effort"), + ): + model = getattr(self, model_field) + if model in FABLE_MODEL_IDS: + new_model, new_effort = gate_fable_model( + model, getattr(self, effort_field), fable_enabled=False + ) + setattr(self, model_field, new_model) + setattr(self, effort_field, new_effort) return self @@ -171,6 +194,7 @@ def _default_settings() -> dict[str, Any]: "pr_summaries": True, "review_trace_links": True, "gateway_enabled": None, + "fable_enabled": False, "review_tracing_project": None, "org_guidelines": DEFAULT_ORG_REVIEW_GUIDELINES, "default_agent_model": fallback_model, @@ -226,6 +250,7 @@ async def upsert_team_settings(update: TeamSettingsUpdate) -> dict[str, Any]: "pr_summaries": update.pr_summaries, "review_trace_links": update.review_trace_links, "gateway_enabled": update.gateway_enabled, + "fable_enabled": update.fable_enabled, "review_tracing_project": update.review_tracing_project, "org_guidelines": update.org_guidelines, "default_agent_model": update.default_agent_model, @@ -353,6 +378,13 @@ async def get_team_gateway_enabled() -> bool | None: return value if isinstance(value, bool) else None +async def get_team_fable_enabled() -> bool: + """Return whether Fable models are enabled for the team.""" + settings = await get_team_settings() + value = settings.get("fable_enabled") + return bool(value) if isinstance(value, bool) else False + + async def get_effective_gateway_enabled() -> bool: """Resolve whether LLM Gateway routing is on: team setting, else env default.""" return resolve_gateway_enabled(await get_team_gateway_enabled()) diff --git a/agent/dashboard/thread_api.py b/agent/dashboard/thread_api.py index 5bb69e42..a6bf0af6 100644 --- a/agent/dashboard/thread_api.py +++ b/agent/dashboard/thread_api.py @@ -31,12 +31,13 @@ from .agent_overrides import normalize_profile_overrides from .options import ( SUPPORTED_MODEL_IDS, default_vision_model_pair, + gate_fable_model, model_supports_effort, model_supports_images, ) from .pr_diff import build_pr_diff_files from .profiles import get_profile, get_valid_access_token -from .team_settings import get_team_default_model +from .team_settings import get_team_default_model, get_team_fable_enabled from .user_mappings import email_for_login logger = logging.getLogger(__name__) @@ -153,6 +154,9 @@ async def _resolve_agent_model_choice( chosen_model, chosen_effort = _normalize_model_choice(model_id, effort) if chosen_model and chosen_effort: resolved_model, resolved_effort = chosen_model, chosen_effort + resolved_model, resolved_effort = gate_fable_model( + resolved_model, resolved_effort, fable_enabled=await get_team_fable_enabled() + ) return resolved_model, resolved_effort diff --git a/agent/reviewer.py b/agent/reviewer.py index 18151c0b..4c27b349 100644 --- a/agent/reviewer.py +++ b/agent/reviewer.py @@ -33,11 +33,13 @@ from deepagents import create_deep_agent from langchain.agents.middleware import ModelCallLimitMiddleware from langchain_core.language_models.chat_models import BaseChatModel +from .dashboard.options import gate_fable_model from .dashboard.team_settings import ( get_effective_gateway_enabled, get_org_review_guidelines, get_team_default_grouping_model, get_team_default_model_pair, + get_team_fable_enabled, ) from .middleware import ( RepairOrphanedToolCallsMiddleware, @@ -822,6 +824,9 @@ async def _resolve_grouping_model( effort = configured_effort if isinstance(configured_effort, str) else None else: model_id, effort = await get_team_default_grouping_model() + model_id, effort = gate_fable_model( + model_id, effort, fable_enabled=await get_team_fable_enabled() + ) model_kwargs = provider_model_kwargs( model_id, effort, @@ -1124,6 +1129,13 @@ async def get_reviewer_agent(config: RunnableConfig) -> Pregel: subagent_effort = ( configured_subagent_effort if isinstance(configured_subagent_effort, str) else None ) + fable_enabled = await get_team_fable_enabled() + model_id, reasoning_effort = gate_fable_model( + model_id, reasoning_effort, fable_enabled=fable_enabled + ) + subagent_model_id, subagent_effort = gate_fable_model( + subagent_model_id, subagent_effort, fable_enabled=fable_enabled + ) model_kwargs = provider_model_kwargs( model_id, reasoning_effort, diff --git a/agent/server.py b/agent/server.py index 619d3cfa..628c1135 100644 --- a/agent/server.py +++ b/agent/server.py @@ -41,12 +41,18 @@ from .dashboard.agent_overrides import ( resolve_github_login, ) from .dashboard.agent_usage import record_agent_thread_usage -from .dashboard.options import DEFAULT_MODEL_ID, SUPPORTED_MODEL_IDS, model_supports_effort +from .dashboard.options import ( + DEFAULT_MODEL_ID, + SUPPORTED_MODEL_IDS, + gate_fable_model, + model_supports_effort, +) from .dashboard.repo_snapshots import resolve_repo_snapshot_id from .dashboard.team_settings import ( get_effective_gateway_enabled, get_team_default_model_pair, get_team_default_repo, + get_team_fable_enabled, ) from .dashboard.user_mappings import email_for_login from .integrations.corridor_mcp import load_corridor_tools @@ -765,6 +771,7 @@ async def get_agent(config: RunnableConfig) -> Pregel: ) team_defaults_task = asyncio.create_task(get_team_default_model_pair("agent")) gateway_task = asyncio.create_task(get_effective_gateway_enabled()) + fable_task = asyncio.create_task(get_team_fable_enabled()) profile_task = asyncio.create_task(load_profile(profile_login)) if profile_login else None try: ( @@ -772,11 +779,13 @@ async def get_agent(config: RunnableConfig) -> Pregel: sandbox_backend, team_defaults, use_gateway, + fable_enabled, ) = await asyncio.gather( triggering_user_identity_task, sandbox_task, team_defaults_task, gateway_task, + fable_task, ) except SandboxRepoMismatchError as exc: # Repo-binding refusal at the run boundary: log for alarming and surface the @@ -787,6 +796,7 @@ async def get_agent(config: RunnableConfig) -> Pregel: triggering_user_identity_task, team_defaults_task, gateway_task, + fable_task, profile_task, ): if pending is not None and not pending.done(): @@ -860,6 +870,13 @@ async def get_agent(config: RunnableConfig) -> Pregel: if always_create_prs: logger.info("Always Create PRs enabled by profile for %s", profile_login) + model_id, profile_effort = gate_fable_model( + model_id, profile_effort, fable_enabled=fable_enabled + ) + subagent_model_id, subagent_effort = gate_fable_model( + subagent_model_id, subagent_effort, fable_enabled=fable_enabled + ) + model_kwargs = provider_model_kwargs( model_id, profile_effort, diff --git a/docs/upstream-sync/triage.jsonl b/docs/upstream-sync/triage.jsonl index 3f7d848e..d854ac89 100644 --- a/docs/upstream-sync/triage.jsonl +++ b/docs/upstream-sync/triage.jsonl @@ -70,7 +70,7 @@ {"sha": "216cf181", "pr": 1699, "subject": "fix: keep workflow HITL without token downscoping (#1699)", "disposition": "landed", "reason": "DIVERGES-FROM-UPSTREAM: fork deliberately does NOT adopt #1699's standing-token workflows:write broadening. Security review (#159) BLOCKed it — the standing ALWAYS-ON proxy token carrying workflows:write turns the HITL guard's git-push-parser gaps (obfuscated-expansion push, `gh api` REST contents PUT, cross-branch refspecs) into live unapproved-workflow-push exploits. Fork keeps BASE without workflows:write and restores the transient per-approval elevation (_run_with_workflow_token mints WORKFLOW_RUNTIME_PROXY_TOKEN_PERMISSIONS around the approved, guard-normalized fixed_command, then downscopes to RUNTIME then BASE): the token scope is the backstop the parser relies on, so a bypass hits GitHub 403. HITL diff-preview/approval-URL/Slack-card additions from #159 retained; token-model divergence only.", "branch": "plan-approval", "local_sha": null, "updated": "2026-07-09T20:00:00Z"} {"sha": "67abf5b0", "pr": 1659, "subject": "fix: surface attributed PR creation failures (#1659)", "disposition": "deferred", "reason": "PR-attribution-failure guard (new mw, safe imports); heavy conflict on diverged open_pull_request.py", "branch": "pr-attribution", "local_sha": null, "updated": "2026-07-08T20:14:42Z"} {"sha": "3dbc0282", "pr": 1676, "subject": "fix: preserve plan redirects after login (#1676)", "disposition": "landed", "reason": "FLAG-HUMAN: follow-on to landed #1668 refining sanitize_redirect_to (open-redirect auth surface); not a dup", "branch": "plan-approval", "local_sha": null, "updated": "2026-07-09T17:10:22Z"} -{"sha": "c75cbb1f", "pr": 1677, "subject": "feat: re-add Fable 5 with an admin toggle to disable it (#1677)", "disposition": "deferred", "reason": "FLAG-HUMAN: re-adds Fable 5 via anthropic: — contradicts dev's deliberate hide (#1483) + Bedrock migration (#62); wont-merge candidate", "branch": "fable-admin-toggle", "local_sha": null, "updated": "2026-07-08T20:14:42Z"} +{"sha": "c75cbb1f", "pr": 1677, "subject": "feat: re-add Fable 5 with an admin toggle to disable it (#1677)", "disposition": "landed", "reason": "Ported + Bedrock-converted onto dev via feat/readd-fable5-bedrock; anthropic: Fable ID mapped to bedrock_converse:us.anthropic.claude-fable-5.", "branch": "fable-admin-toggle", "local_sha": null, "updated": "2026-07-10T17:07:19Z"} {"sha": "bb104d93", "pr": 1679, "subject": "fix: submit plan comments with cmd enter (#1679)", "disposition": "landed", "reason": "applies clean but edits fork-diverged PlanReview.tsx (#130); needs UI/e2e validation — separate PR", "branch": "plan-approval", "local_sha": null, "updated": "2026-07-09T17:10:22Z"} {"sha": "304032fa", "pr": 1680, "subject": "chore: clarify question answering prompt (#1680)", "disposition": "deferred", "reason": "reword Slack info-only answer guidance; conflicts w/ fork's customized Slack prompt", "branch": "prompt-tweaks", "local_sha": null, "updated": "2026-07-08T20:14:42Z"} {"sha": "5003c953", "pr": 1683, "subject": "feat: open Linear-triggered PRs as the triggering user (#1683)", "disposition": "deferred", "reason": "FLAG-HUMAN: adds linear to resolve_github_token per-user OAuth branch (auth surface); depends on #1626 linear.py", "branch": "linear-pr-as-user", "local_sha": null, "updated": "2026-07-08T20:14:42Z"} diff --git a/docs/upstream-sync/triage.md b/docs/upstream-sync/triage.md index bc053790..101342a7 100644 --- a/docs/upstream-sync/triage.md +++ b/docs/upstream-sync/triage.md @@ -54,6 +54,7 @@ Rows key on the **upstream SHA** (stable across local cherry-picks). Deferred ro | `5f7f5fbd` | #1697 | fix: Reduce graph import and loader startup latency (#1697) | Landed | import-hygiene refactor; cross-cutting, references many deferred upstream-only modules | durable-dispatch | | `216cf181` | #1699 | fix: keep workflow HITL without token downscoping (#1699) | Landed | DIVERGES-FROM-UPSTREAM: fork deliberately does NOT adopt #1699's standing-token workflows:write broadening. Security review (#159) BLOCKed it — the standing ALWAYS-ON proxy token carrying workflows:write turns the HITL guard's git-push-parser gaps (obfuscated-expansion push, `gh api` REST contents PUT, cross-branch refspecs) into live unapproved-workflow-push exploits. Fork keeps BASE without workflows:write and restores the transient per-approval elevation (_run_with_workflow_token mints WORKFLOW_RUNTIME_PROXY_TOKEN_PERMISSIONS around the approved, guard-normalized fixed_command, then downscopes to RUNTIME then BASE): the token scope is the backstop the parser relies on, so a bypass hits GitHub 403. HITL diff-preview/approval-URL/Slack-card additions from #159 retained; token-model divergence only. | plan-approval | | `3dbc0282` | #1676 | fix: preserve plan redirects after login (#1676) | Landed | FLAG-HUMAN: follow-on to landed #1668 refining sanitize_redirect_to (open-redirect auth surface); not a dup | plan-approval | +| `c75cbb1f` | #1677 | feat: re-add Fable 5 with an admin toggle to disable it (#1677) | Landed | Ported + Bedrock-converted onto dev via feat/readd-fable5-bedrock; anthropic: Fable ID mapped to bedrock_converse:us.anthropic.claude-fable-5. | fable-admin-toggle | | `bb104d93` | #1679 | fix: submit plan comments with cmd enter (#1679) | Landed | applies clean but edits fork-diverged PlanReview.tsx (#130); needs UI/e2e validation — separate PR | plan-approval | | `feb7ac98` | #1689 | feat(web): surface thread sandbox ID with touch-friendly menu (#1689) | Landed | Ported to dev via feat/thread-sandbox-id-sidebar. | dashboard-ui | | `7f7af715` | #1684 | feat: auto-load scoped AGENTS on reads (#1684) | Landed | ported in #129 (SubdirAgentsReadMiddleware) | subdir-agents | @@ -89,7 +90,6 @@ Rows key on the **upstream SHA** (stable across local cherry-picks). Deferred ro | `4f8bc2dd` | #1692 | refactor: simplify open-swe agent sandbox lifecycle (#1692) | Deferred | FLAG-HUMAN: structural rewrite of ensure_sandbox_for_thread (drops __creating__ 4-case sentinel) | sandbox-refactor | | `48217b68` | #1489 | feat(open-swe): add E2B sandbox provider (#1489) | Deferred | additive E2B provider; separable but ships on the async sandbox.py base | sandbox-refactor | | `67abf5b0` | #1659 | fix: surface attributed PR creation failures (#1659) | Deferred | PR-attribution-failure guard (new mw, safe imports); heavy conflict on diverged open_pull_request.py | pr-attribution | -| `c75cbb1f` | #1677 | feat: re-add Fable 5 with an admin toggle to disable it (#1677) | Deferred | FLAG-HUMAN: re-adds Fable 5 via anthropic: — contradicts dev's deliberate hide (#1483) + Bedrock migration (#62); wont-merge candidate | fable-admin-toggle | | `304032fa` | #1680 | chore: clarify question answering prompt (#1680) | Deferred | reword Slack info-only answer guidance; conflicts w/ fork's customized Slack prompt | prompt-tweaks | | `5003c953` | #1683 | feat: open Linear-triggered PRs as the triggering user (#1683) | Deferred | FLAG-HUMAN: adds linear to resolve_github_token per-user OAuth branch (auth surface); depends on #1626 linear.py | linear-pr-as-user | | `f53caff1` | #1701 | fix: fall back to core GitHub App scope when optional grants missing (#1701) | Deferred | FLAG-HUMAN: GitHub-App permission-ladder degrade (auth surface); heavy conflict on diverged github_app.py/_resolve_proxy_token | github-app-scope | diff --git a/tests/test_agent_schedules.py b/tests/test_agent_schedules.py index df2509b3..dbde7c42 100644 --- a/tests/test_agent_schedules.py +++ b/tests/test_agent_schedules.py @@ -196,7 +196,7 @@ async def test_update_agent_schedule_clears_slack_report_channel(fake_client) -> assert result["slackReportChannel"] is None -def test_agent_run_config_seeds_slack_thread_channel() -> None: +async def test_agent_run_config_seeds_slack_thread_channel() -> None: record = { "id": "sched_1", "model": "Default", @@ -206,12 +206,12 @@ def test_agent_run_config_seeds_slack_thread_channel() -> None: "slack_report_channel": "C0123ABCD", } - config = schedules._agent_run_config(record, "thread_1") + config = await schedules._agent_run_config(record, "thread_1") assert config["configurable"]["slack_thread"] == {"channel_id": "C0123ABCD"} -def test_agent_run_config_omits_slack_thread_without_channel() -> None: +async def test_agent_run_config_omits_slack_thread_without_channel() -> None: record = { "id": "sched_1", "model": "Default", @@ -221,7 +221,7 @@ def test_agent_run_config_omits_slack_thread_without_channel() -> None: "slack_report_channel": None, } - config = schedules._agent_run_config(record, "thread_1") + config = await schedules._agent_run_config(record, "thread_1") assert "slack_thread" not in config["configurable"] diff --git a/tests/test_agent_subagent_models.py b/tests/test_agent_subagent_models.py index a3f5573c..c3f1ab8d 100644 --- a/tests/test_agent_subagent_models.py +++ b/tests/test_agent_subagent_models.py @@ -161,3 +161,71 @@ async def test_agent_subagent_inherits_profile_model_override_without_explicit_p "thinking": {"type": "adaptive", "display": "summarized"}, "output_config": {"effort": "high"}, } + + +@pytest.mark.asyncio +async def test_agent_gate_swaps_disabled_fable_profile_to_opus() -> None: + config: RunnableConfig = { + "configurable": { + "__is_for_execution__": True, + "thread_id": "thread-123", + "github_login": "octocat", + }, + "metadata": {}, + } + main_model = MagicMock(name="main_model") + subagent_model = MagicMock(name="subagent_model") + captured: dict[str, object] = {} + + def fake_create_deep_agent(**kwargs: object) -> _DummyAgent: + captured.update(kwargs) + return _DummyAgent() + + with ( + patch( + "agent.server.resolve_github_token", new_callable=AsyncMock, return_value=("ghp", None) + ), + patch("agent.server.resolve_triggering_user_identity", return_value=None), + patch( + "agent.server.ensure_sandbox_for_thread", + new_callable=AsyncMock, + return_value=MagicMock(), + ), + patch( + "agent.server.aresolve_sandbox_work_dir", + new_callable=AsyncMock, + return_value="/workspace", + ), + patch( + "agent.server.get_team_default_model_pair", + new_callable=AsyncMock, + return_value=( + ("bedrock_converse:us.anthropic.claude-opus-4-8", "medium"), + ("bedrock_converse:us.anthropic.claude-opus-4-8", "low"), + ), + ), + # Profile selected Fable back when it was allowed; it's now disabled. + patch( + "agent.server.load_profile", + new_callable=AsyncMock, + return_value={ + "default_model": "bedrock_converse:us.anthropic.claude-fable-5", + "reasoning_effort": "high", + }, + ), + patch("agent.server.get_team_fable_enabled", new_callable=AsyncMock, return_value=False), + patch("agent.server.fallback_model_id_for", return_value=None), + patch( + "agent.utils.deferred_model.make_model", side_effect=[main_model, subagent_model] + ) as make_model, + patch("agent.server.construct_system_prompt", return_value="prompt"), + patch("agent.server.create_deep_agent", side_effect=fake_create_deep_agent), + ): + await get_agent(config) + + # Fable was scrubbed to Opus for both main and subagent; effort preserved. + assert make_model.call_args_list[0].args == ("bedrock_converse:us.anthropic.claude-opus-4-8",) + assert make_model.call_args_list[0].kwargs["additional_model_request_fields"][ + "output_config" + ] == {"effort": "high"} + assert make_model.call_args_list[1].args == ("bedrock_converse:us.anthropic.claude-opus-4-8",) diff --git a/tests/test_anthropic_model.py b/tests/test_anthropic_model.py new file mode 100644 index 00000000..aec1777e --- /dev/null +++ b/tests/test_anthropic_model.py @@ -0,0 +1,55 @@ +import pytest + +from agent.dashboard.options import SUPPORTED_MODELS, provider_fallback_pair +from agent.utils.model import provider_model_kwargs + +SONNET_5_ID = "bedrock_converse:us.anthropic.claude-sonnet-5" +FABLE_5_ID = "bedrock_converse:us.anthropic.claude-fable-5" + + +def test_sonnet_5_is_supported_with_documented_efforts() -> None: + sonnet = next(m for m in SUPPORTED_MODELS if m["id"] == SONNET_5_ID) + assert sonnet["label"] == "Sonnet 5 (Bedrock)" + assert sonnet["efforts"] == ["low", "medium", "high", "xhigh", "max"] + assert sonnet["default_effort"] == "high" + assert sonnet["supports_images"] is True + + +@pytest.mark.parametrize("effort", ["low", "medium", "high", "xhigh", "max"]) +def test_sonnet_5_efforts_map_to_bedrock_kwargs(effort: str) -> None: + kwargs = provider_model_kwargs(SONNET_5_ID, effort, max_tokens=16_000) + assert kwargs["max_tokens"] == 16_000 + fields = kwargs["additional_model_request_fields"] + assert fields["output_config"] == {"effort": effort} + assert fields["thinking"] == {"type": "adaptive", "display": "summarized"} + + +def test_sonnet_46_fallback_uses_sonnet_5() -> None: + assert provider_fallback_pair("bedrock_converse:us.anthropic.claude-sonnet-4-6", "xhigh") == ( + SONNET_5_ID, + "xhigh", + ) + + +def test_opus_fallback_stays_on_opus_family() -> None: + assert provider_fallback_pair("bedrock_converse:us.anthropic.claude-opus-4-7", "xhigh") == ( + "bedrock_converse:us.anthropic.claude-opus-4-8", + "xhigh", + ) + + +def test_fable_5_is_supported_with_documented_efforts() -> None: + fable = next(m for m in SUPPORTED_MODELS if m["id"] == FABLE_5_ID) + assert fable["label"] == "Fable 5" + assert fable["efforts"] == ["low", "medium", "high", "xhigh", "max"] + assert fable["default_effort"] == "high" + assert fable["supports_images"] is True + + +@pytest.mark.parametrize("effort", ["low", "medium", "high", "xhigh", "max"]) +def test_fable_5_efforts_map_to_bedrock_kwargs(effort: str) -> None: + kwargs = provider_model_kwargs(FABLE_5_ID, effort, max_tokens=16_000) + assert kwargs["max_tokens"] == 16_000 + fields = kwargs["additional_model_request_fields"] + assert fields["output_config"] == {"effort": effort} + assert fields["thinking"] == {"type": "adaptive", "display": "summarized"} diff --git a/tests/test_dashboard_thread_api.py b/tests/test_dashboard_thread_api.py index 856856f8..067391e7 100644 --- a/tests/test_dashboard_thread_api.py +++ b/tests/test_dashboard_thread_api.py @@ -1,16 +1,19 @@ import base64 import json from types import SimpleNamespace +from unittest.mock import AsyncMock, patch import pytest from fastapi import HTTPException -from agent.dashboard import thread_api +from agent.dashboard import routes, thread_api from agent.dashboard.agent_overrides import resolve_agent_model_id from agent.dashboard.options import model_supports_images _TEXT_ONLY_MODEL = "fireworks:accounts/fireworks/models/deepseek-v4-pro" _VISION_MODEL = "bedrock_converse:us.anthropic.claude-opus-4-8" +_FABLE = "bedrock_converse:us.anthropic.claude-fable-5" +_PAIR = ("bedrock_converse:us.anthropic.claude-opus-4-8", "medium") def _image() -> thread_api.DashboardImageBody: @@ -1427,3 +1430,81 @@ async def test_status_filter_refreshes_threads_missing_run_status(monkeypatch) - assert {item["id"] for item in result["items"]} == {"t0"} assert result["items"][0]["status"] == "finished" assert set(run_list_thread_ids) == {"t0", "t1"} + + +@pytest.mark.asyncio +async def test_options_omits_fable_when_disabled() -> None: + with ( + patch( + "agent.dashboard.routes.get_team_fable_enabled", + new_callable=AsyncMock, + return_value=False, + ), + patch( + "agent.dashboard.routes.get_team_default_model", + new_callable=AsyncMock, + return_value=_PAIR, + ), + patch( + "agent.dashboard.routes.get_team_default_subagent_model", + new_callable=AsyncMock, + return_value=_PAIR, + ), + ): + payload = await routes.options() + assert _FABLE not in [m["id"] for m in payload["models"]] + + +@pytest.mark.asyncio +async def test_options_includes_fable_when_enabled() -> None: + with ( + patch( + "agent.dashboard.routes.get_team_fable_enabled", + new_callable=AsyncMock, + return_value=True, + ), + patch( + "agent.dashboard.routes.get_team_default_model", + new_callable=AsyncMock, + return_value=_PAIR, + ), + patch( + "agent.dashboard.routes.get_team_default_subagent_model", + new_callable=AsyncMock, + return_value=_PAIR, + ), + ): + payload = await routes.options() + assert _FABLE in [m["id"] for m in payload["models"]] + + +@pytest.mark.asyncio +async def test_options_gates_stale_fable_default_when_disabled() -> None: + # A stale Fable team default must not be advertised as the default while Fable + # is omitted from the selectable list, or the Cloud Agents page would offer a + # default that PUT /profile then rejects. + fable_pair = (_FABLE, "high") + with ( + patch( + "agent.dashboard.routes.get_team_fable_enabled", + new_callable=AsyncMock, + return_value=False, + ), + patch( + "agent.dashboard.routes.get_team_default_model", + new_callable=AsyncMock, + return_value=fable_pair, + ), + patch( + "agent.dashboard.routes.get_team_default_subagent_model", + new_callable=AsyncMock, + return_value=fable_pair, + ), + ): + payload = await routes.options() + model_ids = [m["id"] for m in payload["models"]] + assert _FABLE not in model_ids + assert payload["default_agent_model"] != _FABLE + assert payload["default_agent_subagent_model"] != _FABLE + assert payload["default_agent_model"] in model_ids + assert payload["default_agent_subagent_model"] in model_ids diff --git a/tests/test_model_fallback_resolution.py b/tests/test_model_fallback_resolution.py index dc11f9df..8f0305f3 100644 --- a/tests/test_model_fallback_resolution.py +++ b/tests/test_model_fallback_resolution.py @@ -5,7 +5,10 @@ import pytest from agent.dashboard.agent_overrides import normalize_profile_overrides from agent.dashboard.options import ( DEFAULT_MODEL_ID, + FABLE_MODEL_IDS, default_model_pair, + fable_disabled_fallback, + gate_fable_model, provider_fallback_pair, ) from agent.dashboard.team_settings import get_team_default_model @@ -88,3 +91,37 @@ def test_profile_unknown_provider_defers_to_team_default() -> None: def test_global_default_is_supported() -> None: model, _ = default_model_pair() assert model == DEFAULT_MODEL_ID + + +def test_gate_fable_passthrough_when_enabled() -> None: + assert gate_fable_model( + "bedrock_converse:us.anthropic.claude-fable-5", "high", fable_enabled=True + ) == ( + "bedrock_converse:us.anthropic.claude-fable-5", + "high", + ) + + +def test_gate_fable_swaps_to_opus_when_disabled() -> None: + assert gate_fable_model( + "bedrock_converse:us.anthropic.claude-fable-5", "high", fable_enabled=False + ) == ( + "bedrock_converse:us.anthropic.claude-opus-4-8", + "high", + ) + + +def test_gate_fable_leaves_non_fable_ids_alone() -> None: + assert gate_fable_model( + "fireworks:accounts/fireworks/models/deepseek-v4-pro", "high", fable_enabled=False + ) == ( + "fireworks:accounts/fireworks/models/deepseek-v4-pro", + "high", + ) + + +def test_fable_disabled_fallback_is_non_fable_claude() -> None: + model, effort = fable_disabled_fallback("high") + assert model == "bedrock_converse:us.anthropic.claude-opus-4-8" + assert model not in FABLE_MODEL_IDS + assert effort == "high" diff --git a/tests/test_team_settings_fable.py b/tests/test_team_settings_fable.py new file mode 100644 index 00000000..6e4ccefa --- /dev/null +++ b/tests/test_team_settings_fable.py @@ -0,0 +1,122 @@ +from __future__ import annotations + +from unittest.mock import AsyncMock, patch + +import pytest + +from agent.dashboard.options import FABLE_MODEL_IDS +from agent.dashboard.team_settings import TeamSettingsUpdate, get_team_fable_enabled + +_FABLE = "bedrock_converse:us.anthropic.claude-fable-5" + +# --- accessor: get_team_fable_enabled (async, patched store) --- + + +@pytest.mark.asyncio +async def test_fable_enabled_defaults_false_when_absent() -> None: + # Legacy record with no fable_enabled key -> off. + with patch( + "agent.dashboard.team_settings.get_team_settings", + new_callable=AsyncMock, + return_value={}, + ): + assert await get_team_fable_enabled() is False + + +@pytest.mark.asyncio +async def test_fable_enabled_true_when_set() -> None: + with patch( + "agent.dashboard.team_settings.get_team_settings", + new_callable=AsyncMock, + return_value={"fable_enabled": True}, + ): + assert await get_team_fable_enabled() is True + + +@pytest.mark.asyncio +async def test_fable_enabled_false_for_non_bool_value() -> None: + # Fail-closed: any non-bool (e.g. a stray string) resolves to False. + with patch( + "agent.dashboard.team_settings.get_team_settings", + new_callable=AsyncMock, + return_value={"fable_enabled": "true"}, + ): + assert await get_team_fable_enabled() is False + + +# --- validation: TeamSettingsUpdate (sync) --- + + +def test_update_defaults_fable_disabled() -> None: + assert TeamSettingsUpdate().fable_enabled is False + + +def test_update_coerces_fable_model_when_disabled() -> None: + # Disabling Fable must never fail: a lingering Fable default is swapped for a + # safe non-Fable fallback (effort preserved) rather than rejected. + update = TeamSettingsUpdate( + default_agent_model=_FABLE, + default_agent_reasoning_effort="high", + ) + assert update.default_agent_model not in FABLE_MODEL_IDS + assert update.default_agent_reasoning_effort == "high" + + +def test_update_coerces_fable_in_any_role_when_disabled() -> None: + # Same coercion applies to every model field, e.g. review chat. + update = TeamSettingsUpdate( + default_chat_model=_FABLE, + default_chat_reasoning_effort="high", + ) + assert update.default_chat_model not in FABLE_MODEL_IDS + assert update.default_chat_reasoning_effort == "high" + + +def test_disable_transition_coerces_all_fable_defaults() -> None: + # The kill-switch flow: an admin who had picked Fable everywhere flips the + # toggle off, and the UI re-sends the whole settings blob with the Fable + # defaults still attached. The update must succeed and strip every Fable id. + update = TeamSettingsUpdate( + fable_enabled=False, + default_agent_model=_FABLE, + default_agent_reasoning_effort="high", + default_agent_subagent_model=_FABLE, + default_agent_subagent_reasoning_effort="high", + default_reviewer_model=_FABLE, + default_reviewer_reasoning_effort="high", + default_reviewer_subagent_model=_FABLE, + default_reviewer_subagent_reasoning_effort="high", + default_grouping_model=_FABLE, + default_grouping_reasoning_effort="high", + default_chat_model=_FABLE, + default_chat_reasoning_effort="high", + ) + for field in ( + "default_agent_model", + "default_agent_subagent_model", + "default_reviewer_model", + "default_reviewer_subagent_model", + "default_grouping_model", + "default_chat_model", + ): + assert getattr(update, field) not in FABLE_MODEL_IDS, field + + +def test_update_leaves_non_fable_defaults_untouched_when_disabled() -> None: + # Coercion only rewrites Fable ids; other selections pass through unchanged. + update = TeamSettingsUpdate( + default_agent_model="bedrock_converse:us.anthropic.claude-sonnet-5", + default_agent_reasoning_effort="medium", + ) + assert update.default_agent_model == "bedrock_converse:us.anthropic.claude-sonnet-5" + assert update.default_agent_reasoning_effort == "medium" + + +def test_update_accepts_fable_model_when_enabled() -> None: + update = TeamSettingsUpdate( + fable_enabled=True, + default_agent_model=_FABLE, + default_agent_reasoning_effort="high", + ) + assert update.fable_enabled is True + assert update.default_agent_model == _FABLE diff --git a/ui/src/lib/api.ts b/ui/src/lib/api.ts index d3321176..fe0652db 100644 --- a/ui/src/lib/api.ts +++ b/ui/src/lib/api.ts @@ -166,6 +166,7 @@ export interface TeamSettings { review_trace_links: boolean /** Tri-state LLM Gateway toggle; null inherits the LANGSMITH_GATEWAY_ENABLED default. */ gateway_enabled?: boolean | null + fable_enabled?: boolean review_tracing_project?: string | null org_guidelines?: string | null default_agent_model?: string | null diff --git a/ui/src/routes/admin.tsx b/ui/src/routes/admin.tsx index 7ec9b2b4..15b4ece6 100644 --- a/ui/src/routes/admin.tsx +++ b/ui/src/routes/admin.tsx @@ -22,6 +22,7 @@ import { SelectValue, } from "@/components/ui/select" import { Skeleton } from "@/components/ui/skeleton" +import { Switch } from "@/components/ui/switch" import { api } from "@/lib/api" import { RequireLogin } from "@/lib/auth-redirect" import { useSession } from "@/lib/session" @@ -57,6 +58,8 @@ function AdminPage() { + + @@ -694,6 +697,44 @@ function PRTraceResolutionSection() { ) } +function FableSection() { + const qc = useQueryClient() + const settings = useQuery({ queryKey: ["teamSettings"], queryFn: api.getTeamSettings }) + const [error, setError] = useState(null) + const save = useMutation({ + mutationFn: (body: TeamSettings) => api.saveTeamSettings(body), + onSuccess: (saved) => { + qc.setQueryData(["teamSettings"], saved) + qc.invalidateQueries({ queryKey: ["options"] }) // refresh pickers so Fable appears/disappears + setError(null) + }, + onError: (e: Error) => setError(e.message), + }) + return ( + +
+ + settings.data && save.mutate({ ...settings.data, fable_enabled: next }) + } + disabled={!settings.data || save.isPending} + /> + } + /> +
+ {error &&

{error}

} +
+ ) +} + function GlobalDefaultsSection({ models }: { models: Array }) { const qc = useQueryClient() const settings = useQuery({