open-swe/tests/test_team_settings_org_guidelines.py
Adam Moussa b65c3a07db
feat: distill Sea Haven conventions into agent prompt, reviewer, and fork docs (#113)
* Add Dependabot ignore for @types/node semver-major bumps

Prevent Dependabot from proposing wrong-direction @types/node major
bumps (e.g. 24 -> 26). /ui runs on Node 24 on Vercel; a too-new types
major still compiles but describes APIs absent at runtime.

Refs: #110

* feat(agent): seed all-repos custom instructions in default_prompt.md

Distill the universally-applicable Sea Haven authoring conventions into the
team-default Custom Instructions the main agent gets on every repo: secrets/
config placement, keep-docs-in-sync, verify-before-push, re-run-real-gates
after delegating, confirm-a-convention-before-adopting, and house writing
style. Toolchain references are generalized (not tied to a specific stack).

* feat(reviewer): seed Sea Haven review baseline as org-guidelines default

Bake DEFAULT_ORG_REVIEW_GUIDELINES (severity model, secrets, security surface,
tests, naming, deferred-work-needs-an-issue) and default org_guidelines to it
in _default_settings(). The reviewer now applies the Sea Haven baseline on
every repo until a workspace admin overrides it with a non-empty value via
the dashboard. Stack-agnostic and well under the 10k-char cap.

* refactor(prompt): consolidate duplicated COMMIT_PR_SECTION + add fork-sync runbook

COMMIT_PR_SECTION had two overlapping passes with a contradictory PR-title
rule (a fixed 'type: description' form vs the repo-aware detection). Collapse
into one numbered sequence (lint -> commit -> push/PR -> notify), keep the
authoritative repo-aware title rule, and drop the duplicate notify step. All
IMPORTANT directives (force-push ban, workflow-approval, autonomy, 403
handling) are preserved verbatim.

Add a fork-maintenance runbook to CLAUDE.md distilling the durable
upstream-sync methodology (conflict triage, deferred-refactor resolution rule,
the silent re-import/wiring hazards, test-impl-same-side, layered CI).

* refactor(prompt): adopt conventional-commit style

Flip the Sea Haven authoring convention baked into the agent prompt from
imperative/no-prefix to conventional-commit style:
- Commit subjects and the no-gate PR-title default now use
  type(scope): description with the allowed type set (feat, fix, docs,
  style, refactor, perf, test, build, ci, chore, revert, release).
- Branch prefixes expanded to feature/, fix/, hotfix/, chore/, docs/,
  refactor/, release/ (kebab-case description).
- The repo-aware gate detection is preserved: a repo's own title gate
  still wins and may narrow the allowed types/scopes.

Updated test_github_comment_prompts.py to assert the new convention.
2026-07-02 18:29:15 -04:00

154 lines
5.3 KiB
Python

from __future__ import annotations
from unittest.mock import AsyncMock, patch
import pytest
from pydantic import ValidationError
from agent.dashboard.team_settings import (
DEFAULT_ORG_REVIEW_GUIDELINES,
ORG_GUIDELINES_MAX_CHARS,
REVIEW_TRACING_PROJECT_MAX_CHARS,
TeamSettingsUpdate,
_default_settings,
get_org_review_guidelines,
get_team_default_model,
get_team_review_tracing_project,
)
_AGENT_PAIR = ("bedrock_converse:us.anthropic.claude-opus-4-8", "high")
_CHAT_PAIR = ("fireworks:accounts/fireworks/models/kimi-k2p7-code", "low")
def test_org_guidelines_blank_normalizes_to_none() -> None:
assert TeamSettingsUpdate(org_guidelines=" ").org_guidelines is None
assert TeamSettingsUpdate(org_guidelines=None).org_guidelines is None
def test_org_guidelines_trimmed() -> None:
update = TeamSettingsUpdate(org_guidelines=" Flag CI gate removals.\n")
assert update.org_guidelines == "Flag CI gate removals."
def test_org_guidelines_rejects_oversized() -> None:
with pytest.raises(ValidationError):
TeamSettingsUpdate(org_guidelines="x" * (ORG_GUIDELINES_MAX_CHARS + 1))
def test_default_settings_seed_sea_haven_org_guidelines() -> None:
# Unset org guidelines default to the baked Sea Haven review baseline so the
# reviewer applies it on every repo until an admin overrides it.
assert _default_settings()["org_guidelines"] == DEFAULT_ORG_REVIEW_GUIDELINES
assert "Sea Haven review baseline" in DEFAULT_ORG_REVIEW_GUIDELINES
assert len(DEFAULT_ORG_REVIEW_GUIDELINES) <= ORG_GUIDELINES_MAX_CHARS
def test_review_tracing_project_blank_normalizes_to_none() -> None:
assert TeamSettingsUpdate(review_tracing_project=" ").review_tracing_project is None
assert TeamSettingsUpdate(review_tracing_project=None).review_tracing_project is None
def test_review_tracing_project_trimmed() -> None:
update = TeamSettingsUpdate(review_tracing_project=" pajuha\n")
assert update.review_tracing_project == "pajuha"
def test_review_tracing_project_rejects_oversized() -> None:
with pytest.raises(ValidationError):
TeamSettingsUpdate(review_tracing_project="x" * (REVIEW_TRACING_PROJECT_MAX_CHARS + 1))
@pytest.mark.asyncio
async def test_get_team_review_tracing_project_returns_trimmed_text() -> None:
with patch(
"agent.dashboard.team_settings.get_team_settings",
new_callable=AsyncMock,
return_value={"review_tracing_project": " pajuha\n"},
):
assert await get_team_review_tracing_project() == "pajuha"
@pytest.mark.asyncio
async def test_get_org_review_guidelines_returns_trimmed_text() -> None:
with patch(
"agent.dashboard.team_settings.get_team_settings",
new_callable=AsyncMock,
return_value={"org_guidelines": " Always check auth.\n"},
):
assert await get_org_review_guidelines() == "Always check auth."
@pytest.mark.asyncio
async def test_get_org_review_guidelines_returns_none_when_unset() -> None:
with patch(
"agent.dashboard.team_settings.get_team_settings",
new_callable=AsyncMock,
return_value={"org_guidelines": None},
):
assert await get_org_review_guidelines() is None
def _settings(**overrides: object) -> dict[str, object]:
base = {
"default_agent_model": _AGENT_PAIR[0],
"default_agent_reasoning_effort": _AGENT_PAIR[1],
"default_chat_model": None,
"default_chat_reasoning_effort": None,
}
base.update(overrides)
return base
@pytest.mark.asyncio
async def test_chat_default_inherits_agent_when_unset() -> None:
with patch(
"agent.dashboard.team_settings.get_team_settings",
new_callable=AsyncMock,
return_value=_settings(),
):
assert await get_team_default_model("chat") == _AGENT_PAIR
@pytest.mark.asyncio
async def test_chat_default_uses_chat_model_when_set() -> None:
with patch(
"agent.dashboard.team_settings.get_team_settings",
new_callable=AsyncMock,
return_value=_settings(
default_chat_model=_CHAT_PAIR[0],
default_chat_reasoning_effort=_CHAT_PAIR[1],
),
):
assert await get_team_default_model("chat") == _CHAT_PAIR
@pytest.mark.asyncio
async def test_chat_default_inherits_agent_when_chat_model_invalid() -> None:
with patch(
"agent.dashboard.team_settings.get_team_settings",
new_callable=AsyncMock,
return_value=_settings(
default_chat_model="bogus:model",
default_chat_reasoning_effort="high",
),
):
assert await get_team_default_model("chat") == _AGENT_PAIR
def test_team_settings_update_accepts_chat_pair() -> None:
update = TeamSettingsUpdate(
default_chat_model=_CHAT_PAIR[0],
default_chat_reasoning_effort=_CHAT_PAIR[1],
)
assert update.default_chat_model == _CHAT_PAIR[0]
assert update.default_chat_reasoning_effort == _CHAT_PAIR[1]
def test_team_settings_update_rejects_chat_effort_without_model() -> None:
with pytest.raises(ValidationError):
TeamSettingsUpdate(default_chat_reasoning_effort="high")
def test_team_settings_update_rejects_unsupported_chat_effort() -> None:
with pytest.raises(ValidationError):
TeamSettingsUpdate(default_chat_model=_CHAT_PAIR[0], default_chat_reasoning_effort="max")