open-swe/tests/test_team_settings_org_guidelines.py

145 lines
4.7 KiB
Python
Raw Normal View History

from __future__ import annotations
from unittest.mock import AsyncMock, patch
import pytest
from pydantic import ValidationError
from agent.dashboard.team_settings import (
ORG_GUIDELINES_MAX_CHARS,
feat: add PR trace resolution (#1612) * feat: add PR trace resolution Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> * fix: inject reviewer trace context as JSON Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> * fix: address review on PR trace resolution Use the documented LangSmith metadata filter syntax (and(eq(metadata_key,...), eq(metadata_value,...))) instead of has(metadata, '{...}'), which does not match runs — _list_thread_runs was silently returning nothing. Bound full-text searches to a 90-day window so they don't hit LangSmith's large-window rate limit. Also folds in the best-effort branch->head-sha resolver (dropping the weighted scoring/threshold + repo/file evidence + GitHub hydration), sandbox JSON injection, and the admin "Resolve trace" dry-run endpoint. The IDOR findings are moot: resolve_pr_to_threads/summarize_agent_session were removed; resolution now runs deterministically from the trusted run config with no model-controlled pr_url or thread_id. * fix: scope branch trace search to the repo Branch names like fix-tests aren't unique across repos (or older PRs) in a shared tracing project, so an unscoped branch hit could resolve to an unrelated thread and write its runs into the reviewer sandbox. Require the repo slug to co-occur with the branch in matched runs; the full head SHA stays unscoped since it is globally unique. Addresses open-swe review on PR #1612. --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
2026-06-25 14:11:21 -07:00
REVIEW_TRACING_PROJECT_MAX_CHARS,
TeamSettingsUpdate,
get_org_review_guidelines,
feat: chat with your PR on the review page (#1534) * feat: chat with your PR on the review page Add a sandbox-less `chat` graph that answers questions about a single PR from its diff, the published review findings, and read-only GitHub access. - agent/chat.py: deepagents graph, no sandbox (default StateBackend, file mutation + execute tools excluded). PR context is seeded as virtual files under /pr/; a repo-scoped App token is resolved in-graph. - tools: read_repo_file, search_repo_code, list_review_findings. - dashboard/review_chat_api.py + routes: per-user chat thread, LangGraph stream/commands/state/history proxy pinned to the chat assistant, seeds diff/findings/overview on first run. Gated by repo access. - UI: Chat tab wired to a chat-scoped StreamProvider (replaces Coming Soon). * feat: admin setting for review-chat default model Add a 'Open SWE Review Chat' default to team settings (default_chat_model / default_chat_reasoning_effort). get_team_default_model("chat") inherits the Agent default when unset; the chat graph resolves through it. Admin RolePicker gains an 'Agent default' inherit option that clears the override. * feat: multi-conversation review chat (tabs, new chat, history) Replace the single per-PR chat thread with multiple per-user conversations: - threads minted client-side; first message persists with a title derived from the prompt. - list + delete endpoints; chat panel gets a tab strip (history), new-chat (+), close (x), refresh, an intro greeting, and suggested prompts. - get_review_chat now returns availability only (ids are client-minted). * ui fixes * ui: review-chat history dropdown, full-width AI replies, resizable side panel * fix(review-chat): enforce per-user thread ownership on proxy endpoints; reseed PR context on head change --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
2026-06-15 17:17:30 -07:00
get_team_default_model,
feat: add PR trace resolution (#1612) * feat: add PR trace resolution Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> * fix: inject reviewer trace context as JSON Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> * fix: address review on PR trace resolution Use the documented LangSmith metadata filter syntax (and(eq(metadata_key,...), eq(metadata_value,...))) instead of has(metadata, '{...}'), which does not match runs — _list_thread_runs was silently returning nothing. Bound full-text searches to a 90-day window so they don't hit LangSmith's large-window rate limit. Also folds in the best-effort branch->head-sha resolver (dropping the weighted scoring/threshold + repo/file evidence + GitHub hydration), sandbox JSON injection, and the admin "Resolve trace" dry-run endpoint. The IDOR findings are moot: resolve_pr_to_threads/summarize_agent_session were removed; resolution now runs deterministically from the trusted run config with no model-controlled pr_url or thread_id. * fix: scope branch trace search to the repo Branch names like fix-tests aren't unique across repos (or older PRs) in a shared tracing project, so an unscoped branch hit could resolve to an unrelated thread and write its runs into the reviewer sandbox. Require the repo slug to co-occur with the branch in matched runs; the full head SHA stays unscoped since it is globally unique. Addresses open-swe review on PR #1612. --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
2026-06-25 14:11:21 -07:00
get_team_review_tracing_project,
)
feat: chat with your PR on the review page (#1534) * feat: chat with your PR on the review page Add a sandbox-less `chat` graph that answers questions about a single PR from its diff, the published review findings, and read-only GitHub access. - agent/chat.py: deepagents graph, no sandbox (default StateBackend, file mutation + execute tools excluded). PR context is seeded as virtual files under /pr/; a repo-scoped App token is resolved in-graph. - tools: read_repo_file, search_repo_code, list_review_findings. - dashboard/review_chat_api.py + routes: per-user chat thread, LangGraph stream/commands/state/history proxy pinned to the chat assistant, seeds diff/findings/overview on first run. Gated by repo access. - UI: Chat tab wired to a chat-scoped StreamProvider (replaces Coming Soon). * feat: admin setting for review-chat default model Add a 'Open SWE Review Chat' default to team settings (default_chat_model / default_chat_reasoning_effort). get_team_default_model("chat") inherits the Agent default when unset; the chat graph resolves through it. Admin RolePicker gains an 'Agent default' inherit option that clears the override. * feat: multi-conversation review chat (tabs, new chat, history) Replace the single per-PR chat thread with multiple per-user conversations: - threads minted client-side; first message persists with a title derived from the prompt. - list + delete endpoints; chat panel gets a tab strip (history), new-chat (+), close (x), refresh, an intro greeting, and suggested prompts. - get_review_chat now returns availability only (ids are client-minted). * ui fixes * ui: review-chat history dropdown, full-width AI replies, resizable side panel * fix(review-chat): enforce per-user thread ownership on proxy endpoints; reseed PR context on head change --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
2026-06-15 17:17:30 -07:00
_AGENT_PAIR = ("anthropic:claude-opus-4-8", "high")
_CHAT_PAIR = ("google_genai:gemini-3.5-flash", "low")
def test_org_guidelines_blank_normalizes_to_none() -> None:
assert TeamSettingsUpdate(org_guidelines=" ").org_guidelines is None
assert TeamSettingsUpdate(org_guidelines=None).org_guidelines is None
def test_org_guidelines_trimmed() -> None:
update = TeamSettingsUpdate(org_guidelines=" Flag CI gate removals.\n")
assert update.org_guidelines == "Flag CI gate removals."
def test_org_guidelines_rejects_oversized() -> None:
with pytest.raises(ValidationError):
TeamSettingsUpdate(org_guidelines="x" * (ORG_GUIDELINES_MAX_CHARS + 1))
feat: add PR trace resolution (#1612) * feat: add PR trace resolution Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> * fix: inject reviewer trace context as JSON Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com> * fix: address review on PR trace resolution Use the documented LangSmith metadata filter syntax (and(eq(metadata_key,...), eq(metadata_value,...))) instead of has(metadata, '{...}'), which does not match runs — _list_thread_runs was silently returning nothing. Bound full-text searches to a 90-day window so they don't hit LangSmith's large-window rate limit. Also folds in the best-effort branch->head-sha resolver (dropping the weighted scoring/threshold + repo/file evidence + GitHub hydration), sandbox JSON injection, and the admin "Resolve trace" dry-run endpoint. The IDOR findings are moot: resolve_pr_to_threads/summarize_agent_session were removed; resolution now runs deterministically from the trusted run config with no model-controlled pr_url or thread_id. * fix: scope branch trace search to the repo Branch names like fix-tests aren't unique across repos (or older PRs) in a shared tracing project, so an unscoped branch hit could resolve to an unrelated thread and write its runs into the reviewer sandbox. Require the repo slug to co-occur with the branch in matched runs; the full head SHA stays unscoped since it is globally unique. Addresses open-swe review on PR #1612. --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
2026-06-25 14:11:21 -07:00
def test_review_tracing_project_blank_normalizes_to_none() -> None:
assert TeamSettingsUpdate(review_tracing_project=" ").review_tracing_project is None
assert TeamSettingsUpdate(review_tracing_project=None).review_tracing_project is None
def test_review_tracing_project_trimmed() -> None:
update = TeamSettingsUpdate(review_tracing_project=" pajuha\n")
assert update.review_tracing_project == "pajuha"
def test_review_tracing_project_rejects_oversized() -> None:
with pytest.raises(ValidationError):
TeamSettingsUpdate(review_tracing_project="x" * (REVIEW_TRACING_PROJECT_MAX_CHARS + 1))
@pytest.mark.asyncio
async def test_get_team_review_tracing_project_returns_trimmed_text() -> None:
with patch(
"agent.dashboard.team_settings.get_team_settings",
new_callable=AsyncMock,
return_value={"review_tracing_project": " pajuha\n"},
):
assert await get_team_review_tracing_project() == "pajuha"
@pytest.mark.asyncio
async def test_get_org_review_guidelines_returns_trimmed_text() -> None:
with patch(
"agent.dashboard.team_settings.get_team_settings",
new_callable=AsyncMock,
return_value={"org_guidelines": " Always check auth.\n"},
):
assert await get_org_review_guidelines() == "Always check auth."
@pytest.mark.asyncio
async def test_get_org_review_guidelines_returns_none_when_unset() -> None:
with patch(
"agent.dashboard.team_settings.get_team_settings",
new_callable=AsyncMock,
return_value={"org_guidelines": None},
):
assert await get_org_review_guidelines() is None
feat: chat with your PR on the review page (#1534) * feat: chat with your PR on the review page Add a sandbox-less `chat` graph that answers questions about a single PR from its diff, the published review findings, and read-only GitHub access. - agent/chat.py: deepagents graph, no sandbox (default StateBackend, file mutation + execute tools excluded). PR context is seeded as virtual files under /pr/; a repo-scoped App token is resolved in-graph. - tools: read_repo_file, search_repo_code, list_review_findings. - dashboard/review_chat_api.py + routes: per-user chat thread, LangGraph stream/commands/state/history proxy pinned to the chat assistant, seeds diff/findings/overview on first run. Gated by repo access. - UI: Chat tab wired to a chat-scoped StreamProvider (replaces Coming Soon). * feat: admin setting for review-chat default model Add a 'Open SWE Review Chat' default to team settings (default_chat_model / default_chat_reasoning_effort). get_team_default_model("chat") inherits the Agent default when unset; the chat graph resolves through it. Admin RolePicker gains an 'Agent default' inherit option that clears the override. * feat: multi-conversation review chat (tabs, new chat, history) Replace the single per-PR chat thread with multiple per-user conversations: - threads minted client-side; first message persists with a title derived from the prompt. - list + delete endpoints; chat panel gets a tab strip (history), new-chat (+), close (x), refresh, an intro greeting, and suggested prompts. - get_review_chat now returns availability only (ids are client-minted). * ui fixes * ui: review-chat history dropdown, full-width AI replies, resizable side panel * fix(review-chat): enforce per-user thread ownership on proxy endpoints; reseed PR context on head change --------- Co-authored-by: open-swe[bot] <open-swe@users.noreply.github.com>
2026-06-15 17:17:30 -07:00
def _settings(**overrides: object) -> dict[str, object]:
base = {
"default_agent_model": _AGENT_PAIR[0],
"default_agent_reasoning_effort": _AGENT_PAIR[1],
"default_chat_model": None,
"default_chat_reasoning_effort": None,
}
base.update(overrides)
return base
@pytest.mark.asyncio
async def test_chat_default_inherits_agent_when_unset() -> None:
with patch(
"agent.dashboard.team_settings.get_team_settings",
new_callable=AsyncMock,
return_value=_settings(),
):
assert await get_team_default_model("chat") == _AGENT_PAIR
@pytest.mark.asyncio
async def test_chat_default_uses_chat_model_when_set() -> None:
with patch(
"agent.dashboard.team_settings.get_team_settings",
new_callable=AsyncMock,
return_value=_settings(
default_chat_model=_CHAT_PAIR[0],
default_chat_reasoning_effort=_CHAT_PAIR[1],
),
):
assert await get_team_default_model("chat") == _CHAT_PAIR
@pytest.mark.asyncio
async def test_chat_default_inherits_agent_when_chat_model_invalid() -> None:
with patch(
"agent.dashboard.team_settings.get_team_settings",
new_callable=AsyncMock,
return_value=_settings(
default_chat_model="bogus:model",
default_chat_reasoning_effort="high",
),
):
assert await get_team_default_model("chat") == _AGENT_PAIR
def test_team_settings_update_accepts_chat_pair() -> None:
update = TeamSettingsUpdate(
default_chat_model=_CHAT_PAIR[0],
default_chat_reasoning_effort=_CHAT_PAIR[1],
)
assert update.default_chat_model == _CHAT_PAIR[0]
assert update.default_chat_reasoning_effort == _CHAT_PAIR[1]
def test_team_settings_update_rejects_chat_effort_without_model() -> None:
with pytest.raises(ValidationError):
TeamSettingsUpdate(default_chat_reasoning_effort="high")
def test_team_settings_update_rejects_unsupported_chat_effort() -> None:
with pytest.raises(ValidationError):
TeamSettingsUpdate(default_chat_model=_CHAT_PAIR[0], default_chat_reasoning_effort="max")