chore(reviewer): add epic verification coverage

This commit is contained in:
Adam Moussa 2026-07-31 14:30:12 -04:00 • committed by Adam Moussa
parent 5f6d756e0d
commit 99b376bbe2
7 changed files with 440 additions and 23 deletions

View file

@ -1,25 +1,5 @@
{ {
"suppressions": [ "suppressions": [
{
"id": "AUTHZ-CLEAN-AUTOAPPROVE-INJECTION-001",
"title": "Clean-review auto-approve derives APPROVE authority from a model-controlled signal on the untrusted auto-review path (prompt-injection-mintable APPROVE)",
"file": "agent/tools/publish_review.py",
"severity": "high",
"status": "confirmed",
"suppression_justification": "ACCEPTED (Adam, 2026-07-21) as a knowingly-deferred risk merged into the dev integration branch via admin merge in PR #217. /sh-security-review returned BLOCK: moving `approve` authorization from the deterministic verdict_requested flag to open_findings_count==0 lets a prompt-injected auto-review (which never sets verdict_requested) land a real bot APPROVE on an attacker-controlled external/fork PR, potentially satisfying branch protection. This reintroduces the hole PR #214 closed. Merged to dev only (NOT main/prod). Compensating controls: (a) confirm dev does not auto-deploy to sh-openswe; (b) on sensitive repos, esp. payments-dashboard, configure rulesets so the seahaven-openswe[bot] APPROVE does not by itself satisfy required approvals. Tracked in issue #218 with the full redesign checklist. REVISIT TRIGGER: MUST be resolved before this change promotes from dev to main/prod, and immediately if dev is found to auto-deploy. Verified HIGH by the sh-security-review detector fan-out (6 independent detectors converged).",
"owner": "adam@seahavenind.com",
"added": "2026-07-21"
},
{
"id": "AUTHZ-CLEAN-AUTOAPPROVE-THREADLAUNDER-002",
"title": "Clean-review auto-approve gate reads reconcile-derived finding status, so a PR author can launder a dirty PR to clean via GitHub thread resolve/outdate",
"file": "agent/tools/publish_review.py",
"severity": "high",
"status": "confirmed",
"suppression_justification": "ACCEPTED (Adam, 2026-07-21), deferred with AUTHZ-CLEAN-AUTOAPPROVE-INJECTION-001 in PR #217 (admin-merged to dev). The open_findings_count gate reads finding status after reconcile_findings_with_review_threads, which flips open->resolved when the GitHub thread is is_resolved/is_outdated — both author-controllable ('Resolve conversation' or a trivial hunk-outdating commit) without fixing the defect, yielding an auto-approve on an unfixed PR. Same compensating controls and revisit trigger as ...-001. Tracked in issue #218 (redesign: compute the gate from the reviewer's own authoritative finding state, not author-influenced thread state). Verified HIGH by /sh-security-review.",
"owner": "adam@seahavenind.com",
"added": "2026-07-21"
},
{ {
"id": "AUTHZ-SLACK-BOT-DEFAULT-001", "id": "AUTHZ-SLACK-BOT-DEFAULT-001",
"title": "Slack entrypoint lacks a per-user repo-access check; default-bot PR authoring removes the implicit per-user repo boundary", "title": "Slack entrypoint lacks a per-user repo-access check; default-bot PR authoring removes the implicit per-user repo boundary",

View file

@ -82,7 +82,7 @@ The system prompt instructs the agent to call a tool every turn, and `ensure_no_
Other middleware exists in `agent/middleware/` (`ExcludeToolsMiddleware`) but isn't wired into the default agent. The reviewer uses a leaner stack (see `reviewer.py:get_reviewer_agent` for the authoritative order), including `SanitizeToolInputsMiddleware`, `ModelCallLimitMiddleware`, `ToolErrorMiddleware`, `PullRequestVerdictGuardMiddleware`, `SlackAssistantStatusMiddleware`, and `settle_review_check_on_exit`. Other middleware exists in `agent/middleware/` (`ExcludeToolsMiddleware`) but isn't wired into the default agent. The reviewer uses a leaner stack (see `reviewer.py:get_reviewer_agent` for the authoritative order), including `SanitizeToolInputsMiddleware`, `ModelCallLimitMiddleware`, `ToolErrorMiddleware`, `PullRequestVerdictGuardMiddleware`, `SlackAssistantStatusMiddleware`, and `settle_review_check_on_exit`.
**Verdict gating (Sea Haven fork):** `PullRequestVerdictGuardMiddleware` (`agent/middleware/pr_verdict_guard.py`) is wired into BOTH graphs (`server.py:get_agent` after `PullRequestCreationGuardMiddleware`; `reviewer.py:get_reviewer_agent` after `ToolErrorMiddleware`). It blocks shell-path review verdicts (`gh pr review --approve/-a/--request-changes/-r`, `gh api`/`curl` posting `event=APPROVE|REQUEST_CHANGES` to `/pulls/N/reviews`); comment reviews and reads pass through. Verdicts go exclusively through `publish_review(verdict=...)`: `request_changes` is honored only when the run's `configurable["verdict_requested"] is True` — set solely by the explicit-mention dispatch path (`trigger_pr_review_from_ref(request_verdict=True)` → `_build_reviewer_configurable`); auto-review dispatches never set it. `approve` is additionally honored on non-verdict-requested runs when the review is clean (zero open findings — the clean-review auto-approve); with open findings it downgrades to a comment (`verdict_ignored_reason="approve_with_open_findings"`). The tool layer also downgrades self-reviews (PR author in `INTERNAL_BOT_LOGINS`) to comment reviews and best-effort dismisses a recorded stale APPROVE when a later publish surfaces new findings. **Verdict gating (Sea Haven fork):** Verdict authority is set by dispatch, not inferred by the reviewer model. `verdict_requested=True` is reserved for explicit human paths, including a direct `@openswe review` command and the main agent's `request_pr_review` tool when the user explicitly asks for a verdict. `verdict_authorized=True` is reserved for automatic paths after deterministic policy and trust checks. Automatic verdicts default off at both the team and per-repository levels; either opt-in enables policy evaluation, but only a non-fork PR whose author is an internal bot or an active member of the repository organization receives automatic authority. Automatic verdicts must match authoritative finding state: `approve` requires zero blocking findings and `request_changes` requires one or more. Explicit human verdict requests are not subject to that automatic consistency rule, but both paths still re-check the current PR head, GitHub's recorded review state, and self-review constraints. A self-review is downgraded to a comment review; its `Open SWE Review` check remains finding-aware and can still fail when blocking findings exist. `PullRequestVerdictGuardMiddleware` is wired into both the coding and reviewer graphs, and all agent shell verdict paths are blocked: `gh pr review` verdict flags plus direct `gh api`/`curl` `APPROVE` or `REQUEST_CHANGES` submissions. Comment reviews and reads remain allowed. Verdicts must go through `publish_review(verdict=...)`, which also best-effort dismisses a recorded stale approval when a later review surfaces new findings.
There is intentionally no after-agent safety net that opens a PR for the agent. The agent itself is responsible for committing, pushing, opening/updating the draft PR, and replying in the source channel — all via `GH_TOKEN=dummy gh` and `slack_thread_reply` / `linear_comment`. There is intentionally no after-agent safety net that opens a PR for the agent. The agent itself is responsible for committing, pushing, opening/updating the draft PR, and replying in the source channel — all via `GH_TOKEN=dummy gh` and `slack_thread_reply` / `linear_comment`.

View file

@ -85,6 +85,8 @@ OTHER_USER = {"login": TEST_USERS[1]["login"], "email": TEST_USERS[1]["email"]}
for _k, _v in _DEFAULTS.items(): for _k, _v in _DEFAULTS.items():
os.environ.setdefault(_k, _v) os.environ.setdefault(_k, _v)
os.environ["CONFIGURED_ADMINS"] = "alice,alice@example.com"
for _d in (TMP, _GH_DIR, _WORK_DIR): for _d in (TMP, _GH_DIR, _WORK_DIR):
_d.mkdir(parents=True, exist_ok=True) _d.mkdir(parents=True, exist_ok=True)

View file

@ -69,7 +69,10 @@ def slack_messages(channel: str) -> list[dict[str, Any]]:
# --- GitHub ---------------------------------------------------------------- # --- GitHub ----------------------------------------------------------------
PULLS: list[dict[str, Any]] = [] PULLS: list[dict[str, Any]] = []
CHECK_RUNS: list[dict[str, Any]] = []
REVIEW_DISPATCHES: list[dict[str, Any]] = []
_pr_seq = [0] _pr_seq = [0]
_check_seq = [0]
def _git(*args: str, cwd: Path | None = None) -> str: def _git(*args: str, cwd: Path | None = None) -> str:
@ -151,6 +154,8 @@ def create_pull(
"state": "open", "state": "open",
"merged": False, "merged": False,
"author": "open-swe[bot]", "author": "open-swe[bot]",
"head_sha": f"head-{number:04d}",
"base_sha": f"base-{number:04d}",
"files": files, "files": files,
"additions": sum(f["additions"] for f in files), "additions": sum(f["additions"] for f in files),
"deletions": sum(f["deletions"] for f in files), "deletions": sum(f["deletions"] for f in files),
@ -159,12 +164,56 @@ def create_pull(
return pr return pr
def create_review_pull(owner: str, repo: str) -> dict[str, Any]:
pr = create_pull(
owner,
repo,
head="feature/review-me",
base=BASE_BRANCH,
title="Review command fixture",
body="A deterministic pull request for reviewer routing.",
draft=False,
)
pr["author"] = "alice"
return pr
def find_pull(number: int) -> dict[str, Any] | None: def find_pull(number: int) -> dict[str, Any] | None:
return next((p for p in PULLS if p["number"] == number), None) return next((p for p in PULLS if p["number"] == number), None)
def create_check_run(owner: str, repo: str, payload: dict[str, Any]) -> dict[str, Any]:
_check_seq[0] += 1
check = {
"id": _check_seq[0],
"owner": owner,
"repo": repo,
"name": payload.get("name"),
"head_sha": payload.get("head_sha"),
"status": payload.get("status"),
"conclusion": payload.get("conclusion"),
"details_url": payload.get("details_url"),
"output": payload.get("output", {}),
}
CHECK_RUNS.append(check)
return check
def update_check_run(check_run_id: int, payload: dict[str, Any]) -> dict[str, Any] | None:
check = next((item for item in CHECK_RUNS if item["id"] == check_run_id), None)
if check is None:
return None
check.update(
{key: payload[key] for key in ("status", "conclusion", "output") if key in payload}
)
return check
def reset() -> None: def reset() -> None:
SLACK_MESSAGES.clear() SLACK_MESSAGES.clear()
PULLS.clear() PULLS.clear()
CHECK_RUNS.clear()
REVIEW_DISPATCHES.clear()
_pr_seq[0] = 0 _pr_seq[0] = 0
_check_seq[0] = 0
seed_bare_remote() seed_bare_remote()

View file

@ -60,8 +60,11 @@ _SLACK_USERS: dict[str, dict[str, str]] = {
} }
from agent.api.app import app # noqa: E402 from agent.api.app import app # noqa: E402
from agent.dashboard import routes as dashboard_routes # noqa: E402
from agent.dashboard.oauth import COOKIE_NAME, issue_session # noqa: E402 from agent.dashboard.oauth import COOKIE_NAME, issue_session # noqa: E402
from agent.utils import github_checks # noqa: E402
from agent.utils.thread_ids import generate_thread_id_from_slack_thread # noqa: E402 from agent.utils.thread_ids import generate_thread_id_from_slack_thread # noqa: E402
from agent.webhooks import common as webhook_common # noqa: E402
GITHUB_WEBHOOK_SECRET = os.environ["GITHUB_WEBHOOK_SECRET"] GITHUB_WEBHOOK_SECRET = os.environ["GITHUB_WEBHOOK_SECRET"]
SLACK_SIGNING_SECRET = os.environ["SLACK_SIGNING_SECRET"] SLACK_SIGNING_SECRET = os.environ["SLACK_SIGNING_SECRET"]
@ -70,6 +73,86 @@ STATIC_DIR = Path(__file__).parent / "static"
CURRENT_THREAD: dict[str, str | None] = {"channel": DEMO_CHANNEL, "thread_ts": None} CURRENT_THREAD: dict[str, str | None] = {"channel": DEMO_CHANNEL, "thread_ts": None}
fakes.seed_bare_remote() fakes.seed_bare_remote()
_real_dispatch_agent_run = webhook_common.dispatch_agent_run
async def _fake_installation_token(*_args: object, **_kwargs: object) -> str:
return "dummy-installation-token"
async def _fake_installation_token_with_expiry(
*_args: object, **_kwargs: object
) -> tuple[str, None]:
return "dummy-installation-token", None
async def _fake_fetch_pr_metadata(pr_ref: Any, *, token: str) -> dict[str, Any] | None: # noqa: ARG001
pr = fakes.find_pull(pr_ref.number)
return _gh_pr_json(pr) if pr is not None else None
async def _fake_reviewer_token(*_args: object, **_kwargs: object) -> tuple[str, None]:
return "dummy-installation-token", None
async def _fake_reaction(*_args: object, **_kwargs: object) -> bool:
return True
async def _fake_started_comment(*_args: object, **_kwargs: object) -> None:
return None
async def _record_review_dispatch(
thread_id: str,
prompt: str,
configurable: dict[str, Any],
*,
source: str,
assistant_id: str = "agent",
**kwargs: object,
) -> dict[str, str]:
if assistant_id != "reviewer":
return await _real_dispatch_agent_run(
thread_id,
prompt,
configurable,
source=source,
assistant_id=assistant_id,
**kwargs,
)
run_id = f"review-run-{len(fakes.REVIEW_DISPATCHES) + 1}"
fakes.REVIEW_DISPATCHES.append(
{
"thread_id": thread_id,
"prompt": prompt,
"configurable": configurable,
"source": source,
"assistant_id": assistant_id,
"run_id": run_id,
}
)
return {"run_id": run_id}
async def _fake_installations_and_repos(
_login: str,
) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
return (
[{"id": 1, "account": {"login": e2e_env.OWNER, "type": "Organization"}}],
[{"full_name": f"{e2e_env.OWNER}/{e2e_env.REPO}", "private": False}],
)
webhook_common.get_github_app_installation_token = _fake_installation_token
webhook_common.get_github_app_installation_token_with_expiry = _fake_installation_token_with_expiry
webhook_common.fetch_github_pr_metadata = _fake_fetch_pr_metadata
webhook_common._reviewer_token_for_repo = _fake_reviewer_token
webhook_common.react_to_github_comment = _fake_reaction
webhook_common.post_review_started_comment = _fake_started_comment
webhook_common.dispatch_agent_run = _record_review_dispatch
dashboard_routes._fetch_user_installations_and_repos = _fake_installations_and_repos
github_checks._GITHUB_API_BASE = e2e_env.FAKE_GITHUB_API
# --- control + Slack compose (the test driver) ----------------------------- # --- control + Slack compose (the test driver) -----------------------------
@ -80,6 +163,53 @@ async def control_reset() -> JSONResponse:
return JSONResponse({"ok": True}) return JSONResponse({"ok": True})
@app.post("/control/review-pr")
async def control_review_pr() -> JSONResponse:
fakes.reset()
pr = fakes.create_review_pull(e2e_env.OWNER, e2e_env.REPO)
return JSONResponse(_gh_pr_json(pr))
@app.get("/control/review-dispatches")
async def control_review_dispatches() -> JSONResponse:
return JSONResponse(fakes.REVIEW_DISPATCHES)
@app.get("/control/check-runs")
async def control_check_runs() -> JSONResponse:
return JSONResponse(fakes.CHECK_RUNS)
@app.post("/control/review-check/evaluate")
async def control_review_check_evaluate(request: Request) -> JSONResponse:
body = await request.json()
outcome = body.get("outcome")
if not isinstance(outcome, dict):
raise HTTPException(400, "outcome must be an object")
head_sha = str(body.get("head_sha") or f"evaluation-{len(fakes.CHECK_RUNS) + 1}")
check_run_id = await github_checks.create_review_check_run(
owner=e2e_env.OWNER,
repo=e2e_env.REPO,
head_sha=head_sha,
token="dummy-installation-token",
)
if check_run_id is None:
raise HTTPException(500, "failed to create check run")
conclusion, title, summary = github_checks.review_check_conclusion(outcome)
completed = await github_checks.complete_review_check_run(
owner=e2e_env.OWNER,
repo=e2e_env.REPO,
check_run_id=check_run_id,
token="dummy-installation-token",
conclusion=conclusion,
title=title,
summary=summary,
)
if not completed:
raise HTTPException(500, "failed to complete check run")
return JSONResponse({"id": check_run_id, "conclusion": conclusion, "title": title})
@app.get("/control/state") @app.get("/control/state")
async def control_state() -> JSONResponse: async def control_state() -> JSONResponse:
return JSONResponse( return JSONResponse(
@ -313,6 +443,16 @@ async def ui_agents_plan(thread_id: str) -> FileResponse: # noqa: ARG001
return _ui_file("_shell.html") return _ui_file("_shell.html")
@app.get("/review", response_class=HTMLResponse)
async def ui_review() -> FileResponse:
return _ui_file("_shell.html")
@app.get("/review/repositories/{owner}", response_class=HTMLResponse)
async def ui_review_repositories(owner: str) -> FileResponse: # noqa: ARG001
return _ui_file("_shell.html")
@app.get("/login", response_class=HTMLResponse) @app.get("/login", response_class=HTMLResponse)
async def ui_login() -> FileResponse: async def ui_login() -> FileResponse:
return _ui_file("_shell.html") return _ui_file("_shell.html")
@ -406,6 +546,8 @@ async def mock_github_pr(owner: str, repo: str, number: int) -> HTMLResponse: #
# --- fake GitHub REST API (open_pull_request hits this) -------------------- # --- fake GitHub REST API (open_pull_request hits this) --------------------
def _gh_pr_json(pr: dict[str, Any]) -> dict[str, Any]: def _gh_pr_json(pr: dict[str, Any]) -> dict[str, Any]:
full_name = f"{pr['owner']}/{pr['repo']}"
repo = {"id": 1, "full_name": full_name, "private": False}
return { return {
"number": pr["number"], "number": pr["number"],
"html_url": _pr_html_url(pr), "html_url": _pr_html_url(pr),
@ -415,8 +557,8 @@ def _gh_pr_json(pr: dict[str, Any]) -> dict[str, Any]:
"title": pr["title"], "title": pr["title"],
"body": pr["body"], "body": pr["body"],
"user": {"login": pr["author"]}, "user": {"login": pr["author"]},
"head": {"ref": pr["head"]}, "head": {"ref": pr["head"], "sha": pr["head_sha"], "repo": repo},
"base": {"ref": pr["base"]}, "base": {"ref": pr["base"], "sha": pr["base_sha"], "repo": repo},
"additions": pr["additions"], "additions": pr["additions"],
"deletions": pr["deletions"], "deletions": pr["deletions"],
"changed_files": len(pr["files"]), "changed_files": len(pr["files"]),
@ -463,6 +605,26 @@ async def gh_get_pull(owner: str, repo: str, number: int) -> JSONResponse: # no
return JSONResponse(_gh_pr_json(pr)) return JSONResponse(_gh_pr_json(pr))
@app.post("/fake-gh/repos/{owner}/{repo}/check-runs")
async def gh_create_check_run(owner: str, repo: str, request: Request) -> JSONResponse:
payload = await request.json()
return JSONResponse(fakes.create_check_run(owner, repo, payload), status_code=201)
@app.patch("/fake-gh/repos/{owner}/{repo}/check-runs/{check_run_id}")
async def gh_update_check_run(
owner: str,
repo: str,
check_run_id: int,
request: Request, # noqa: ARG001
) -> JSONResponse:
payload = await request.json()
check = fakes.update_check_run(check_run_id, payload)
if check is None:
return JSONResponse({"message": "Not Found"}, status_code=404)
return JSONResponse(check)
# --- fake Slack API (real slack code hits this) ---------------------------- # --- fake Slack API (real slack code hits this) ----------------------------
def _ok(extra: dict[str, Any] | None = None) -> JSONResponse: def _ok(extra: dict[str, Any] | None = None) -> JSONResponse:
return JSONResponse({"ok": True, **(extra or {})}) return JSONResponse({"ok": True, **(extra or {})})

View file

@ -56,6 +56,85 @@ async function expectTranscriptVisible(page: Page) {
}).toPass({ timeout: 60000 }); }).toPass({ timeout: 60000 });
} }
test.describe("Review verdict settings (real dashboard UI and API)", () => {
test("admin controls persist team and per-repository auto-verdict policy", async ({
page,
baseURL,
}) => {
await loginAs(page, SAME_USER);
const mutationHeaders = { origin: baseURL ?? "" };
const resetTeam = await page.request.put("/dashboard/api/team-settings", {
data: { auto_verdict: false },
headers: mutationHeaders,
});
expect(resetTeam.ok(), await resetTeam.text()).toBeTruthy();
const resetRepo = await page.request.put(
"/dashboard/api/auto-verdict-repos",
{
data: { full_name: "fakeorg/demo", enabled: false },
headers: mutationHeaders,
},
);
expect(resetRepo.ok()).toBeTruthy();
await page.goto("/review");
await expect(
page.getByRole("heading", { name: "Open SWE Review" }),
).toBeVisible();
const teamToggle = page.getByRole("switch").first();
await expect(teamToggle).not.toBeChecked();
const teamSaved = page.waitForResponse(
(response) =>
response.url().endsWith("/dashboard/api/team-settings") &&
response.request().method() === "PUT",
);
await teamToggle.click();
expect((await teamSaved).ok()).toBeTruthy();
await expect(teamToggle).toBeChecked();
const teamContract = await (
await page.request.get("/dashboard/api/team-settings")
).json();
expect(teamContract.auto_verdict).toBe(true);
await page.goto("/review/repositories/fakeorg");
const repoToggle = page.getByRole("switch", {
name: "Allow automatic verdicts for fakeorg/demo",
});
await expect(repoToggle).not.toBeChecked();
const repoSaved = page.waitForResponse(
(response) =>
response.url().endsWith("/dashboard/api/auto-verdict-repos") &&
response.request().method() === "PUT",
);
await repoToggle.click();
expect((await repoSaved).ok()).toBeTruthy();
await expect(repoToggle).toBeChecked();
const repoContract = await (
await page.request.get("/dashboard/api/auto-verdict-repos")
).json();
expect(repoContract).toEqual({ repos: ["fakeorg/demo"] });
await loginAs(page, OTHER_USER);
const forbidden = await page.request.put(
"/dashboard/api/auto-verdict-repos",
{
data: { full_name: "fakeorg/demo", enabled: false },
headers: mutationHeaders,
},
);
expect(forbidden.status()).toBe(403);
await page.reload();
await expect(
page.getByRole("switch", {
name: "Allow automatic verdicts for fakeorg/demo",
}),
).toBeDisabled();
});
});
test.describe("Slack → web handoff (real dashboard UI)", () => { test.describe("Slack → web handoff (real dashboard UI)", () => {
test("the SAME user continues the conversation in the web app", async ({ test("the SAME user continues the conversation in the web app", async ({
page, page,

View file

@ -0,0 +1,145 @@
import { createHmac } from "node:crypto";
import { expect, test } from "@playwright/test";
const WEBHOOK_SECRET = "test-github-secret";
test.describe("Reviewer verification contracts", () => {
test("direct @openswe review routes to the reviewer without user mapping", async ({
page,
}) => {
const seeded = await page.request.post("/control/review-pr");
expect(seeded.ok()).toBeTruthy();
const pr = await seeded.json();
const payload = {
action: "created",
repository: {
id: 1,
name: "demo",
full_name: "fakeorg/demo",
private: true,
owner: { login: "fakeorg" },
},
issue: {
number: pr.number,
html_url: pr.html_url,
pull_request: { html_url: pr.html_url },
},
comment: {
id: 24601,
node_id: "IC_kwDO_e2e",
body: "@openswe review: focus on authorization boundaries",
user: { login: "unmapped-reviewer" },
},
sender: { id: 9001, login: "unmapped-reviewer" },
};
const raw = JSON.stringify(payload);
const signature = `sha256=${createHmac("sha256", WEBHOOK_SECRET)
.update(raw)
.digest("hex")}`;
const webhook = await page.request.post("/webhooks/github", {
data: raw,
headers: {
"Content-Type": "application/json",
"X-GitHub-Event": "issue_comment",
"X-Hub-Signature-256": signature,
},
});
expect(webhook.ok()).toBeTruthy();
expect(await webhook.json()).toEqual({
status: "accepted",
message: "Processing on-demand PR review",
});
await expect
.poll(async () => {
const response = await page.request.get("/control/review-dispatches");
return (await response.json()).length;
})
.toBe(1);
const dispatches = await (
await page.request.get("/control/review-dispatches")
).json();
expect(dispatches).toHaveLength(1);
expect(dispatches[0].assistant_id).toBe("reviewer");
expect(dispatches[0].source).toBe("github_comment");
expect(dispatches[0].configurable).toMatchObject({
github_login: "unmapped-reviewer",
github_user_id: 9001,
review_requested: true,
verdict_requested: true,
repo: { owner: "fakeorg", name: "demo" },
pr_number: pr.number,
});
expect(dispatches[0].configurable).not.toHaveProperty("email");
expect(dispatches[0].prompt).toContain(
"focus on authorization boundaries",
);
});
test("review outcomes settle checks as success, failure, or neutral", async ({
page,
}) => {
await page.request.post("/control/reset");
const cases = [
{
outcome: {
verdict_submitted: true,
verdict_event: "APPROVE",
blocking_finding_count: 0,
},
conclusion: "success",
title: "Review approved",
},
{
outcome: {
verdict_submitted: true,
verdict_event: "REQUEST_CHANGES",
blocking_finding_count: 1,
},
conclusion: "failure",
title: "Changes requested",
},
{
outcome: {
verdict_submitted: false,
verdict_ignored_reason: "verdict_not_requested",
blocking_finding_count: 0,
},
conclusion: "neutral",
title: "Verdict withheld",
},
];
for (const [index, item] of cases.entries()) {
const response = await page.request.post(
"/control/review-check/evaluate",
{
data: {
head_sha: `tri-state-${index + 1}`,
outcome: item.outcome,
},
},
);
expect(response.ok()).toBeTruthy();
expect(await response.json()).toMatchObject({
conclusion: item.conclusion,
title: item.title,
});
}
const checks = await (
await page.request.get("/control/check-runs")
).json();
expect(checks.map((check: { conclusion: string }) => check.conclusion)).toEqual([
"success",
"failure",
"neutral",
]);
expect(
checks.map((check: { status: string }) => check.status),
).toEqual(["completed", "completed", "completed"]);
});
});