From 99b376bbe2e4805ceabd909476dfa55650731613 Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Fri, 31 Jul 2026 14:30:12 -0400 Subject: [PATCH] chore(reviewer): add epic verification coverage --- .security-review/suppressions.json | 20 --- CLAUDE.md | 2 +- tests/e2e/e2e_env.py | 2 + tests/e2e/fakes.py | 49 ++++++ tests/e2e/harness.py | 166 +++++++++++++++++- tests/e2e/tests/dashboard.spec.ts | 79 +++++++++ tests/e2e/tests/reviewer_verification.spec.ts | 145 +++++++++++++++ 7 files changed, 440 insertions(+), 23 deletions(-) create mode 100644 tests/e2e/tests/reviewer_verification.spec.ts diff --git a/.security-review/suppressions.json b/.security-review/suppressions.json index aaaa23fe..ef384a42 100644 --- a/.security-review/suppressions.json +++ b/.security-review/suppressions.json @@ -1,25 +1,5 @@ { "suppressions": [ - { - "id": "AUTHZ-CLEAN-AUTOAPPROVE-INJECTION-001", - "title": "Clean-review auto-approve derives APPROVE authority from a model-controlled signal on the untrusted auto-review path (prompt-injection-mintable APPROVE)", - "file": "agent/tools/publish_review.py", - "severity": "high", - "status": "confirmed", - "suppression_justification": "ACCEPTED (Adam, 2026-07-21) as a knowingly-deferred risk merged into the dev integration branch via admin merge in PR #217. /sh-security-review returned BLOCK: moving `approve` authorization from the deterministic verdict_requested flag to open_findings_count==0 lets a prompt-injected auto-review (which never sets verdict_requested) land a real bot APPROVE on an attacker-controlled external/fork PR, potentially satisfying branch protection. This reintroduces the hole PR #214 closed. Merged to dev only (NOT main/prod). Compensating controls: (a) confirm dev does not auto-deploy to sh-openswe; (b) on sensitive repos, esp. payments-dashboard, configure rulesets so the seahaven-openswe[bot] APPROVE does not by itself satisfy required approvals. Tracked in issue #218 with the full redesign checklist. REVISIT TRIGGER: MUST be resolved before this change promotes from dev to main/prod, and immediately if dev is found to auto-deploy. Verified HIGH by the sh-security-review detector fan-out (6 independent detectors converged).", - "owner": "adam@seahavenind.com", - "added": "2026-07-21" - }, - { - "id": "AUTHZ-CLEAN-AUTOAPPROVE-THREADLAUNDER-002", - "title": "Clean-review auto-approve gate reads reconcile-derived finding status, so a PR author can launder a dirty PR to clean via GitHub thread resolve/outdate", - "file": "agent/tools/publish_review.py", - "severity": "high", - "status": "confirmed", - "suppression_justification": "ACCEPTED (Adam, 2026-07-21), deferred with AUTHZ-CLEAN-AUTOAPPROVE-INJECTION-001 in PR #217 (admin-merged to dev). The open_findings_count gate reads finding status after reconcile_findings_with_review_threads, which flips open->resolved when the GitHub thread is is_resolved/is_outdated — both author-controllable ('Resolve conversation' or a trivial hunk-outdating commit) without fixing the defect, yielding an auto-approve on an unfixed PR. Same compensating controls and revisit trigger as ...-001. Tracked in issue #218 (redesign: compute the gate from the reviewer's own authoritative finding state, not author-influenced thread state). Verified HIGH by /sh-security-review.", - "owner": "adam@seahavenind.com", - "added": "2026-07-21" - }, { "id": "AUTHZ-SLACK-BOT-DEFAULT-001", "title": "Slack entrypoint lacks a per-user repo-access check; default-bot PR authoring removes the implicit per-user repo boundary", diff --git a/CLAUDE.md b/CLAUDE.md index 613c84f2..a99b6f7d 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -82,7 +82,7 @@ The system prompt instructs the agent to call a tool every turn, and `ensure_no_ Other middleware exists in `agent/middleware/` (`ExcludeToolsMiddleware`) but isn't wired into the default agent. The reviewer uses a leaner stack (see `reviewer.py:get_reviewer_agent` for the authoritative order), including `SanitizeToolInputsMiddleware`, `ModelCallLimitMiddleware`, `ToolErrorMiddleware`, `PullRequestVerdictGuardMiddleware`, `SlackAssistantStatusMiddleware`, and `settle_review_check_on_exit`. -**Verdict gating (Sea Haven fork):** `PullRequestVerdictGuardMiddleware` (`agent/middleware/pr_verdict_guard.py`) is wired into BOTH graphs (`server.py:get_agent` after `PullRequestCreationGuardMiddleware`; `reviewer.py:get_reviewer_agent` after `ToolErrorMiddleware`). It blocks shell-path review verdicts (`gh pr review --approve/-a/--request-changes/-r`, `gh api`/`curl` posting `event=APPROVE|REQUEST_CHANGES` to `/pulls/N/reviews`); comment reviews and reads pass through. Verdicts go exclusively through `publish_review(verdict=...)`: `request_changes` is honored only when the run's `configurable["verdict_requested"] is True` — set solely by the explicit-mention dispatch path (`trigger_pr_review_from_ref(request_verdict=True)` → `_build_reviewer_configurable`); auto-review dispatches never set it. `approve` is additionally honored on non-verdict-requested runs when the review is clean (zero open findings — the clean-review auto-approve); with open findings it downgrades to a comment (`verdict_ignored_reason="approve_with_open_findings"`). The tool layer also downgrades self-reviews (PR author in `INTERNAL_BOT_LOGINS`) to comment reviews and best-effort dismisses a recorded stale APPROVE when a later publish surfaces new findings. +**Verdict gating (Sea Haven fork):** Verdict authority is set by dispatch, not inferred by the reviewer model. `verdict_requested=True` is reserved for explicit human paths, including a direct `@openswe review` command and the main agent's `request_pr_review` tool when the user explicitly asks for a verdict. `verdict_authorized=True` is reserved for automatic paths after deterministic policy and trust checks. Automatic verdicts default off at both the team and per-repository levels; either opt-in enables policy evaluation, but only a non-fork PR whose author is an internal bot or an active member of the repository organization receives automatic authority. Automatic verdicts must match authoritative finding state: `approve` requires zero blocking findings and `request_changes` requires one or more. Explicit human verdict requests are not subject to that automatic consistency rule, but both paths still re-check the current PR head, GitHub's recorded review state, and self-review constraints. A self-review is downgraded to a comment review; its `Open SWE Review` check remains finding-aware and can still fail when blocking findings exist. `PullRequestVerdictGuardMiddleware` is wired into both the coding and reviewer graphs, and all agent shell verdict paths are blocked: `gh pr review` verdict flags plus direct `gh api`/`curl` `APPROVE` or `REQUEST_CHANGES` submissions. Comment reviews and reads remain allowed. Verdicts must go through `publish_review(verdict=...)`, which also best-effort dismisses a recorded stale approval when a later review surfaces new findings. There is intentionally no after-agent safety net that opens a PR for the agent. The agent itself is responsible for committing, pushing, opening/updating the draft PR, and replying in the source channel — all via `GH_TOKEN=dummy gh` and `slack_thread_reply` / `linear_comment`. diff --git a/tests/e2e/e2e_env.py b/tests/e2e/e2e_env.py index c3451cf4..a6ad1c5a 100644 --- a/tests/e2e/e2e_env.py +++ b/tests/e2e/e2e_env.py @@ -85,6 +85,8 @@ OTHER_USER = {"login": TEST_USERS[1]["login"], "email": TEST_USERS[1]["email"]} for _k, _v in _DEFAULTS.items(): os.environ.setdefault(_k, _v) +os.environ["CONFIGURED_ADMINS"] = "alice,alice@example.com" + for _d in (TMP, _GH_DIR, _WORK_DIR): _d.mkdir(parents=True, exist_ok=True) diff --git a/tests/e2e/fakes.py b/tests/e2e/fakes.py index ba57ee82..e7fc1efb 100644 --- a/tests/e2e/fakes.py +++ b/tests/e2e/fakes.py @@ -69,7 +69,10 @@ def slack_messages(channel: str) -> list[dict[str, Any]]: # --- GitHub ---------------------------------------------------------------- PULLS: list[dict[str, Any]] = [] +CHECK_RUNS: list[dict[str, Any]] = [] +REVIEW_DISPATCHES: list[dict[str, Any]] = [] _pr_seq = [0] +_check_seq = [0] def _git(*args: str, cwd: Path | None = None) -> str: @@ -151,6 +154,8 @@ def create_pull( "state": "open", "merged": False, "author": "open-swe[bot]", + "head_sha": f"head-{number:04d}", + "base_sha": f"base-{number:04d}", "files": files, "additions": sum(f["additions"] for f in files), "deletions": sum(f["deletions"] for f in files), @@ -159,12 +164,56 @@ def create_pull( return pr +def create_review_pull(owner: str, repo: str) -> dict[str, Any]: + pr = create_pull( + owner, + repo, + head="feature/review-me", + base=BASE_BRANCH, + title="Review command fixture", + body="A deterministic pull request for reviewer routing.", + draft=False, + ) + pr["author"] = "alice" + return pr + + def find_pull(number: int) -> dict[str, Any] | None: return next((p for p in PULLS if p["number"] == number), None) +def create_check_run(owner: str, repo: str, payload: dict[str, Any]) -> dict[str, Any]: + _check_seq[0] += 1 + check = { + "id": _check_seq[0], + "owner": owner, + "repo": repo, + "name": payload.get("name"), + "head_sha": payload.get("head_sha"), + "status": payload.get("status"), + "conclusion": payload.get("conclusion"), + "details_url": payload.get("details_url"), + "output": payload.get("output", {}), + } + CHECK_RUNS.append(check) + return check + + +def update_check_run(check_run_id: int, payload: dict[str, Any]) -> dict[str, Any] | None: + check = next((item for item in CHECK_RUNS if item["id"] == check_run_id), None) + if check is None: + return None + check.update( + {key: payload[key] for key in ("status", "conclusion", "output") if key in payload} + ) + return check + + def reset() -> None: SLACK_MESSAGES.clear() PULLS.clear() + CHECK_RUNS.clear() + REVIEW_DISPATCHES.clear() _pr_seq[0] = 0 + _check_seq[0] = 0 seed_bare_remote() diff --git a/tests/e2e/harness.py b/tests/e2e/harness.py index 97baeaaa..27a05c5e 100644 --- a/tests/e2e/harness.py +++ b/tests/e2e/harness.py @@ -60,8 +60,11 @@ _SLACK_USERS: dict[str, dict[str, str]] = { } from agent.api.app import app # noqa: E402 +from agent.dashboard import routes as dashboard_routes # noqa: E402 from agent.dashboard.oauth import COOKIE_NAME, issue_session # noqa: E402 +from agent.utils import github_checks # noqa: E402 from agent.utils.thread_ids import generate_thread_id_from_slack_thread # noqa: E402 +from agent.webhooks import common as webhook_common # noqa: E402 GITHUB_WEBHOOK_SECRET = os.environ["GITHUB_WEBHOOK_SECRET"] SLACK_SIGNING_SECRET = os.environ["SLACK_SIGNING_SECRET"] @@ -70,6 +73,86 @@ STATIC_DIR = Path(__file__).parent / "static" CURRENT_THREAD: dict[str, str | None] = {"channel": DEMO_CHANNEL, "thread_ts": None} fakes.seed_bare_remote() +_real_dispatch_agent_run = webhook_common.dispatch_agent_run + + +async def _fake_installation_token(*_args: object, **_kwargs: object) -> str: + return "dummy-installation-token" + + +async def _fake_installation_token_with_expiry( + *_args: object, **_kwargs: object +) -> tuple[str, None]: + return "dummy-installation-token", None + + +async def _fake_fetch_pr_metadata(pr_ref: Any, *, token: str) -> dict[str, Any] | None: # noqa: ARG001 + pr = fakes.find_pull(pr_ref.number) + return _gh_pr_json(pr) if pr is not None else None + + +async def _fake_reviewer_token(*_args: object, **_kwargs: object) -> tuple[str, None]: + return "dummy-installation-token", None + + +async def _fake_reaction(*_args: object, **_kwargs: object) -> bool: + return True + + +async def _fake_started_comment(*_args: object, **_kwargs: object) -> None: + return None + + +async def _record_review_dispatch( + thread_id: str, + prompt: str, + configurable: dict[str, Any], + *, + source: str, + assistant_id: str = "agent", + **kwargs: object, +) -> dict[str, str]: + if assistant_id != "reviewer": + return await _real_dispatch_agent_run( + thread_id, + prompt, + configurable, + source=source, + assistant_id=assistant_id, + **kwargs, + ) + run_id = f"review-run-{len(fakes.REVIEW_DISPATCHES) + 1}" + fakes.REVIEW_DISPATCHES.append( + { + "thread_id": thread_id, + "prompt": prompt, + "configurable": configurable, + "source": source, + "assistant_id": assistant_id, + "run_id": run_id, + } + ) + return {"run_id": run_id} + + +async def _fake_installations_and_repos( + _login: str, +) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + return ( + [{"id": 1, "account": {"login": e2e_env.OWNER, "type": "Organization"}}], + [{"full_name": f"{e2e_env.OWNER}/{e2e_env.REPO}", "private": False}], + ) + + +webhook_common.get_github_app_installation_token = _fake_installation_token +webhook_common.get_github_app_installation_token_with_expiry = _fake_installation_token_with_expiry +webhook_common.fetch_github_pr_metadata = _fake_fetch_pr_metadata +webhook_common._reviewer_token_for_repo = _fake_reviewer_token +webhook_common.react_to_github_comment = _fake_reaction +webhook_common.post_review_started_comment = _fake_started_comment +webhook_common.dispatch_agent_run = _record_review_dispatch +dashboard_routes._fetch_user_installations_and_repos = _fake_installations_and_repos +github_checks._GITHUB_API_BASE = e2e_env.FAKE_GITHUB_API # --- control + Slack compose (the test driver) ----------------------------- @@ -80,6 +163,53 @@ async def control_reset() -> JSONResponse: return JSONResponse({"ok": True}) +@app.post("/control/review-pr") +async def control_review_pr() -> JSONResponse: + fakes.reset() + pr = fakes.create_review_pull(e2e_env.OWNER, e2e_env.REPO) + return JSONResponse(_gh_pr_json(pr)) + + +@app.get("/control/review-dispatches") +async def control_review_dispatches() -> JSONResponse: + return JSONResponse(fakes.REVIEW_DISPATCHES) + + +@app.get("/control/check-runs") +async def control_check_runs() -> JSONResponse: + return JSONResponse(fakes.CHECK_RUNS) + + +@app.post("/control/review-check/evaluate") +async def control_review_check_evaluate(request: Request) -> JSONResponse: + body = await request.json() + outcome = body.get("outcome") + if not isinstance(outcome, dict): + raise HTTPException(400, "outcome must be an object") + head_sha = str(body.get("head_sha") or f"evaluation-{len(fakes.CHECK_RUNS) + 1}") + check_run_id = await github_checks.create_review_check_run( + owner=e2e_env.OWNER, + repo=e2e_env.REPO, + head_sha=head_sha, + token="dummy-installation-token", + ) + if check_run_id is None: + raise HTTPException(500, "failed to create check run") + conclusion, title, summary = github_checks.review_check_conclusion(outcome) + completed = await github_checks.complete_review_check_run( + owner=e2e_env.OWNER, + repo=e2e_env.REPO, + check_run_id=check_run_id, + token="dummy-installation-token", + conclusion=conclusion, + title=title, + summary=summary, + ) + if not completed: + raise HTTPException(500, "failed to complete check run") + return JSONResponse({"id": check_run_id, "conclusion": conclusion, "title": title}) + + @app.get("/control/state") async def control_state() -> JSONResponse: return JSONResponse( @@ -313,6 +443,16 @@ async def ui_agents_plan(thread_id: str) -> FileResponse: # noqa: ARG001 return _ui_file("_shell.html") +@app.get("/review", response_class=HTMLResponse) +async def ui_review() -> FileResponse: + return _ui_file("_shell.html") + + +@app.get("/review/repositories/{owner}", response_class=HTMLResponse) +async def ui_review_repositories(owner: str) -> FileResponse: # noqa: ARG001 + return _ui_file("_shell.html") + + @app.get("/login", response_class=HTMLResponse) async def ui_login() -> FileResponse: return _ui_file("_shell.html") @@ -406,6 +546,8 @@ async def mock_github_pr(owner: str, repo: str, number: int) -> HTMLResponse: # # --- fake GitHub REST API (open_pull_request hits this) -------------------- def _gh_pr_json(pr: dict[str, Any]) -> dict[str, Any]: + full_name = f"{pr['owner']}/{pr['repo']}" + repo = {"id": 1, "full_name": full_name, "private": False} return { "number": pr["number"], "html_url": _pr_html_url(pr), @@ -415,8 +557,8 @@ def _gh_pr_json(pr: dict[str, Any]) -> dict[str, Any]: "title": pr["title"], "body": pr["body"], "user": {"login": pr["author"]}, - "head": {"ref": pr["head"]}, - "base": {"ref": pr["base"]}, + "head": {"ref": pr["head"], "sha": pr["head_sha"], "repo": repo}, + "base": {"ref": pr["base"], "sha": pr["base_sha"], "repo": repo}, "additions": pr["additions"], "deletions": pr["deletions"], "changed_files": len(pr["files"]), @@ -463,6 +605,26 @@ async def gh_get_pull(owner: str, repo: str, number: int) -> JSONResponse: # no return JSONResponse(_gh_pr_json(pr)) +@app.post("/fake-gh/repos/{owner}/{repo}/check-runs") +async def gh_create_check_run(owner: str, repo: str, request: Request) -> JSONResponse: + payload = await request.json() + return JSONResponse(fakes.create_check_run(owner, repo, payload), status_code=201) + + +@app.patch("/fake-gh/repos/{owner}/{repo}/check-runs/{check_run_id}") +async def gh_update_check_run( + owner: str, + repo: str, + check_run_id: int, + request: Request, # noqa: ARG001 +) -> JSONResponse: + payload = await request.json() + check = fakes.update_check_run(check_run_id, payload) + if check is None: + return JSONResponse({"message": "Not Found"}, status_code=404) + return JSONResponse(check) + + # --- fake Slack API (real slack code hits this) ---------------------------- def _ok(extra: dict[str, Any] | None = None) -> JSONResponse: return JSONResponse({"ok": True, **(extra or {})}) diff --git a/tests/e2e/tests/dashboard.spec.ts b/tests/e2e/tests/dashboard.spec.ts index cce7a994..e38e0a10 100644 --- a/tests/e2e/tests/dashboard.spec.ts +++ b/tests/e2e/tests/dashboard.spec.ts @@ -56,6 +56,85 @@ async function expectTranscriptVisible(page: Page) { }).toPass({ timeout: 60000 }); } +test.describe("Review verdict settings (real dashboard UI and API)", () => { + test("admin controls persist team and per-repository auto-verdict policy", async ({ + page, + baseURL, + }) => { + await loginAs(page, SAME_USER); + const mutationHeaders = { origin: baseURL ?? "" }; + + const resetTeam = await page.request.put("/dashboard/api/team-settings", { + data: { auto_verdict: false }, + headers: mutationHeaders, + }); + expect(resetTeam.ok(), await resetTeam.text()).toBeTruthy(); + const resetRepo = await page.request.put( + "/dashboard/api/auto-verdict-repos", + { + data: { full_name: "fakeorg/demo", enabled: false }, + headers: mutationHeaders, + }, + ); + expect(resetRepo.ok()).toBeTruthy(); + + await page.goto("/review"); + await expect( + page.getByRole("heading", { name: "Open SWE Review" }), + ).toBeVisible(); + const teamToggle = page.getByRole("switch").first(); + await expect(teamToggle).not.toBeChecked(); + const teamSaved = page.waitForResponse( + (response) => + response.url().endsWith("/dashboard/api/team-settings") && + response.request().method() === "PUT", + ); + await teamToggle.click(); + expect((await teamSaved).ok()).toBeTruthy(); + await expect(teamToggle).toBeChecked(); + + const teamContract = await ( + await page.request.get("/dashboard/api/team-settings") + ).json(); + expect(teamContract.auto_verdict).toBe(true); + + await page.goto("/review/repositories/fakeorg"); + const repoToggle = page.getByRole("switch", { + name: "Allow automatic verdicts for fakeorg/demo", + }); + await expect(repoToggle).not.toBeChecked(); + const repoSaved = page.waitForResponse( + (response) => + response.url().endsWith("/dashboard/api/auto-verdict-repos") && + response.request().method() === "PUT", + ); + await repoToggle.click(); + expect((await repoSaved).ok()).toBeTruthy(); + await expect(repoToggle).toBeChecked(); + + const repoContract = await ( + await page.request.get("/dashboard/api/auto-verdict-repos") + ).json(); + expect(repoContract).toEqual({ repos: ["fakeorg/demo"] }); + + await loginAs(page, OTHER_USER); + const forbidden = await page.request.put( + "/dashboard/api/auto-verdict-repos", + { + data: { full_name: "fakeorg/demo", enabled: false }, + headers: mutationHeaders, + }, + ); + expect(forbidden.status()).toBe(403); + await page.reload(); + await expect( + page.getByRole("switch", { + name: "Allow automatic verdicts for fakeorg/demo", + }), + ).toBeDisabled(); + }); +}); + test.describe("Slack → web handoff (real dashboard UI)", () => { test("the SAME user continues the conversation in the web app", async ({ page, diff --git a/tests/e2e/tests/reviewer_verification.spec.ts b/tests/e2e/tests/reviewer_verification.spec.ts new file mode 100644 index 00000000..839dec98 --- /dev/null +++ b/tests/e2e/tests/reviewer_verification.spec.ts @@ -0,0 +1,145 @@ +import { createHmac } from "node:crypto"; + +import { expect, test } from "@playwright/test"; + +const WEBHOOK_SECRET = "test-github-secret"; + +test.describe("Reviewer verification contracts", () => { + test("direct @openswe review routes to the reviewer without user mapping", async ({ + page, + }) => { + const seeded = await page.request.post("/control/review-pr"); + expect(seeded.ok()).toBeTruthy(); + const pr = await seeded.json(); + const payload = { + action: "created", + repository: { + id: 1, + name: "demo", + full_name: "fakeorg/demo", + private: true, + owner: { login: "fakeorg" }, + }, + issue: { + number: pr.number, + html_url: pr.html_url, + pull_request: { html_url: pr.html_url }, + }, + comment: { + id: 24601, + node_id: "IC_kwDO_e2e", + body: "@openswe review: focus on authorization boundaries", + user: { login: "unmapped-reviewer" }, + }, + sender: { id: 9001, login: "unmapped-reviewer" }, + }; + const raw = JSON.stringify(payload); + const signature = `sha256=${createHmac("sha256", WEBHOOK_SECRET) + .update(raw) + .digest("hex")}`; + + const webhook = await page.request.post("/webhooks/github", { + data: raw, + headers: { + "Content-Type": "application/json", + "X-GitHub-Event": "issue_comment", + "X-Hub-Signature-256": signature, + }, + }); + expect(webhook.ok()).toBeTruthy(); + expect(await webhook.json()).toEqual({ + status: "accepted", + message: "Processing on-demand PR review", + }); + + await expect + .poll(async () => { + const response = await page.request.get("/control/review-dispatches"); + return (await response.json()).length; + }) + .toBe(1); + const dispatches = await ( + await page.request.get("/control/review-dispatches") + ).json(); + expect(dispatches).toHaveLength(1); + expect(dispatches[0].assistant_id).toBe("reviewer"); + expect(dispatches[0].source).toBe("github_comment"); + expect(dispatches[0].configurable).toMatchObject({ + github_login: "unmapped-reviewer", + github_user_id: 9001, + review_requested: true, + verdict_requested: true, + repo: { owner: "fakeorg", name: "demo" }, + pr_number: pr.number, + }); + expect(dispatches[0].configurable).not.toHaveProperty("email"); + expect(dispatches[0].prompt).toContain( + "focus on authorization boundaries", + ); + + }); + + test("review outcomes settle checks as success, failure, or neutral", async ({ + page, + }) => { + await page.request.post("/control/reset"); + const cases = [ + { + outcome: { + verdict_submitted: true, + verdict_event: "APPROVE", + blocking_finding_count: 0, + }, + conclusion: "success", + title: "Review approved", + }, + { + outcome: { + verdict_submitted: true, + verdict_event: "REQUEST_CHANGES", + blocking_finding_count: 1, + }, + conclusion: "failure", + title: "Changes requested", + }, + { + outcome: { + verdict_submitted: false, + verdict_ignored_reason: "verdict_not_requested", + blocking_finding_count: 0, + }, + conclusion: "neutral", + title: "Verdict withheld", + }, + ]; + + for (const [index, item] of cases.entries()) { + const response = await page.request.post( + "/control/review-check/evaluate", + { + data: { + head_sha: `tri-state-${index + 1}`, + outcome: item.outcome, + }, + }, + ); + expect(response.ok()).toBeTruthy(); + expect(await response.json()).toMatchObject({ + conclusion: item.conclusion, + title: item.title, + }); + } + + const checks = await ( + await page.request.get("/control/check-runs") + ).json(); + expect(checks.map((check: { conclusion: string }) => check.conclusion)).toEqual([ + "success", + "failure", + "neutral", + ]); + expect( + checks.map((check: { status: string }) => check.status), + ).toEqual(["completed", "completed", "completed"]); + }); +});