mirror of
https://github.com/Sea-Haven-Industries/open-swe.git
synced 2026-09-30 04:33:12 +00:00
chore(reviewer): add epic verification coverage
This commit is contained in:
parent
5f6d756e0d
commit
99b376bbe2
7 changed files with 440 additions and 23 deletions
|
|
@ -1,25 +1,5 @@
|
|||
{
|
||||
"suppressions": [
|
||||
{
|
||||
"id": "AUTHZ-CLEAN-AUTOAPPROVE-INJECTION-001",
|
||||
"title": "Clean-review auto-approve derives APPROVE authority from a model-controlled signal on the untrusted auto-review path (prompt-injection-mintable APPROVE)",
|
||||
"file": "agent/tools/publish_review.py",
|
||||
"severity": "high",
|
||||
"status": "confirmed",
|
||||
"suppression_justification": "ACCEPTED (Adam, 2026-07-21) as a knowingly-deferred risk merged into the dev integration branch via admin merge in PR #217. /sh-security-review returned BLOCK: moving `approve` authorization from the deterministic verdict_requested flag to open_findings_count==0 lets a prompt-injected auto-review (which never sets verdict_requested) land a real bot APPROVE on an attacker-controlled external/fork PR, potentially satisfying branch protection. This reintroduces the hole PR #214 closed. Merged to dev only (NOT main/prod). Compensating controls: (a) confirm dev does not auto-deploy to sh-openswe; (b) on sensitive repos, esp. payments-dashboard, configure rulesets so the seahaven-openswe[bot] APPROVE does not by itself satisfy required approvals. Tracked in issue #218 with the full redesign checklist. REVISIT TRIGGER: MUST be resolved before this change promotes from dev to main/prod, and immediately if dev is found to auto-deploy. Verified HIGH by the sh-security-review detector fan-out (6 independent detectors converged).",
|
||||
"owner": "adam@seahavenind.com",
|
||||
"added": "2026-07-21"
|
||||
},
|
||||
{
|
||||
"id": "AUTHZ-CLEAN-AUTOAPPROVE-THREADLAUNDER-002",
|
||||
"title": "Clean-review auto-approve gate reads reconcile-derived finding status, so a PR author can launder a dirty PR to clean via GitHub thread resolve/outdate",
|
||||
"file": "agent/tools/publish_review.py",
|
||||
"severity": "high",
|
||||
"status": "confirmed",
|
||||
"suppression_justification": "ACCEPTED (Adam, 2026-07-21), deferred with AUTHZ-CLEAN-AUTOAPPROVE-INJECTION-001 in PR #217 (admin-merged to dev). The open_findings_count gate reads finding status after reconcile_findings_with_review_threads, which flips open->resolved when the GitHub thread is is_resolved/is_outdated — both author-controllable ('Resolve conversation' or a trivial hunk-outdating commit) without fixing the defect, yielding an auto-approve on an unfixed PR. Same compensating controls and revisit trigger as ...-001. Tracked in issue #218 (redesign: compute the gate from the reviewer's own authoritative finding state, not author-influenced thread state). Verified HIGH by /sh-security-review.",
|
||||
"owner": "adam@seahavenind.com",
|
||||
"added": "2026-07-21"
|
||||
},
|
||||
{
|
||||
"id": "AUTHZ-SLACK-BOT-DEFAULT-001",
|
||||
"title": "Slack entrypoint lacks a per-user repo-access check; default-bot PR authoring removes the implicit per-user repo boundary",
|
||||
|
|
|
|||
|
|
@ -82,7 +82,7 @@ The system prompt instructs the agent to call a tool every turn, and `ensure_no_
|
|||
|
||||
Other middleware exists in `agent/middleware/` (`ExcludeToolsMiddleware`) but isn't wired into the default agent. The reviewer uses a leaner stack (see `reviewer.py:get_reviewer_agent` for the authoritative order), including `SanitizeToolInputsMiddleware`, `ModelCallLimitMiddleware`, `ToolErrorMiddleware`, `PullRequestVerdictGuardMiddleware`, `SlackAssistantStatusMiddleware`, and `settle_review_check_on_exit`.
|
||||
|
||||
**Verdict gating (Sea Haven fork):** `PullRequestVerdictGuardMiddleware` (`agent/middleware/pr_verdict_guard.py`) is wired into BOTH graphs (`server.py:get_agent` after `PullRequestCreationGuardMiddleware`; `reviewer.py:get_reviewer_agent` after `ToolErrorMiddleware`). It blocks shell-path review verdicts (`gh pr review --approve/-a/--request-changes/-r`, `gh api`/`curl` posting `event=APPROVE|REQUEST_CHANGES` to `/pulls/N/reviews`); comment reviews and reads pass through. Verdicts go exclusively through `publish_review(verdict=...)`: `request_changes` is honored only when the run's `configurable["verdict_requested"] is True` — set solely by the explicit-mention dispatch path (`trigger_pr_review_from_ref(request_verdict=True)` → `_build_reviewer_configurable`); auto-review dispatches never set it. `approve` is additionally honored on non-verdict-requested runs when the review is clean (zero open findings — the clean-review auto-approve); with open findings it downgrades to a comment (`verdict_ignored_reason="approve_with_open_findings"`). The tool layer also downgrades self-reviews (PR author in `INTERNAL_BOT_LOGINS`) to comment reviews and best-effort dismisses a recorded stale APPROVE when a later publish surfaces new findings.
|
||||
**Verdict gating (Sea Haven fork):** Verdict authority is set by dispatch, not inferred by the reviewer model. `verdict_requested=True` is reserved for explicit human paths, including a direct `@openswe review` command and the main agent's `request_pr_review` tool when the user explicitly asks for a verdict. `verdict_authorized=True` is reserved for automatic paths after deterministic policy and trust checks. Automatic verdicts default off at both the team and per-repository levels; either opt-in enables policy evaluation, but only a non-fork PR whose author is an internal bot or an active member of the repository organization receives automatic authority. Automatic verdicts must match authoritative finding state: `approve` requires zero blocking findings and `request_changes` requires one or more. Explicit human verdict requests are not subject to that automatic consistency rule, but both paths still re-check the current PR head, GitHub's recorded review state, and self-review constraints. A self-review is downgraded to a comment review; its `Open SWE Review` check remains finding-aware and can still fail when blocking findings exist. `PullRequestVerdictGuardMiddleware` is wired into both the coding and reviewer graphs, and all agent shell verdict paths are blocked: `gh pr review` verdict flags plus direct `gh api`/`curl` `APPROVE` or `REQUEST_CHANGES` submissions. Comment reviews and reads remain allowed. Verdicts must go through `publish_review(verdict=...)`, which also best-effort dismisses a recorded stale approval when a later review surfaces new findings.
|
||||
|
||||
There is intentionally no after-agent safety net that opens a PR for the agent. The agent itself is responsible for committing, pushing, opening/updating the draft PR, and replying in the source channel — all via `GH_TOKEN=dummy gh` and `slack_thread_reply` / `linear_comment`.
|
||||
|
||||
|
|
|
|||
|
|
@ -85,6 +85,8 @@ OTHER_USER = {"login": TEST_USERS[1]["login"], "email": TEST_USERS[1]["email"]}
|
|||
for _k, _v in _DEFAULTS.items():
|
||||
os.environ.setdefault(_k, _v)
|
||||
|
||||
os.environ["CONFIGURED_ADMINS"] = "alice,alice@example.com"
|
||||
|
||||
for _d in (TMP, _GH_DIR, _WORK_DIR):
|
||||
_d.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
|
|
|
|||
|
|
@ -69,7 +69,10 @@ def slack_messages(channel: str) -> list[dict[str, Any]]:
|
|||
|
||||
# --- GitHub ----------------------------------------------------------------
|
||||
PULLS: list[dict[str, Any]] = []
|
||||
CHECK_RUNS: list[dict[str, Any]] = []
|
||||
REVIEW_DISPATCHES: list[dict[str, Any]] = []
|
||||
_pr_seq = [0]
|
||||
_check_seq = [0]
|
||||
|
||||
|
||||
def _git(*args: str, cwd: Path | None = None) -> str:
|
||||
|
|
@ -151,6 +154,8 @@ def create_pull(
|
|||
"state": "open",
|
||||
"merged": False,
|
||||
"author": "open-swe[bot]",
|
||||
"head_sha": f"head-{number:04d}",
|
||||
"base_sha": f"base-{number:04d}",
|
||||
"files": files,
|
||||
"additions": sum(f["additions"] for f in files),
|
||||
"deletions": sum(f["deletions"] for f in files),
|
||||
|
|
@ -159,12 +164,56 @@ def create_pull(
|
|||
return pr
|
||||
|
||||
|
||||
def create_review_pull(owner: str, repo: str) -> dict[str, Any]:
|
||||
pr = create_pull(
|
||||
owner,
|
||||
repo,
|
||||
head="feature/review-me",
|
||||
base=BASE_BRANCH,
|
||||
title="Review command fixture",
|
||||
body="A deterministic pull request for reviewer routing.",
|
||||
draft=False,
|
||||
)
|
||||
pr["author"] = "alice"
|
||||
return pr
|
||||
|
||||
|
||||
def find_pull(number: int) -> dict[str, Any] | None:
|
||||
return next((p for p in PULLS if p["number"] == number), None)
|
||||
|
||||
|
||||
def create_check_run(owner: str, repo: str, payload: dict[str, Any]) -> dict[str, Any]:
|
||||
_check_seq[0] += 1
|
||||
check = {
|
||||
"id": _check_seq[0],
|
||||
"owner": owner,
|
||||
"repo": repo,
|
||||
"name": payload.get("name"),
|
||||
"head_sha": payload.get("head_sha"),
|
||||
"status": payload.get("status"),
|
||||
"conclusion": payload.get("conclusion"),
|
||||
"details_url": payload.get("details_url"),
|
||||
"output": payload.get("output", {}),
|
||||
}
|
||||
CHECK_RUNS.append(check)
|
||||
return check
|
||||
|
||||
|
||||
def update_check_run(check_run_id: int, payload: dict[str, Any]) -> dict[str, Any] | None:
|
||||
check = next((item for item in CHECK_RUNS if item["id"] == check_run_id), None)
|
||||
if check is None:
|
||||
return None
|
||||
check.update(
|
||||
{key: payload[key] for key in ("status", "conclusion", "output") if key in payload}
|
||||
)
|
||||
return check
|
||||
|
||||
|
||||
def reset() -> None:
|
||||
SLACK_MESSAGES.clear()
|
||||
PULLS.clear()
|
||||
CHECK_RUNS.clear()
|
||||
REVIEW_DISPATCHES.clear()
|
||||
_pr_seq[0] = 0
|
||||
_check_seq[0] = 0
|
||||
seed_bare_remote()
|
||||
|
|
|
|||
|
|
@ -60,8 +60,11 @@ _SLACK_USERS: dict[str, dict[str, str]] = {
|
|||
}
|
||||
|
||||
from agent.api.app import app # noqa: E402
|
||||
from agent.dashboard import routes as dashboard_routes # noqa: E402
|
||||
from agent.dashboard.oauth import COOKIE_NAME, issue_session # noqa: E402
|
||||
from agent.utils import github_checks # noqa: E402
|
||||
from agent.utils.thread_ids import generate_thread_id_from_slack_thread # noqa: E402
|
||||
from agent.webhooks import common as webhook_common # noqa: E402
|
||||
|
||||
GITHUB_WEBHOOK_SECRET = os.environ["GITHUB_WEBHOOK_SECRET"]
|
||||
SLACK_SIGNING_SECRET = os.environ["SLACK_SIGNING_SECRET"]
|
||||
|
|
@ -70,6 +73,86 @@ STATIC_DIR = Path(__file__).parent / "static"
|
|||
CURRENT_THREAD: dict[str, str | None] = {"channel": DEMO_CHANNEL, "thread_ts": None}
|
||||
|
||||
fakes.seed_bare_remote()
|
||||
_real_dispatch_agent_run = webhook_common.dispatch_agent_run
|
||||
|
||||
|
||||
async def _fake_installation_token(*_args: object, **_kwargs: object) -> str:
|
||||
return "dummy-installation-token"
|
||||
|
||||
|
||||
async def _fake_installation_token_with_expiry(
|
||||
*_args: object, **_kwargs: object
|
||||
) -> tuple[str, None]:
|
||||
return "dummy-installation-token", None
|
||||
|
||||
|
||||
async def _fake_fetch_pr_metadata(pr_ref: Any, *, token: str) -> dict[str, Any] | None: # noqa: ARG001
|
||||
pr = fakes.find_pull(pr_ref.number)
|
||||
return _gh_pr_json(pr) if pr is not None else None
|
||||
|
||||
|
||||
async def _fake_reviewer_token(*_args: object, **_kwargs: object) -> tuple[str, None]:
|
||||
return "dummy-installation-token", None
|
||||
|
||||
|
||||
async def _fake_reaction(*_args: object, **_kwargs: object) -> bool:
|
||||
return True
|
||||
|
||||
|
||||
async def _fake_started_comment(*_args: object, **_kwargs: object) -> None:
|
||||
return None
|
||||
|
||||
|
||||
async def _record_review_dispatch(
|
||||
thread_id: str,
|
||||
prompt: str,
|
||||
configurable: dict[str, Any],
|
||||
*,
|
||||
source: str,
|
||||
assistant_id: str = "agent",
|
||||
**kwargs: object,
|
||||
) -> dict[str, str]:
|
||||
if assistant_id != "reviewer":
|
||||
return await _real_dispatch_agent_run(
|
||||
thread_id,
|
||||
prompt,
|
||||
configurable,
|
||||
source=source,
|
||||
assistant_id=assistant_id,
|
||||
**kwargs,
|
||||
)
|
||||
run_id = f"review-run-{len(fakes.REVIEW_DISPATCHES) + 1}"
|
||||
fakes.REVIEW_DISPATCHES.append(
|
||||
{
|
||||
"thread_id": thread_id,
|
||||
"prompt": prompt,
|
||||
"configurable": configurable,
|
||||
"source": source,
|
||||
"assistant_id": assistant_id,
|
||||
"run_id": run_id,
|
||||
}
|
||||
)
|
||||
return {"run_id": run_id}
|
||||
|
||||
|
||||
async def _fake_installations_and_repos(
|
||||
_login: str,
|
||||
) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
|
||||
return (
|
||||
[{"id": 1, "account": {"login": e2e_env.OWNER, "type": "Organization"}}],
|
||||
[{"full_name": f"{e2e_env.OWNER}/{e2e_env.REPO}", "private": False}],
|
||||
)
|
||||
|
||||
|
||||
webhook_common.get_github_app_installation_token = _fake_installation_token
|
||||
webhook_common.get_github_app_installation_token_with_expiry = _fake_installation_token_with_expiry
|
||||
webhook_common.fetch_github_pr_metadata = _fake_fetch_pr_metadata
|
||||
webhook_common._reviewer_token_for_repo = _fake_reviewer_token
|
||||
webhook_common.react_to_github_comment = _fake_reaction
|
||||
webhook_common.post_review_started_comment = _fake_started_comment
|
||||
webhook_common.dispatch_agent_run = _record_review_dispatch
|
||||
dashboard_routes._fetch_user_installations_and_repos = _fake_installations_and_repos
|
||||
github_checks._GITHUB_API_BASE = e2e_env.FAKE_GITHUB_API
|
||||
|
||||
|
||||
# --- control + Slack compose (the test driver) -----------------------------
|
||||
|
|
@ -80,6 +163,53 @@ async def control_reset() -> JSONResponse:
|
|||
return JSONResponse({"ok": True})
|
||||
|
||||
|
||||
@app.post("/control/review-pr")
|
||||
async def control_review_pr() -> JSONResponse:
|
||||
fakes.reset()
|
||||
pr = fakes.create_review_pull(e2e_env.OWNER, e2e_env.REPO)
|
||||
return JSONResponse(_gh_pr_json(pr))
|
||||
|
||||
|
||||
@app.get("/control/review-dispatches")
|
||||
async def control_review_dispatches() -> JSONResponse:
|
||||
return JSONResponse(fakes.REVIEW_DISPATCHES)
|
||||
|
||||
|
||||
@app.get("/control/check-runs")
|
||||
async def control_check_runs() -> JSONResponse:
|
||||
return JSONResponse(fakes.CHECK_RUNS)
|
||||
|
||||
|
||||
@app.post("/control/review-check/evaluate")
|
||||
async def control_review_check_evaluate(request: Request) -> JSONResponse:
|
||||
body = await request.json()
|
||||
outcome = body.get("outcome")
|
||||
if not isinstance(outcome, dict):
|
||||
raise HTTPException(400, "outcome must be an object")
|
||||
head_sha = str(body.get("head_sha") or f"evaluation-{len(fakes.CHECK_RUNS) + 1}")
|
||||
check_run_id = await github_checks.create_review_check_run(
|
||||
owner=e2e_env.OWNER,
|
||||
repo=e2e_env.REPO,
|
||||
head_sha=head_sha,
|
||||
token="dummy-installation-token",
|
||||
)
|
||||
if check_run_id is None:
|
||||
raise HTTPException(500, "failed to create check run")
|
||||
conclusion, title, summary = github_checks.review_check_conclusion(outcome)
|
||||
completed = await github_checks.complete_review_check_run(
|
||||
owner=e2e_env.OWNER,
|
||||
repo=e2e_env.REPO,
|
||||
check_run_id=check_run_id,
|
||||
token="dummy-installation-token",
|
||||
conclusion=conclusion,
|
||||
title=title,
|
||||
summary=summary,
|
||||
)
|
||||
if not completed:
|
||||
raise HTTPException(500, "failed to complete check run")
|
||||
return JSONResponse({"id": check_run_id, "conclusion": conclusion, "title": title})
|
||||
|
||||
|
||||
@app.get("/control/state")
|
||||
async def control_state() -> JSONResponse:
|
||||
return JSONResponse(
|
||||
|
|
@ -313,6 +443,16 @@ async def ui_agents_plan(thread_id: str) -> FileResponse: # noqa: ARG001
|
|||
return _ui_file("_shell.html")
|
||||
|
||||
|
||||
@app.get("/review", response_class=HTMLResponse)
|
||||
async def ui_review() -> FileResponse:
|
||||
return _ui_file("_shell.html")
|
||||
|
||||
|
||||
@app.get("/review/repositories/{owner}", response_class=HTMLResponse)
|
||||
async def ui_review_repositories(owner: str) -> FileResponse: # noqa: ARG001
|
||||
return _ui_file("_shell.html")
|
||||
|
||||
|
||||
@app.get("/login", response_class=HTMLResponse)
|
||||
async def ui_login() -> FileResponse:
|
||||
return _ui_file("_shell.html")
|
||||
|
|
@ -406,6 +546,8 @@ async def mock_github_pr(owner: str, repo: str, number: int) -> HTMLResponse: #
|
|||
|
||||
# --- fake GitHub REST API (open_pull_request hits this) --------------------
|
||||
def _gh_pr_json(pr: dict[str, Any]) -> dict[str, Any]:
|
||||
full_name = f"{pr['owner']}/{pr['repo']}"
|
||||
repo = {"id": 1, "full_name": full_name, "private": False}
|
||||
return {
|
||||
"number": pr["number"],
|
||||
"html_url": _pr_html_url(pr),
|
||||
|
|
@ -415,8 +557,8 @@ def _gh_pr_json(pr: dict[str, Any]) -> dict[str, Any]:
|
|||
"title": pr["title"],
|
||||
"body": pr["body"],
|
||||
"user": {"login": pr["author"]},
|
||||
"head": {"ref": pr["head"]},
|
||||
"base": {"ref": pr["base"]},
|
||||
"head": {"ref": pr["head"], "sha": pr["head_sha"], "repo": repo},
|
||||
"base": {"ref": pr["base"], "sha": pr["base_sha"], "repo": repo},
|
||||
"additions": pr["additions"],
|
||||
"deletions": pr["deletions"],
|
||||
"changed_files": len(pr["files"]),
|
||||
|
|
@ -463,6 +605,26 @@ async def gh_get_pull(owner: str, repo: str, number: int) -> JSONResponse: # no
|
|||
return JSONResponse(_gh_pr_json(pr))
|
||||
|
||||
|
||||
@app.post("/fake-gh/repos/{owner}/{repo}/check-runs")
|
||||
async def gh_create_check_run(owner: str, repo: str, request: Request) -> JSONResponse:
|
||||
payload = await request.json()
|
||||
return JSONResponse(fakes.create_check_run(owner, repo, payload), status_code=201)
|
||||
|
||||
|
||||
@app.patch("/fake-gh/repos/{owner}/{repo}/check-runs/{check_run_id}")
|
||||
async def gh_update_check_run(
|
||||
owner: str,
|
||||
repo: str,
|
||||
check_run_id: int,
|
||||
request: Request, # noqa: ARG001
|
||||
) -> JSONResponse:
|
||||
payload = await request.json()
|
||||
check = fakes.update_check_run(check_run_id, payload)
|
||||
if check is None:
|
||||
return JSONResponse({"message": "Not Found"}, status_code=404)
|
||||
return JSONResponse(check)
|
||||
|
||||
|
||||
# --- fake Slack API (real slack code hits this) ----------------------------
|
||||
def _ok(extra: dict[str, Any] | None = None) -> JSONResponse:
|
||||
return JSONResponse({"ok": True, **(extra or {})})
|
||||
|
|
|
|||
|
|
@ -56,6 +56,85 @@ async function expectTranscriptVisible(page: Page) {
|
|||
}).toPass({ timeout: 60000 });
|
||||
}
|
||||
|
||||
test.describe("Review verdict settings (real dashboard UI and API)", () => {
|
||||
test("admin controls persist team and per-repository auto-verdict policy", async ({
|
||||
page,
|
||||
baseURL,
|
||||
}) => {
|
||||
await loginAs(page, SAME_USER);
|
||||
const mutationHeaders = { origin: baseURL ?? "" };
|
||||
|
||||
const resetTeam = await page.request.put("/dashboard/api/team-settings", {
|
||||
data: { auto_verdict: false },
|
||||
headers: mutationHeaders,
|
||||
});
|
||||
expect(resetTeam.ok(), await resetTeam.text()).toBeTruthy();
|
||||
const resetRepo = await page.request.put(
|
||||
"/dashboard/api/auto-verdict-repos",
|
||||
{
|
||||
data: { full_name: "fakeorg/demo", enabled: false },
|
||||
headers: mutationHeaders,
|
||||
},
|
||||
);
|
||||
expect(resetRepo.ok()).toBeTruthy();
|
||||
|
||||
await page.goto("/review");
|
||||
await expect(
|
||||
page.getByRole("heading", { name: "Open SWE Review" }),
|
||||
).toBeVisible();
|
||||
const teamToggle = page.getByRole("switch").first();
|
||||
await expect(teamToggle).not.toBeChecked();
|
||||
const teamSaved = page.waitForResponse(
|
||||
(response) =>
|
||||
response.url().endsWith("/dashboard/api/team-settings") &&
|
||||
response.request().method() === "PUT",
|
||||
);
|
||||
await teamToggle.click();
|
||||
expect((await teamSaved).ok()).toBeTruthy();
|
||||
await expect(teamToggle).toBeChecked();
|
||||
|
||||
const teamContract = await (
|
||||
await page.request.get("/dashboard/api/team-settings")
|
||||
).json();
|
||||
expect(teamContract.auto_verdict).toBe(true);
|
||||
|
||||
await page.goto("/review/repositories/fakeorg");
|
||||
const repoToggle = page.getByRole("switch", {
|
||||
name: "Allow automatic verdicts for fakeorg/demo",
|
||||
});
|
||||
await expect(repoToggle).not.toBeChecked();
|
||||
const repoSaved = page.waitForResponse(
|
||||
(response) =>
|
||||
response.url().endsWith("/dashboard/api/auto-verdict-repos") &&
|
||||
response.request().method() === "PUT",
|
||||
);
|
||||
await repoToggle.click();
|
||||
expect((await repoSaved).ok()).toBeTruthy();
|
||||
await expect(repoToggle).toBeChecked();
|
||||
|
||||
const repoContract = await (
|
||||
await page.request.get("/dashboard/api/auto-verdict-repos")
|
||||
).json();
|
||||
expect(repoContract).toEqual({ repos: ["fakeorg/demo"] });
|
||||
|
||||
await loginAs(page, OTHER_USER);
|
||||
const forbidden = await page.request.put(
|
||||
"/dashboard/api/auto-verdict-repos",
|
||||
{
|
||||
data: { full_name: "fakeorg/demo", enabled: false },
|
||||
headers: mutationHeaders,
|
||||
},
|
||||
);
|
||||
expect(forbidden.status()).toBe(403);
|
||||
await page.reload();
|
||||
await expect(
|
||||
page.getByRole("switch", {
|
||||
name: "Allow automatic verdicts for fakeorg/demo",
|
||||
}),
|
||||
).toBeDisabled();
|
||||
});
|
||||
});
|
||||
|
||||
test.describe("Slack → web handoff (real dashboard UI)", () => {
|
||||
test("the SAME user continues the conversation in the web app", async ({
|
||||
page,
|
||||
|
|
|
|||
145
tests/e2e/tests/reviewer_verification.spec.ts
Normal file
145
tests/e2e/tests/reviewer_verification.spec.ts
Normal file
|
|
@ -0,0 +1,145 @@
|
|||
import { createHmac } from "node:crypto";
|
||||
|
||||
import { expect, test } from "@playwright/test";
|
||||
|
||||
const WEBHOOK_SECRET = "test-github-secret";
|
||||
|
||||
test.describe("Reviewer verification contracts", () => {
|
||||
test("direct @openswe review routes to the reviewer without user mapping", async ({
|
||||
page,
|
||||
}) => {
|
||||
const seeded = await page.request.post("/control/review-pr");
|
||||
expect(seeded.ok()).toBeTruthy();
|
||||
const pr = await seeded.json();
|
||||
const payload = {
|
||||
action: "created",
|
||||
repository: {
|
||||
id: 1,
|
||||
name: "demo",
|
||||
full_name: "fakeorg/demo",
|
||||
private: true,
|
||||
owner: { login: "fakeorg" },
|
||||
},
|
||||
issue: {
|
||||
number: pr.number,
|
||||
html_url: pr.html_url,
|
||||
pull_request: { html_url: pr.html_url },
|
||||
},
|
||||
comment: {
|
||||
id: 24601,
|
||||
node_id: "IC_kwDO_e2e",
|
||||
body: "@openswe review: focus on authorization boundaries",
|
||||
user: { login: "unmapped-reviewer" },
|
||||
},
|
||||
sender: { id: 9001, login: "unmapped-reviewer" },
|
||||
};
|
||||
const raw = JSON.stringify(payload);
|
||||
const signature = `sha256=${createHmac("sha256", WEBHOOK_SECRET)
|
||||
.update(raw)
|
||||
.digest("hex")}`;
|
||||
|
||||
const webhook = await page.request.post("/webhooks/github", {
|
||||
data: raw,
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
"X-GitHub-Event": "issue_comment",
|
||||
"X-Hub-Signature-256": signature,
|
||||
},
|
||||
});
|
||||
expect(webhook.ok()).toBeTruthy();
|
||||
expect(await webhook.json()).toEqual({
|
||||
status: "accepted",
|
||||
message: "Processing on-demand PR review",
|
||||
});
|
||||
|
||||
await expect
|
||||
.poll(async () => {
|
||||
const response = await page.request.get("/control/review-dispatches");
|
||||
return (await response.json()).length;
|
||||
})
|
||||
.toBe(1);
|
||||
const dispatches = await (
|
||||
await page.request.get("/control/review-dispatches")
|
||||
).json();
|
||||
expect(dispatches).toHaveLength(1);
|
||||
expect(dispatches[0].assistant_id).toBe("reviewer");
|
||||
expect(dispatches[0].source).toBe("github_comment");
|
||||
expect(dispatches[0].configurable).toMatchObject({
|
||||
github_login: "unmapped-reviewer",
|
||||
github_user_id: 9001,
|
||||
review_requested: true,
|
||||
verdict_requested: true,
|
||||
repo: { owner: "fakeorg", name: "demo" },
|
||||
pr_number: pr.number,
|
||||
});
|
||||
expect(dispatches[0].configurable).not.toHaveProperty("email");
|
||||
expect(dispatches[0].prompt).toContain(
|
||||
"focus on authorization boundaries",
|
||||
);
|
||||
|
||||
});
|
||||
|
||||
test("review outcomes settle checks as success, failure, or neutral", async ({
|
||||
page,
|
||||
}) => {
|
||||
await page.request.post("/control/reset");
|
||||
const cases = [
|
||||
{
|
||||
outcome: {
|
||||
verdict_submitted: true,
|
||||
verdict_event: "APPROVE",
|
||||
blocking_finding_count: 0,
|
||||
},
|
||||
conclusion: "success",
|
||||
title: "Review approved",
|
||||
},
|
||||
{
|
||||
outcome: {
|
||||
verdict_submitted: true,
|
||||
verdict_event: "REQUEST_CHANGES",
|
||||
blocking_finding_count: 1,
|
||||
},
|
||||
conclusion: "failure",
|
||||
title: "Changes requested",
|
||||
},
|
||||
{
|
||||
outcome: {
|
||||
verdict_submitted: false,
|
||||
verdict_ignored_reason: "verdict_not_requested",
|
||||
blocking_finding_count: 0,
|
||||
},
|
||||
conclusion: "neutral",
|
||||
title: "Verdict withheld",
|
||||
},
|
||||
];
|
||||
|
||||
for (const [index, item] of cases.entries()) {
|
||||
const response = await page.request.post(
|
||||
"/control/review-check/evaluate",
|
||||
{
|
||||
data: {
|
||||
head_sha: `tri-state-${index + 1}`,
|
||||
outcome: item.outcome,
|
||||
},
|
||||
},
|
||||
);
|
||||
expect(response.ok()).toBeTruthy();
|
||||
expect(await response.json()).toMatchObject({
|
||||
conclusion: item.conclusion,
|
||||
title: item.title,
|
||||
});
|
||||
}
|
||||
|
||||
const checks = await (
|
||||
await page.request.get("/control/check-runs")
|
||||
).json();
|
||||
expect(checks.map((check: { conclusion: string }) => check.conclusion)).toEqual([
|
||||
"success",
|
||||
"failure",
|
||||
"neutral",
|
||||
]);
|
||||
expect(
|
||||
checks.map((check: { status: string }) => check.status),
|
||||
).toEqual(["completed", "completed", "completed"]);
|
||||
});
|
||||
});
|
||||
Loading…
Add table
Reference in a new issue