diff --git a/.claude/workflows/domain-reorg-adoption.mjs b/.claude/workflows/domain-reorg-adoption.mjs new file mode 100644 index 00000000..e3005038 --- /dev/null +++ b/.claude/workflows/domain-reorg-adoption.mjs @@ -0,0 +1,229 @@ +export const meta = { + name: 'domain-reorg-adoption', + description: 'Execute the gate-approved C1-C7 domain-reorg build plan: moves, webapp split, security review, PR — no merge', + whenToUse: 'Run once, after Adam gives the go for the reorg build. Requires clean dev; plan + scoping artifacts live in docs/upstream-sync/domain-reorg/.', + phases: [ + { title: 'Preflight', detail: 'clean tree, artifacts present, branch created', model: 'haiku' }, + { title: 'C1 docs/resources', detail: 'doc moves + prompt loader + wheel check', model: 'sonnet' }, + { title: 'C2 review package', detail: '9 module moves + 38 importer rewrites', model: 'sonnet' }, + { title: 'C3 shims+manifest', detail: '21 shims, ci_monitor, langgraph.json, bun pin', model: 'sonnet' }, + { title: 'C4 webapp split', detail: 'the critical auth-surface split', model: 'fable' }, + { title: 'C4 security review', detail: 'detector fan-out + proof-or-kill verify' }, + { title: 'C5 test moves', detail: 'tests// mechanical moves', model: 'sonnet' }, + { title: 'C6 UI moves', detail: 'ui/src/features/ moves + import rewrites', model: 'sonnet' }, + { title: 'C7 docs/ledger/memory', detail: 'docs, triage ledger, memory, Confluence check', model: 'opus' }, + { title: 'PR', detail: 'push branch, open PR, final report — NO merge' }, + ], +} + +// --------------------------------------------------------------------------- +// Shared context strings +// --------------------------------------------------------------------------- +const REPO = '/Users/adammoussa/Documents/repositories/seahaven/open-swe' +const SCRATCH = '/Users/adammoussa/Documents/repositories/seahaven/open-swe/docs/upstream-sync/domain-reorg' +const BRANCH = 'refactor/domain-reorg-adoption' + +const COMMON = ` +You are executing one commit of a gate-approved build plan in ${REPO} (branch ${BRANCH}). +Plan: ${SCRATCH}/reorg-build-plan.md — READ IT FIRST, plus the section of ${SCRATCH}/reorg-scoping-report.md relevant to your commit. +Move-map artifacts: ${SCRATCH}/movemap-m50.txt, ${SCRATCH}/cross.json, ${SCRATCH}/forkonly.json. +HARD RULES (violations = abort and report, do not improvise): +- Fork file contents, upstream layout. Upstream content ONLY for the 21 structural "A" shims listed in the scoping report. +- NEVER adopt upstream content for agent/webhooks/{github,linear,slack}.py, tests/conftest.py, tests/e2e/harness.py. +- Use "git mv" for moves so rename tracking survives. One commit for your cascade class only; commit message follows the repo's conventional style (see recent git log), body references the plan step. +- No pushes from your stage. No force operations. No edits outside your commit's scope. Do not touch docs/upstream-sync/* unless your stage says so. +- Gate ladder before committing: ruff check && ruff format --check, then uv run pytest --co -q, then the stage-specific gates. If a gate fails, fix within scope; if you cannot, ABORT: git reset --hard to the pre-stage SHA you recorded at start, and report the failure. +Return JSON: {ok, commit_sha, gates: {name: pass|fail|skipped}, notes, aborted_reason}` + +const RESULT_SCHEMA = { + type: 'object', + properties: { + ok: { type: 'boolean' }, + commit_sha: { type: 'string' }, + gates: { type: 'object' }, + notes: { type: 'string' }, + aborted_reason: { type: 'string' }, + }, + required: ['ok', 'notes'], +} + +const results = { commits: [], securityReview: null, pr: null } + +function ensure(stage, r) { + if (!r || !r.ok) { + throw new Error(`${stage} failed: ${r ? r.aborted_reason || r.notes : 'agent returned null'}`) + } + results.commits.push({ stage, sha: r.commit_sha, gates: r.gates, notes: r.notes }) + return r +} + +// --------------------------------------------------------------------------- +// Preflight — cheap checks, haiku +// --------------------------------------------------------------------------- +phase('Preflight') +const pre = await agent( + `${COMMON} +STAGE: Preflight (no commit). +1. Verify ${REPO} is on dev and dev == origin/dev (fetch first is allowed here). Tree must be clean EXCEPT untracked files under docs/upstream-sync/domain-reorg/ and .claude/workflows/, which are expected (they are committed in C7). +2. Verify the plan file, scoping report, and all three move-map artifacts exist and are non-empty. +3. Verify branch ${BRANCH} does not already exist locally or on origin. +4. Create ${BRANCH} off dev. Record the base SHA. +Return JSON with ok, notes, and base_sha in notes.`, + { label: 'preflight', phase: 'Preflight', model: 'haiku', effort: 'low', schema: RESULT_SCHEMA }, +) +if (!pre || !pre.ok) throw new Error(`Preflight failed: ${pre ? pre.notes : 'null'}`) +log('Preflight OK — branch created') + +// --------------------------------------------------------------------------- +// C1–C3: independent-of-C4 stages, sequential commits on the shared branch +// --------------------------------------------------------------------------- +phase('C1 docs/resources') +ensure('C1', await agent( + `${COMMON} +STAGE: C1 — docs/resources/assets (plan section "C1"). +Moves: INSTALLATION.md and CUSTOMIZATION.md -> docs/, static/ -> assets/ (fix README refs), default_prompt.md -> agent/resources/ (+ __init__.py). Apply the importlib.resources loader hunk to the fork's prompt.py exactly as described in scoping §2b (upstream's hunk pattern, fork's file). +Extra gate: build the wheel (uv build or python -m build) and verify agent/resources/default_prompt.md is inside it; if missing, add the explicit hatchling include and note it.`, + { label: 'C1', phase: 'C1 docs/resources', model: 'sonnet', schema: RESULT_SCHEMA }, +)) + +phase('C2 review package') +ensure('C2', await agent( + `${COMMON} +STAGE: C2 — agent/review/ package (plan section "C2", scoping §2a/2b reviewer rows). +git mv the 9 reviewer/style modules per the move-map (reviewer_findings.py -> agent/review/findings.py etc.), replicate upstream's import-rewire pattern on FORK content for the R<100 ones, rewrite all 38 importer files (grep for reviewer_ and review_style_ imports to find them — do not trust the count blindly), create agent/review/__init__.py exporting the fork's existing public surface. +Extra gate: full reviewer/findings unit suites pass (uv run pytest tests/ -k "review or finding" plus any suite the moved modules own).`, + { label: 'C2', phase: 'C2 review package', model: 'sonnet', schema: RESULT_SCHEMA }, +)) + +phase('C3 shims+manifest') +ensure('C3', await agent( + `${COMMON} +STAGE: C3 — graphs/runtime/providers shims + langgraph.json + CI hardening (plan section "C3"). +Adopt the 21 upstream "A" shim files verbatim-then-verify: for each, confirm every import it makes resolves against FORK modules (they delegate to agent.server etc.); fix delegation targets if fork names differ. Add fork-only agent/graphs/ci_monitor.py shim following the same pattern. Retarget langgraph.json graph entrypoints to agent.graphs.* (http.app stays agent.webapp:app). Pin bun-version in .github/workflows/ci.yml to the version in ui/ (check .bun-version / package.json engines; else current stable — note which). +Extra gate: timeout-bounded "make dev" boot check — all six graphs (agent, reviewer, analyzer, chat, scheduler, ci_monitor) register and the FastAPI app mounts; kill it after verifying startup logs.`, + { label: 'C3', phase: 'C3 shims+manifest', model: 'sonnet', schema: RESULT_SCHEMA }, +)) + +// --------------------------------------------------------------------------- +// C4 — the critical commit. Strongest model, high effort. +// --------------------------------------------------------------------------- +phase('C4 webapp split') +ensure('C4', await agent( + `${COMMON} +STAGE: C4 — FastAPI split (plan section "C4" + approved decisions 1-2). THE critical auth-surface commit. +Split the fork's 2,590-line agent/webapp.py into: agent/webhooks/common.py (shared verify/dispatch helpers), agent/api/app.py (composition), agent/api/health.py (/health + /webhooks/run-complete), and per-source github_routes.py / linear_routes.py / slack_routes.py / jira_routes.py / confluence_routes.py. Atlassian Connect lifecycle + descriptor routes (/connect/*) go INTO confluence_routes.py (approved decision 2). webapp.py becomes the compatibility shim re-exporting app (mirror upstream's shim shape). +Preserve EXACTLY: all signature-verification logic (GitHub HMAC, Slack, Linear, verify_jira_secret + optional HMAC/timestamp, Connect JWT/qsh), token-attribution gating, TID-COLLIDE-01 binding, the renamed _is_repo_auto_review_enabled gates, ci-autofix trigger wiring. Handlers keep module-attribute access style (common.X / service.X). +Retarget the ~240 monkeypatch.setattr(webapp, ...) sites across the 11 test files + tests/conftest.py + tests/e2e/harness.py per upstream's retarget pattern (webhook_common / handler modules) — find every site by grep, not by count. +Extra gates before committing: full unit suite; FULL E2E (Playwright + real langgraph dev server); residual-importer sweep — git grep for "agent.webapp"/"from agent import webapp" must return only the shim and intentional compat references (list what remains in notes).`, + { label: 'C4', phase: 'C4 webapp split', model: 'fable', effort: 'high', schema: RESULT_SCHEMA }, +)) +log('C4 committed — entering security review (hard gate before C5)') + +// --------------------------------------------------------------------------- +// C4 security review — detector fan-out (sonnet lenses) + proof-or-kill verifier (opus), +// with a bounded fix loop (fable) per the approved failure path (new commits, no rewrites). +// --------------------------------------------------------------------------- +phase('C4 security review') +const LENSES = [ + { key: 'authz', focus: 'broken object-level auth, missing access checks, webhook signature verification gaps (GitHub HMAC / Slack / Linear / Jira shared-secret+HMAC / Connect JWT+qsh), token-attribution gating, TID-COLLIDE-01 thread binding, the _is_repo_auto_review_enabled gates' }, + { key: 'injection', focus: 'untrusted webhook payload handling: parsing, header trust, SSRF-prone fetches, path handling in the moved dispatch code' }, + { key: 'logic', focus: 'multi-step invariants broken by the split: dispatch ordering, dedup/replay protection, race conditions across the new module boundaries, dropped code paths (diff every moved function against its pre-split source)' }, +] +const FINDINGS_SCHEMA = { + type: 'object', + properties: { findings: { type: 'array', items: { type: 'object' } } }, + required: ['findings'], +} +const VERDICT_SCHEMA = { + type: 'object', + properties: { confirmed: { type: 'array', items: { type: 'object' } }, unverified: { type: 'array', items: { type: 'object' } } }, + required: ['confirmed', 'unverified'], +} + +let reviewRound = 0 +let confirmedHighs = [] +while (true) { + reviewRound += 1 + const c4diff = `the C4 split commit(s) on ${BRANCH} in ${REPO} (git show for each C4/C4a commit; compare moved code against pre-split agent/webapp.py at the merge base)` + const found = (await parallel(LENSES.map(l => () => + agent( + `You are a hostile ${l.key} security auditor. Audit ONLY ${l.key} issues in ${c4diff}. Focus: ${l.focus}. Read the actual files at both revisions; for each checklist area either cite the specific safe line or file a finding {id,title,claimed_severity,file,line,data_flow,proof:{input,outcome},recommendation}. Findings must be about the SPLIT (regressions vs pre-split behavior), not pre-existing issues. Return {findings:[...]}.`, + { label: `detect:${l.key}:r${reviewRound}`, phase: 'C4 security review', model: 'sonnet', schema: FINDINGS_SCHEMA }, + )))).filter(Boolean).flatMap(r => r.findings) + + if (!found.length) { results.securityReview = { rounds: reviewRound, confirmed: [], outcome: 'clean' }; break } + + const verdict = await agent( + `You are a skeptical exploitation verifier — you did NOT find these and are rewarded for killing weak claims. For each candidate finding on ${c4diff}, demand a concrete proof-of-exploit (specific input -> specific bad outcome, consistent with the code at HEAD of ${BRANCH}). Default to unverified when uncertain. Candidates: ${JSON.stringify(found)}. Return {confirmed:[...], unverified:[...]} keeping severity only on confirmed items.`, + { label: `verify:r${reviewRound}`, phase: 'C4 security review', model: 'opus', effort: 'high', schema: VERDICT_SCHEMA }, + ) + confirmedHighs = (verdict?.confirmed || []).filter(f => ['critical', 'high'].includes((f.claimed_severity || f.severity || '').toLowerCase())) + if (!confirmedHighs.length) { + results.securityReview = { rounds: reviewRound, confirmed: verdict?.confirmed || [], unverified: verdict?.unverified || [], outcome: 'pass' } + break + } + if (reviewRound >= 3) { + results.securityReview = { rounds: reviewRound, confirmed: confirmedHighs, outcome: 'BLOCKED' } + throw new Error(`Security review still has ${confirmedHighs.length} confirmed critical/high after ${reviewRound} rounds — plan failure path: fundamental-flaw handling, return to Adam. Branch left intact for inspection.`) + } + log(`Security round ${reviewRound}: ${confirmedHighs.length} confirmed critical/high — dispatching fix commit C4a`) + ensure(`C4a-r${reviewRound}`, await agent( + `${COMMON} +STAGE: C4a — address confirmed security-review findings as a NEW commit on ${BRANCH} (plan failure path: no history rewrite, no force-push). Findings with proofs: ${JSON.stringify(confirmedHighs)}. Fix each within the split's scope; re-run full unit + E2E before committing.`, + { label: `C4a:r${reviewRound}`, phase: 'C4 security review', model: 'fable', effort: 'high', schema: RESULT_SCHEMA }, + )) +} +log(`Security review outcome: ${results.securityReview.outcome} after ${results.securityReview.rounds} round(s)`) + +// --------------------------------------------------------------------------- +// HARD GATE passed — C5 and C6 (sequential commits; C6 depends only on branch order, not C5 content) +// --------------------------------------------------------------------------- +phase('C5 test moves') +ensure('C5', await agent( + `${COMMON} +STAGE: C5 — tests// moves (plan section "C5", scoping §2a + §2c placements). +git mv every test per the move-map; apply remaining R<100 monkeypatch retargets not already done in C4; place the 13 fork-only tests per the approved table (webhooks/, auth/, tools/, sandbox/, github/). Path-only commit — zero logic edits. +Extra gate: uv run pytest --co -q collects everything (no import errors, count matches pre-move) then full unit green.`, + { label: 'C5', phase: 'C5 test moves', model: 'sonnet', effort: 'low', schema: RESULT_SCHEMA }, +)) + +phase('C6 UI moves') +ensure('C6', await agent( + `${COMMON} +STAGE: C6 — ui/src/features/ moves (plan section "C6"). +git mv per the move-map including ported/ -> features/agents/experiments|chat; rewrite @/components/agents and @/lib/agents imports across the 54 importer files AND the moved files themselves (grep-driven); swap the tsconfig/eslint exclude lists for the single experiments glob; retarget the AgentPromptBar shim; delete ui/pnpm-workspace.yaml. Fork-diverged files (PlanReview.tsx, WorkflowApprovalCard.tsx, AgentsSidebar.tsx, SidebarFilterMenu.tsx, AgentGitPanel.tsx, AutomationEditor.tsx, DiffView.tsx, CloudPromptBar.tsx) keep fork content — imports only. +Extra gate: cd ui && bun install && bunx tsc --noEmit && bun run build, then Playwright E2E.`, + { label: 'C6', phase: 'C6 UI moves', model: 'sonnet', schema: RESULT_SCHEMA }, +)) + +phase('C7 docs/ledger/memory') +ensure('C7', await agent( + `${COMMON} +STAGE: C7 — docs + ledger + memory (plan section "C7" — read its obligations verbatim; you own them). +1. Update fork CLAUDE.md / README.md / AGENTS.md path references (architecture sections naming agent/webapp.py, old test paths, ui paths). +2. Ledger: flip 8356eb34 to landed in docs/upstream-sync/triage.jsonl with branch ${BRANCH}; add re-triage notes to the 8 unblocked rows per scoping §5; make triage-render; make triage-check must pass. +3. Memory: update /Users/adammoussa/.claude/projects/-Users-adammoussa-Documents-repositories-seahaven-open-swe/memory/open-swe-upstream-triage-status.md (+ MEMORY.md hook if needed) with the new module layout summary (agent/{graphs,runtime,api,review,resources,webhooks/*_routes}, tests//, ui/src/features/) and the reorg-landed status. Memory edits are NOT part of the git commit. +4. Confluence check: via ToolSearch load the Atlassian MCP search tools; search the IT space for any page documenting this repo's module map. If found, update it; if none, or if Confluence is unreachable, record the exact outcome string ("updated page " / "no Confluence change required" / "OUTSTANDING as of — Confluence inaccessible") in your notes AND in the memory file. +5. Stage docs/upstream-sync/domain-reorg/ and .claude/workflows/domain-reorg-adoption.mjs so the plan, scoping artifacts, and this workflow ship with the exercise. +Commit all repo-file changes as the C7 commit; run ruff + triage-check as gates.`, + { label: 'C7', phase: 'C7 docs/ledger/memory', model: 'opus', schema: RESULT_SCHEMA }, +)) + +// --------------------------------------------------------------------------- +// PR — push and open, NO merge. Merge is Adam's (after @openswe review). +// --------------------------------------------------------------------------- +phase('PR') +results.pr = await agent( + `${COMMON} +STAGE: PR (no commit). Push ${BRANCH} to origin (plain push — the pre-push hook runs security scanners; if it blocks, report, never bypass). Open a PR against dev with gh pr create --base dev. Title: "refactor: adopt upstream domain reorg (8356eb34) — fork content, upstream layout". Body MUST include: the commit-by-commit summary from this build (paste from your reading of git log), explicit review-scope callouts for C2 (reviewer-module moves) and C4 (auth-surface split), the security-review outcome (${JSON.stringify(results.securityReview?.outcome)}), the C7 Confluence-check outcome, the post-deploy verification checklist from the plan (as unchecked boxes), and "MERGE-COMMIT ONLY — do not squash/rebase". Do NOT merge, do NOT approve, do NOT comment @openswe — Adam drives the review. +Return JSON: {ok, notes} with the PR URL in notes.`, + { label: 'open-pr', phase: 'PR', model: 'sonnet', schema: RESULT_SCHEMA }, +) + +return { + branch: BRANCH, + commits: results.commits, + securityReview: results.securityReview, + pr: results.pr?.notes || 'PR step failed — branch is pushed or local; open manually', + next: 'Adam: review PR (C2+C4 scope), run @openswe review, merge with a MERGE COMMIT, then run the post-deploy checklist and post results to the PR.', +} diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 8a3a52a8..ba8ad643 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -45,6 +45,8 @@ jobs: steps: - uses: actions/checkout@v7 - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2 + with: + bun-version: "1.3.14" - name: Install UI dependencies working-directory: ui run: bun install --frozen-lockfile @@ -74,6 +76,8 @@ jobs: with: node-version: "24" - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2 + with: + bun-version: "1.3.14" - name: Install Python deps (langgraph dev runtime) run: uv sync --locked - name: Install Playwright + Chromium @@ -132,6 +136,8 @@ jobs: steps: - uses: actions/checkout@v7 - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2 + with: + bun-version: "1.3.14" - name: ui/package.json and ui/bun.lock agree working-directory: ui run: bun install --frozen-lockfile diff --git a/AGENTS.md b/AGENTS.md index 5e1aacb7..a022b123 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -14,25 +14,27 @@ Dependencies are managed with **uv**. Tests use pytest (`asyncio_mode = "auto"`) ```bash make install # uv pip install -e . -make dev # uv run langgraph dev — serves all three graphs + the FastAPI app from langgraph.json +make dev # uv run langgraph dev — serves all six graphs + the FastAPI app from langgraph.json make run # uvicorn agent.webapp:app --reload --port 8000 (FastAPI only, no LangGraph runtime) make test # uv run pytest -vvv tests/ -make test TEST_FILE=tests/test_open_pr_middleware.py # single test file -uv run pytest -vvv tests/test_open_pr_middleware.py::test_name # single test +make test TEST_FILE=tests/github/test_open_pull_request.py # single test file +uv run pytest -vvv tests/github/test_open_pull_request.py::test_name # single test make lint # ruff check + ruff format --diff make format # ruff format + ruff check --fix ``` -`langgraph.json` declares three graph entrypoints and the FastAPI app, all served together by `langgraph dev`: +`langgraph.json` declares six graph entrypoints and the FastAPI app, all served together by `langgraph dev`. Every graph entrypoint targets a thin `agent.graphs.*` re-export shim that delegates to the unmoved factory module (domain reorg): | Graph | Entrypoint | Purpose | |---|---|---| -| `agent` | `agent.server:get_agent` | Main coding agent (Slack/Linear/Jira/Confluence/GitHub-triggered). | -| `reviewer` | `agent.reviewer:get_reviewer_agent` | Read-only PR reviewer. Findings model + `publish_review`. | -| `analyzer` | `agent.analyzer:get_analyzer` | Learns per-repo reviewer style from historical PRs and this reviewer's own finding outcomes. | -| `ci_monitor` | `agent.ci_monitor:get_ci_monitor` | Polling fallback for CI auto-fix: each tick sweeps open agent-authored PRs for failing checks / merge conflicts via `agent.ci_autofix.sweep_open_prs`. | +| `agent` | `agent.graphs.agent:traced_agent` (shim → `agent.server:get_agent`) | Main coding agent (Slack/Linear/Jira/Confluence/GitHub-triggered). | +| `reviewer` | `agent.graphs.reviewer:traced_reviewer_agent` (shim → `agent.reviewer:get_reviewer_agent`) | Read-only PR reviewer. Findings model + `publish_review`. | +| `analyzer` | `agent.graphs.analyzer:traced_analyzer` (shim → `agent.analyzer:get_analyzer`) | Learns per-repo reviewer style from historical PRs and this reviewer's own finding outcomes. | +| `chat` | `agent.graphs.chat:traced_chat_agent` | Dashboard Agents chat graph. | +| `scheduler` | `agent.graphs.scheduler:get_scheduler` | Reconcile sweep for stragglers. | +| `ci_monitor` | `agent.graphs.ci_monitor:get_ci_monitor` (shim → `agent.ci_monitor:get_ci_monitor`) | Polling fallback for CI auto-fix: each tick sweeps open agent-authored PRs for failing checks / merge conflicts via `agent.ci_autofix.sweep_open_prs`. | -The FastAPI app is `agent.webapp:app`. +The FastAPI app is `agent.webapp:app` — now a compatibility shim re-exporting `agent.api.app:app`. CI auto-fix ("PR babysitting") lives in `agent/ci_autofix.py`: when a CI check fails (webhook `check_run` / `check_suite` / `workflow_run` / `status`) or a reviewer leaves actionable feedback on a PR Open SWE opened, it locates the originating agent thread (by `pr_url` metadata) and dispatches a confidence-gated fix run on the `agent` graph. Gated by the per-user `auto_fix_ci` profile flag, the enabled-repos opt-in, and a per-PR `@open-swe autofix on|off` toggle (`agent/dashboard/autofix_state.py`). Skip-rules (base-branch failures, human commits, same-head dedupe, batching while runs are active, loop cap) all live in `ci_autofix.py`. @@ -41,9 +43,9 @@ CI auto-fix ("PR babysitting") lives in `agent/ci_autofix.py`: when a CI check f ### Entrypoints - **`agent/server.py` → `get_agent(config)`** — main graph factory. Called per-thread. Resolves the GitHub token, gets-or-creates the sandbox for the thread, resolves the team/profile/per-thread model + effort, then constructs a fresh `create_deep_agent(...)` with the curated tool list and middleware stack. The agent itself is stateless — all per-thread state lives in the sandbox + thread metadata. -- **`agent/reviewer.py` → `get_reviewer_agent(config)`** — reviewer graph factory. Shares `ensure_sandbox_for_thread` with the main agent but wires a reviewer-only toolset (`add_finding`, `update_finding`, `list_findings`, `publish_review`, `web_search`, `fetch_url`, `http_request`) and a different system prompt that pins the single-evolving-findings model and the diff-anchored bar for filing a finding. Read-only: no commit/push/PR-opening tools. +- **`agent/reviewer.py` → `get_reviewer_agent(config)`** — reviewer graph factory. Shares `ensure_sandbox_for_thread` with the main agent but wires a reviewer-only toolset (`add_finding`, `update_finding`, `list_findings`, `publish_review`, `web_search`, `fetch_url`, `http_request`) and a different system prompt that pins the single-evolving-findings model and the diff-anchored bar for filing a finding. Read-only: no commit/push/PR-opening tools. Its supporting modules live in the `agent/review/` package (domain reorg): `findings.py`, `publish.py`, `reconcile.py`, `trace_context.py`, `diff.py`, `groups.py`, `eval_store.py`, `style_collector.py`, `style_guidance.py`. - **`agent/analyzer.py` → `get_analyzer(config)`** — small graph that emits a per-repo style prompt via the `save_review_style_prompt` tool, consumed by the reviewer as a "repository-specific review style" appendix. It runs in one of two modes (`analyzer_mode` in `configurable`): **bootstrap** (cold-start: crawl historical PR reviews) and **continual** (nightly: refine using this reviewer's own finding outcomes via `read_finding_outcomes`). Each mode's procedure lives in a deepagents **skill** (`agent/skills/bootstrap-repo-analysis/`, `agent/skills/continual-learning/`) served as virtual files via a `CompositeBackend` `/skills/` route + `StateBackend` (seeded into the run's `files` channel by the launcher — never written to the sandbox). Launchers and the per-repo nightly cron live in `agent/dashboard/review_style_jobs.py` and `agent/dashboard/analyzer_cron.py`; the cron is registered when bootstrap completes. -- **`agent/webapp.py`** — custom FastAPI routes mounted alongside the LangGraph server. Webhooks land here (GitHub, Linear, Slack, Jira, and the Confluence Atlassian Connect `/connect/*` routes). Each webhook resolves a deterministic `thread_id` (so follow-up messages route to the same agent run) and triggers/streams a run via the `langgraph_sdk` client. Also auto-reviews PRs on `opened` / `ready_for_review` events when the repo+author opt in. The Atlassian triggers (`agent/webhooks/{jira,confluence}.py`, `agent/utils/atlassian_connect.py`) verify webhook trust — a Jira Automation shared secret / optional HMAC, and a Confluence Connect HS256-JWT + `qsh` with RS256 signed-install — then re-fetch the triggering comment server-side before deriving identity. +- **FastAPI layer (`agent/api/` + `agent/webhooks/`)** — custom FastAPI routes mounted alongside the LangGraph server. The domain reorg split the fork's former `agent/webapp.py` monolith into `agent/api/app.py` (app composition + router mounts), `agent/api/health.py` (`/health`, `/webhooks/run-complete`), shared helpers in `agent/webhooks/common.py`, and per-source route modules `agent/webhooks/{github,linear,slack,jira,confluence}_routes.py`; `agent/webapp.py` remains a compatibility shim re-exporting `app`. Webhooks land in those route modules (GitHub, Linear, Slack, Jira, and the Confluence Atlassian Connect `/connect/*` routes, which fold into `confluence_routes.py`). Each webhook resolves a deterministic `thread_id` (so follow-up messages route to the same agent run) and triggers/streams a run via the `langgraph_sdk` client. Also auto-reviews PRs on `opened` / `ready_for_review` events when the repo+author opt in. The Atlassian triggers (`agent/webhooks/{jira,confluence}.py`, `agent/utils/atlassian_connect.py`) verify webhook trust — a Jira Automation shared secret / optional HMAC, and a Confluence Connect HS256-JWT + `qsh` with RS256 signed-install — then re-fetch the triggering comment server-side before deriving identity. - **`agent/dashboard/`** — `router` mounted under the FastAPI app at startup (`app.include_router(dashboard_router)`). Owns GitHub OAuth, per-user profiles, admin endpoints, team defaults, enabled-repo lists, review-style management, and the Agents chat thread API used by the UI in `ui/`. ### Sandbox lifecycle (the tricky part) @@ -114,10 +116,10 @@ Webhooks compute deterministic thread ids so the same Linear issue / Slack threa ## Conventions -- Tests are unit-only by default (`tests/`). Integration tests would go under `tests/integration_tests/` (currently empty — `make integration_tests` no-ops if missing). +- Tests are unit-only by default and organized by domain under `tests//` (`tests/agent/`, `tests/auth/`, `tests/github/`, `tests/webhooks/`, `tests/reviewer/`, `tests/sandbox/`, `tests/middleware/`, `tests/models/`, `tests/slack/`, `tests/tools/`, `tests/dashboard/`, `tests/analyzer/`); `tests/conftest.py` and the `tests/e2e/` harness stay at the top level. Integration tests would go under `tests/integration_tests/` (currently empty — `make integration_tests` no-ops if missing). - New sandbox providers: add a module under `agent/integrations/` and wire it into `SANDBOX_FACTORIES` in `agent/utils/sandbox.py`. See `CUSTOMIZATION.md`. - New tools: add to `agent/tools/`, export from `agent/tools/__init__.py`, add to the `tools=[...]` list in `server.py:get_agent` (or `reviewer.py` for reviewer-only tools). - New middleware: add to `agent/middleware/`, export from `agent/middleware/__init__.py`, add to the `middleware=[...]` list in `server.py:get_agent` — order is significant (see the stack above). - New dashboard endpoints: add to `agent/dashboard/routes.py`. The router is auto-mounted on the FastAPI app. -- New graphs: register the entrypoint in `langgraph.json` under `graphs`. +- New graphs: add an `agent/graphs/.py` re-export shim that delegates to the factory module, then register the shim entrypoint (`agent.graphs.:`) in `langgraph.json` under `graphs`. - Minimal-to-no code comments — only when the *why* isn't obvious from the code. diff --git a/CLAUDE.md b/CLAUDE.md index 2d9c2a68..d5d541fa 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -14,33 +14,40 @@ Dependencies are managed with **uv**. Tests use pytest (`asyncio_mode = "auto"`) ```bash make install # uv pip install -e . -make dev # uv run langgraph dev — serves all three graphs + the FastAPI app from langgraph.json +make dev # uv run langgraph dev — serves all six graphs + the FastAPI app from langgraph.json make run # uvicorn agent.webapp:app --reload --port 8000 (FastAPI only, no LangGraph runtime) make test # uv run pytest -vvv tests/ -make test TEST_FILE=tests/test_open_pr_middleware.py # single test file -uv run pytest -vvv tests/test_open_pr_middleware.py::test_name # single test +make test TEST_FILE=tests/github/test_open_pull_request.py # single test file +uv run pytest -vvv tests/github/test_open_pull_request.py::test_name # single test make lint # ruff check + ruff format --diff make format # ruff format + ruff check --fix ``` -`langgraph.json` declares three graph entrypoints and the FastAPI app, all served together by `langgraph dev`: +`langgraph.json` declares six graph entrypoints and the FastAPI app, all served together by `langgraph dev`. Since the domain reorg, every graph entrypoint targets a thin `agent.graphs.*` re-export shim (`agent/graphs/.py`) rather than the factory module directly; the shims delegate to the unmoved factories (`agent/server.py`, `agent/reviewer.py`, `agent/analyzer.py`, `agent/chat.py`, `agent/scheduler.py`, `agent/ci_monitor.py`): | Graph | Entrypoint | Purpose | |---|---|---| -| `agent` | `agent.server:traced_agent` (wraps `get_agent`) | Main coding agent (Slack/Linear/Jira/Confluence/GitHub-triggered). | -| `reviewer` | `agent.reviewer:traced_reviewer_agent` (wraps `get_reviewer_agent`) | Read-only PR reviewer. Findings model + `publish_review`. | -| `analyzer` | `agent.analyzer:traced_analyzer` (wraps `get_analyzer`) | Learns per-repo reviewer style from historical PRs and this reviewer's own finding outcomes. | +| `agent` | `agent.graphs.agent:traced_agent` (shim → `agent.server:get_agent`) | Main coding agent (Slack/Linear/Jira/Confluence/GitHub-triggered). | +| `reviewer` | `agent.graphs.reviewer:traced_reviewer_agent` (shim → `agent.reviewer:get_reviewer_agent`) | Read-only PR reviewer. Findings model + `publish_review`. | +| `analyzer` | `agent.graphs.analyzer:traced_analyzer` (shim → `agent.analyzer:get_analyzer`) | Learns per-repo reviewer style from historical PRs and this reviewer's own finding outcomes. | +| `chat` | `agent.graphs.chat:traced_chat_agent` | Dashboard Agents chat graph. | +| `scheduler` | `agent.graphs.scheduler:get_scheduler` | Reconcile sweep for stragglers. | +| `ci_monitor` | `agent.graphs.ci_monitor:get_ci_monitor` | Fork-only polling fallback for the CI auto-fix flow. | -The FastAPI app is `agent.webapp:app`. +The FastAPI app is `agent.webapp:app` — now a compatibility shim that re-exports `agent.api.app:app` (see the FastAPI split under "Entrypoints"). ## Architecture ### Entrypoints - **`agent/server.py` → `get_agent(config)`** — main graph factory. Called per-thread. Resolves the GitHub token, gets-or-creates the sandbox for the thread, resolves the team/profile/per-thread model + effort, then constructs a fresh `create_deep_agent(...)` with the curated tool list and middleware stack. The agent itself is stateless — all per-thread state lives in the sandbox + thread metadata. -- **`agent/reviewer.py` → `get_reviewer_agent(config)`** — reviewer graph factory. Shares `ensure_sandbox_for_thread` with the main agent but wires a reviewer-only toolset (`add_finding`, `update_finding`, `list_findings`, `publish_review`, `web_search`, `fetch_url`, `http_request`) and a different system prompt that pins the single-evolving-findings model and the diff-anchored bar for filing a finding. Read-only: no commit/push/PR-opening tools. +- **`agent/reviewer.py` → `get_reviewer_agent(config)`** — reviewer graph factory. Shares `ensure_sandbox_for_thread` with the main agent but wires a reviewer-only toolset (`add_finding`, `update_finding`, `list_findings`, `publish_review`, `web_search`, `fetch_url`, `http_request`) and a different system prompt that pins the single-evolving-findings model and the diff-anchored bar for filing a finding. Read-only: no commit/push/PR-opening tools. Its supporting modules live in the **`agent/review/`** package (domain reorg): `findings.py`, `publish.py`, `reconcile.py`, `trace_context.py`, `diff.py`, `groups.py`, `eval_store.py`, `style_collector.py`, `style_guidance.py`. - **`agent/analyzer.py` → `get_analyzer(config)`** — small graph that emits a per-repo style prompt via the `save_review_style_prompt` tool, consumed by the reviewer as a "repository-specific review style" appendix. It runs in one of two modes (`analyzer_mode` in `configurable`): **bootstrap** (cold-start: crawl historical PR reviews) and **continual** (nightly: refine using this reviewer's own finding outcomes via `read_finding_outcomes`). Each mode's procedure lives in a deepagents **skill** (`agent/skills/bootstrap-repo-analysis/`, `agent/skills/continual-learning/`) served as virtual files via a `CompositeBackend` `/skills/` route + `StateBackend` (seeded into the run's `files` channel by the launcher — never written to the sandbox). Launchers and the per-repo nightly cron live in `agent/dashboard/review_style_jobs.py` and `agent/dashboard/analyzer_cron.py`; the cron is registered when bootstrap completes. -- **`agent/webapp.py`** — thin FastAPI routing layer mounted alongside the LangGraph server. Defines the webhook routes (GitHub, Linear, Slack, Jira, Confluence Connect `/connect/*`) plus `/webhooks/run-complete`, and keeps the shared helpers/constants; the per-source handlers live in **`agent/webhooks/{github,slack,linear,jira,confluence}.py`** (re-exported from `webapp` so existing call sites and tests keep working). Each webhook resolves a deterministic `thread_id` (so follow-up messages route to the same agent run) and triggers a run through the single durable dispatch contract in **`agent/dispatch.py`** (`dispatch_agent_run`: `multitask_strategy="interrupt"` + `durability="sync"` + completion webhook); `agent/completion.py` posts a failure reply if a run dies, and `agent/reconcile.py` (a `scheduler`-graph sweep) catches stragglers. The GitHub handler also auto-reviews PRs on `opened` / `ready_for_review` and drives the CI auto-fix flow (`agent/ci_autofix.py`). +- **FastAPI layer (`agent/api/` + `agent/webhooks/`)** — the domain reorg split the fork's former 2,590-line `agent/webapp.py` monolith into a per-source layout; `agent/webapp.py` is now a 4-line compatibility shim (`from .api.app import app`). The pieces: + - **`agent/api/app.py`** composes the FastAPI app and mounts every router; **`agent/api/health.py`** owns `/health` and `/webhooks/run-complete`. + - **`agent/webhooks/common.py`** holds the shared verify/dispatch helpers and constants (signature verification, thread-id derivation, the dispatch entry). Route modules and handlers reach these via **module-attribute access** (`common.X`) so the ~240 `monkeypatch.setattr` test sites retarget cleanly. + - Per-source **route modules** `agent/webhooks/{github,linear,slack,jira,confluence}_routes.py` define the HTTP routes (GitHub, Linear, Slack, Jira, and the Confluence Atlassian Connect `/connect/*` + `/webhooks/jira` routes; the fork-only `jira_routes.py`/`confluence_routes.py` mirror upstream's `github_routes.py` pattern, and the Connect lifecycle/descriptor routes fold into `confluence_routes.py`). The **handler modules** `agent/webhooks/{github,slack,linear,jira,confluence}.py` carry the per-source logic. + - Each webhook resolves a deterministic `thread_id` (so follow-up messages route to the same agent run) and triggers a run through the single durable dispatch contract in **`agent/dispatch.py`** (`dispatch_agent_run`: `multitask_strategy="interrupt"` + `durability="sync"` + completion webhook); `agent/completion.py` posts a failure reply if a run dies, and `agent/reconcile.py` (a `scheduler`-graph sweep) catches stragglers. The GitHub handler also auto-reviews PRs on `opened` / `ready_for_review` and drives the CI auto-fix flow (`agent/ci_autofix.py`). - **Atlassian triggers.** The Jira trigger is a Jira **Automation** rule POSTing to `/webhooks/jira` with a shared-secret header (`verify_jira_secret`; optional HMAC-body+timestamp via `JIRA_WEBHOOK_REQUIRE_SIGNATURE`), since Jira Cloud has no native webhook signing. The Confluence trigger is a private **Atlassian Connect app** (`agent/utils/atlassian_connect.py`): the `comment_created` webhook is HS256-JWT-verified against the per-tenant stored `sharedSecret` with a hand-rolled `qsh` (query-string-hash) check; `signed-install` is on, so install/uninstall lifecycle callbacks are RS256-verified against Atlassian's published keys (no trust-on-first-use). Install secrets are stored **encrypted** in the LangGraph store, keyed by `clientKey`. Both Atlassian webhook bodies are treated as pointers only — the triggering comment's real author/text is re-fetched server-side via the Basic-auth service account before anything security-relevant is derived, and attribution is gated on an active user mapping (mirrors the GitHub-login / token-attribution flow). Descriptor served at `GET /connect/atlassian-connect.json`. - **`agent/dashboard/`** — `router` mounted under the FastAPI app at startup (`app.include_router(dashboard_router)`). Owns GitHub OAuth, per-user profiles, admin endpoints, team defaults, enabled-repo lists, review-style management, and the Agents chat thread API used by the UI in `ui/`. @@ -112,17 +119,17 @@ Webhooks compute deterministic thread ids so the same Linear issue / Slack threa ## Conventions -- Tests are unit-only by default (`tests/`). Integration tests would go under `tests/integration_tests/` (currently empty — `make integration_tests` no-ops if missing). +- Tests are unit-only by default and organized by domain under `tests//` (`tests/agent/`, `tests/auth/`, `tests/github/`, `tests/webhooks/`, `tests/reviewer/`, `tests/sandbox/`, `tests/middleware/`, `tests/models/`, `tests/slack/`, `tests/tools/`, `tests/dashboard/`, `tests/analyzer/`); `tests/conftest.py` and the `tests/e2e/` harness stay at the top level. Integration tests would go under `tests/integration_tests/` (currently empty — `make integration_tests` no-ops if missing). - New sandbox providers: add a module under `agent/integrations/` and wire it into `SANDBOX_FACTORIES` in `agent/utils/sandbox.py`. See `CUSTOMIZATION.md`. - New tools: add to `agent/tools/`, export from `agent/tools/__init__.py`, add to the `tools=[...]` list in `server.py:get_agent` (or `reviewer.py` for reviewer-only tools). - New middleware: add to `agent/middleware/`, export from `agent/middleware/__init__.py`, add to the `middleware=[...]` list in `server.py:get_agent` — order is significant (see the stack above). - New dashboard endpoints: add to `agent/dashboard/routes.py`. The router is auto-mounted on the FastAPI app. -- New graphs: register the entrypoint in `langgraph.json` under `graphs`. +- New graphs: add an `agent/graphs/.py` re-export shim that delegates to the factory module, then register the shim entrypoint (`agent.graphs.:`) in `langgraph.json` under `graphs`. - Minimal-to-no code comments — only when the *why* isn't obvious from the code. ## Fork maintenance — syncing `upstream/main` -This is a long-lived fork of `langchain-ai/open-swe` with Sea Haven customizations woven into upstream-owned files (notably `agent/prompt.py` prompt constants, `agent/webapp.py`, and the tool/middleware wiring). Merging upstream is a triage exercise, not a fast-forward. When you want upstream's clean changes but must **defer a large structural refactor** (and its entangled features), work in this order: +This is a long-lived fork of `langchain-ai/open-swe` with Sea Haven customizations woven into upstream-owned files (notably `agent/prompt.py` prompt constants, the `agent/api/` + `agent/webhooks/` FastAPI layer — the former `agent/webapp.py` monolith, now split and left as a shim — and the tool/middleware wiring). Merging upstream is a triage exercise, not a fast-forward. When you want upstream's clean changes but must **defer a large structural refactor** (and its entangled features), work in this order: 1. **Triage before resolving.** Merge-base is `git merge-base HEAD upstream/main`. The truthful conflict set is the combined merge, `git merge-tree --write-tree --name-only HEAD upstream/main` — a per-commit probe against each commit's parent *overstates* conflicts (a file a refactor merely added shows up as a phantom `modify/delete`). Decide keep-baseline vs adopt-refactor **before** resolving, and surface the choice to a human for any auth/webhook/IAM surface. 2. **Chase the cascade, not just the textual conflicts.** The hard part is the non-conflicting files the refactor also touched. Get the refactor's file set (`git diff-tree --no-commit-id --name-status -r `) and cross-reference the files this fork modified (`git diff --name-only HEAD`). Files in both = hand-resolve; files only the refactor touched = mechanical. diff --git a/README.md b/README.md index 4e34f72d..8232d10c 100644 --- a/README.md +++ b/README.md @@ -1,9 +1,9 @@ @@ -54,7 +54,7 @@ create_deep_agent( Every task runs in its own **isolated cloud sandbox** — a remote Linux environment with full shell access. The repo is cloned in, the agent gets full permissions, and the blast radius of any mistake is fully contained. No production access, no confirmation prompts. -Open SWE supports multiple sandbox providers out of the box — [Modal](https://modal.com/), [Daytona](https://www.daytona.io/), [Runloop](https://www.runloop.ai/), [E2B](https://e2b.dev/), and [LangSmith](https://smith.langchain.com/) — and you can plug in your own. See the [Customization Guide](CUSTOMIZATION.md#1-sandbox) for details. +Open SWE supports multiple sandbox providers out of the box — [Modal](https://modal.com/), [Daytona](https://www.daytona.io/), [Runloop](https://www.runloop.ai/), [E2B](https://e2b.dev/), and [LangSmith](https://smith.langchain.com/) — and you can plug in your own. See the [Customization Guide](docs/CUSTOMIZATION.md#1-sandbox) for details. This follows the principle all three companies converge on: **isolate first, then give full permissions inside the boundary.** @@ -112,7 +112,7 @@ All three companies in the article converge on **Slack as the primary invocation - **Confluence** — Comment `@openswe` on a page. A private Atlassian Connect app delivers the `comment_created` event; the agent acts and replies on the page. - **GitHub** — Tag `@openswe` in PR comments on agent-created PRs to have it address review feedback and push fixes to the same branch. -See **[INSTALLATION.md](./INSTALLATION.md) §5** for per-surface trigger setup. +See **[INSTALLATION.md](./docs/INSTALLATION.md) §5** for per-surface trigger setup. Each invocation creates a deterministic thread ID, so follow-up messages on the same issue or thread route to the same running agent. @@ -123,7 +123,7 @@ Each invocation creates a deterministic thread ID, so follow-up messages on the ### 7. Validation — Prompt-Driven The agent is instructed to run linters, formatters, and tests before committing, and is responsible end-to-end for committing, pushing, opening/updating the draft PR, and replying in the source channel. -This is an area where you can extend Open SWE for your org: add deterministic CI checks, visual verification, or review gates as additional middleware. See the [Customization Guide](CUSTOMIZATION.md#6-middleware) for how. +This is an area where you can extend Open SWE for your org: add deterministic CI checks, visual verification, or review gates as additional middleware. See the [Customization Guide](docs/CUSTOMIZATION.md#6-middleware) for how. --- @@ -156,8 +156,8 @@ This is an area where you can extend Open SWE for your org: add deterministic CI ## Getting Started -- **[Installation Guide](INSTALLATION.md)** — local dev (backend + dashboard), GitHub App creation, LangSmith, Linear/Slack/GitHub triggers, and production deployment -- **[Customization Guide](CUSTOMIZATION.md)** — swap the sandbox, model, tools, triggers, system prompt, and middleware for your org +- **[Installation Guide](docs/INSTALLATION.md)** — local dev (backend + dashboard), GitHub App creation, LangSmith, Linear/Slack/GitHub triggers, and production deployment +- **[Customization Guide](docs/CUSTOMIZATION.md)** — swap the sandbox, model, tools, triggers, system prompt, and middleware for your org ## Deployment (Sea Haven fork) @@ -169,7 +169,7 @@ and secrets live in the LangGraph deployment config and Vercel environment variables. Promotion from `dev` to `prod` (`main`) is handled by [`.github/workflows/promote-to-main.yml`](.github/workflows/promote-to-main.yml). -See **[INSTALLATION.md § 10 "Production deployment"](INSTALLATION.md#10-production-deployment)** +See **[INSTALLATION.md § 10 "Production deployment"](docs/INSTALLATION.md#10-production-deployment)** for the full backend + dashboard setup. > The earlier self-hosted AWS stack (CDK under `infra/`, an ARM64 EC2 box + nginx diff --git a/agent/analyzer.py b/agent/analyzer.py index 8f16934b..de5a6243 100644 --- a/agent/analyzer.py +++ b/agent/analyzer.py @@ -36,7 +36,7 @@ from .middleware import ( TimeoutWrapupMiddleware, ToolErrorMiddleware, ) -from .review_style_guidance import REVIEWER_STYLE_THEMES +from .review.style_guidance import REVIEWER_STYLE_THEMES from .server import ( DEFAULT_LLM_MAX_TOKENS, DEFAULT_LLM_MODEL_ID, diff --git a/agent/api/__init__.py b/agent/api/__init__.py new file mode 100644 index 00000000..0c87b0fd --- /dev/null +++ b/agent/api/__init__.py @@ -0,0 +1 @@ +"""FastAPI application composition.""" diff --git a/agent/api/app.py b/agent/api/app.py new file mode 100644 index 00000000..77d3f02d --- /dev/null +++ b/agent/api/app.py @@ -0,0 +1,57 @@ +"""FastAPI application composition.""" + +import os +from collections.abc import AsyncIterator +from contextlib import asynccontextmanager + +from fastapi import FastAPI +from fastapi.middleware.cors import CORSMiddleware + +from ..dashboard import router as dashboard_router +from ..dashboard.plan_api import plan_router +from ..dashboard.workflow_approval_api import workflow_approval_router +from ..webhooks.confluence_routes import router as confluence_webhook_router +from ..webhooks.github_routes import router as github_webhook_router +from ..webhooks.jira_routes import router as jira_webhook_router +from ..webhooks.linear_routes import router as linear_webhook_router +from ..webhooks.slack_routes import router as slack_webhook_router +from .health import router as health_router + + +@asynccontextmanager +async def lifespan(_app: FastAPI) -> AsyncIterator[None]: + from ..utils.model import validate_local_dev_llm_config + from ..utils.sandbox import validate_sandbox_startup_config + + validate_sandbox_startup_config() + validate_local_dev_llm_config() + yield + + +app = FastAPI(lifespan=lifespan) + +DASHBOARD_ALLOWED_ORIGINS: list[str] = [ + o.strip() for o in os.environ.get("DASHBOARD_ALLOWED_ORIGINS", "").split(",") if o.strip() +] +if DASHBOARD_ALLOWED_ORIGINS: + if "*" in DASHBOARD_ALLOWED_ORIGINS: + raise RuntimeError( + "DASHBOARD_ALLOWED_ORIGINS must not include '*' when allow_credentials=True" + ) + app.add_middleware( + CORSMiddleware, + allow_origins=DASHBOARD_ALLOWED_ORIGINS, + allow_credentials=True, + allow_methods=["GET", "POST", "PUT", "PATCH", "DELETE", "OPTIONS"], + allow_headers=["*"], + ) + +app.include_router(dashboard_router) +app.include_router(plan_router) +app.include_router(workflow_approval_router) +app.include_router(linear_webhook_router) +app.include_router(jira_webhook_router) +app.include_router(confluence_webhook_router) +app.include_router(slack_webhook_router) +app.include_router(health_router) +app.include_router(github_webhook_router) diff --git a/agent/api/health.py b/agent/api/health.py new file mode 100644 index 00000000..e3b1fc7a --- /dev/null +++ b/agent/api/health.py @@ -0,0 +1,27 @@ +"""Health and run-completion routes.""" + +from fastapi import APIRouter, HTTPException, Request + +from ..completion import handle_run_completion, verify_run_complete_token + +router = APIRouter() + + +@router.get("/health") +async def health_check() -> dict[str, str]: + """Health check endpoint.""" + return {"status": "healthy"} + + +@router.post("/webhooks/run-complete") +async def run_complete_webhook(request: Request) -> dict[str, str]: + """Platform run-completion webhook: post a failure reply for runs that died.""" + if not verify_run_complete_token(request.query_params.get("token")): + raise HTTPException(status_code=401, detail="Invalid run-complete token") + try: + payload = await request.json() + except Exception: # noqa: BLE001 + return {"status": "error", "message": "Invalid JSON"} + if not isinstance(payload, dict): + return {"status": "ignored", "reason": "payload not an object"} + return await handle_run_completion(payload) diff --git a/agent/ci_autofix.py b/agent/ci_autofix.py index 13571f58..22cefe15 100644 --- a/agent/ci_autofix.py +++ b/agent/ci_autofix.py @@ -4,7 +4,7 @@ This is the shared core for "PR babysitting": when a CI check fails (or a reviewer leaves actionable feedback) on a PR that Open SWE opened, locate the originating agent thread and dispatch a confidence-gated fix run on it. -Both the GitHub webhook path (:mod:`agent.webapp`) and the polling fallback +Both the GitHub webhook path (:mod:`agent.webhooks.github_routes`) and the polling fallback (:mod:`agent.ci_monitor`) call into here, so all the skip-rules, dedupe, and loop-capping live in one place. Skip-rules mirror Cursor/Claude Code: @@ -26,7 +26,7 @@ from .dashboard.agent_overrides import load_profile, resolve_login_from_email_as from .dashboard.autofix_state import is_pr_autofix_disabled from .dashboard.enabled_repos import is_review_repo_enabled from .dispatch import dispatch_agent_run -from .reviewer_findings import REVIEWER_THREAD_KIND +from .review.findings import REVIEWER_THREAD_KIND from .utils.dashboard_links import dashboard_thread_url from .utils.github_app import get_github_app_installation_token from .utils.github_checks import post_autofix_status_check diff --git a/agent/dashboard/__init__.py b/agent/dashboard/__init__.py index e6a42bb4..b5b01fd2 100644 --- a/agent/dashboard/__init__.py +++ b/agent/dashboard/__init__.py @@ -3,7 +3,7 @@ ``router`` is loaded lazily (PEP 562): importing any dashboard submodule (e.g. ``agent.dashboard.options`` from middleware) executes this __init__, and it must NOT drag in routes.py + FastAPI + every API/job module. Only the -webapp, which actually mounts the router, pays that cost. +API app (``agent.api.app``), which actually mounts the router, pays that cost. """ from typing import Any diff --git a/agent/dashboard/agent_usage.py b/agent/dashboard/agent_usage.py index f3a305c7..01840345 100644 --- a/agent/dashboard/agent_usage.py +++ b/agent/dashboard/agent_usage.py @@ -12,7 +12,7 @@ from typing import Any, Literal import httpx from langgraph_sdk import get_client -from ..reviewer_findings import REVIEWER_THREAD_KIND +from ..review.findings import REVIEWER_THREAD_KIND from ..utils.github_app import get_github_app_installation_token USAGE_THREAD_NAMESPACE: list[str] = ["agent_usage", "threads"] diff --git a/agent/dashboard/eval_jobs.py b/agent/dashboard/eval_jobs.py index d64c7ea1..91293ff4 100644 --- a/agent/dashboard/eval_jobs.py +++ b/agent/dashboard/eval_jobs.py @@ -17,13 +17,13 @@ from typing import Any, Literal, TypedDict from langgraph_sdk import get_client -from agent.reviewer_eval_store import ( +from agent.review.eval_store import ( _HEARTBEAT_STALE_SECONDS, DEFAULT_EVAL_PROJECT, EVALS_NAMESPACE, REVIEWER_EVAL_KEY, ) -from agent.reviewer_findings import REVIEW_FINDING_CAP +from agent.review.findings import REVIEW_FINDING_CAP logger = logging.getLogger(__name__) diff --git a/agent/dashboard/review_api.py b/agent/dashboard/review_api.py index 1c84aca6..f7cbd69f 100644 --- a/agent/dashboard/review_api.py +++ b/agent/dashboard/review_api.py @@ -19,7 +19,7 @@ from urllib.parse import urljoin, urlparse import httpx from fastapi import HTTPException, Response -from ..reviewer_findings import REVIEWER_THREAD_KIND +from ..review.findings import REVIEWER_THREAD_KIND from ..utils.github_app import get_github_app_installation_token from ..utils.github_checks import github_headers from ..utils.thread_ops import langgraph_client @@ -443,7 +443,7 @@ async def create_review_comment( _HTML_COMMENT_RE = re.compile(r"", re.DOTALL) -# Inline comments the reviewer posts carry this hidden marker (see reviewer_publish). +# Inline comments the reviewer posts carry this hidden marker (see review.publish). _OPEN_SWE_COMMENT_RE = re.compile(r"