From a30ce2ab4055c87699e1cea15f156c3b31448acc Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Mon, 29 Jun 2026 19:58:38 -0400 Subject: [PATCH] =?UTF-8?q?feat:=20managed=20LangGraph=20Cloud=20+=20Verce?= =?UTF-8?q?l=20migration=20(PR2=20=E2=80=94=20code=20fixes=20+=20docs)=20(?= =?UTF-8?q?#65)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(dashboard): managed-cloud OAuth hardening + admin user-mapping endpoint Prepare the dashboard backend for the managed LangGraph Cloud + Vercel runtime, where the API is HTTPS and cross-site from the UI. - OAuth redirect_uri (#2): coerce a schemeless DASHBOARD_API_BASE_URL to https:// in _api_base_url() so GitHub stops rejecting login with "redirect_uri not associated with this application". _cookie_security() now treats a schemeless (managed) value as Secure; SameSite=None too, consistent with the coerced scheme. - OAuth state cookie (#3): document that osw_oauth_state is host-only by design (a Domain cookie is unsafe across *.vercel.app, a public suffix), so login must always start on the stable alias to avoid "oauth state mismatch". Operational contract; no behavioral change. - Admin user mappings (#4): add POST /admin/user-mappings so an admin can set the github_login -> work_email link from the dashboard instead of a raw Store write. New "admin" MappingSource provenance value. * fix(webapp): refresh user-mapping cache on GitHub webhook paths On managed LangGraph Cloud the backend runs multiple replicas, so the per-process GitHub<->work-email mapping cache can be stale on the replica handling a webhook (a mapping created on another replica is invisible until refresh). process_github_pr_comment and process_github_issue now refresh the cache from the durable Store before resolving the author's email, matching the existing Slack mention path (process_slack_mention). * perf(webapp): defer deepagents import to speed custom-app cold start The custom FastAPI app (agent.webapp:app, the langgraph.json http.app) pulled deepagents -> langchain_anthropic -> anthropic into its import graph via dashboard.routes, only to build skill/chat seed files. Defer those create_file_data imports into the functions that use them. Removes deepagents/langchain_anthropic/anthropic from app import entirely and roughly halves module-import wall time (~0.6-0.8s -> ~0.35s warm; larger cold-start saving since native anthropic init is skipped). Behavior identical. (reviewer_diff already imports deepagents under TYPE_CHECKING.) * feat(ui): set work_email user mappings from the admin dashboard Add an "Add / update" form to the admin User mappings section and the adminUpsertUserMapping API client method, wiring the new POST /admin/user-mappings endpoint. Admins can now create or update a github_login -> work_email mapping directly instead of waiting for the user to self-connect Slack. * docs: document managed LangGraph Cloud + Vercel deployment - INSTALLATION §10: add the managed production env triad (LANGGRAPH_URL, DASHBOARD_BASE_URL + DASHBOARD_API_BASE_URL with https://, empty VITE_DASHBOARD_API_BASE_URL for same-origin), the stable-alias login and vercel.json stable-deployment-URL requirements, multi-replica cache note, plus redirect_uri-scheme and oauth-state-mismatch troubleshooting. Refresh the langgraph.json snippet to all six graphs. - README: reframe deployment around the managed migration; link the plan. - deploy/MIGRATION.md: import the self-hosted -> managed migration plan. --- INSTALLATION.md | 27 ++- agent/dashboard/review_chat_api.py | 7 +- agent/dashboard/routes.py | 54 ++++- agent/dashboard/user_mappings.py | 2 +- agent/utils/analyzer_skills.py | 9 +- agent/webapp.py | 18 ++ deploy/MIGRATION.md | 361 +++++++++++++++++++++++++++++ ui/src/lib/api.ts | 9 + ui/src/routes/admin.tsx | 51 +++- 9 files changed, 527 insertions(+), 11 deletions(-) create mode 100644 deploy/MIGRATION.md diff --git a/INSTALLATION.md b/INSTALLATION.md index 419c663f..d4984448 100644 --- a/INSTALLATION.md +++ b/INSTALLATION.md @@ -638,14 +638,17 @@ Production runs the backend and dashboard separately. 3. Set all environment variables from step 6 in the deployment config. Set `DASHBOARD_BASE_URL` and `LANGGRAPH_URL` to your production URLs (all `https://`). Set `DASHBOARD_API_BASE_URL` to the URL browsers use for dashboard API requests and OAuth callbacks: either the backend URL for direct cross-origin calls, or the dashboard/Vercel URL when a same-origin rewrite proxies `/dashboard/api/*`. 4. Update your webhook URLs (Linear, Slack, GitHub App) and the GitHub App / Slack OAuth callback URLs to your production URLs (replace the ngrok / localhost values). The dashboard GitHub App callback must be `/dashboard/api/auth/callback`. -The `langgraph.json` at the project root defines the three graphs and the HTTP app: +The `langgraph.json` at the project root defines the graphs and the custom HTTP app: ```json { "graphs": { "agent": "agent.server:traced_agent", "reviewer": "agent.reviewer:traced_reviewer_agent", - "analyzer": "agent.analyzer:traced_analyzer" + "analyzer": "agent.analyzer:traced_analyzer", + "chat": "agent.chat:traced_chat_agent", + "scheduler": "agent.scheduler:get_scheduler", + "ci_monitor": "agent.ci_monitor:get_ci_monitor" }, "http": { "app": "agent.webapp:app" @@ -657,6 +660,23 @@ The `langgraph.json` at the project root defines the three graphs and the HTTP a Alternatively, you can run the dashboard as a direct cross-origin client: set `VITE_DASHBOARD_API_BASE_URL` to the hosted backend origin, set `DASHBOARD_API_BASE_URL` to that same backend origin, and include the dashboard origin in `DASHBOARD_ALLOWED_ORIGINS`. +### Managed LangGraph Cloud + Vercel — the production env triad + +The production runtime is **managed LangGraph Cloud** (backend) + **Vercel** (UI), not a self-hosted `langgraph dev` process. Get these three right, in the recommended same-origin setup: + +| Variable | Where | Value | Why | +|---|---|---|---| +| `LANGGRAPH_URL` | backend (LangGraph Cloud) | the deployment's own `https://…us.langgraph.app` URL | The server-side `langgraph_client()` (thread sidebar + run creation) connects here. It defaults to `http://localhost:2024` for local dev; **on managed it must be set** or every dashboard/thread call `ConnectError`s. | +| `DASHBOARD_BASE_URL` | backend | the dashboard origin, `https://…` | Where token-free settings links point. | +| `DASHBOARD_API_BASE_URL` | backend | the **same dashboard/Vercel origin**, `https://…` | Builds the OAuth `redirect_uri` and drives the session-cookie `Secure; SameSite=None` flags. **Must include the `https://` scheme** — a schemeless value yields a schemeless `redirect_uri` and GitHub rejects login with "redirect_uri not associated with this application." (The server now coerces a schemeless value to `https://` as a backstop, but set it explicitly.) | +| `VITE_DASHBOARD_API_BASE_URL` | UI build (Vercel) | **empty** | Leave empty so the SPA calls relative `/dashboard/api/*` same-origin; `ui/vercel.json` rewrites those to the backend so the `osw_session` cookie rides along. | + +`ui/vercel.json`'s rewrite `destination` must be the **stable LangGraph deployment URL** for the target environment (e.g. `https://-.us.langgraph.app`) — a per-deployment alias that is stable across revisions. Never point it at an immutable per-deploy / preview URL. Vercel does not interpolate env vars into `vercel.json` rewrites, so this hostname is committed per Vercel project; set the dev project's rewrite to the dev deployment and the prod project's to the prod deployment. + +**Always start dashboard login on the stable alias / custom domain**, not on an immutable per-deploy Vercel URL. The OAuth state cookie (`osw_oauth_state`) is intentionally host-only (a `Domain` cookie is unsafe across `*.vercel.app`, which is on the public-suffix list). If login starts on one host and GitHub redirects back to another, the cookie is dropped and the callback fails with "oauth state mismatch." + +On managed the backend runs as **multiple replicas**. In-process caches that must stay coherent across replicas (notably the GitHub⇄work-email user mapping) are refreshed from the durable Store on the webhook hot paths (Slack and GitHub) before they are read. + ## Troubleshooting ### Webhook not receiving events @@ -675,7 +695,8 @@ Alternatively, you can run the dashboard as a direct cross-origin client: set `V ### Dashboard login fails or won't stay logged in - `500 GITHUB_APP_CLIENT_ID not configured` (or client secret): set `GITHUB_APP_CLIENT_ID` / `GITHUB_APP_CLIENT_SECRET` (step 3c) and `DASHBOARD_JWT_SECRET`. -- OAuth `redirect_uri` mismatch: the GitHub App must list `/dashboard/api/auth/callback` as a callback URL (step 3b). Locally that's `http://localhost:2024/dashboard/api/auth/callback`. +- OAuth `redirect_uri` mismatch / "redirect_uri not associated with this application": the GitHub App must list `/dashboard/api/auth/callback` as a callback URL (step 3b). Locally that's `http://localhost:2024/dashboard/api/auth/callback`. On managed, a common cause is a **schemeless** `DASHBOARD_API_BASE_URL` (e.g. `dash.example.com` instead of `https://dash.example.com`), which produces a schemeless `redirect_uri`; always set the full `https://` URL. +- "oauth state mismatch" after the GitHub redirect: login was started on a different host than the one GitHub redirected back to (typically an immutable per-deploy Vercel URL vs the stable alias). The `osw_oauth_state` cookie is host-only by design, so it isn't presented to the other host. Always start login on the stable alias / custom domain. - Login redirects but the session doesn't stick: this is almost always a cookie problem. Locally, keep `DASHBOARD_API_BASE_URL` on `http://` (so cookies are `SameSite=Lax`); in prod use `https://` for both API and frontend and add the frontend origin to `DASHBOARD_ALLOWED_ORIGINS`. - Login rejected with an org error: `ALLOWED_GITHUB_ORGS` gates dashboard login (and requires the App's Organization → Members: Read-only permission). See step 5. - Admin pages 403: add your GitHub login or email to `CONFIGURED_ADMINS`. diff --git a/agent/dashboard/review_chat_api.py b/agent/dashboard/review_chat_api.py index a918870a..2fb23f98 100644 --- a/agent/dashboard/review_chat_api.py +++ b/agent/dashboard/review_chat_api.py @@ -17,7 +17,6 @@ from datetime import UTC, datetime from typing import Any import httpx -from deepagents.backends.utils import create_file_data from fastapi import HTTPException from ..reviewer_diff import fetch_pr_diff @@ -223,6 +222,12 @@ async def _build_pr_context( Accepts an already-fetched ``review`` to avoid re-fetching it when the caller has just read it to decide whether a reseed is needed. """ + # Deferred import: deepagents pulls langchain_anthropic / anthropic (~0.7s) + # into the import graph. It is otherwise dragged into the custom FastAPI + # app's cold start via dashboard.routes; only needed when seeding chat + # files, so import it lazily here. + from deepagents.backends.utils import create_file_data + if review is None: review = await get_review(owner, repo, pr_number) findings = review.get("findings") if isinstance(review.get("findings"), list) else [] diff --git a/agent/dashboard/routes.py b/agent/dashboard/routes.py index d38d0c34..ab51f511 100644 --- a/agent/dashboard/routes.py +++ b/agent/dashboard/routes.py @@ -236,6 +236,13 @@ def _api_base_url() -> str: v = os.environ.get("DASHBOARD_API_BASE_URL", "").rstrip("/") if not v: raise HTTPException(500, "DASHBOARD_API_BASE_URL not configured") + if not v.startswith(("http://", "https://")): + # A schemeless value (e.g. "open-swe-prod.us.langgraph.app") produces a + # schemeless OAuth redirect_uri, which GitHub rejects with + # "redirect_uri not associated with this application". Managed + # deployments are served over HTTPS, so default to https:// when the + # operator omitted the scheme. + v = f"https://{v}" return v @@ -255,9 +262,14 @@ def _cookie_security() -> tuple[bool, str]: rejected and the frontend/API are same-site, so fall back to ``SameSite=Lax`` without ``Secure``. """ - if os.environ.get("DASHBOARD_API_BASE_URL", "").startswith("https://"): - return True, "none" - return False, "lax" + api = os.environ.get("DASHBOARD_API_BASE_URL", "") + # Only an explicit ``http://`` (local dev) or an unconfigured value falls + # back to the insecure same-site cookie. ``https://`` *and* a schemeless + # managed host (which ``_api_base_url`` coerces to https) are cross-site + # over TLS and must use ``Secure; SameSite=None``. + if not api or api.startswith("http://"): + return False, "lax" + return True, "none" def _set_session_cookie(response: Response, jwt_token: str) -> None: @@ -277,6 +289,15 @@ def _set_state_cookie(response: Response, nonce: str) -> None: # SameSite=Lax so GitHub's top-level redirect back to /auth/callback # still presents this cookie; the cookie is single-purpose and lives # only for the duration of one OAuth round-trip. + # + # This cookie is intentionally host-only (no Domain attribute): scoping it + # to a shared parent domain is not safe across Vercel's immutable per-deploy + # hostnames (``*.vercel.app`` is on the public-suffix list, so a Domain + # cookie there is rejected). The operational contract is therefore to + # *always start login on the stable alias / custom domain* so the host that + # sets this cookie is the same host GitHub redirects back to. Starting the + # flow on an immutable per-deploy URL and finishing on the alias (or vice + # versa) drops the cookie and surfaces as "oauth state mismatch". secure, _ = _cookie_security() response.set_cookie( key=STATE_COOKIE_NAME, @@ -803,6 +824,33 @@ async def admin_list_user_mappings( } +class UserMappingUpsert(BaseModel): + github_login: str + work_email: str + slack_user_id: str | None = None + + +@router.post("/admin/user-mappings") +async def admin_upsert_user_mapping( + body: UserMappingUpsert, + _admin: dict[str, Any] = _ADMIN_DEP, +) -> dict[str, Any]: + """Create or update a GitHub↔work-email mapping from the admin dashboard. + + Lets an admin set the ``work_email`` link directly instead of waiting for + the user to self-connect Slack (or doing a raw Store write). + """ + try: + return await upsert_mapping( + github_login=body.github_login, + work_email=body.work_email, + slack_user_id=body.slack_user_id or None, + source="admin", + ) + except ValueError as e: + raise HTTPException(400, str(e)) from e + + @router.delete("/admin/user-mappings/{github_login}") async def admin_delete_user_mapping( github_login: str, diff --git a/agent/dashboard/user_mappings.py b/agent/dashboard/user_mappings.py index 8568bdc4..2644e8b3 100644 --- a/agent/dashboard/user_mappings.py +++ b/agent/dashboard/user_mappings.py @@ -35,7 +35,7 @@ logger = logging.getLogger(__name__) USER_MAPPINGS_NAMESPACE: list[str] = ["user_mappings"] -MappingSource = Literal["slack_oauth"] +MappingSource = Literal["slack_oauth", "admin"] MappingStatus = Literal["active", "pending"] diff --git a/agent/utils/analyzer_skills.py b/agent/utils/analyzer_skills.py index 69ef019e..fbf343d2 100644 --- a/agent/utils/analyzer_skills.py +++ b/agent/utils/analyzer_skills.py @@ -17,8 +17,6 @@ from __future__ import annotations from pathlib import Path from typing import Any -from deepagents.backends.utils import create_file_data - SKILLS_DIR = Path(__file__).resolve().parent.parent / "skills" SKILLS_ROUTE = "/skills/" @@ -41,6 +39,13 @@ def build_skill_files() -> dict[str, Any]: can serve them. Keys omit the ``/skills`` prefix (stripped by the composite route); values are ``FileData`` v2 entries. """ + # Deferred import: deepagents (and its langchain_anthropic / anthropic + # transitive deps) is heavy (~0.7s) and is otherwise pulled into the custom + # FastAPI app's import chain via dashboard.routes, slowing cold start. It's + # only needed when a skill bundle is actually built (analyzer launch), so + # import it lazily here. + from deepagents.backends.utils import create_file_data + files: dict[str, Any] = {} for skill in ANALYZER_MODES.values(): skill_md = SKILLS_DIR / skill / "SKILL.md" diff --git a/agent/webapp.py b/agent/webapp.py index 2577b804..445fa1d3 100644 --- a/agent/webapp.py +++ b/agent/webapp.py @@ -3049,6 +3049,16 @@ async def process_github_pr_comment(payload: dict[str, Any], event_type: str) -> else: logger.warning("Failed to persist branch_name metadata for thread %s", thread_id) + # Refresh the per-process user-mapping cache from the Store before + # resolving the author's email. On a multi-replica managed deployment this + # replica's cache may be stale (a mapping created on another replica is not + # otherwise visible), which would drop a legitimately-mapped user. Mirrors + # the Slack mention path (process_slack_mention). + try: + await refresh_user_mapping_cache() + except Exception: # noqa: BLE001 + logger.debug("Could not refresh user mapping cache for GitHub PR comment", exc_info=True) + email = await email_for_login(github_login) or "" if email: github_token = await _get_or_resolve_thread_github_token(thread_id, email, repo=repo_config) @@ -3328,6 +3338,14 @@ async def process_github_issue(payload: dict[str, Any], event_type: str) -> None logger.warning("Missing GitHub issue id/number, skipping") return + # Refresh the per-process user-mapping cache from the Store before + # resolving the author's email (multi-replica staleness; mirrors the Slack + # mention path in process_slack_mention). + try: + await refresh_user_mapping_cache() + except Exception: # noqa: BLE001 + logger.debug("Could not refresh user mapping cache for GitHub issue", exc_info=True) + email = await email_for_login(github_login) or "" if not email: logger.warning("No email mapping for GitHub user '%s', skipping", github_login) diff --git a/deploy/MIGRATION.md b/deploy/MIGRATION.md new file mode 100644 index 00000000..c64d95bf --- /dev/null +++ b/deploy/MIGRATION.md @@ -0,0 +1,361 @@ +# Open SWE — Migration Plan: Self-Hosted AWS → Managed LangGraph Cloud + Vercel + +**Repo:** `Sea-Haven-Industries/open-swe` (private) · **AWS:** 328440206208 / us-east-1 +**Author:** Adam Moussa · **Date:** 2026-06-29 · **Status:** DRAFT — owes a `/sh-plan-review` before prod cutover (Phase C gate) + +--- + +## 1. Executive Summary + +### Decision (settled — not re-litigated here) +Move the Open SWE deployment off the bespoke self-hosted AWS stack (stock `langgraph dev`, in-memory store, no durability — issue **#9**) onto the **managed runtime**: + +- **Backend → LangGraph Cloud** (a.k.a. "LangSmith Deployment"): git-connected, auto-builds a revision on push, zero-downtime, revision rollback, durable Postgres-backed store/checkpointer. +- **UI → Vercel**: `ui/` SPA, atomic deploys, instant rollback, per-PR previews (net-new capability). + +### Why +1. **Dissolves the licensing blocker.** Self-hosted `langgraph up` on the LangSmith **Plus** key is "Self-Hosted Lite" (1M node-exec/yr cap, Elastic License 2.0, production legally ambiguous). Managed **IS** the licensed product with **no node cap** — runs billed flat at $0.005, nodes not charged. +2. **Deletes the entire durable-runtime build** (RDS/Redis/Docker/AMI) **and most of the bespoke AWS CD** (S3 release pipeline, SSM deploy docs, packer, CDK box/ALB stacks). +3. **It's the upstream-canonical deployment** and the repo is **already wired for it**: `ui/vercel.json` has the same-origin `/dashboard/api/*` rewrite; `langgraph.json` is cloud-format (6 graphs + `http.app`). + +### Cost +≈ **$160/mo incremental** for one always-on prod deployment: +- $0.0036/min prod uptime ≈ **$155/mo** (always-on) +- + **$0.005/run** +- + traces pay-as-you-go above 10k/mo +- The **$39/mo Plus seat is already paid** (not incremental) +- **Dev deployment is FREE** (1 included on Plus, preemptible) + +A self-hosted `langgraph up` + RDS plan (already `/sh-plan-review`'d to APPROVE-after-revision this session) is the documented **FALLBACK** if managed is ever rejected (see §11). + +--- + +## 2. Architecture: Before → After + +### Before (self-hosted AWS — LIVE as of 2026-06-29) +``` +GitHub/Slack/Linear ──webhook──▶ hooks.seahaven.com ─┐ +Browser (dashboard) ─────────────▶ openswe.seahaven.com ─┤ + ▼ + shared seahaven-com ALB (:443, host+path rules) + ▼ + EC2 (prod i-08a729e50779c4b07 t4g.large ARM64, private subnet) + nginx :80 ──proxy /dashboard/api + /webhooks──▶ langgraph dev :2024 (loopback) + in-memory store (reseeded by seed_store.sh ExecStartPost) + ▼ + Config: Secrets Manager open-swe-prod/* + SSM /open-swe-prod/* + Release: GitHub Actions → S3 open-swe-prod-assets/releases/* → SSM doc deploy.sh + IaC: CDK OpenSweIamStack + OpenSweDevStack + OpenSweProdStack + Sandbox: LangSmith cloud (DEFAULT_SANDBOX_SNAPSHOT_ID + GitHub proxy) +``` + +### After (managed) +``` +GitHub/Slack/Linear ──webhook──▶ *.langgraph.app (or hooks.seahaven.com CNAME → TODO §10) +Browser (dashboard) ─────────────▶ open-swe-dashboard.vercel.app (stable alias / custom domain) + │ same-origin rewrite /dashboard/api/* (ui/vercel.json) + ▼ + LangGraph Cloud "Deployment" (managed, git-connected to `main` for prod / `dev` for dev) + ├─ serves the 6 graphs (agent, reviewer, analyzer, chat, scheduler, ci_monitor) + ├─ serves the custom http.app (agent.webapp:app = webhooks + dashboard API + OAuth) + ├─ durable Postgres store + checkpointer (issue #9 SOLVED) — autoscaled 1→10 replicas + └─ env/secrets in the Deployment config (NOT Secrets Manager) — Adam accepted deviation + ▼ + Sandbox: LangSmith cloud (UNCHANGED — DEFAULT_SANDBOX_SNAPSHOT_ID + GitHub-App proxy) + Auth: GitHub App seahaven-openswe (UNCHANGED — App 4146115 / Install 142615168) +``` + +**What changes shape:** runtime host (EC2 → managed PaaS), durability (in-memory → managed Postgres), CD (bespoke S3/SSM/packer → git-connected auto-build), config home (Secrets Manager/SSM → Deployment+Vercel env). **What stays:** the GitHub App, the LangSmith sandbox plane, CI (lint/format/unit/Playwright), the app code itself. + +--- + +## 3. Phased Plan + +### Phase A — Dev spike (MOSTLY DONE) +Goal: prove managed serves our custom app + durability, at $0, before committing prod $. + +**Proven this session:** +- ✅ Dev backend deployed to LangGraph Cloud, connected to branch `dev`: + `https://open-swe-dev-hosted-e76c2b0e8a7955fe8ad3110a7a54e5d0.us.langgraph.app` + (Bedrock + a Fireworks key set; AWS creds for Bedrock deferred — see §10 open decision). +- ✅ UI deployed to Vercel — team `sea-haven`, project `open-swe-dashboard`, `https://open-swe-dashboard.vercel.app`; `ui/vercel.json` rewrite repointed at the dev deployment. +- ✅ Managed serves the custom `http.app` (dashboard API + webhooks) — **no platform auth gate** in front of our routes (webhook returns 401 sig-enforced, so signatures still govern). +- ✅ Vercel same-origin rewrite → backend works. +- ✅ GitHub OAuth dashboard login end-to-end. +- ✅ **DURABILITY** — team-default model survived a revision redeploy (this is issue **#9**'s core acceptance goal, proven on managed). + +**Remaining Phase A items (the GitHub `@openswe` run-trigger loop):** +- [ ] A1. Get an `@openswe` GitHub comment to dispatch a run end-to-end on dev. Blocked by the user-mapping + cache gotchas (see fixes #4 and #5 in §5). +- [ ] A2. Create the owner user-mapping in the managed Store (namespace `["user_mappings"]`, key = lowercased login `amoussa1229`, value `{github_login, work_email}`) — the Admin UI can't set `work_email` yet (fix #4). +- [ ] A3. Verify a freshly-added mapping is seen without a redeploy across managed's multi-replica autoscaling (fix #5). +- [ ] A4. Confirm `LANGGRAPH_URL` is set to the deployment URL on the dev deployment so `langgraph_client()` calls resolve (fix #1 — config now, code later). + +### Phase B — Harden (the code fixes + env codification) +Land the 6 code fixes (§5), codify env/config, and resolve the Bedrock-auth decision **before** standing up prod. + +- [ ] B1. Land code fixes #1–#6 (§5) as a PR into `dev` (or split into focused PRs). Re-run `make lint` + `make test`. +- [ ] B2. **Codify the env contract for managed.** Update `ui/vercel.json` rewrite to the *prod* deployment URL (Phase C) but keep dev pointing at dev. Add a documented LangGraph-Cloud env list to the repo (see §4) — NOT secrets, just the variable inventory + which are excluded. +- [ ] B3. **Resolve the Bedrock-auth decision** (§10 open decision): static AWS keys for a Bedrock-scoped IAM user in the deployment env, OR run the agent on Fireworks. This intersects the #62 model work. +- [x] B4. ~~Merge #62 first~~ — **DONE / N/A.** #62 (Bedrock + Fireworks) **IS** merged to `dev` and is what the dev deployment runs — verified on `origin/dev` (`options.py` → `DEFAULT_MODEL_ID = "bedrock_converse:us.anthropic.claude-opus-4-8"`) and confirmed by the live deployment's `git_ref_sha: a4ed19ba…` (= the #62 merge commit). The contrary note came from a STALE local checkout; no merge action needed. +- [ ] B5. Audit ALL in-process caches for the single-process → multi-replica assumption (generalization of fix #5; see §5). +- [ ] B6. Measure + reduce custom-app import time (fix #6) so the deployment isn't flagged unhealthy / slow to scale. + +### Phase C — Prod deployment +- [ ] C1. Create a **prod LangGraph Cloud deployment** tracking branch `main` (the durable autoscaled 1→10 tier, not the free Dev tier). Record its `*.langgraph.app` URL. +- [ ] C2. Create the **prod Vercel project/target** (or promote the existing `open-swe-dashboard` to production); set its env (same-origin mode: `VITE_DASHBOARD_API_BASE_URL` empty). +- [ ] C3. Set the **prod env triad** (§4) on the prod deployment + Vercel: + - `LANGGRAPH_URL` = the prod `*.langgraph.app` URL + - `DASHBOARD_BASE_URL` + `DASHBOARD_API_BASE_URL` = the prod Vercel origin (with `https://` scheme — fix #2) +- [ ] C4. **Establish the prod approval gate** (§6): git-connected auto-deploy needs an explicit gate. Preserve the prod manual-approval that exists today (the GitHub `prod` Environment reviewer). Mechanism = protected `main` ruleset + the platform's "require manual promotion to production" if available (§10 open decision). +- [ ] C5. **Repoint webhooks + OAuth URLs** to prod: + - GitHub App `seahaven-openswe` webhook URL → prod backend webhook URL (`*.langgraph.app/webhooks/github` or `hooks.seahaven.com` — §10 custom-domain decision) + - Slack Event Subscriptions + Interactivity Request URLs → prod + - Linear webhook URL → prod (if/when wired) + - GitHub App OAuth callback → `/dashboard/api/auth/callback` (the Vercel prod origin; fixes #2, #3) +- [ ] C6. **Custom-domain decision** (§10): dashboard custom domain solved by Vercel; webhooks `hooks.seahaven.com` either re-point to `*.langgraph.app` directly or front via CNAME. Managed issues `*.langgraph.app` URLs ONLY (no documented custom-domain support on the backend). + +### Phase D — Cutover + verify +- [ ] D1. Run the issue **#9 acceptance criteria on PROD**: durable store survives a revision redeploy; team defaults / user mappings persist; a paused/in-flight run survives a redeploy (the durability win self-host never had). +- [ ] D2. End-to-end prod smoke: `@openswe` GitHub comment → run dispatch → sandbox → draft PR → reply in source channel. +- [ ] D3. Dashboard prod smoke: GitHub OAuth login on the stable Vercel alias (fix #3); admin pages reachable for `CONFIGURED_ADMINS`. +- [ ] D4. Webhook sig-enforcement smoke: unsigned POST to each `/webhooks/*` returns 401; signed returns 200. +- [ ] D5. **Soak** for an agreed window (recommend 3–7 days) with the AWS prod stack still standing as the rollback target (§11) before any teardown. + +### Phase E — AWS decommission (AFTER soak) +Only after managed prod is proven + soaked. See §7 for the precise retire-vs-keep list. Phased teardown: +- [ ] E1. Export the live ALB listener-rule / target-group config for `open-swe-prod` and `open-swe-dev` to JSON (the export is the rollback source — these rules were partly hand-built). +- [ ] E2. Disable the bespoke CD (build-artifacts / cd-infra workflows) so nothing re-deploys the EC2 boxes. +- [ ] E3. `cdk destroy OpenSweProdStack` then `OpenSweDevStack` (mind the **RETAIN** secret shells — they survive and hold the global `open-swe-/*` names; force-delete only EMPTY shells once env is fully migrated; keep populated ones transitionally as the env source — §7). +- [ ] E4. Remove ALB rules/target groups, S3 assets buckets, SSM deploy docs, and the per-env CDK bootstrap qualifier `oswedev` (`CDKToolkit-oswedev`) — once nothing references them. +- [ ] E5. Decide DNS: repoint `openswe.seahaven.com` → Vercel; resolve `hooks.seahaven.com` → `*.langgraph.app` (or CNAME). +- [ ] E6. Leave `OpenSweIamStack` until last — the promotion App + any retained roles may still be referenced. + +### Phase F — Docs / memory +- [ ] F1. Rework Confluence "AWS Architecture Map" (page **1540098**) + the open-swe child page (**26116098**): the RDS/EC2/ALB subgraph is replaced by an **external-services view** (LangGraph Cloud + Vercel + the retained GitHub App + LangSmith sandbox). +- [ ] F2. Update `README.md` + `INSTALLATION.md` (§10 is already canonical-correct; align Sea Haven specifics). +- [ ] F3. Update project memories `project_open_swe_migration.md` + `reference_open_swe_deployment.md` to mark the managed cutover and the AWS teardown. + +--- + +## 4. Env / Config Reference + +### The prod triad (per INSTALLATION.md §10, lines 630–658) +| Var | Prod value | Notes | +|---|---|---| +| `LANGGRAPH_URL` | the deployment URL (`https://...langgraph.app`) | **NOT** localhost. Drives `thread_ops.langgraph_url()` (fix #1). | +| `DASHBOARD_BASE_URL` | the Vercel origin | same-origin rewrite mode | +| `DASHBOARD_API_BASE_URL` | the Vercel origin, **with `https://` scheme** | scheme required or OAuth `redirect_uri` is schemeless and GitHub rejects (fix #2) | +| `VITE_DASHBOARD_API_BASE_URL` | **empty** | same-origin mode; UI calls relative `/dashboard/api/*`, Vercel rewrites to backend | + +GitHub App dashboard OAuth callback = `/dashboard/api/auth/callback` (the Vercel prod origin). + +### Where secrets/env live now +**LangGraph Cloud Deployment config + Vercel env** — NOT AWS Secrets Manager. This **deviates from the Sea Haven `secrets-and-config.md` "Secrets Manager for all sensitive" handbook rule** — **Adam ACCEPTED this deviation** (managed has no instance role / no fetch-config boot hook; the platform's own secret store is the mechanism). + +### Sourcing env from the existing AWS stack (transitional) +Pull from the existing `open-swe-dev` Secrets Manager + SSM via a documented CLI dump: +- `aws secretsmanager list-secrets --filters Key=name,Values=open-swe-dev/` then per-name `get-secret-value` — **NOT** `batch-get-secret-value` (its pagination silently drops values past page 1; this bit the box twice — see memory). +- SSM: `aws ssm get-parameters-by-path --path /open-swe-dev/`. + +### EXCLUDE when copying to managed +| Category | Vars | +|---|---| +| Box-/self-host-specific | `LANGGRAPH_URL` (set fresh to deployment URL), `LANGGRAPH_URL_PROD`, `LANGSMITH_ENDPOINT`, `LANGSMITH_ENDPOINT_PROD`, `LANGSMITH_URL_PROD`, `LANGSMITH_TENANT_ID_PROD`, `LANGSMITH_HOST_API_URL`, `LANGCHAIN_REVISION_ID` | +| Dropped providers (PR #62) | `OPENAI_API_KEY`, `GOOGLE_API_KEY`, `GROQ_API_KEY` (these 6 Secrets Manager shells deleted 2026-06-29, final purge 2026-07-06) | +| Unused sandbox providers | `DAYTONA_API_KEY`, `RUNLOOP_API_KEY` | +| Eval-only | `JUDGE_ANTHROPIC_API_KEY` (kept on AWS for `evals/reviewer/judge.py`, not needed in the runtime deployment) | + +### KEY mappings on managed +- `LANGSMITH_API_KEY` = the **`LANGSMITH_API_KEY_PROD`** value (and `LANGCHAIN_API_KEY` = same). +- The full secret inventory is `SECRET_VARS` in `infra/lib/constructs/config-store.ts:46` (28 shells) — use it as the checklist of what to carry, minus the EXCLUDE rows above. + +### Models (post-#62) + Bedrock auth +Post-#62 `SUPPORTED_MODELS` = `bedrock_converse:us.anthropic.claude-opus-4-8` (**DEFAULT**) + 3 Fireworks models; **no direct `anthropic:` option**. +✅ **#62 is merged to `dev` and deployed** — verified on `origin/dev` (`options.py` → `bedrock_converse` default) and the live deployment `git_ref_sha a4ed19ba`. (The pre-#62 values appear only on a stale LOCAL checkout — ignore.) + +**Bedrock on managed has NO EC2 instance role** (the #62 IAM design used the EC2 instance role as the Bedrock principal — that breaks on managed). Options (OPEN DECISION, §10): +- (a) Static `AWS_ACCESS_KEY_ID` / `AWS_SECRET_ACCESS_KEY` / `AWS_REGION` for a **Bedrock-scoped IAM user** in the deployment env, or +- (b) Run the agent on **Fireworks** and avoid Bedrock entirely on managed. + +--- + +## 5. Required Code Fixes (the 6 gotchas) + +These are self-host → managed assumption breaks discovered on the spike. Each gets a config workaround now and a code fix to land in Phase B. + +### Fix #1 — `langgraph_url()` localhost fallback +**File:** `agent/utils/thread_ops.py:38-41` +```python +def langgraph_url() -> str: + return os.environ.get("LANGGRAPH_URL") or os.environ.get( + "LANGGRAPH_URL_PROD", "http://localhost:2024" + ) +``` +On managed, every `langgraph_client()` call (thread sidebar, run creation, the same pattern in `agent/dashboard/user_mappings.py:43` `get_client()` with no URL) `ConnectError`s unless `LANGGRAPH_URL` is set to the deployment URL. +- **Config fix (now):** set `LANGGRAPH_URL` = the deployment URL on every deployment (Phase A4 / C3). +- **Code fix (Phase B):** default to the in-process / deployment URL rather than `http://localhost:2024` when running inside a managed deployment (e.g. honor the platform's own URL env, or fail loud instead of silently hitting localhost). + +### Fix #2 — OAuth `DASHBOARD_API_BASE_URL` must include scheme +**Where:** OAuth redirect construction (dashboard `oauth.py` / `routes.py`), driven by `DASHBOARD_API_BASE_URL`. +A schemeless value yields a schemeless `redirect_uri` → GitHub rejects with "redirect_uri not associated." +- **Fix:** always set `DASHBOARD_API_BASE_URL` **with `https://`** (Phase C3). Code hardening (Phase B): assert/normalize a scheme at startup and fail loud if missing. + +### Fix #3 — OAuth state cookie is host-only +**Cookie:** `osw_oauth_state` (host-only). Login must START on the SAME host as `DASHBOARD_API_BASE_URL` — the **stable Vercel alias**, never the immutable per-deploy URL — or you get "oauth state mismatch." +- **Fix:** pin `DASHBOARD_API_BASE_URL` + the login entry point to the stable alias / custom domain (Phase C3, D3). Document this so a future per-deploy preview URL isn't used for login. + +### Fix #4 — Admin "User mappings" UI can't set `work_email` +**Files:** the Admin → User mappings UI (`ui/`) + `agent/dashboard/user_mappings.py` (`upsert_mapping` at line 257 takes `work_email` but the UI form doesn't expose it). +Mappings can't be fully created from the dashboard → must write the Store directly: namespace `["user_mappings"]`, key = **lowercased login**, value `{github_login, work_email}` (matching `_index_record` at `user_mappings.py:73-84`, which keys `_by_login` on `login.lower()`). +- **Fix (Phase B):** add the `work_email` field to the Admin mappings form so mappings are fully creatable from the UI. + +### Fix #5 — GitHub webhook path doesn't refresh the user-mapping cache (multi-replica break) +**Files:** `agent/webapp.py:3052` (GitHub path) vs `agent/webapp.py:1091` (Slack path). +The Slack path refreshes before lookup: +```python +# agent/webapp.py:1089-1093 (Slack) +await refresh_user_mapping_cache() +... +``` +The GitHub path does **not** — it calls `email = await email_for_login(github_login)` (`webapp.py:3052`, again at `:3331`) cold. The cache (`user_mappings.py` `_ensure_cache_loaded`, line 197) is **one-shot per process** (`_cache_loaded` flag). On self-host single-process this was fine; on managed's **multi-replica autoscaling**, a freshly-added mapping isn't seen by a replica whose cache loaded earlier — until restart. +- **Fix:** refresh-before-lookup on the GitHub path (mirror the Slack path), or add a TTL / cross-replica invalidation to the cache. +- **Generalize (Phase B5):** audit ALL in-process caches for the single-process → multi-replica assumption — `SANDBOX_BACKENDS` dict (`agent/utils/sandbox_state.py`), `_THREAD_RUN_LOCKS` (`thread_ops.py:18`), `_by_login`/`_by_email`/`_by_slack_id` (`user_mappings.py:67-69`). Sandbox affinity is already thread-keyed + persisted in thread metadata (`sandbox_id`), so it's the cache/lock state that needs the multi-replica review. + +### Fix #6 — Slow custom-app import (~8s startup) +**Symptom:** "exceeded expected startup time" → risks the deployment being marked unhealthy / slow to scale out. +- **Fix (Phase B6):** lazy imports / reduce import-time work in `agent/webapp.py` and the graph factories. Profile with `FF_PROFILE_IMPORTS` (the import-profiling flag) to find the heavy modules. + +--- + +## 6. CD / Ops Changes + +### Retires (bespoke pipeline) +- GitHub Actions **build → S3 releases → SSM doc → `deploy.sh` → health-gate** (`.github/scripts/`, `deploy/ami/deploy.sh`) +- `roll-box.sh` / `publish-and-deploy.sh` / `rollback.sh` +- AMI baking (packer `deploy/ami/open-swe-base.pkr.hcl`) + `cdk.context.json` AMI pin +- `cd-infra.yml` CDK deploys + OIDC bootstrap qualifiers +- `fetch-config.sh` / `seed_store.sh` (boot-time config materialization + Store reseed) + +### Replaced by git-connected PaaS +- **LangGraph Cloud** auto-builds a revision on push (first-party zero-downtime + revision rollback). +- **Vercel** builds / atomic-deploys / instant-rollback + per-PR previews (**net-new** capability the AWS stack never had). + +### Stays +- **CI** (lint / format / unit / Playwright E2E) — still matters and still gates merges. `make lint`, `make test`. + +### Gate re-homing +- The **dev → main promotion gate** (`check-dev-green.sh` + protected-`main` ruleset 18238334) **re-homes**: the **dev deployment tracks `dev`**, the **prod deployment tracks `main`**, so the gate governs exactly what reaches prod. +- Today's promotion machinery: `promote-dev-to-prod.yml` + the `seahaven-promotion` GitHub App (app_id 4170147, in ruleset 18238334 bypass_actors) FF-pushes `dev → main`. Under managed, a push to `main` is what triggers the prod build — so the promotion gate IS the prod deploy gate. + +### Preserve the PROD manual-approval gate (LOAD-BEARING) +Today the `prod` GitHub Environment (required reviewer `amoussa1229`) is the manual approval. Git-connected auto-deploy removes the CD job that consulted that Environment, so the approval must be re-established explicitly: +- **Mechanism options (OPEN DECISION §10):** protected `main` (only the promotion App can FF-push, and that push is itself the gate) AND/OR the platform's "require manual promotion to production" if LangGraph Cloud exposes it. +- **Norm shift:** deploy-then-merge → **merge-to-deploy**. Lean on the dev deployment + Vercel previews to verify *before* promoting `dev → main`. (This inverts the Sea Haven handbook deploy-then-merge default — call it out in the README + handbook note.) + +--- + +## 7. AWS Decommission — Retire vs Keep + +### RETIRE (becomes vestigial once managed prod is live + soaked) +| Item | Path / resource | +|---|---| +| AMI bake + cloud-init | `deploy/ami/` (`open-swe-base.pkr.hcl`, `provision.sh`, `user-data.sh`, `deploy.sh`, `templates/`) | +| Self-host boot/config scripts | `deploy/seahaven/fetch-config.sh`, `seed_store.sh`, `put-config.sh`, `nginx/openswe.conf`, `systemd/open-swe.service`, `aegra/` (already deferred) | +| CDK box/ALB/AMI stacks | `infra/lib/open-swe-stack.ts`, `constructs/app-service.ts`, `assets-bucket.ts`, `ami-cache.ts`, `instance-role.ts`, `github-deploy-roles.ts` | +| EC2 instances | prod `i-08a729e50779c4b07` (t4g.large), dev `i-0a3bb8e0ddd36c29b` | +| S3 release buckets | `open-swe-dev-assets`, `open-swe-prod-assets` | +| SSM deploy docs | `open-swe-dev-deploy`, `open-swe-prod-deploy` | +| ALB plumbing | `open-swe-prod-tg` / `open-swe-dev-tg` target groups + the `seahaven-com` ALB host/path listener rules for openswe/hooks | +| Per-env bootstrap qualifier | dev `oswedev` (`CDKToolkit-oswedev`) | +| RDS / Redis durable plan | **never built** — fully dropped (managed provides durability) | +| Bespoke CD workflows | `cd-infra.yml`, `build-artifacts.yml`, `roll-box.sh`/`publish-and-deploy.sh`/`rollback.sh` | + +### KEEP +| Item | Why | +|---|---| +| **GitHub App `seahaven-openswe`** (App 4146115 / Install 142615168) | unchanged auth + webhook source; only the webhook/OAuth URLs repoint | +| **LangSmith sandbox setup** incl. `DEFAULT_SANDBOX_SNAPSHOT_ID` (`dc36e509-d4d2-4efc-8a4e-61f74f3446ec`) | the only sandbox with working in-sandbox git/gh auth (`_configure_github_proxy`); managed default `SANDBOX_TYPE=langsmith` uses it | +| **Secrets Manager `open-swe-{dev,prod}/*` values** | the **source you copy env FROM**, at least transitionally — do NOT force-delete populated shells until managed env is fully cut over and verified | +| **DNS** | repoint `openswe.seahaven.com` → Vercel; decide `hooks.seahaven.com` → `*.langgraph.app` or a CNAME (§10) | +| **`seahaven-promotion` App** + main/dev rulesets (18238334 / 18238542) | still govern `dev → main` promotion = the prod deploy gate | +| **`evals/` + `JUDGE_ANTHROPIC_API_KEY`** | eval pipeline unaffected (separate from runtime) | + +**Phased teardown rule:** remove AWS infra ONLY after managed prod is proven + soaked (Phase D5). The exported ALB JSON (E1) is the rollback source for the hand-built listener rules. + +--- + +## 8. Cost + +| Line item | Monthly | +|---|---| +| Prod deployment uptime ($0.0036/min, always-on) | ≈ $155 | +| Runs ($0.005/run) | usage-based | +| Traces above 10k/mo | pay-as-you-go | +| LangSmith **Plus** seat ($39) | already paid (not incremental) | +| Dev deployment | **$0** (1 free on Plus, preemptible) | +| **Incremental total** | **≈ $160/mo** | + +Compare to the retired AWS stack: 2× EC2 (t4g.large prod + t4g.medium/large dev), 2× S3 buckets, ALB share, NAT egress, plus the **never-built** RDS/Redis durable tier the self-host path would have added. Managed nets out roughly cost-neutral-to-cheaper while *adding* durability + previews. **TODO: confirm current AWS run-rate for an apples-to-apples delta** (not verified in repo). + +--- + +## 9. Sea Haven Gates & Obligations + +- [ ] **`/sh-plan-review`** (GPT-4.1 `cross_reviewer`) on THIS plan **BEFORE prod cutover** (Phase C gate). ⚠️ `run.py` misroutes reviewer-framed prompts to the no-op `done` route — call `models.get_cross_reviewer()` directly (load orchestrator `.env`, `ChatOpenAI("gpt-4.1")`) per `feedback_orchestrator_usage`. +- [ ] **`/sh-security-review`** on the sensitive surface: this migration touches **auth/webhook signature verification** (the custom `http.app` is now publicly reachable on `*.langgraph.app` with no platform gate in front) and **secrets handling** (secrets move from Secrets Manager to the Deployment/Vercel env). Required, not opt-in — resolve confirmed critical/high before cutover. +- [ ] **GPT-4.1 cross-family review** if the Bedrock-auth fix changes IAM (new Bedrock-scoped IAM user + policy = IAM change → mandatory cross-review). +- [ ] **Confluence** — rework "AWS Architecture Map" (**1540098**) + open-swe child page (**26116098**): replace the RDS/EC2/ALB subgraph with an external-services view (LangGraph Cloud + Vercel + retained GitHub App + LangSmith sandbox). Do it in the same conversation as the cutover, not as a follow-up. +- [ ] **README + INSTALLATION** updated (INSTALLATION §10 already canonical; align Sea Haven specifics + the merge-to-deploy norm shift). +- [ ] **Memories** — update `project_open_swe_migration.md` + `reference_open_swe_deployment.md`. + +--- + +## 10. Decisions — RESOLVED (2026-06-29, Adam) + +1. **Bedrock auth on managed → STATIC AWS KEYS.** Create a **Bedrock-scoped IAM user**; set `AWS_ACCESS_KEY_ID` / `AWS_SECRET_ACCESS_KEY` / `AWS_REGION=us-east-1` in the deployment env; reuse the #62 instance-role Bedrock policy (`bedrock:InvokeModel[WithResponseStream]` on the `us.anthropic.claude-opus-4-8` inference-profile ARN + per-region foundation-model ARNs in us-east-1/2 + us-west-2). Caveat: long-lived keys → rotation reminder + keep the policy tightly scoped. **New IAM user+policy = mandatory GPT-4.1 IAM cross-review + `/sh-security-review` before merge.** +2. **Webhook domain → RAW `*.langgraph.app`** (the "easier" path — no CNAME/DNS/TLS). Point GitHub/Slack/Linear webhook URLs straight at the deployment. Dashboard gets a **Vercel custom domain** (e.g. `openswe.seahaven.com`) since that's trivial on Vercel. +3. **Prod approval gate → GATED PROMOTE-TO-MAIN WORKFLOW.** Re-home `promote-dev-to-prod.yml`: gate the promote job on the **`prod` GitHub Environment reviewer (`amoussa1229`)**; on approval it FF-pushes `dev → main`; the push triggers the managed prod build. Preserves today's manual-approval semantics — approval is on the promote job, the merge to `main` is the deploy trigger. +4. **Cost/trace monitoring → YES, in LangSmith** (usage/trace budget + alert in the LangSmith workspace). +5. **Dev/prod home → CONFIRMED:** prod = LangChain managed (LangGraph Cloud) + Vercel UI; dev = managed **free** Dev tier (same runtime as prod). +6. ~~#62 merge status~~ — RESOLVED (merged + deployed). + +## 10b. Execution approach — 2-PR SPLIT (Adam's call, 2026-06-29, post-plan-review) + +The plan-review BLOCKed a literal single PR and recommended a 3-PR minimum (isolate cache fix #5 + the promotion-gate workflow). Adam chose a **2-PR split** — isolating the prod-deploy-gate change (highest risk to prod), bundling the rest: +- **PR1 — promotion-gate workflow + ruleset.** Re-home `promote-dev-to-prod.yml` to the gated promote-to-main (decision 3), and lock `main` so ONLY the promote App can push (BLOCK 3). Isolated because it changes *how prod deploys* — must be independently reviewable/revertible. +- **PR2 — the 6 code fixes (§5) + `ui/vercel.json` prod repoint + README/INSTALLATION.** (Accepted deviation from the reviewer's 3-PR rec: the multi-replica cache fix #5 stays bundled in PR2 rather than isolated — Adam's call.) +- **NOT in either PR (rollback safety):** AWS **infra-code deletion** (`deploy/ami`, `deploy/seahaven`, `infra/` self-host stack) + `cdk destroy` stay a **POST-SOAK cleanup (Phase E)**. +- **Not PR content (operational, sequenced around the merges):** LangGraph Cloud prod deployment + env, Vercel prod + custom domain, webhook/OAuth repoint, Bedrock IAM user, LangSmith budget, Confluence/memory. + +## 10c. Plan-review resolution (GPT-4.1 cross_reviewer, 2026-06-29) — verdict REQUEST CHANGES, all BLOCKs addressed + +- **BLOCK 1 (single PR)** → addressed via the **2-PR split** above (conscious deviation: cache fix #5 not isolated — Adam accepted). +- **BLOCK 2 (raw LangGraph API public?)** → **EMPIRICALLY RESOLVED.** Unauthenticated probes of the dev deployment returned **403 "Missing authentication headers"** for `/threads/search`, `/assistants/search`, `/store/items`; `/ok`=200, `/dashboard/api/me`=401. The platform gates the raw control-plane API. **Keep as a pre-prod gate (Phase D4):** re-probe the PROD deployment URL before cutover. +- **BLOCK 3 (prod gate enforcement)** → PR1 must **lock `main`** so only the promote App can push (verify ruleset 18238334: no direct-push path, PR-required, promote App is the sole FF bypass). Any non-gated push to `main` would auto-deploy prod. +- **BLOCK 4 (secrets posture)** → add to Phase B/E: **rotate** all secrets after migrating them to the platform env stores and BEFORE deleting from AWS; **document who can read/write** the LangGraph Cloud + Vercel env (access audit); delete AWS shells only post-cutover; no dual-homed/stale secrets; record in memory. +- **FIX (rollback integrity)** → Phase E checklist: "no IaC/DNS/config the rollback needs is altered in PR1/PR2 or during cutover." +- **FIX (gates resolved-before-merge)** → make explicit: do NOT merge PR1/PR2 until the GPT-4.1 IAM cross-review (Bedrock IAM user) + `/sh-security-review` (auth/webhook/secrets surface) are **resolved with no critical/high**; Confluence (1540098 + 26116098) + memory updated in the cutover conversation. +- **NITs/QUESTIONs** → carried into §8/§10 TODOs (cost delta, `FF_PROFILE_IMPORTS`, platform feature availability, key-rotation owner, env-store access logging, cache race-review under autoscaling, atomic webhook repoint). + +--- + +## 11. Rollback Story + +**Pre-cutover:** the self-host `langgraph up` + RDS plan (RDS+Redis+Docker; `/sh-plan-review`'d to APPROVE-after-revision this session) is the documented **fallback** if managed is rejected before cutover. Its durable design wins (env-scoped RDS physical names; `Credentials.fromGeneratedSecret({secretName:"open-swe-/rds-credentials"})` under the existing instance-role secret prefix → no new IAM; DESTROY-on-rollback RDS-managed secret; derive `DATABASE_URI` in fetch-config) are captured in the project memory. + +**Post-cutover (managed is live, AWS still standing during soak):** if managed prod fails, **rollback = repoint webhooks + DNS back** to the AWS stack: +- GitHub App / Slack / Linear webhook URLs → `hooks.seahaven.com` +- `openswe.seahaven.com` DNS → the `seahaven-com` ALB (restore from the E1 export) +- `ui/vercel.json` rewrite → the AWS dashboard origin (or stop using Vercel) +- Do NOT run Phase E teardown until the soak passes — the AWS stack IS the rollback target. + +**Revision-level rollback (within managed):** LangGraph Cloud revision rollback (backend) + Vercel instant rollback (UI) cover bad deploys without leaving the platform. + +--- + +## TODOs flagged (couldn't verify in repo) +- ~~#62 merge state~~ — RESOLVED: merged to `dev` + deployed (the draft read a stale local checkout; `origin/dev` and the deployment `git_ref_sha a4ed19ba` confirm Bedrock+Fireworks). +- **Current AWS run-rate** for the cost delta (§8) — not derivable from repo. +- **LangGraph Cloud custom-domain + "manual promotion to production" feature availability** (§10.2, §10.3) — platform features, verify in the LangGraph Cloud console/docs at execution time. +- **`FF_PROFILE_IMPORTS`** exact flag name/usage (§5 fix #6) — referenced from session context; confirm the flag exists in `agent/` before relying on it. +- Exact webhook path on `*.langgraph.app` (whether the custom `http.app` mounts at root so `/webhooks/github` is reachable as-is) — proven reachable on the dev spike; re-verify the exact path on prod. diff --git a/ui/src/lib/api.ts b/ui/src/lib/api.ts index e60ee8f9..5e1923db 100644 --- a/ui/src/lib/api.ts +++ b/ui/src/lib/api.ts @@ -732,6 +732,15 @@ export const api = { request( `/admin/user-mappings?page=${page}&page_size=${pageSize}` ), + adminUpsertUserMapping: (input: { + github_login: string + work_email: string + slack_user_id?: string | null + }) => + request("/admin/user-mappings", { + method: "POST", + body: JSON.stringify(input), + }), adminDeleteUserMapping: (github_login: string) => request<{ deleted: boolean }>( `/admin/user-mappings/${encodeURIComponent(github_login)}`, diff --git a/ui/src/routes/admin.tsx b/ui/src/routes/admin.tsx index e21941cc..db57b51d 100644 --- a/ui/src/routes/admin.tsx +++ b/ui/src/routes/admin.tsx @@ -171,6 +171,8 @@ const PAGE_SIZE = 20 function UserMappingsSection({ enabled }: { enabled: boolean }) { const [error, setError] = useState(null) const [page, setPage] = useState(1) + const [newLogin, setNewLogin] = useState("") + const [newEmail, setNewEmail] = useState("") const mappings = useQuery({ queryKey: ["adminUserMappings", page], @@ -193,16 +195,63 @@ function UserMappingsSection({ enabled }: { enabled: boolean }) { onError: (e: Error) => setError(e.message), }) + const upsert = useMutation({ + mutationFn: () => + api.adminUpsertUserMapping({ + github_login: newLogin.trim(), + work_email: newEmail.trim(), + }), + onSuccess: () => { + setNewLogin("") + setNewEmail("") + setError(null) + void mappings.refetch() + }, + onError: (e: Error) => setError(e.message), + }) + + const canSubmit = + newLogin.trim().length > 0 && + newEmail.trim().length > 0 && + !upsert.isPending + const items = mappings.data?.items ?? [] return (
{error && {error}} +
{ + e.preventDefault() + if (canSubmit) upsert.mutate() + }} + > + setNewLogin(e.target.value)} + placeholder="github-login" + aria-label="GitHub login" + className="h-8 text-xs" + /> + setNewEmail(e.target.value)} + placeholder="work@example.com" + aria-label="Work email" + type="email" + className="h-8 text-xs" + /> + +
+
{mappings.isLoading ? (