From 404b3f6f7558f0ca8be746692e332c5aad5c3b3e Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Fri, 26 Jun 2026 18:49:09 -0400 Subject: [PATCH] =?UTF-8?q?feat:=20stand=20up=20dev=20properly=20=E2=80=94?= =?UTF-8?q?=20assets=20bucket=20+=20artifact=20CD=20+=20baked=20AMI=20+=20?= =?UTF-8?q?on-box=20uv=20sync=20(T7+T19+T14)=20(#18)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(infra): build + pin the baked open-swe-base-arm64 AMI (T12 AMI / item 3) Packer-build the custom base image and repoint AppService off the AL2023 placeholder onto it. deploy/ami/open-swe-base.pkr.hcl — fix two bugs that blocked the first real `packer build` (the config had only ever been `packer validate`'d at T8): - the file provisioner failed uploading the templates dir ('scp: …: Is a directory') — a trailing-slash contents-upload needs the dest dir to exist; added a 'mkdir -p /tmp/open-swe-templates' shell provisioner + dropped the dest trailing slash. - the shell provisioner's custom execute_command omitted {{ .Vars }}, so the environment_vars never reached provision.sh (which runs under set -u and aborted on CLOUDWATCH_AGENT_DEB_URL). Added {{ .Vars }}. infra: - ami-cache.ts: BAKED_OPEN_SWE_AMI_ID = ami-0545363bb147229ff (built 2026-06-26 from open-swe-base-arm64-20260626-201929) + bakedOpenSweArm64() pinning it by exact id via MachineImage.genericLinux (offline, deterministic). Dropped the now-dead AL2023 cachedInContext helper + context key; kept the EBS/replacement discipline docs. - app-service.ts: machineImage → bakedOpenSweArm64(). - open-swe-stack.ts: output BakedAmiId (was the AL2023 PinnedAmiId guard). - cdk.context.json → {} (AMI is a static id pin; no context lookups remain). - README: Baked AMI + EBS-replacement-discipline section. tsc + cdk synth(dev+prod) + jest(16) clean; template ImageId = the baked AMI. NOTE: held — do NOT merge until the open-swe-dev secret values are populated (put-config.sh). The infra CD is live, so merging this to dev auto-deploys OpenSweDevStack; without secrets the box boots but fetch-config fail-fasts → unhealthy ALB target on the shared prod ALB. Merge once secrets are set (T14). * fix(ami): ASCII-only AMI description + re-pin to ami-00080084502093021 Third packer bug: ami_description had an em-dash (non-ASCII); AWS rejects non-ASCII in the AMI Description attribute, so packer registered then DEREGISTERED the first AMI (ami-0545…) on the ModifyImageAttribute error. Replaced with an ASCII '-'. Rebuilt clean → ami-00080084502093021 (available). Re-pinned BAKED_OPEN_SWE_AMI_ID. * fix(deploy): GitHub App + Slack required for prod only, not dev Per the migration decision: do NOT create/duplicate a separate dev GitHub App or Slack app — only prod owns the single shared app. So fetch-config.sh no longer hard-requires the GitHub App quintet (ID/PRIVATE_KEY/INSTALLATION_ID/CLIENT_ID/ CLIENT_SECRET) + Slack/webhook secrets for dev; they move into the prod-only block alongside the existing GITHUB_WEBHOOK_SECRET/SLACK_SIGNING_SECRET. Dev now boots with just DASHBOARD_JWT_SECRET + TOKEN_ENCRYPTION_KEY + the active provider key(s) + the langsmith sandbox keys. Dev is a deployment-validation env (boot/health/boundary) with no GitHub/Slack/webhook integration; prod parity is unchanged (prod still requires everything). * feat: stand up dev properly — S3 assets bucket + artifact CD + on-box uv sync (T7+T19) Make the dev/prod box deployable end-to-end: a real artifact pipeline and a re-runnable on-box deploy, so OpenSweDevStack can come up genuinely healthy. Infra (T7): - assets-bucket.ts: open-swe--assets S3 bucket — BLOCK_ALL public access, SSE-S3, enforceSSL (deny non-TLS), versioned, lifecycle (expire noncurrent + abort MPU), RETAIN. Wired into OpenSweStack + CfnOutput. - app-service.ts: open-swe--deploy SSM document that runs the baked /opt/open-swe/bin/deploy.sh (tag-scoped roll-the-box). machineImage is the baked open-swe-base-arm64 AMI (folds in the held #16). IAM (app deploy role — cross-review gated): - github-deploy-roles.ts: app role gains s3:PutObject/DeleteObject scoped to open-swe--assets/releases/* (CI uploads releases). Drops the generic AWS-RunShellScript grant now that the dedicated open-swe--deploy document is the only SendCommand path — closes the T4 BLOCK#3 arbitrary-shell timebox. Boot/deploy (T19): - deploy/ami/deploy.sh: single, re-runnable app-deploy procedure — pull app.tar.gz/spa.tar.gz from S3, `uv sync --frozen --no-dev` (native ARM64 venv at the real path, py3.12 pre-baked), restart open-swe.service + reload nginx. - user-data.sh: nginx starts BEFORE the app deploy (static /healthz -> the ALB target is healthy even before the first release); deploy.sh is base64-rendered by CDK into user-data (a normal reviewable repo file, not a heredoc) and the first-boot deploy is NON-FATAL (no release yet -> wait for the first SSM deploy). CI (T7+T19): - build-artifacts.yml (+ .github/scripts): build the SPA with bun (vite -> ui/.output/public -> spa.tar.gz), package the Python source via git archive (app.tar.gz, no ui/ no .venv), upload to releases// + releases/latest/ via the githubdeploy-open-swe-app- OIDC role, then fire open-swe--deploy. push dev -> dev (auto); push main -> prod (env "prod" approval gate). Local: ruff/shellcheck clean, tsc clean, jest 16/16, cdk synth offline OK, deploy.sh base64 round-trips exact. * harden(sec-review): tar extraction, deploy gating, least-privilege, secret guard Address the /sh-security-review fan-out + proof-or-kill verifier pass. Only one confirmed-high surfaced and it is PRE-EXISTING and out-of-diff (OSWE-IAC-AUDIT-01, the account-wide CDK cfn-exec residual already documented in config.ts; recorded in .security-review/suppressions.json with justification + flagged for the per-env bootstrap-qualifier follow-up). The rest were verifier-downgraded to unverified; these are the cheap defense-in-depth fixes worth taking regardless: - deploy.sh: extract tarballs with --no-same-owner --no-same-permissions (root never honors an archive's uid/mode → no setuid/foreign-owned file can land); and treat "no release in S3 yet" as a benign exit 0, distinct from a real deploy failure (set -e stays loud once a release exists). - publish-and-deploy.sh: gate on the AGGREGATE SSM Command.Status (+ TargetCount), not CommandInvocations[0], so a partial failure across the brief 2-instance replacement window can't be reported as success. - instance-role.ts: scope the box's s3:GetObject to releases/* (mirrors the app role's write scope) instead of the whole bucket. - package-artifacts.sh: fail-closed secret-shaped-file guard on app.tar.gz (defense in depth over .gitignore; scoped to data extensions so *_credentials.py source is not a false positive — verified against the real tree). Deferred as documented follow-ups (verifier: unverified, supply-chain-gated to the CI OIDC writer; bucket is BLOCK_ALL + enforceSSL + versioned): SHA-pinned immutable releases// pulls + signed checksum (vs mutable latest/), single-tarball release to remove the torn-read window, and app-aware ALB health (vs static nginx /healthz). shellcheck/tsc/jest(16) clean; both stacks synth offline. * fix(infra): ASCII-only EC2 SecurityGroup descriptions + synth-time guard The instance-SG GroupDescription + ingress/egress rule descriptions carried an em-dash / arrow (—, →). `tsc` and `cdk synth` accept them, but the EC2 API rejects non-ASCII in GroupDescription ("Character sets beyond ASCII are not supported"), so OpenSweDevStack's first deploy failed at the SG and rolled back. (Pre-existing from #14; same class as the AMI-description ASCII bug.) - app-service.ts: replace —/→ with ASCII (- / ->) in the SG GroupDescription, the ingress/egress rule descriptions, and the Route53 comment. - test/ascii-aws-fields.test.ts: synth-time guard asserting EC2 SecurityGroup GroupDescription + rule descriptions are pure ASCII, so this fails the build instead of a deploy next time. jest 18/18; tsc clean. * fix(infra): SG rule descriptions use ASCII-charset-safe text (no `>`) The first ASCII fix replaced the arrow with `->`, but EC2 SecurityGroup *rule* descriptions allow a stricter set than ASCII — `a-zA-Z0-9. _-:/()#,@[]+=&;{}!$*`, which EXCLUDES `<`/`>`. So OpenSweDevStack's second deploy still failed at the ingress rule. Use "to" instead of "->", and tighten the guard test from "ASCII only" to the exact EC2 allowed charset so it catches `>` (and `<`) too. jest 18/18; tsc clean. * fix(infra): minify embedded deploy.sh so user-data fits EC2's 25.6 KB limit The base64 deploy.sh embedded in user-data pushed the encoded boot script to 27184 bytes, over EC2's 25600-byte cap, so OpenSweDevStack's instance failed with "Encoded User data is limited to 25600 bytes". Strip full-line comments + blank lines from deploy.sh before base64-embedding it (repo file keeps comments; only the on-box copy is minified; the script is opaque base64 so user-data heredocs are unaffected) -> rendered user-data drops to 16424 bytes (9 KB margin). Add a synth-time guard test asserting EC2 user-data stays under 25600 bytes encoded. jest 19/19; minified deploy.sh passes bash -n + shellcheck. --- .github/scripts/package-artifacts.sh | 46 ++++++++ .github/scripts/publish-and-deploy.sh | 75 +++++++++++++ .github/workflows/build-artifacts.yml | 114 ++++++++++++++++++++ .gitignore | 3 + .security-review/suppressions.json | 14 +++ deploy/ami/deploy.sh | 102 ++++++++++++++++++ deploy/ami/open-swe-base.pkr.hcl | 18 +++- deploy/ami/user-data.sh | 53 +++++---- deploy/seahaven/fetch-config.sh | 25 +++-- infra/README.md | 38 +++---- infra/cdk.context.json | 4 +- infra/lib/constructs/ami-cache.ts | 83 +++++--------- infra/lib/constructs/app-service.ts | 102 +++++++++++++----- infra/lib/constructs/assets-bucket.ts | 58 ++++++++++ infra/lib/constructs/github-deploy-roles.ts | 31 +++--- infra/lib/constructs/instance-role.ts | 5 +- infra/lib/open-swe-stack.ts | 38 ++++--- infra/test/ascii-aws-fields.test.ts | 74 +++++++++++++ 18 files changed, 718 insertions(+), 165 deletions(-) create mode 100755 .github/scripts/package-artifacts.sh create mode 100755 .github/scripts/publish-and-deploy.sh create mode 100644 .github/workflows/build-artifacts.yml create mode 100644 .security-review/suppressions.json create mode 100755 deploy/ami/deploy.sh create mode 100644 infra/lib/constructs/assets-bucket.ts create mode 100644 infra/test/ascii-aws-fields.test.ts diff --git a/.github/scripts/package-artifacts.sh b/.github/scripts/package-artifacts.sh new file mode 100755 index 00000000..e9771517 --- /dev/null +++ b/.github/scripts/package-artifacts.sh @@ -0,0 +1,46 @@ +#!/usr/bin/env bash +# Package the two release artifacts (run from the repo root by build-artifacts.yml): +# +# spa.tar.gz = CONTENTS of the built SPA dir (ui/.output/public/*), so it extracts +# straight into the nginx web root with _shell.html at the root. +# app.tar.gz = the Python source the box runs `uv sync` against. `git archive` +# gives a clean tree (no node_modules, no .venv, no local cruft); +# ui/ is intentionally excluded (it ships as spa.tar.gz). +set -euo pipefail + +SPA_DIR="ui/.output/public" +[ -f "${SPA_DIR}/_shell.html" ] || { + echo "ERROR: SPA build output missing ${SPA_DIR}/_shell.html (did 'bun run build' run?)" >&2 + exit 1 +} + +echo "==> spa.tar.gz from ${SPA_DIR}" +tar -C "${SPA_DIR}" -czf spa.tar.gz . + +echo "==> app.tar.gz from source (git archive HEAD)" +git archive --format=tar.gz -o app.tar.gz HEAD \ + agent deploy langgraph.json pyproject.toml uv.lock README.md + +# Sanity: the box's `uv sync --frozen` needs pyproject.toml + uv.lock at the root, +# and the package itself (agent/). Fail loudly here rather than on the box. +for required in pyproject.toml uv.lock agent/server.py langgraph.json; do + tar -tzf app.tar.gz | grep -qx "${required}" || { + echo "ERROR: app.tar.gz is missing ${required}" >&2 + exit 1 + } +done + +# Fail-closed secret guard: the source is git-archived wholesale, so reject the +# release if a secret-shaped FILE slipped into the tracked tree (defense in depth +# on top of .gitignore — the artifact lands on the box + in S3). Scoped to data +# extensions so credential-handling *source* (e.g. team_credentials.py) is not a +# false positive. +SECRET_RE='(^|/)(\.env(\..+)?|id_rsa|.*\.(pem|key|p12|pfx)|.*(secret|credential|password|token)s?\.(json|ya?ml|txt|env|ini|cfg))$' +if tar -tzf app.tar.gz | grep -qiE "${SECRET_RE}"; then + echo "ERROR: app.tar.gz contains a secret-shaped file — refusing to publish:" >&2 + tar -tzf app.tar.gz | grep -iE "${SECRET_RE}" >&2 + exit 1 +fi + +echo "==> artifacts:" +ls -la spa.tar.gz app.tar.gz diff --git a/.github/scripts/publish-and-deploy.sh b/.github/scripts/publish-and-deploy.sh new file mode 100755 index 00000000..0fc9cca3 --- /dev/null +++ b/.github/scripts/publish-and-deploy.sh @@ -0,0 +1,75 @@ +#!/usr/bin/env bash +# Publish the packaged artifacts to the env's S3 bucket and roll the box to them. +# Run by build-artifacts.yml AFTER aws creds are configured (env: ENV, BUCKET, +# DEPLOY_DOC). Each release is stored immutably under releases// and mirrored +# to releases/latest/ (what the box's deploy.sh pulls). +# +# The deploy is fired by TAG (project=open-swe,env=), which is exactly what +# the app deploy role's tag-scoped ssm:SendCommand allows — so this needs no +# ec2:DescribeInstances and no instance id up front. +set -euo pipefail + +: "${ENV:?}" "${BUCKET:?}" "${DEPLOY_DOC:?}" +SHA="${GITHUB_SHA:?}" +[ -f app.tar.gz ] && [ -f spa.tar.gz ] || { echo "ERROR: artifacts not built" >&2; exit 1; } + +echo "==> upload release ${SHA} to s3://${BUCKET}/releases/${SHA}/" +for f in app.tar.gz spa.tar.gz; do + aws s3 cp "${f}" "s3://${BUCKET}/releases/${SHA}/${f}" +done + +echo "==> mirror to releases/latest/ (server-side copy)" +for f in app.tar.gz spa.tar.gz; do + aws s3 cp "s3://${BUCKET}/releases/${SHA}/${f}" "s3://${BUCKET}/releases/latest/${f}" +done + +echo "==> fire ${DEPLOY_DOC} via SSM (tag-targeted: project=open-swe, env=${ENV})" +CMD_ID="$(aws ssm send-command \ + --document-name "${DEPLOY_DOC}" \ + --targets "Key=tag:project,Values=open-swe" "Key=tag:env,Values=${ENV}" \ + --comment "release ${SHA}" \ + --query 'Command.CommandId' --output text)" +echo "command: ${CMD_ID}" + +echo "==> wait for the deploy to finish" +STATUS="Pending" +IID="" +for _ in $(seq 1 60); do + sleep 10 + # CommandInvocations is empty until SSM registers the target invocation. + IID="$(aws ssm list-command-invocations --command-id "${CMD_ID}" \ + --query 'CommandInvocations[0].InstanceId' --output text 2>/dev/null || echo None)" + [ -z "${IID}" ] || [ "${IID}" = "None" ] && continue + STATUS="$(aws ssm list-command-invocations --command-id "${CMD_ID}" \ + --query 'CommandInvocations[0].Status' --output text 2>/dev/null || echo Pending)" + case "${STATUS}" in + Success | Failed | Cancelled | TimedOut) break ;; + esac +done + +if [ -z "${IID}" ] || [ "${IID}" = "None" ]; then + echo "ERROR: no box picked up the deploy command (is a running open-swe ${ENV} box registered with SSM?)" >&2 + exit 1 +fi + +echo "==> deploy.sh output from ${IID}:" +echo "----- stdout -----" +aws ssm get-command-invocation --command-id "${CMD_ID}" --instance-id "${IID}" \ + --query 'StandardOutputContent' --output text || true +echo "----- stderr -----" +aws ssm get-command-invocation --command-id "${CMD_ID}" --instance-id "${IID}" \ + --query 'StandardErrorContent' --output text || true + +# Gate on the AGGREGATE command status (Success only if EVERY targeted invocation +# succeeded), not CommandInvocations[0] — during a userDataCausesReplacement window +# two instances can briefly share the project/env tags, and a partial failure on +# the other instance must not be reported as success. +TARGETS="$(aws ssm list-commands --command-id "${CMD_ID}" \ + --query 'Commands[0].TargetCount' --output text 2>/dev/null || echo 1)" +[ "${TARGETS}" = "1" ] || echo "WARNING: deploy fanned out to ${TARGETS} instances (expected 1)" +AGG="$(aws ssm list-commands --command-id "${CMD_ID}" \ + --query 'Commands[0].Status' --output text 2>/dev/null || echo Failed)" + +echo "==> aggregate deploy status: ${AGG} (across ${TARGETS} target(s))" +[ "${AGG}" = "Success" ] || { echo "ERROR: deploy did not succeed (${AGG})" >&2; exit 1; } +echo "==> ${ENV} rolled to release ${SHA}" diff --git a/.github/workflows/build-artifacts.yml b/.github/workflows/build-artifacts.yml new file mode 100644 index 00000000..4da0d95a --- /dev/null +++ b/.github/workflows/build-artifacts.yml @@ -0,0 +1,114 @@ +name: Build & publish app artifacts + +# T7 + T19 — build the release (SPA + Python source) and publish it to the per-env +# S3 artifact bucket, then roll the box to it. +# +# push to dev → publish to open-swe-dev-assets → deploy dev box (AUTO) +# push to main → publish to open-swe-prod-assets → deploy prod box (manual approval: env "prod") +# +# Two artifacts (the box's deploy.sh pulls both from releases/latest/): +# spa.tar.gz = the built dashboard SPA (vite -> ui/.output/public). Built HERE +# (not on the box) — the build is memory-heavy and the box is small. +# app.tar.gz = the Python source tree (NO ui/, NO .venv). The box runs +# `uv sync` to build a native-ARM64 venv at the real runtime path. +# +# Each release is uploaded under releases// (immutable, auditable) AND mirrored +# to releases/latest/ (what the box pulls). Then the open-swe--deploy SSM +# document is fired (tag-scoped to project=open-swe,env=) to roll the box. +# +# OIDC subject alignment (matches the per-env app-role trust in infra/lib/config.ts): +# - publish-dev declares NO `environment:` → sub = repo:…:ref:refs/heads/dev +# - publish-prod declares `environment: prod` → sub = repo:…:environment:prod +# (also triggers the prod Environment's required-reviewer approval gate). +# +# Prerequisites: +# - repo variables AWS_DEPLOY_ROLE_APP_DEV / AWS_DEPLOY_ROLE_APP_PROD = the +# githubdeploy-open-swe-app- role ARNs (open-swe-iam CfnOutputs). +# - the open-swe- stack deployed (creates the bucket + the SSM deploy doc). + +permissions: + contents: read + +on: + push: + branches: [dev, main] + paths: + - "agent/**" + - "ui/**" + - "deploy/**" + - "langgraph.json" + - "pyproject.toml" + - "uv.lock" + - ".github/workflows/build-artifacts.yml" + workflow_dispatch: + +concurrency: + # one publish+deploy per branch at a time; never cancel an in-flight release. + group: build-artifacts-${{ github.ref }} + cancel-in-progress: false + +jobs: + publish-dev: + name: Publish + deploy (dev) + if: ${{ github.ref == 'refs/heads/dev' }} + runs-on: ubuntu-latest + timeout-minutes: 30 + permissions: + id-token: write + contents: read + env: + ENV: dev + BUCKET: open-swe-dev-assets + DEPLOY_DOC: open-swe-dev-deploy + steps: + - uses: actions/checkout@v6 + - uses: oven-sh/setup-bun@v2 + with: + bun-version: latest + - name: Build SPA (vite -> ui/.output/public) + working-directory: ui + run: | + bun install --frozen-lockfile + bun run build + - name: Package artifacts + run: bash .github/scripts/package-artifacts.sh + - uses: aws-actions/configure-aws-credentials@v6 + with: + role-to-assume: ${{ vars.AWS_DEPLOY_ROLE_APP_DEV }} + aws-region: us-east-1 + - name: Publish to S3 + roll the box + run: bash .github/scripts/publish-and-deploy.sh + + publish-prod: + name: Publish + deploy (prod) + if: ${{ github.ref == 'refs/heads/main' }} + runs-on: ubuntu-latest + timeout-minutes: 30 + # Manual-approval gate: the "prod" Environment requires a reviewer (Adam). Also + # makes the OIDC sub …:environment:prod (matches the prod app-role trust). + environment: prod + permissions: + id-token: write + contents: read + env: + ENV: prod + BUCKET: open-swe-prod-assets + DEPLOY_DOC: open-swe-prod-deploy + steps: + - uses: actions/checkout@v6 + - uses: oven-sh/setup-bun@v2 + with: + bun-version: latest + - name: Build SPA (vite -> ui/.output/public) + working-directory: ui + run: | + bun install --frozen-lockfile + bun run build + - name: Package artifacts + run: bash .github/scripts/package-artifacts.sh + - uses: aws-actions/configure-aws-credentials@v6 + with: + role-to-assume: ${{ vars.AWS_DEPLOY_ROLE_APP_PROD }} + aws-region: us-east-1 + - name: Publish to S3 + roll the box + run: bash .github/scripts/publish-and-deploy.sh diff --git a/.gitignore b/.gitignore index 4b61765b..f3850d6f 100644 --- a/.gitignore +++ b/.gitignore @@ -68,4 +68,7 @@ __pycache__/ *.egg-info/ .eggs/ +# Local working docs (gitignored — survives upstream merges, never pushed) +TODO.md + # \ No newline at end of file diff --git a/.security-review/suppressions.json b/.security-review/suppressions.json new file mode 100644 index 00000000..722fc906 --- /dev/null +++ b/.security-review/suppressions.json @@ -0,0 +1,14 @@ +{ + "suppressions": [ + { + "id": "OSWE-IAC-AUDIT-01", + "title": "Dev-branch infra OIDC role can assume the account-wide CDK cfn-exec admin role (cdk-hnb659fds-*), a path to mutating prod", + "file": "infra/lib/constructs/github-deploy-roles.ts", + "severity": "high", + "status": "confirmed", + "suppression_justification": "PRE-EXISTING and NOT introduced or worsened by the T7+T19 change (the assets bucket / app-role PutObject / SSM deploy doc). This is the known single-account-wide CDK cfn-exec residual already documented in infra/lib/config.ts:31-34 and the github-deploy-roles.ts construct comment, accepted at the T4 GPT-4.1 IAM cross-review and the v5 plan-review. WHO can assume each env's infra role is exact-subject scoped (StringEquals on the dev ref / prod environment); the residual is the shared account-wide cfn-exec-role that every env's infra role can reach. The tracked fix is per-env CDK bootstrap qualifiers so each env's infra role assumes its own env-scoped cfn-exec-role. Suppressed for THIS change's gate because it is out-of-diff and unchanged; surfaced to Adam for scheduling the per-env-bootstrap remediation.", + "owner": "adam@seahavenind.com", + "added": "2026-06-26" + } + ] +} diff --git a/deploy/ami/deploy.sh b/deploy/ami/deploy.sh new file mode 100755 index 00000000..2839503c --- /dev/null +++ b/deploy/ami/deploy.sh @@ -0,0 +1,102 @@ +#!/usr/bin/env bash +# Open SWE app deploy — RE-RUNNABLE (first boot + every subsequent release). +# +# Pulls the current release from S3 (open-swe--assets), builds the venv +# natively on the box, and restarts the service. This is the SINGLE source of the +# app-deploy procedure; it runs in two places: +# +# 1. first boot — user-data.sh decodes this script to /opt/open-swe/bin and +# calls it ONCE (non-fatal: if no release is published yet, +# nginx is already up and the box waits for the first deploy). +# 2. every release — the `open-swe--deploy` SSM document (CI fires it after +# uploading app.tar.gz / spa.tar.gz) runs this same script. +# +# It deploys CODE + STATIC ASSETS only. Secrets/config are NOT fetched here: the +# systemd unit's ExecStartPre=fetch-config.sh materializes the tmpfs .env on every +# (re)start, fail-fast — so `systemctl restart` below is what reloads config too. +# +# Contract: +# app.tar.gz = the Python source tree (pyproject.toml + uv.lock + agent/ + +# deploy/ + langgraph.json + README.md, NO ui/, NO .venv). The venv +# is built HERE with `uv sync` so it is native ARM64 and lives at +# the real runtime path (no cross-built / non-relocatable venv). +# spa.tar.gz = the built dashboard SPA (vite output: _shell.html + assets), +# extracted to the nginx web root. +set -euo pipefail +exec > >(tee -a /var/log/open-swe/deploy.log) 2>&1 +echo "==> open-swe deploy start $(date -u +%FT%TZ)" + +# Non-secret pointers written by user-data.sh (env, region, bucket, artifact prefix). +# shellcheck disable=SC1091 +. /etc/open-swe/boot.env +export AWS_DEFAULT_REGION="${AWS_REGION:?boot.env missing AWS_REGION}" +: "${ASSETS_BUCKET:?boot.env missing ASSETS_BUCKET}" +: "${ARTIFACT_PREFIX:?boot.env missing ARTIFACT_PREFIX}" + +# Fixed layout — must match provision.sh + user-data.sh + the templates. +SERVICE_USER="openswe" +APP_DIR="/opt/open-swe/app" +WWW_ROOT="/var/www/open-swe" +ENV_FILE="/run/open-swe/.env" +UV_BIN="/usr/local/bin/uv" +UV_PYTHON_INSTALL_DIR="/opt/uv/python" # where provision.sh pre-installed py3.12 +SERVICE_HOME="/opt/open-swe" + +# Benign-vs-failure distinction: on a brand-new env no release is published yet. +# Treat "app.tar.gz absent in S3" as a benign no-op (exit 0) so first boot is not a +# scary failure; ONCE a release exists, any later step failing is loud (set -e). +if ! aws s3 ls "s3://${ASSETS_BUCKET}/${ARTIFACT_PREFIX}/app.tar.gz" >/dev/null 2>&1; then + echo "==> no release published at s3://${ASSETS_BUCKET}/${ARTIFACT_PREFIX}/ yet — nothing to deploy" + exit 0 +fi + +echo "==> pull release from s3://${ASSETS_BUCKET}/${ARTIFACT_PREFIX}/" +tmp="$(mktemp -d)" +trap 'rm -rf "$tmp"' EXIT +aws s3 cp "s3://${ASSETS_BUCKET}/${ARTIFACT_PREFIX}/app.tar.gz" "${tmp}/app.tar.gz" +aws s3 cp "s3://${ASSETS_BUCKET}/${ARTIFACT_PREFIX}/spa.tar.gz" "${tmp}/spa.tar.gz" + +# Replace app source + SPA atomically-ish: clear the dirs (drops files removed in +# this release) then extract. The venv is rebuilt below, so wiping .venv too is +# fine — uv's cache (in the service home) makes the rebuild fast. +# +# Hardening: deploy.sh runs as root, so extract with --no-same-owner +# --no-same-permissions — files take root:root + umask perms (NOT the archive's +# uid/mode), so a tarball cannot land a setuid/setgid binary or a foreign-owned +# file; the chown -R below then hands the tree to the service user. (GNU tar also +# refuses `..`-escaping members by default.) Defense-in-depth: the only writer of +# this bucket is the CI OIDC app role, but the box never trusts the archive's +# ownership/mode regardless. +echo "==> install app source -> ${APP_DIR}" +install -d -o "$SERVICE_USER" -g "$SERVICE_USER" -m 0755 "$APP_DIR" "$WWW_ROOT" +find "$APP_DIR" -mindepth 1 -delete +find "$WWW_ROOT" -mindepth 1 -delete +tar --no-same-owner --no-same-permissions -xzf "${tmp}/app.tar.gz" -C "$APP_DIR" +tar --no-same-owner --no-same-permissions -xzf "${tmp}/spa.tar.gz" -C "$WWW_ROOT" +# langgraph reads ./.env from WorkingDirectory; point it at the tmpfs file the +# systemd ExecStartPre materializes. +ln -sfn "$ENV_FILE" "${APP_DIR}/.env" +chown -R "$SERVICE_USER":"$SERVICE_USER" "$APP_DIR" "$WWW_ROOT" + +echo "==> build venv natively (uv sync --frozen --no-dev)" +# Run as the service user so the venv + uv cache are owned by it. Pin the +# pre-baked interpreter dir so uv never reaches out to download Python at deploy. +cd "$APP_DIR" +sudo -u "$SERVICE_USER" env \ + HOME="$SERVICE_HOME" \ + UV_PYTHON_INSTALL_DIR="$UV_PYTHON_INSTALL_DIR" \ + UV_CACHE_DIR="${SERVICE_HOME}/.cache/uv" \ + "$UV_BIN" sync --frozen --no-dev + +echo "==> restart open-swe.service + reload nginx" +# ExecStartPre=fetch-config.sh fails-fast if secrets/config are missing, so a +# restart here surfaces a bad config as a failed unit (non-zero exit below). +systemctl restart open-swe.service +nginx -t && systemctl reload nginx + +if systemctl is-active --quiet open-swe.service; then + echo "==> open-swe deploy OK $(date -u +%FT%TZ)" +else + echo "!! open-swe.service is not active after deploy (check fetch-config/secrets)" + exit 1 +fi diff --git a/deploy/ami/open-swe-base.pkr.hcl b/deploy/ami/open-swe-base.pkr.hcl index d5c30737..0c37d72a 100644 --- a/deploy/ami/open-swe-base.pkr.hcl +++ b/deploy/ami/open-swe-base.pkr.hcl @@ -73,7 +73,8 @@ source "amazon-ebs" "open-swe" { ssh_username = "ubuntu" ami_name = "${var.ami_name_prefix}-${local.timestamp}" - ami_description = "Open SWE base — Ubuntu 24.04 arm64 + uv/py3.12 + nginx + CW agent (templates only, no secrets)" + # ASCII only — AWS rejects non-ASCII in the AMI Description attribute. + ami_description = "Open SWE base - Ubuntu 24.04 arm64 + uv/py3.12 + nginx + CW agent (templates only, no secrets)" source_ami_filter { filters = { @@ -113,10 +114,17 @@ build { name = "open-swe-base" sources = ["source.amazon-ebs.open-swe"] - # Stage the boot-time templates and helper scripts into the image. + # Stage the boot-time templates into the image. The destination dir must exist + # BEFORE a trailing-slash (contents-only) file upload — packer's file provisioner + # does not create it, and uploading the directory itself trips scp ("Is a + # directory"). So mkdir first, then upload the contents into it. + provisioner "shell" { + inline = ["mkdir -p /tmp/open-swe-templates"] + } + provisioner "file" { source = "${path.root}/templates/" - destination = "/tmp/open-swe-templates/" + destination = "/tmp/open-swe-templates" } provisioner "shell" { @@ -127,7 +135,9 @@ build { "CLOUDWATCH_AGENT_DEB_URL=${var.cloudwatch_agent_deb_url}", "AWSCLI_ZIP_URL=${var.awscli_zip_url}", ] - execute_command = "chmod +x {{ .Path }}; sudo -E bash '{{ .Path }}'" + # {{ .Vars }} MUST be included or the environment_vars above never reach the + # script (provision.sh runs under `set -u` and fails on the first reference). + execute_command = "chmod +x {{ .Path }}; {{ .Vars }} sudo -E bash '{{ .Path }}'" script = "${path.root}/scripts/provision.sh" } } diff --git a/deploy/ami/user-data.sh b/deploy/ami/user-data.sh index b204be19..d67a4a5f 100755 --- a/deploy/ami/user-data.sh +++ b/deploy/ami/user-data.sh @@ -52,6 +52,7 @@ cat >/etc/open-swe/boot.env < fetch app artifact from s3://${ASSETS_BUCKET}/${ARTIFACT_PREFIX}/" -tmp="$(mktemp -d)" -aws s3 cp "s3://${ASSETS_BUCKET}/${ARTIFACT_PREFIX}/app.tar.gz" "${tmp}/app.tar.gz" -aws s3 cp "s3://${ASSETS_BUCKET}/${ARTIFACT_PREFIX}/spa.tar.gz" "${tmp}/spa.tar.gz" - -install -d -o "$SERVICE_USER" -g "$SERVICE_USER" -m 0755 "$APP_DIR" "$WWW_ROOT" -tar -xzf "${tmp}/app.tar.gz" -C "$APP_DIR" -tar -xzf "${tmp}/spa.tar.gz" -C "$WWW_ROOT" -chown -R "$SERVICE_USER":"$SERVICE_USER" "$APP_DIR" "$WWW_ROOT" -rm -rf "$tmp" - -# The app reads ./.env from WorkingDirectory (langgraph.json "env": ".env"). -# Point it at the tmpfs file fetch-config.sh materializes. -ln -sfn "$ENV_FILE" "${APP_DIR}/.env" +# --- install the deploy script (single source of the app-deploy procedure) --- +# deploy.sh (deploy/ami/deploy.sh) pulls the release from S3, builds the venv with +# `uv sync`, and restarts the service. CDK base64-renders the file into the +# @@DEPLOY_SH_B64@@ token below so it is a normal reviewable repo file, not an +# inline heredoc. The `open-swe--deploy` SSM document runs this same script +# for every subsequent release. +echo "==> install /opt/open-swe/bin/deploy.sh" +install -d -o root -g root -m 0755 /opt/open-swe/bin +base64 -d >/opt/open-swe/bin/deploy.sh <<'DEPLOY_SH_B64' +@@DEPLOY_SH_B64@@ +DEPLOY_SH_B64 +chmod 0755 /opt/open-swe/bin/deploy.sh # --- render + install the systemd unit --------------------------------------- echo "==> install systemd unit" @@ -115,14 +113,29 @@ sed -e "s|@@OPENSWE_ENV@@|${OPENSWE_ENV}|g" \ -a fetch-config -m ec2 -s \ -c file:/opt/aws/amazon-cloudwatch-agent/etc/open-swe-cw.json -# --- start services ---------------------------------------------------------- +# --- start nginx FIRST (the security boundary + health surface) -------------- # NOTE: intentionally NO swapfile here. The 8 GB-swapfile OOM hack existed only # for the on-box Nitro SPA build, which now runs in GitHub Actions -> S3. -echo "==> start nginx + open-swe.service" +# nginx is brought up BEFORE the app is deployed so the ALB target-group health +# check (static `/healthz` -> 200) passes and the box is a healthy target even on +# the very first boot, before any release is published. open-swe.service is +# enabled (boot persistence) but STARTED by deploy.sh once the app is on disk. +echo "==> start nginx" systemctl enable --now nginx systemctl reload nginx -# open-swe.service ExecStartPre=fetch-config.sh fails-fast if config is missing, -# so a bad secrets/SSM setup surfaces as a failed unit (not a half-up box). -systemctl enable --now open-swe.service +systemctl enable open-swe.service + +# --- deploy the app (NON-FATAL on first boot) -------------------------------- +# deploy.sh pulls the release, builds the venv, and starts open-swe.service. On a +# brand-new env no release exists yet, so this is allowed to fail WITHOUT aborting +# user-data: nginx is already up (healthy target), and the first `build-artifacts` +# run + `open-swe--deploy` SSM command will bring the app up. A failure here +# is logged, not fatal. +echo "==> initial app deploy (non-fatal if no release is published yet)" +if /opt/open-swe/bin/deploy.sh; then + echo "==> initial app deploy succeeded" +else + echo "==> no release yet (or deploy failed): open-swe.service deferred to the next SSM deploy" +fi echo "==> open-swe user-data done $(date -u +%FT%TZ)" diff --git a/deploy/seahaven/fetch-config.sh b/deploy/seahaven/fetch-config.sh index 8cabcde0..72dd2332 100755 --- a/deploy/seahaven/fetch-config.sh +++ b/deploy/seahaven/fetch-config.sh @@ -190,12 +190,12 @@ VARS[DEFAULT_REPO_OWNER]="$SH_REPO_OWNER" required=( DASHBOARD_JWT_SECRET # RuntimeError on startup if missing (oauth.py) TOKEN_ENCRYPTION_KEY # Fernet key(s); decrypts per-user GitHub tokens - GITHUB_APP_ID # GitHub App trio (installation-token minting) ... - GITHUB_APP_PRIVATE_KEY # ... multiline PEM ... - GITHUB_APP_INSTALLATION_ID # ... used by utils/github_app.py - GITHUB_APP_CLIENT_ID # dashboard OAuth login (prod parity) - GITHUB_APP_CLIENT_SECRET # dashboard OAuth login (prod parity) ) +# NOTE: the GitHub App is NOT created/duplicated for dev — only prod owns the +# (single, shared) GitHub App + Slack app. So the GitHub App quintet + Slack + +# webhook-signing secrets are required for PROD only (see the prod block below). +# Dev boots without them: it has no GitHub-App/Slack/webhook integration — it is a +# deployment-validation env (boot/health/boundary), not a live-triggered agent. # Active model-provider key(s): model selection is store-driven (team_settings), # so fetch-config cannot infer it from .env. Default to the seeded cross-family @@ -216,10 +216,19 @@ case "$sandbox_type" in *) log "WARNING: unknown SANDBOX_TYPE='${sandbox_type}' — not enforcing a sandbox key" ;; esac -# Prod parity: webhook-signing secrets default empty in code but are required in -# prod. Linear's is required only when the Linear integration is wired. +# Prod-only: the GitHub App (installation-token minting + dashboard OAuth) and the +# webhook-signing secrets. Dev has no GitHub/Slack app, so none of these are +# required there; prod owns the single shared app and must have all of them. if [ "$ENV" = "prod" ]; then - required+=(GITHUB_WEBHOOK_SECRET SLACK_SIGNING_SECRET) + required+=( + GITHUB_APP_ID # GitHub App trio (installation-token minting) ... + GITHUB_APP_PRIVATE_KEY # ... multiline PEM ... + GITHUB_APP_INSTALLATION_ID # ... used by utils/github_app.py + GITHUB_APP_CLIENT_ID # dashboard OAuth login + GITHUB_APP_CLIENT_SECRET # dashboard OAuth login + GITHUB_WEBHOOK_SECRET # webhook signature verification + SLACK_SIGNING_SECRET # Slack webhook signature verification + ) if [ -n "${VARS[LINEAR_API_KEY]:-}" ] && [ "${OPENSWE_REQUIRE_LINEAR:-1}" = "1" ]; then required+=(LINEAR_WEBHOOK_SECRET) fi diff --git a/infra/README.md b/infra/README.md index e1d4b0a6..256cb868 100644 --- a/infra/README.md +++ b/infra/README.md @@ -21,11 +21,11 @@ infra/ │ ├── instance-role.ts # open-swe--instance-role (least-privilege) │ ├── config-store.ts # Secrets Manager + SSM Parameter Store shells (T11) │ ├── app-service.ts # EC2 box + imported-ALB ingress + Route53 + logs (T12) -│ └── ami-cache.ts # cached ARM64 AL2023 helper + EBS/AMI discipline docs +│ └── ami-cache.ts # baked open-swe AMI pin (by id) + EBS/replacement docs ├── test/ │ └── kebab-naming-aspect.test.ts # jest: Aspect passes conforming names, flags bad ones ├── cdk.json -├── cdk.context.json # COMMITTED — pins the AMI (see AMI cache discipline) +├── cdk.context.json # COMMITTED — {} (AMI is a static id pin; no lookups) ├── package.json # aws-cdk-lib pinned EXACT (2.260.0) ├── tsconfig.json ├── jest.config.js @@ -197,17 +197,21 @@ hooks hostname is scoped to `/webhooks/*` only (OSWE-T12-02 hygiene). An X-Forwarded-For spoof candidate was **killed** — no code trusts the leftmost XFF. No confirmed critical/high; no block. -## AMI cache discipline (EBS-fix plumbing — consumed by T12 `AppService`) +## Baked AMI + EBS-replacement discipline -`cachedArm64AmazonLinux2023()` (in `lib/constructs/ami-cache.ts`) returns an -ARM64 Amazon Linux 2023 image with `cachedInContext: true`, so the resolved AMI -id is pinned in the committed `cdk.context.json`. Without the pin, every deploy -could pick up a newer AL2023 release → AMI change → **EC2 instance replacement** -(the file-share data-loss root cause — memory `feedback_inline_ebs_volumes`). +`bakedOpenSweArm64()` (in `lib/constructs/ami-cache.ts`) pins the custom +**open-swe-base-arm64** image by EXACT id (`BAKED_OPEN_SWE_AMI_ID`) via +`MachineImage.genericLinux({ "us-east-1": "" })` — no SSM lookup, so synth +and deploy are fully offline/deterministic. The image is built by +`deploy/ami/open-swe-base.pkr.hcl` (ARM64 Ubuntu 24.04 + uv/py3.12 + nginx + CW +agent + boot templates, **no secrets**); the box's `user-data.sh` assumes that +baked layout (`/opt/open-swe`, `openswe` user, nginx, CW agent). -Design intent (now consumed by `AppService`): +Pinning by exact id (vs a `most_recent` name filter) is what prevents a routine +deploy from silently swapping the AMI → **EC2 instance replacement** (the +file-share data-loss root cause — memory `feedback_inline_ebs_volumes`). -- `userDataCausesReplacement: true` is the **deliberate** choice — user-data is +- `userDataCausesReplacement: true` is **deliberate** — user-data is provisioning-only and carries no durable state. - **No durable state on the box → no RETAIN volume.** The in-memory langgraph store is rebuilt on every boot from S3 + Secrets Manager / SSM, so there is @@ -217,18 +221,16 @@ Design intent (now consumed by `AppService`): deploy snapshot the root volume and wait `state=completed`, and re-verify "no local-only durable state" first. -Refresh the AMI pin deliberately: +Refresh the AMI deliberately: ```bash -cdk context --reset 'ssm:account=328440206208:parameterName=/aws/service/ami-amazon-linux-latest/al2023-ami-kernel-default-arm64:region=us-east-1' -cdk synth # review the diff — it WILL show "requires replacement" +cd deploy/ami && packer build open-swe-base.pkr.hcl # prints the new ami-… id +# update BAKED_OPEN_SWE_AMI_ID in infra/lib/constructs/ami-cache.ts +cd infra && npx cdk diff OpenSweDevStack # WILL show "requires replacement" ``` -> The committed `cdk.context.json` ships a dummy-but-valid-shaped AMI id -> (`ami-00000000000000000`) so `cdk synth` resolves the cache locally without any -> live AWS call. **Before the first real deploy**, repoint `AppService` to the -> baked `open-swe-base-arm64` AMI and pin its real id — the placeholder is -> intentionally un-bootable on stock AL2023. +> `cdk.context.json` is `{}` — nothing is resolved via context anymore (the AMI is +> a static id pin), so synth makes no live AWS call. ## Commands diff --git a/infra/cdk.context.json b/infra/cdk.context.json index 5e1a4249..0967ef42 100644 --- a/infra/cdk.context.json +++ b/infra/cdk.context.json @@ -1,3 +1 @@ -{ - "ssm:account=328440206208:parameterName=/aws/service/ami-amazon-linux-latest/al2023-ami-kernel-6.1-arm64:region=us-east-1": "ami-00000000000000000" -} +{} diff --git a/infra/lib/constructs/ami-cache.ts b/infra/lib/constructs/ami-cache.ts index dea21aa0..7687ac07 100644 --- a/infra/lib/constructs/ami-cache.ts +++ b/infra/lib/constructs/ami-cache.ts @@ -1,62 +1,35 @@ import * as ec2 from "aws-cdk-lib/aws-ec2"; +import { REGION } from "../config"; /** - * cdk.context.json key for the cached AL2023 ARM64 AMI. `latestAmazonLinux2023` - * + ARM_64 in the pinned aws-cdk-lib resolves the public SSM parameter below - * (the default-kernel alias for this CDK version is the `kernel-6.1` line); with - * `cachedInContext: true` CDK stores the resolved id under this exact key. - * Exported so the env stack can check "is the AMI already pinned?" and skip the - * resolve at synth when it is not — guaranteeing T3 synth makes no live AWS call. + * The baked open-swe base AMI (ARM64 Ubuntu 24.04 + uv/py3.12 + nginx + CW agent + * + boot templates — NO secrets), produced by `deploy/ami/open-swe-base.pkr.hcl`. + * Pinned by EXACT id (not a name filter) so synth/deploy is fully offline and + * deterministic. * - * If an aws-cdk-lib bump changes the default kernel alias, `cdk synth` will write - * a new key into cdk.context.json — update this constant + the committed pin to - * match (the AMI cache naturally tracks the CDK version). + * Built 2026-06-26 from open-swe-base-arm64-20260626-203433. + * + * ── EBS / AMI replacement discipline (memory feedback_inline_ebs_volumes) ── + * + * Refresh DELIBERATELY: `cd deploy/ami && packer build open-swe-base.pkr.hcl`, + * then update this id. A new id → EC2 instance REPLACEMENT. Pinning by exact id + * (vs a `most_recent` name filter) is what prevents a routine deploy from silently + * swapping the AMI — the root cause of the file-share data-loss incidents + * (5/15, 5/27, 6/5). + * + * `userDataCausesReplacement: true` (AppService) is likewise DELIBERATE: user-data + * is provisioning-only and the box holds NO durable state (the langgraph store is + * in-memory, rebuilt every boot from S3 + Secrets Manager / SSM), so there is + * intentionally no standalone `ec2.Volume` + `removalPolicy.RETAIN`. The design + * goal is replacement-TOLERANCE, not avoidance. + * + * Operational guard before ANY replacing deploy (AMI / userData / instance-type): + * snapshot the root volume AND wait `state=completed`, re-verify "no local-only + * durable state", and review the `cdk diff` replacement at PR time. */ -export const AL2023_ARM64_SSM_CONTEXT_KEY = - "ssm:account=328440206208:parameterName=/aws/service/ami-amazon-linux-latest/al2023-ami-kernel-6.1-arm64:region=us-east-1"; +export const BAKED_OPEN_SWE_AMI_ID = "ami-00080084502093021"; -/** - * Cached ARM64 Amazon Linux 2023 machine image. - * - * ── EBS / AMI cache discipline (see memory feedback_inline_ebs_volumes) ── - * - * `cachedInContext: true` PINS the resolved AMI id into the committed - * cdk.context.json. Without it, `latestAmazonLinux2023()` resolves the NEWEST - * AL2023 release on every synth/deploy, so a routine deploy can swap the AMI → - * EC2 instance REPLACEMENT whenever AWS ships a release. That was the root cause - * of the file-share data-loss incidents (5/15, 5/27, 6/5). Refresh the pin - * DELIBERATELY: - * - * cdk context --reset '' && cdk synth - * - * then review the `cdk diff` (it WILL report "requires replacement") before - * deploying. - * - * ── userDataCausesReplacement intent (consumed at T12) ── - * - * open-swe user-data is provisioning-only: install the langgraph runtime, pull - * config from Secrets Manager / SSM, pull the build artifact from S3, start the - * service. It holds NO durable state. T12 sets `userDataCausesReplacement: true` - * DELIBERATELY so a config/bootstrap change rolls a fresh, known-good box. - * - * ── "No durable state on box → no RETAIN volume" assertion ── - * - * The langgraph store is in-memory and is reconstructed on every boot from S3 - * (artifact) + Secrets Manager / SSM (config). Nothing of record lives only on - * the instance's disk. Therefore there is intentionally NO standalone - * `ec2.Volume` + `removalPolicy.RETAIN` here: the design goal is replacement- - * TOLERANCE, not replacement-avoidance. - * - * ── Operational guard still applies at T12 (feedback_inline_ebs_volumes) ── - * - * Before ANY replacing deploy (AMI / userData / instance-type change): snapshot - * the root volume AND wait for `state=completed`, re-verify "no local-only - * durable state" first, and ensure cdk-diff-on-PR surfaces the replacement at - * review time. - */ -export function cachedArm64AmazonLinux2023(): ec2.IMachineImage { - return ec2.MachineImage.latestAmazonLinux2023({ - cpuType: ec2.AmazonLinuxCpuType.ARM_64, - cachedInContext: true, - }); +/** The baked open-swe base image, pinned by id (offline, deterministic). */ +export function bakedOpenSweArm64(): ec2.IMachineImage { + return ec2.MachineImage.genericLinux({ [REGION]: BAKED_OPEN_SWE_AMI_ID }); } diff --git a/infra/lib/constructs/app-service.ts b/infra/lib/constructs/app-service.ts index ae6faf8d..c49fec23 100644 --- a/infra/lib/constructs/app-service.ts +++ b/infra/lib/constructs/app-service.ts @@ -7,15 +7,16 @@ import * as elbTargets from "aws-cdk-lib/aws-elasticloadbalancingv2-targets"; import * as logs from "aws-cdk-lib/aws-logs"; import * as route53 from "aws-cdk-lib/aws-route53"; import * as iam from "aws-cdk-lib/aws-iam"; +import * as ssm from "aws-cdk-lib/aws-ssm"; import { Construct } from "constructs"; import { EnvName, prefix } from "../config"; -import { cachedArm64AmazonLinux2023 } from "./ami-cache"; +import { bakedOpenSweArm64 } from "./ami-cache"; /** * Shared seahaven-vpc + internet-facing ALB facts (read-only recon 2026-06-26; * scratchpad/T12-infra-facts.md). A SINGLE VPC and a SINGLE shared ALB front * both the on-prem `seahaven-site` stack and open-swe. We IMPORT every one of - * these and NEVER own them — open-swe only ADDS its own instance SG, a standalone + * these and NEVER own them - open-swe only ADDS its own instance SG, a standalone * ALB-egress rule, listener rules, a target group, and DNS records. * * ── Cross-stack coordination on the SHARED listener + ALB SG (T13 review) ── @@ -28,7 +29,7 @@ import { cachedArm64AmazonLinux2023 } from "./ami-cache"; * * The ONE shared namespace that REQUIRES coordination is listener-rule PRIORITY * (globally unique per listener; a collision is a fail-SAFE deploy error, not - * silent drift). Ownership map — keep these disjoint when editing either stack: + * silent drift). Ownership map - keep these disjoint when editing either stack: * - seahaven-site (on-prem): priorities 4-7 + default. * - open-swe: priorities 2, 3 (webhooks) and 10, 11 (site). * open-swe's webhook rules are HOST-scoped to its own *.seahaven.com hosts, so @@ -102,10 +103,10 @@ export interface AppServiceProps { * The open-swe compute + ingress wiring for one env (T12): * - one ARM64 EC2 box in private1 (1a), replacement-tolerant (no RETAIN volume), * - a standalone instance SG reachable ONLY from the shared ALB SG on :80, - * - a target group → instance:80 (nginx is the sole ingress; :2024 stays loopback), + * - a target group -> instance:80 (nginx is the sole ingress; :2024 stays loopback), * - two listener rules on the imported :443 listener (webhooks below the on-prem - * path rule; site catch-all above it), both → the same TG, - * - Route53 alias records for both hostnames → the shared ALB, + * path rule; site catch-all above it), both -> the same TG, + * - Route53 alias records for both hostnames -> the shared ALB, * - IaC-owned CloudWatch log groups at 30-day retention. * * Everything ALB/VPC/zone-side is IMPORTED. Synth is offline: the AMI is the @@ -115,6 +116,8 @@ export interface AppServiceProps { export class AppService extends Construct { public readonly instance: ec2.Instance; public readonly targetGroup: elbv2.ApplicationTargetGroup; + /** Name of the SSM document CI fires to roll the box to the latest release. */ + public readonly deployDocumentName: string; constructor(scope: Construct, id: string, props: AppServiceProps) { super(scope, id); @@ -123,7 +126,7 @@ export class AppService extends Construct { const net = ENV_NET[env]; const artifactPrefix = props.artifactPrefix ?? "releases/latest"; - // Import the shared VPC with explicit attributes (no fromLookup → offline synth). + // Import the shared VPC with explicit attributes (no fromLookup -> offline synth). const vpc = ec2.Vpc.fromVpcAttributes(this, "Vpc", { vpcId: SHARED.vpcId, availabilityZones: [...SHARED.availabilityZones], @@ -134,13 +137,13 @@ export class AppService extends Construct { const instanceSg = new ec2.SecurityGroup(this, "InstanceSg", { vpc, securityGroupName: `${p}-instance-sg`, - description: `${p} instance SG — ingress only from the shared ALB SG on :80; egress via NAT.`, + description: `${p} instance SG - ingress only from the shared ALB SG on :80; egress via NAT.`, allowAllOutbound: true, }); instanceSg.addIngressRule( ec2.Peer.securityGroupId(SHARED.albSecurityGroupId), ec2.Port.tcp(80), - `${p}: shared ALB SG → nginx :80`, + `${p}: shared ALB SG to nginx :80`, ); // Open the IMPORTED ALB SG to our instance via a STANDALONE egress rule, so we // never mutate the ALB SG's own (on-prem-owned) definition. @@ -150,12 +153,32 @@ export class AppService extends Construct { fromPort: 80, toPort: 80, destinationSecurityGroupId: instanceSg.securityGroupId, - description: `${p}: ALB → instance nginx :80`, + description: `${p}: ALB to instance nginx :80`, }); + // The app-deploy procedure (deploy/ami/deploy.sh) is a normal reviewable repo + // file; CDK base64-encodes it (single line - no `$`/regex-special chars in the + // base64 alphabet) and renders it into user-data's @@DEPLOY_SH_B64@@ token, so + // user-data writes it verbatim to /opt/open-swe/bin/deploy.sh at first boot. + // The same file is run by the open-swe--deploy SSM document on every + // release - a single source of truth for "pull release, build venv, restart". + const deployShPath = path.join(__dirname, "..", "..", "..", "deploy", "ami", "deploy.sh"); + // Minify before embedding: strip full-line comments + blank lines (keep the + // shebang) so the base64 fits EC2's 25.6 KB user-data limit. The repo file + // keeps its comments; only the on-box copy is minified. deploy.sh becomes + // opaque base64 here, so this never affects user-data's heredoc parsing. + const deployShMin = fs + .readFileSync(deployShPath, "utf8") + .split("\n") + .filter((line, i) => i === 0 || (!/^\s*#/.test(line) && line.trim() !== "")) + .join("\n"); + const deployShB64 = Buffer.from(deployShMin, "utf8").toString("base64"); + // Render the provisioning script's @@tokens@@ into the instance user-data. // userDataCausesReplacement makes a bootstrap change roll a fresh box (the box - // holds no durable state — see ami-cache.ts / user-data.sh). + // holds no durable state - see ami-cache.ts / user-data.sh). Editing deploy.sh + // therefore also rolls the box (its base64 is embedded here) - acceptable: the + // box is replacement-tolerant, and ongoing releases never touch user-data. const userDataPath = path.join(__dirname, "..", "..", "..", "deploy", "ami", "user-data.sh"); const userData = ec2.UserData.custom( fs @@ -163,7 +186,8 @@ export class AppService extends Construct { .replace(/@@OPENSWE_ENV@@/g, env) .replace(/@@ASSETS_BUCKET@@/g, `${p}-assets`) .replace(/@@SERVER_NAME@@/g, net.dashboardHost) - .replace(/@@ARTIFACT_PREFIX@@/g, artifactPrefix), + .replace(/@@ARTIFACT_PREFIX@@/g, artifactPrefix) + .replace(/@@DEPLOY_SH_B64@@/g, deployShB64), ); this.instance = new ec2.Instance(this, "Instance", { @@ -177,13 +201,11 @@ export class AppService extends Construct { ], }, instanceType: new ec2.InstanceType(net.instanceType), - // TODO(T12-deploy): repoint to the custom open-swe-base-arm64 AMI baked by - // packer (deploy/ami). The AL2023 ARM64 cache is a synth-time placeholder - // (pinned in cdk.context.json) so the box DEFINITION synths offline; the - // baked AMI id is pinned (cdk context) before the first real deploy. - // user-data.sh assumes the baked layout (/opt/open-swe, openswe user, nginx, - // CloudWatch agent) — it is NOT runnable on the stock AL2023 placeholder. - machineImage: cachedArm64AmazonLinux2023(), + // The baked open-swe base AMI (deploy/ami packer build) - ARM64 Ubuntu 24.04 + // with the /opt/open-swe layout, openswe user, nginx, and CW agent that + // user-data.sh assumes. Pinned by exact id (see ami-cache.ts); refresh by + // rebuilding and updating BAKED_OPEN_SWE_AMI_ID. + machineImage: bakedOpenSweArm64(), role: props.instanceRole, securityGroup: instanceSg, userData, @@ -193,7 +215,7 @@ export class AppService extends Construct { blockDevices: [ { deviceName: "/dev/xvda", - // gp3 encrypted root; deleteOnTermination (no durable on-box state → + // gp3 encrypted root; deleteOnTermination (no durable on-box state -> // intentionally NO standalone RETAIN volume; see ami-cache.ts). volume: ec2.BlockDeviceVolume.ebs(30, { volumeType: ec2.EbsDeviceVolumeType.GP3, @@ -204,7 +226,35 @@ export class AppService extends Construct { ], }); - // Target group → instance:80 (nginx). Health check hits nginx's /healthz + // SSM deploy document (open-swe--deploy): runs the baked + // /opt/open-swe/bin/deploy.sh to pull the latest release + restart. CI fires it + // (tag-scoped to project=open-swe,env=) after uploading a release, so the + // app deploy role needs SendCommand ONLY on this document - NOT on the generic + // AWS-RunShellScript (closes the T4 BLOCK#3 arbitrary-shell timebox). + this.deployDocumentName = `${p}-deploy`; + new ssm.CfnDocument(this, "DeployDoc", { + name: this.deployDocumentName, + documentType: "Command", + documentFormat: "YAML", + updateMethod: "NewVersion", + content: { + schemaVersion: "2.2", + description: `Roll the ${p} box to the latest published release (runs /opt/open-swe/bin/deploy.sh).`, + mainSteps: [ + { + action: "aws:runShellScript", + name: "deploy", + inputs: { + // Fixed command - no parameters, so nothing untrusted is interpolated + // into the shell. The script itself reads /etc/open-swe/boot.env. + runCommand: ["bash /opt/open-swe/bin/deploy.sh"], + }, + }, + ], + }, + }); + + // Target group -> instance:80 (nginx). Health check hits nginx's /healthz // (returns 200; the dashboard TG health path defined in open-swe.nginx.conf). this.targetGroup = new elbv2.ApplicationTargetGroup(this, "Tg", { vpc, @@ -233,7 +283,7 @@ export class AppService extends Construct { securityGroup: albSg, }); - // (1) Webhooks — accepted on EITHER host (integrations may target either), and + // (1) Webhooks - accepted on EITHER host (integrations may target either), and // MUST be below the on-prem path-only rule 5 (see ENV_NET note). new elbv2.ApplicationListenerRule(this, "WebhooksRule", { listener, @@ -244,7 +294,7 @@ export class AppService extends Construct { ], action: elbv2.ListenerAction.forward([this.targetGroup]), }); - // (2) Dashboard SPA + /dashboard/api/ (OAuth) — DASHBOARD host ONLY. The hooks + // (2) Dashboard SPA + /dashboard/api/ (OAuth) - DASHBOARD host ONLY. The hooks // host intentionally serves nothing but /webhooks/* (rule 1), so the OAuth / // dashboard surface stays single-origin (OSWE-T12-02). Non-webhook paths on the // hooks host fall through to the on-prem default. @@ -255,7 +305,7 @@ export class AppService extends Construct { action: elbv2.ListenerAction.forward([this.targetGroup]), }); - // Route53 ALIAS records → the shared ALB, for both hostnames. + // Route53 ALIAS records -> the shared ALB, for both hostnames. const zone = route53.HostedZone.fromHostedZoneAttributes(this, "PublicZone", { hostedZoneId: SHARED.publicZoneId, zoneName: SHARED.publicZoneName, @@ -274,14 +324,14 @@ export class AppService extends Construct { zone, recordName: host, target: route53.RecordTarget.fromAlias(albAlias), - comment: `${p} ${label.toLowerCase()} → shared seahaven-com ALB`, + comment: `${p} ${label.toLowerCase()} -> shared seahaven-com ALB`, }); } // IaC-owned CloudWatch log groups at 30-day retention. Names mirror the // CloudWatch-agent config (deploy/ami/templates/amazon-cloudwatch-agent.json); // owning them here makes retention declarative rather than agent-set. Logs are - // not durable state → DESTROY on stack delete. + // not durable state -> DESTROY on stack delete. for (const suffix of ["app", "user-data", "nginx-access", "nginx-error"]) { new logs.LogGroup(this, `Log-${suffix}`, { logGroupName: `/open-swe/${env}/${suffix}`, diff --git a/infra/lib/constructs/assets-bucket.ts b/infra/lib/constructs/assets-bucket.ts new file mode 100644 index 00000000..aa409ad3 --- /dev/null +++ b/infra/lib/constructs/assets-bucket.ts @@ -0,0 +1,58 @@ +import * as cdk from "aws-cdk-lib"; +import * as s3 from "aws-cdk-lib/aws-s3"; +import { Construct } from "constructs"; +import { EnvName, prefix } from "../config"; + +/** + * The per-env S3 artifact bucket (`open-swe--assets`) the box pulls its + * release from (T7). CI builds the SPA + packages the app source and uploads + * `app.tar.gz` / `spa.tar.gz` under `releases//` + `releases/latest/` + * (`build-artifacts.yml`, via the `githubdeploy-open-swe-app-` OIDC role); + * the box pulls `releases/latest/*` at boot / on deploy via its instance role. + * + * The bucket holds ONLY build artifacts — no secrets (those live in Secrets + * Manager + SSM), no durable runtime state (the langgraph store is in-memory and + * rebuilt every boot). It is therefore safe to treat as reproducible-from-CI, but + * we RETAIN it on stack delete so an accidental `cdk destroy` cannot strand the + * box with no artifact to pull on its next replacement. + * + * Security posture (locked, reviewed in T7): + * - `BLOCK_ALL` public access (this is an internal artifact store; ALB/nginx is + * the only public surface — never S3 directly). + * - SSE-S3 encryption at rest + `enforceSSL` (deny any non-TLS request). + * - versioned, so a bad release can be rolled back to the previous object + * version (the last-good-artifact story in T19); a lifecycle rule expires + * NONcurrent versions after 30 days so history does not grow unbounded. + * - aborts incomplete multipart uploads after 7 days (cost hygiene). + * + * The name is the load-bearing contract: `instance-role.ts` (read), the app + * deploy role in `github-deploy-roles.ts` (write), and `user-data.sh` / + * `deploy.sh` (`@@ASSETS_BUCKET@@`) all reference `open-swe--assets` by + * literal name, so it is set explicitly here rather than auto-generated. + */ +export class AssetsBucket extends Construct { + public readonly bucket: s3.Bucket; + + constructor(scope: Construct, id: string, envName: EnvName) { + super(scope, id); + const p = prefix(envName); + + this.bucket = new s3.Bucket(this, "Bucket", { + bucketName: `${p}-assets`, + blockPublicAccess: s3.BlockPublicAccess.BLOCK_ALL, + encryption: s3.BucketEncryption.S3_MANAGED, + enforceSSL: true, + versioned: true, + // Artifacts are reproducible from CI, but RETAIN protects against an + // accidental stack delete leaving the box with nothing to pull (see above). + removalPolicy: cdk.RemovalPolicy.RETAIN, + lifecycleRules: [ + { + id: "expire-noncurrent-artifact-versions", + noncurrentVersionExpiration: cdk.Duration.days(30), + abortIncompleteMultipartUploadAfter: cdk.Duration.days(7), + }, + ], + }); + } +} diff --git a/infra/lib/constructs/github-deploy-roles.ts b/infra/lib/constructs/github-deploy-roles.ts index cf6f811e..a53333f5 100644 --- a/infra/lib/constructs/github-deploy-roles.ts +++ b/infra/lib/constructs/github-deploy-roles.ts @@ -104,19 +104,18 @@ export class GithubDeployRoles extends Construct { ); // SendCommand also has to reference the command document. Scope to this env's - // open-swe deploy document. - // T4 BLOCK#3: GPT-4.1 flagged AWS-RunShellScript as an arbitrary-shell escalation - // path. TIMEBOXED (accepted until T19): the current SSM deploy runs deploy.sh via - // AWS-RunShellScript; it is already tag-scoped to env= (statement above). - // TODO(T19): drop AWS-RunShellScript once `open-swe--deploy` is the only path. + // open-swe deploy document ONLY. + // T4 BLOCK#3 (CLOSED at T19): GPT-4.1 flagged AWS-RunShellScript as an + // arbitrary-shell escalation path. The dedicated `open-swe-${envName}-deploy` + // SSM document (app-service.ts) now runs the fixed, parameter-less command + // `bash /opt/open-swe/bin/deploy.sh`, so AWS-RunShellScript is dropped here: + // this role can run ONLY that one document, and only on its own env's box + // (tag-scoped by the SsmSendCommandTagScoped statement above). this.appRole.addToPolicy( new iam.PolicyStatement({ sid: "SsmSendCommandDocuments", actions: ["ssm:SendCommand"], - resources: [ - `arn:aws:ssm:${REGION}:${ACCOUNT}:document/open-swe-${envName}-deploy`, - `arn:aws:ssm:${REGION}::document/AWS-RunShellScript`, - ], + resources: [`arn:aws:ssm:${REGION}:${ACCOUNT}:document/open-swe-${envName}-deploy`], }), ); @@ -134,13 +133,17 @@ export class GithubDeployRoles extends Construct { }), ); - // Read-only access to THIS env's artifact bucket only (CI uploads; the box - // pulls via its instance role — the app deploy role only reads to verify). + // Read+WRITE access to THIS env's artifact bucket only (T19): the + // build-artifacts workflow uploads app.tar.gz / spa.tar.gz under releases/*, + // then fires the deploy document so the box pulls them via its instance role. + // Object actions are scoped to releases/* (the only prefix CI writes), and to + // THIS env's bucket — a dev token can never write the prod bucket. No + // bucket-level mutation (no PutBucket*/Delete bucket) — that stays with CDK. this.appRole.addToPolicy( new iam.PolicyStatement({ - sid: "ReadArtifactBucket", - actions: ["s3:GetObject"], - resources: [`arn:aws:s3:::open-swe-${envName}-assets/*`], + sid: "ReadWriteArtifactObjects", + actions: ["s3:GetObject", "s3:PutObject", "s3:DeleteObject"], + resources: [`arn:aws:s3:::open-swe-${envName}-assets/releases/*`], }), ); this.appRole.addToPolicy( diff --git a/infra/lib/constructs/instance-role.ts b/infra/lib/constructs/instance-role.ts index 399efada..7f5eab96 100644 --- a/infra/lib/constructs/instance-role.ts +++ b/infra/lib/constructs/instance-role.ts @@ -33,11 +33,14 @@ export class InstanceRole extends Construct { ); // Read the build artifact from the env's S3 asset bucket (deploy = pull). + // Scoped to releases/* — the only prefix CI writes and the box pulls — so a + // compromised box (or stolen IMDS creds) cannot read anything else that might + // ever land in the bucket (least-privilege; mirrors the app role's write scope). this.role.addToPolicy( new iam.PolicyStatement({ sid: "ReadArtifactObjects", actions: ["s3:GetObject"], - resources: [`arn:aws:s3:::${p}-assets/*`], + resources: [`arn:aws:s3:::${p}-assets/releases/*`], }), ); this.role.addToPolicy( diff --git a/infra/lib/open-swe-stack.ts b/infra/lib/open-swe-stack.ts index 4d4f7377..78fccc94 100644 --- a/infra/lib/open-swe-stack.ts +++ b/infra/lib/open-swe-stack.ts @@ -2,12 +2,10 @@ import * as cdk from "aws-cdk-lib"; import { Construct } from "constructs"; import { EnvName, prefix } from "./config"; import { AppService } from "./constructs/app-service"; +import { AssetsBucket } from "./constructs/assets-bucket"; import { ConfigStore } from "./constructs/config-store"; import { InstanceRole } from "./constructs/instance-role"; -import { - AL2023_ARM64_SSM_CONTEXT_KEY, - cachedArm64AmazonLinux2023, -} from "./constructs/ami-cache"; +import { BAKED_OPEN_SWE_AMI_ID } from "./constructs/ami-cache"; export interface OpenSweStackProps extends cdk.StackProps { /** open-swe environment — drives the `open-swe--*` resource naming. */ @@ -27,6 +25,7 @@ export interface OpenSweStackProps extends cdk.StackProps { export class OpenSweStack extends cdk.Stack { public readonly instanceRole: InstanceRole; public readonly configStore: ConfigStore; + public readonly assetsBucket: AssetsBucket; public readonly appService: AppService; constructor(scope: Construct, id: string, props: OpenSweStackProps) { @@ -48,18 +47,17 @@ export class OpenSweStack extends cdk.Stack { // The instance role already grants read on open-swe-/* + /open-swe-/*. this.configStore = new ConfigStore(this, "Config", { envName }); - // AMI cache discipline (see lib/constructs/ami-cache.ts). T3 is synth-only: - // resolve + surface the pinned AMI id ONLY when it is already cached in - // cdk.context.json, so synth never makes a live SSM call. T12 consumes - // `cachedArm64AmazonLinux2023()` for the actual ec2.Instance. - if (this.node.tryGetContext(AL2023_ARM64_SSM_CONTEXT_KEY) !== undefined) { - const amiId = cachedArm64AmazonLinux2023().getImage(this).imageId; - new cdk.CfnOutput(this, "PinnedAmiId", { - value: amiId, - description: - "Cached AL2023 ARM64 AMI id (pinned in cdk.context.json; consumed by the T12 EC2 instance).", - }); - } + // T7: the S3 artifact bucket (open-swe--assets) CI uploads releases to + // and the box pulls app.tar.gz / spa.tar.gz from. The instance role already + // grants read on it by name; the app deploy role grants write. + this.assetsBucket = new AssetsBucket(this, "Assets", envName); + + // Surface the baked open-swe base AMI id the box runs on (pinned by id in + // ami-cache.ts; refreshed by a deliberate packer rebuild → replacement). + new cdk.CfnOutput(this, "BakedAmiId", { + value: BAKED_OPEN_SWE_AMI_ID, + description: "Baked open-swe-base-arm64 AMI id consumed by the EC2 instance.", + }); // T12: compute + ingress. Imports the shared seahaven-vpc + ALB and adds the // env's EC2 box, instance SG, target group, listener rules, DNS, log groups. @@ -80,5 +78,13 @@ export class OpenSweStack extends cdk.Stack { value: this.appService.targetGroup.targetGroupArn, description: `${p} ALB target group ARN (→ instance:80 nginx).`, }); + new cdk.CfnOutput(this, "AssetsBucketName", { + value: this.assetsBucket.bucket.bucketName, + description: `${p} S3 artifact bucket (CI uploads releases; box pulls).`, + }); + new cdk.CfnOutput(this, "DeployDocumentName", { + value: this.appService.deployDocumentName, + description: `${p} SSM document that rolls the box to the latest release.`, + }); } } diff --git a/infra/test/ascii-aws-fields.test.ts b/infra/test/ascii-aws-fields.test.ts new file mode 100644 index 00000000..89fb56bc --- /dev/null +++ b/infra/test/ascii-aws-fields.test.ts @@ -0,0 +1,74 @@ +import * as cdk from "aws-cdk-lib"; +import { Template } from "aws-cdk-lib/assertions"; +import { OpenSweStack } from "../lib/open-swe-stack"; + +const ENV = { account: "328440206208", region: "us-east-1" }; + +/** + * Several AWS APIs reject non-ASCII in fields that `cdk synth` happily emits and + * `tsc` happily compiles — so a stray em-dash/arrow only blows up at DEPLOY time + * (e.g. EC2 SecurityGroup GroupDescription: "Character sets beyond ASCII are not + * supported"). This has bitten us twice (the AMI Description, then the instance-SG + * description). This test fails the build at synth time instead. + * + * Scope: the EC2 fields with a documented ASCII/restricted-charset constraint — + * SecurityGroup GroupDescription and ingress/egress rule descriptions. (CloudFormation + * Output descriptions + Route53 comments accept UTF-8, so they are not asserted.) + * + * EC2 rule descriptions are stricter than ASCII: the allowed set is + * `a-zA-Z0-9. _-:/()#,@[]+=&;{}!$*` — note it EXCLUDES `<` and `>`, which is why a + * naive em-dash -> "->" replacement still fails at deploy. We assert that exact set. + */ +// Characters NOT in the EC2 description allowed set. +const DISALLOWED = /[^a-zA-Z0-9. _:/()#,@[\]+=&;{}!$*-]/; + +function synthDev(): Record }> { + const app = new cdk.App(); + const dev = new OpenSweStack(app, "OpenSweDevStack", { + stackName: "open-swe-dev", + env: ENV, + envName: "dev", + }); + return Template.fromStack(dev).toJSON().Resources; +} + +describe("ASCII-only EC2 description fields", () => { + const resources = synthDev(); + + it("SecurityGroup GroupDescription is ASCII", () => { + for (const [id, r] of Object.entries(resources)) { + if (r.Type !== "AWS::EC2::SecurityGroup") continue; + const desc = (r.Properties?.GroupDescription as string) ?? ""; + expect(DISALLOWED.test(desc) ? `${id}: ${desc}` : "ascii").toBe("ascii"); + } + }); + + it("SecurityGroup ingress/egress rule descriptions are ASCII", () => { + for (const [id, r] of Object.entries(resources)) { + const props = r.Properties ?? {}; + const groups: Array<{ Description?: string }> = []; + if (Array.isArray(props.SecurityGroupIngress)) groups.push(...props.SecurityGroupIngress); + if (Array.isArray(props.SecurityGroupEgress)) groups.push(...props.SecurityGroupEgress); + // Standalone AWS::EC2::SecurityGroupEgress / ...Ingress resources. + if (r.Type === "AWS::EC2::SecurityGroupEgress" || r.Type === "AWS::EC2::SecurityGroupIngress") { + groups.push(props as { Description?: string }); + } + for (const rule of groups) { + const desc = rule.Description ?? ""; + expect(DISALLOWED.test(desc) ? `${id}: ${desc}` : "ascii").toBe("ascii"); + } + } + }); + + // EC2 caps base64-encoded user-data at 25600 bytes; CDK + tsc don't check it, so + // an oversized boot script (e.g. an embedded deploy.sh) only fails at deploy. + it("EC2 user-data fits the 25600-byte encoded limit", () => { + for (const [id, r] of Object.entries(resources)) { + if (r.Type !== "AWS::EC2::Instance") continue; + const ud = (r.Properties?.UserData as { "Fn::Base64"?: string }) ?? {}; + const script = typeof ud["Fn::Base64"] === "string" ? ud["Fn::Base64"] : ""; + const encoded = Buffer.from(script, "utf8").toString("base64").length; + expect(`${id}: ${encoded} bytes`).toBe(encoded < 25600 ? `${id}: ${encoded} bytes` : "OVER 25600"); + } + }); +});