#!/usr/bin/env bash # Open SWE EC2 user-data — PROVISIONING-ONLY (runs once, at first boot). # # This is the rationale for `userDataCausesReplacement: true` in CDK: user-data # does FIRST-BOOT provisioning, never durable runtime config. Editing it is a # deliberate instance replacement. Durable runtime config is fetched fresh on # every service start by deploy/seahaven/fetch-config.sh (ExecStartPre). # # The box holds NO durable state of its own: # - secrets/config -> Secrets Manager + SSM, materialized to a tmpfs .env at boot # - app artifact -> pulled from S3 (open-swe--assets) via the instance role # - store state -> reseeded by seed_store.sh (ExecStartPost) on every start # => there is no RETAIN volume to protect; replacement is tolerated. The EBS # discipline (snapshot root + wait state=completed BEFORE any replacing deploy) # is the safety net, not durable on-box state. See README "EBS discipline". # # Tokens (@@...@@) are substituted by CDK when it renders this script into the # launch template. region is read from IMDSv2 as a fallback. set -euo pipefail exec > >(tee -a /var/log/open-swe-user-data.log) 2>&1 echo "==> open-swe user-data start $(date -u +%FT%TZ)" # --- CDK-rendered values ----------------------------------------------------- # NOTE: CDK-substituted tokens use %%...%% (rendered by app-service.ts), DISTINCT # from the @@...@@ tokens this script seds into the baked systemd/nginx templates. # The two MUST NOT share a delimiter: a shared @@OPENSWE_ENV@@ / @@SERVER_NAME@@ # let CDK clobber the sed PATTERN, leaving the unit's token unsubstituted. OPENSWE_ENV="%%OPENSWE_ENV%%" # dev | prod ASSETS_BUCKET="%%ASSETS_BUCKET%%" # open-swe--assets SERVER_NAME="%%SERVER_NAME%%" # openswe[-dev].seahaven.com ARTIFACT_PREFIX="%%ARTIFACT_PREFIX%%" # e.g. releases/latest # --- fixed layout (must match provision.sh + templates) ---------------------- SERVICE_USER="openswe" APP_DIR="/opt/open-swe/app" VENV="${APP_DIR}/.venv" WWW_ROOT="/var/www/open-swe" TEMPLATE_DIR="/opt/open-swe/templates" ENV_FILE="/run/open-swe/.env" PORT="2024" FETCH_CONFIG="${APP_DIR}/deploy/seahaven/fetch-config.sh" SEED_STORE="${APP_DIR}/deploy/seahaven/seed_store.sh" # region from IMDSv2 TOKEN="$(curl -fsS -X PUT "http://169.254.169.254/latest/api/token" \ -H "X-aws-ec2-metadata-token-ttl-seconds: 300" || true)" AWS_REGION="$(curl -fsS -H "X-aws-ec2-metadata-token: ${TOKEN}" \ http://169.254.169.254/latest/meta-data/placement/region || echo us-east-1)" export AWS_DEFAULT_REGION="$AWS_REGION" echo "env=${OPENSWE_ENV} region=${AWS_REGION} bucket=${ASSETS_BUCKET} host=${SERVER_NAME}" # --- boot.env: non-secret pointers fetch-config.sh reads --------------------- install -d -o root -g root -m 0755 /etc/open-swe cat >/etc/open-swe/boot.env <-deploy` SSM document runs this same script # for every subsequent release. echo "==> install /opt/open-swe/bin/deploy.sh" install -d -o root -g root -m 0755 /opt/open-swe/bin base64 -d >/opt/open-swe/bin/deploy.sh <<'DEPLOY_SH_B64' %%DEPLOY_SH_B64%% DEPLOY_SH_B64 chmod 0755 /opt/open-swe/bin/deploy.sh # --- render + install the systemd unit --------------------------------------- echo "==> install systemd unit" sed \ -e "s|@@SERVICE_USER@@|${SERVICE_USER}|g" \ -e "s|@@APP_DIR@@|${APP_DIR}|g" \ -e "s|@@VENV@@|${VENV}|g" \ -e "s|@@PORT@@|${PORT}|g" \ -e "s|@@ENV_FILE@@|${ENV_FILE}|g" \ -e "s|@@OPENSWE_ENV@@|${OPENSWE_ENV}|g" \ -e "s|@@FETCH_CONFIG@@|${FETCH_CONFIG}|g" \ -e "s|@@SEED_STORE@@|${SEED_STORE}|g" \ "${TEMPLATE_DIR}/open-swe.service" >/etc/systemd/system/open-swe.service systemctl daemon-reload # --- render + install the nginx site ----------------------------------------- echo "==> install nginx site" sed \ -e "s|@@SERVER_NAME@@|${SERVER_NAME}|g" \ -e "s|@@WWW_ROOT@@|${WWW_ROOT}|g" \ -e "s|@@BACKEND_ADDR@@|127.0.0.1:${PORT}|g" \ "${TEMPLATE_DIR}/open-swe.nginx.conf" >/etc/nginx/sites-available/open-swe ln -sfn /etc/nginx/sites-available/open-swe /etc/nginx/sites-enabled/open-swe rm -f /etc/nginx/sites-enabled/default nginx -t # --- CloudWatch agent: 30-day log retention ---------------------------------- echo "==> configure CloudWatch agent (30-day retention)" sed -e "s|@@OPENSWE_ENV@@|${OPENSWE_ENV}|g" \ "${TEMPLATE_DIR}/amazon-cloudwatch-agent.json" \ >/opt/aws/amazon-cloudwatch-agent/etc/open-swe-cw.json /opt/aws/amazon-cloudwatch-agent/bin/amazon-cloudwatch-agent-ctl \ -a fetch-config -m ec2 -s \ -c file:/opt/aws/amazon-cloudwatch-agent/etc/open-swe-cw.json # --- start nginx FIRST (the security boundary + health surface) -------------- # NOTE: intentionally NO swapfile here. The 8 GB-swapfile OOM hack existed only # for the on-box Nitro SPA build, which now runs in GitHub Actions -> S3. # nginx is brought up BEFORE the app is deployed so the ALB target-group health # check (static `/healthz` -> 200) passes and the box is a healthy target even on # the very first boot, before any release is published. open-swe.service is # enabled (boot persistence) but STARTED by deploy.sh once the app is on disk. echo "==> start nginx" systemctl enable --now nginx systemctl reload nginx systemctl enable open-swe.service # --- deploy the app (NON-FATAL on first boot) -------------------------------- # deploy.sh pulls the release, builds the venv, and starts open-swe.service. On a # brand-new env no release exists yet, so this is allowed to fail WITHOUT aborting # user-data: nginx is already up (healthy target), and the first `build-artifacts` # run + `open-swe--deploy` SSM command will bring the app up. A failure here # is logged, not fatal. echo "==> initial app deploy (non-fatal if no release is published yet)" if /opt/open-swe/bin/deploy.sh; then echo "==> initial app deploy succeeded" else echo "==> no release yet (or deploy failed): open-swe.service deferred to the next SSM deploy" fi echo "==> open-swe user-data done $(date -u +%FT%TZ)"