From 3224d7abf4c162366bfac13ffca528d80ff2cd05 Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Sat, 27 Jun 2026 20:21:59 -0400 Subject: [PATCH] =?UTF-8?q?ci:=20gate=20dev=E2=86=92prod=20promotion=20on?= =?UTF-8?q?=20green=20checks=20+=20add=20rollback=20safety=20net=20(#28)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit T20 CD safety nets. Two gaps closed before the first real prod deploy: 1. Promotion gate. promote_dev_to_prod.yml previously fast-forwarded dev→main unconditionally. It now hard-gates on check-dev-green.sh: every check-run on the dev HEAD must be completed+passing AND the Agent CI suite (lint/format/unit/E2E) must be present+success, or the promotion blocks (fails safe on a missing/renamed check). The promote run excludes its OWN check-run by run-id (unforgeable), never by the mutable name "promote", so a colliding red check cannot hide. Fields are read with a 0x1F separator so an empty conclusion (every in_progress check) cannot shift columns. ci.yml now also runs on push:dev so dev HEAD actually carries that signal (a PR check alone can be admin-merged past). 2. Rollback + last-good. publish-and-deploy.sh advances releases/last-good/ only after a successful roll (deploy.sh gates on `systemctl is-active`), and makes releases/latest/ transactional — reverting to the prior release if the roll fails so a replaced box never self-deploys a broken release. New rollback.yml + rollback.sh re-point latest at last-good (or an explicit sha) and re-fire the deploy; prod is gated by the `prod` Environment approval, same as a deploy. The shared fire/wait/aggregate-gate logic is factored into roll-box.sh (used by both forward and backward rolls). Least-privilege: drop the unused s3:DeleteObject from the app deploy role — publish/rollback/deploy only Get+Put (S3-to-S3 copy), and the rollback fallback now depends on immutable release history staying intact. Lifecycle expiry (not CI) handles old-version cleanup. Gate logic unit-tested (7 cases + jq round-trip). IAM change + release-safety control cross-reviewed by GPT-4.1: APPROVE, no blocks. Claude-Session: https://claude.ai/code/session_01DMhLf4G5V8MStJQyAW95hi --- .github/scripts/check-dev-green.sh | 85 ++++++++++++++++++ .github/scripts/publish-and-deploy.sh | 99 +++++++++------------ .github/scripts/roll-box.sh | 59 ++++++++++++ .github/scripts/rollback.sh | 66 ++++++++++++++ .github/workflows/ci.yml | 5 +- .github/workflows/promote_dev_to_prod.yml | 18 ++++ .github/workflows/rollback.yml | 84 +++++++++++++++++ infra/lib/constructs/github-deploy-roles.ts | 6 +- 8 files changed, 365 insertions(+), 57 deletions(-) create mode 100755 .github/scripts/check-dev-green.sh create mode 100755 .github/scripts/roll-box.sh create mode 100755 .github/scripts/rollback.sh create mode 100644 .github/workflows/rollback.yml diff --git a/.github/scripts/check-dev-green.sh b/.github/scripts/check-dev-green.sh new file mode 100755 index 00000000..fa96bf23 --- /dev/null +++ b/.github/scripts/check-dev-green.sh @@ -0,0 +1,85 @@ +#!/usr/bin/env bash +# Promotion gate: refuse to promote a dev HEAD that is not fully green. +# +# Reads check-runs on stdin — one +# namestatusconclusiondetails_url +# per line, fields separated by ASCII Unit Separator (0x1F) — so it is unit-testable +# WITHOUT GitHub. promote_dev_to_prod.yml pipes the live `gh api .../check-runs` +# output in. 0x1F (not TAB) is used deliberately: TAB is IFS-whitespace, so an empty +# conclusion (every in_progress check has a null conclusion) would collapse and shift +# the columns — which would make the promote run fail to exclude itself. 0x1F is +# non-whitespace, so `read` preserves empty fields and the columns stay aligned. +# +# A dev HEAD is promotable ONLY when BOTH hold: +# 1. every present check-run is completed with a passing conclusion +# (success/neutral/skipped); any pending/failed/cancelled/timed_out check BLOCKS. +# 2. every check named in REQUIRED_CHECKS is present AND concluded "success". +# This positive allow-list is what stops a partial-signal promotion — e.g. a +# commit pushed with [skip ci] (no CI check-runs) that still carries one +# unrelated green check, or a required check silently renamed/dropped. The gate +# FAILS SAFE: a missing required check blocks rather than promotes. +# +# Self-exclusion: the promotion workflow's OWN in-progress check-run is dropped by +# EXCLUDE_RUN_ID (its github.run_id, matched in the check-run details_url) — an +# unforgeable identity, NOT a mutable check name. A check that merely happens to be +# named "promote" can no longer hide a real failing/required check. +set -euo pipefail + +EXCLUDE_RUN_ID="${EXCLUDE_RUN_ID:-}" +# Mandatory checks (one per line). Defaults to the Agent CI suite, which runs on +# every push to dev (see ci.yml). Keep in sync with those job names; if a name +# drifts the gate blocks (fails safe) until the list is updated. +REQUIRED_CHECKS="${REQUIRED_CHECKS:-Agent lint +Agent format check +Agent unit tests +Playwright E2E}" + +declare -A GREEN +seen=0 +bad=0 +while IFS=$'\037' read -r name status conclusion details_url; do + [ -n "${name:-}" ] || continue + # drop the promotion run's own check-run by run id (never by name). + if [ -n "${EXCLUDE_RUN_ID}" ] && \ + [ "${details_url}" != "${details_url#*/runs/${EXCLUDE_RUN_ID}/}" ]; then + continue + fi + seen=$((seen + 1)) + if [ "${status}" != "completed" ]; then + echo "BLOCK: check '${name}' is '${status}' (not completed)" >&2 + bad=1 + continue + fi + case "${conclusion}" in + success) + GREEN["${name}"]=1 + echo "ok: ${name} (success)" ;; + neutral | skipped) + echo "ok: ${name} (${conclusion})" ;; + *) + echo "BLOCK: check '${name}' concluded '${conclusion:-}'" >&2 + bad=1 ;; + esac +done + +if [ "${seen}" -eq 0 ]; then + echo "BLOCK: no check-runs found for this commit — refusing to promote an unverified dev HEAD" >&2 + exit 1 +fi + +missing=0 +while IFS= read -r req; do + [ -n "${req}" ] || continue + if [ -z "${GREEN[${req}]:-}" ]; then + echo "BLOCK: required check '${req}' is missing or not successful on dev HEAD" >&2 + missing=1 + fi +done <&2 + exit 1 +fi +echo "PASS: ${seen} check(s) present, all green; all required checks present + successful." diff --git a/.github/scripts/publish-and-deploy.sh b/.github/scripts/publish-and-deploy.sh index 0fc9cca3..2551aa54 100755 --- a/.github/scripts/publish-and-deploy.sh +++ b/.github/scripts/publish-and-deploy.sh @@ -1,75 +1,64 @@ #!/usr/bin/env bash # Publish the packaged artifacts to the env's S3 bucket and roll the box to them. # Run by build-artifacts.yml AFTER aws creds are configured (env: ENV, BUCKET, -# DEPLOY_DOC). Each release is stored immutably under releases// and mirrored -# to releases/latest/ (what the box's deploy.sh pulls). +# DEPLOY_DOC). Each release is stored immutably under releases//. # -# The deploy is fired by TAG (project=open-swe,env=), which is exactly what -# the app deploy role's tag-scoped ssm:SendCommand allows — so this needs no -# ec2:DescribeInstances and no instance id up front. +# releases/latest/ (what the box's deploy.sh pulls) is advanced TRANSACTIONALLY: +# it is pointed at the new release, the box is rolled, and ONLY on a successful +# roll is it kept — a failed roll reverts releases/latest/ to the prior release so +# a later box boot / replacement never self-deploys a release that failed to come +# up. releases/last-good/ (rollback fallback) is advanced only after success and +# means "last release whose deploy.sh brought the service up active" (deploy.sh +# gates on `systemctl is-active`), not merely "last uploaded". +# +# The fire/wait/gate against the box lives in roll-box.sh (shared with rollback.sh); +# the deploy is fired by TAG (project=open-swe,env=), exactly what the app +# deploy role's tag-scoped ssm:SendCommand allows. set -euo pipefail : "${ENV:?}" "${BUCKET:?}" "${DEPLOY_DOC:?}" SHA="${GITHUB_SHA:?}" [ -f app.tar.gz ] && [ -f spa.tar.gz ] || { echo "ERROR: artifacts not built" >&2; exit 1; } +HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" + +# Re-point releases/latest/ at the release stored under releases//, recording +# the sha so the pointer is self-describing (used to revert on failure). +point_latest() { + local s="$1" f + for f in app.tar.gz spa.tar.gz; do + aws s3 cp "s3://${BUCKET}/releases/${s}/${f}" "s3://${BUCKET}/releases/latest/${f}" + done + printf '%s\n' "${s}" | aws s3 cp - "s3://${BUCKET}/releases/latest/sha.txt" +} + +# Which release does latest point at right now? (empty on the very first deploy.) +PREV_SHA="$(aws s3 cp "s3://${BUCKET}/releases/latest/sha.txt" - 2>/dev/null | tr -d '[:space:]' || true)" echo "==> upload release ${SHA} to s3://${BUCKET}/releases/${SHA}/" for f in app.tar.gz spa.tar.gz; do aws s3 cp "${f}" "s3://${BUCKET}/releases/${SHA}/${f}" done -echo "==> mirror to releases/latest/ (server-side copy)" -for f in app.tar.gz spa.tar.gz; do - aws s3 cp "s3://${BUCKET}/releases/${SHA}/${f}" "s3://${BUCKET}/releases/latest/${f}" -done +echo "==> point releases/latest/ -> ${SHA} (was ${PREV_SHA:-})" +point_latest "${SHA}" -echo "==> fire ${DEPLOY_DOC} via SSM (tag-targeted: project=open-swe, env=${ENV})" -CMD_ID="$(aws ssm send-command \ - --document-name "${DEPLOY_DOC}" \ - --targets "Key=tag:project,Values=open-swe" "Key=tag:env,Values=${ENV}" \ - --comment "release ${SHA}" \ - --query 'Command.CommandId' --output text)" -echo "command: ${CMD_ID}" - -echo "==> wait for the deploy to finish" -STATUS="Pending" -IID="" -for _ in $(seq 1 60); do - sleep 10 - # CommandInvocations is empty until SSM registers the target invocation. - IID="$(aws ssm list-command-invocations --command-id "${CMD_ID}" \ - --query 'CommandInvocations[0].InstanceId' --output text 2>/dev/null || echo None)" - [ -z "${IID}" ] || [ "${IID}" = "None" ] && continue - STATUS="$(aws ssm list-command-invocations --command-id "${CMD_ID}" \ - --query 'CommandInvocations[0].Status' --output text 2>/dev/null || echo Pending)" - case "${STATUS}" in - Success | Failed | Cancelled | TimedOut) break ;; - esac -done - -if [ -z "${IID}" ] || [ "${IID}" = "None" ]; then - echo "ERROR: no box picked up the deploy command (is a running open-swe ${ENV} box registered with SSM?)" >&2 +# Roll the box to releases/latest/. On a non-Success aggregate, roll-box.sh exits +# non-zero; revert latest to the prior release so no later boot pulls the bad one. +if ! ROLL_COMMENT="release ${SHA}" bash "${HERE}/roll-box.sh"; then + if [ -n "${PREV_SHA}" ]; then + echo "!! deploy failed — reverting releases/latest/ -> ${PREV_SHA}" >&2 + point_latest "${PREV_SHA}" + else + echo "!! deploy failed on the FIRST release — leaving releases/latest/ = ${SHA} (no prior release to revert to)" >&2 + fi exit 1 fi -echo "==> deploy.sh output from ${IID}:" -echo "----- stdout -----" -aws ssm get-command-invocation --command-id "${CMD_ID}" --instance-id "${IID}" \ - --query 'StandardOutputContent' --output text || true -echo "----- stderr -----" -aws ssm get-command-invocation --command-id "${CMD_ID}" --instance-id "${IID}" \ - --query 'StandardErrorContent' --output text || true +# Roll succeeded (service came up active) -> this release is now the known-good one. +echo "==> mark releases/last-good/ = ${SHA} (rollback fallback target)" +for f in app.tar.gz spa.tar.gz; do + aws s3 cp "s3://${BUCKET}/releases/${SHA}/${f}" "s3://${BUCKET}/releases/last-good/${f}" +done +printf '%s\n' "${SHA}" | aws s3 cp - "s3://${BUCKET}/releases/last-good/sha.txt" -# Gate on the AGGREGATE command status (Success only if EVERY targeted invocation -# succeeded), not CommandInvocations[0] — during a userDataCausesReplacement window -# two instances can briefly share the project/env tags, and a partial failure on -# the other instance must not be reported as success. -TARGETS="$(aws ssm list-commands --command-id "${CMD_ID}" \ - --query 'Commands[0].TargetCount' --output text 2>/dev/null || echo 1)" -[ "${TARGETS}" = "1" ] || echo "WARNING: deploy fanned out to ${TARGETS} instances (expected 1)" -AGG="$(aws ssm list-commands --command-id "${CMD_ID}" \ - --query 'Commands[0].Status' --output text 2>/dev/null || echo Failed)" - -echo "==> aggregate deploy status: ${AGG} (across ${TARGETS} target(s))" -[ "${AGG}" = "Success" ] || { echo "ERROR: deploy did not succeed (${AGG})" >&2; exit 1; } -echo "==> ${ENV} rolled to release ${SHA}" +echo "==> ${ENV} rolled to release ${SHA} (now last-good)" diff --git a/.github/scripts/roll-box.sh b/.github/scripts/roll-box.sh new file mode 100755 index 00000000..a7dc42fc --- /dev/null +++ b/.github/scripts/roll-box.sh @@ -0,0 +1,59 @@ +#!/usr/bin/env bash +# Fire the env's SSM deploy document (tag-targeted) and wait for it to finish, +# gating on the AGGREGATE command status. Shared by publish-and-deploy.sh (forward +# roll) and rollback.sh (backward roll) so the fire/wait/gate logic lives in ONE +# place. Requires ENV + DEPLOY_DOC in the environment and aws creds already set. +# +# Tag-targeting (project=open-swe,env=) is exactly what the app deploy role's +# tag-scoped ssm:SendCommand allows — no ec2:DescribeInstances, no instance id. +set -euo pipefail +: "${ENV:?}" "${DEPLOY_DOC:?}" +COMMENT="${ROLL_COMMENT:-roll ${ENV}}" + +echo "==> fire ${DEPLOY_DOC} via SSM (tag-targeted: project=open-swe, env=${ENV})" +CMD_ID="$(aws ssm send-command \ + --document-name "${DEPLOY_DOC}" \ + --targets "Key=tag:project,Values=open-swe" "Key=tag:env,Values=${ENV}" \ + --comment "${COMMENT}" \ + --query 'Command.CommandId' --output text)" +echo "command: ${CMD_ID}" + +echo "==> wait for the deploy to finish" +IID="" +for _ in $(seq 1 60); do + sleep 10 + IID="$(aws ssm list-command-invocations --command-id "${CMD_ID}" \ + --query 'CommandInvocations[0].InstanceId' --output text 2>/dev/null || echo None)" + [ -z "${IID}" ] || [ "${IID}" = "None" ] && continue + STATUS="$(aws ssm list-command-invocations --command-id "${CMD_ID}" \ + --query 'CommandInvocations[0].Status' --output text 2>/dev/null || echo Pending)" + case "${STATUS}" in + Success | Failed | Cancelled | TimedOut) break ;; + esac +done + +if [ -z "${IID}" ] || [ "${IID}" = "None" ]; then + echo "ERROR: no box picked up the deploy command (is a running open-swe ${ENV} box registered with SSM?)" >&2 + exit 1 +fi + +echo "==> deploy.sh output from ${IID}:" +echo "----- stdout -----" +aws ssm get-command-invocation --command-id "${CMD_ID}" --instance-id "${IID}" \ + --query 'StandardOutputContent' --output text || true +echo "----- stderr -----" +aws ssm get-command-invocation --command-id "${CMD_ID}" --instance-id "${IID}" \ + --query 'StandardErrorContent' --output text || true + +# Gate on the AGGREGATE command status (Success only if EVERY targeted invocation +# succeeded), not CommandInvocations[0] — during a userDataCausesReplacement window +# two instances can briefly share the project/env tags, and a partial failure on the +# other instance must not be reported as success. +TARGETS="$(aws ssm list-commands --command-id "${CMD_ID}" \ + --query 'Commands[0].TargetCount' --output text 2>/dev/null || echo 1)" +[ "${TARGETS}" = "1" ] || echo "WARNING: deploy fanned out to ${TARGETS} instances (expected 1)" +AGG="$(aws ssm list-commands --command-id "${CMD_ID}" \ + --query 'Commands[0].Status' --output text 2>/dev/null || echo Failed)" + +echo "==> aggregate deploy status: ${AGG} (across ${TARGETS} target(s))" +[ "${AGG}" = "Success" ] || { echo "ERROR: deploy did not succeed (${AGG})" >&2; exit 1; } diff --git a/.github/scripts/rollback.sh b/.github/scripts/rollback.sh new file mode 100755 index 00000000..1c277047 --- /dev/null +++ b/.github/scripts/rollback.sh @@ -0,0 +1,66 @@ +#!/usr/bin/env bash +# Roll an env BACK to a prior release: re-point releases/latest/ at a chosen release +# and re-fire the deploy. Run by rollback.yml after aws creds are configured. +# +# ENV dev | prod (required) +# BUCKET open-swe--assets (required) +# DEPLOY_DOC open-swe--deploy (required) +# TARGET_SHA release sha to restore; blank => releases/last-good/ (optional) +# +# releases/latest/ is moved TRANSACTIONALLY (same as the forward deploy): pointed at +# the target, the box rolled, and on a failed roll latest is reverted to whatever it +# was before the rollback attempt — so a failed rollback never leaves latest at a +# release the box could not bring up. releases/last-good/ is left untouched; advance +# it by running a forward deploy. +# +# Reuses the app deploy role's existing releases/* write + tag-scoped ssm:SendCommand +# — no new IAM. The fire/wait/gate is the SAME roll-box.sh the forward deploy uses. +set -euo pipefail +: "${ENV:?}" "${BUCKET:?}" "${DEPLOY_DOC:?}" +HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" + +# Copy a release prefix into releases/latest/ AND record the resolved sha, so +# releases/latest/sha.txt is always a real commit a later revert can resolve. +point_latest() { # $1 = source prefix (releases/|releases/last-good); $2 = sha to record + local src="$1" sha="$2" f + for f in app.tar.gz spa.tar.gz; do + aws s3 cp "s3://${BUCKET}/${src}/${f}" "s3://${BUCKET}/releases/latest/${f}" + done + printf '%s\n' "${sha}" | aws s3 cp - "s3://${BUCKET}/releases/latest/sha.txt" +} + +if [ -n "${TARGET_SHA:-}" ]; then + SRC="releases/${TARGET_SHA}" + LABEL="${TARGET_SHA}" + RESOLVED_SHA="${TARGET_SHA}" +else + SRC="releases/last-good" + LABEL="last-good" + # resolve last-good's real sha so latest/sha.txt records a commit, not a label. + RESOLVED_SHA="$(aws s3 cp "s3://${BUCKET}/releases/last-good/sha.txt" - 2>/dev/null | tr -d '[:space:]' || true)" + RESOLVED_SHA="${RESOLVED_SHA:-last-good}" +fi +echo "==> rollback ${ENV} to ${LABEL} (s3://${BUCKET}/${SRC}/, sha=${RESOLVED_SHA})" + +# Refuse to roll back to a release that is not fully present. +for f in app.tar.gz spa.tar.gz; do + aws s3 ls "s3://${BUCKET}/${SRC}/${f}" >/dev/null 2>&1 \ + || { echo "ERROR: ${SRC}/${f} not found in s3://${BUCKET} — cannot roll back to ${LABEL}" >&2; exit 1; } +done + +# Capture what latest points at now, so a failed rollback can be reverted. +PREV_SHA="$(aws s3 cp "s3://${BUCKET}/releases/latest/sha.txt" - 2>/dev/null | tr -d '[:space:]' || true)" + +echo "==> point releases/latest/ -> ${SRC} (was ${PREV_SHA:-})" +point_latest "${SRC}" "${RESOLVED_SHA}" + +if ! ROLL_COMMENT="rollback ${ENV} to ${LABEL}" bash "${HERE}/roll-box.sh"; then + if [ -n "${PREV_SHA}" ]; then + echo "!! rollback deploy failed — reverting releases/latest/ -> ${PREV_SHA}" >&2 + point_latest "releases/${PREV_SHA}" "${PREV_SHA}" + fi + exit 1 +fi + +echo "==> ${ENV} rolled back to ${LABEL}" +echo "NOTE: releases/last-good/ is left unchanged; re-run a forward deploy to advance it." diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index ff062325..c79fd149 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -5,7 +5,10 @@ permissions: on: push: - branches: ["main"] + # dev as well as main so every dev HEAD carries the full Agent CI signal that + # the dev->main promotion gate (check-dev-green.sh) reads. PR checks alone are + # not enough: an admin-merge can land a red PR onto dev. + branches: ["main", "dev"] pull_request: workflow_dispatch: diff --git a/.github/workflows/promote_dev_to_prod.yml b/.github/workflows/promote_dev_to_prod.yml index 3677d42f..8c38e699 100644 --- a/.github/workflows/promote_dev_to_prod.yml +++ b/.github/workflows/promote_dev_to_prod.yml @@ -15,11 +15,29 @@ concurrency: jobs: promote: runs-on: ubuntu-latest + permissions: + contents: write + checks: read steps: - uses: actions/checkout@v6 with: ref: dev fetch-depth: 0 + - name: Require dev HEAD fully green + # Hard precondition: every check-run on the dev HEAD commit must be + # completed + passing before we let it become prod. A red OR still-pending + # check blocks the promotion. The promote job's own in-progress check-run + # is excluded by name so the gate can't deadlock on itself. + env: + GH_TOKEN: ${{ github.token }} + # exclude THIS run's own check-run by its run id (not by name). + EXCLUDE_RUN_ID: ${{ github.run_id }} + run: | + SHA="$(git rev-parse HEAD)" + echo "dev HEAD = ${SHA}" + gh api --paginate "repos/${GITHUB_REPOSITORY}/commits/${SHA}/check-runs" \ + -q '.check_runs[] | [.name, .status, (.conclusion // ""), (.details_url // "")] | join("\u001f")' \ + | bash .github/scripts/check-dev-green.sh - name: Fast-forward main (PROD) to dev # main is the production branch; a plain ref push is fast-forward-only # (branch protection rejects non-FF), so a diverged main fails loudly. diff --git a/.github/workflows/rollback.yml b/.github/workflows/rollback.yml new file mode 100644 index 00000000..1a51e49c --- /dev/null +++ b/.github/workflows/rollback.yml @@ -0,0 +1,84 @@ +name: Rollback (re-point env to a prior release) + +# Roll an env back to a previously published release without rebuilding. Re-points +# releases/latest/ at the chosen release and re-fires the open-swe--deploy SSM +# document — same fire/wait/gate path as a forward deploy (roll-box.sh). +# +# env=dev, sha blank → restore open-swe-dev-assets/releases/last-good/ (AUTO) +# env=prod, sha blank → restore open-swe-prod-assets/releases/last-good/ (manual +# approval: Environment "prod", same gate as a prod deploy) +# sha= → restore that exact releases// instead of last-good. +# +# No new IAM: reuses the githubdeploy-open-swe-app- role's existing releases/* +# write + tag-scoped ssm:SendCommand. OIDC subject alignment matches build-artifacts: +# - rollback-dev declares NO `environment:` → sub = repo:…:ref:refs/heads/ +# - rollback-prod declares `environment: prod` → sub = repo:…:environment:prod + +permissions: + contents: read + +on: + workflow_dispatch: + inputs: + env: + description: "Which environment to roll back" + required: true + type: choice + options: [dev, prod] + sha: + description: "Release SHA to restore (blank = releases/last-good)" + required: false + type: string + +concurrency: + # never overlap a rollback with another rollback/deploy of the same env. + group: rollback-${{ inputs.env }} + cancel-in-progress: false + +jobs: + rollback-dev: + name: Rollback (dev) + if: ${{ inputs.env == 'dev' }} + runs-on: ubuntu-latest + timeout-minutes: 20 + permissions: + id-token: write + contents: read + env: + ENV: dev + BUCKET: open-swe-dev-assets + DEPLOY_DOC: open-swe-dev-deploy + TARGET_SHA: ${{ inputs.sha }} + steps: + - uses: actions/checkout@v6 + - uses: aws-actions/configure-aws-credentials@v6 + with: + role-to-assume: ${{ vars.AWS_DEPLOY_ROLE_APP_DEV }} + aws-region: us-east-1 + - name: Re-point releases/latest + redeploy + run: bash .github/scripts/rollback.sh + + rollback-prod: + name: Rollback (prod) + if: ${{ inputs.env == 'prod' }} + runs-on: ubuntu-latest + timeout-minutes: 20 + # Manual-approval gate: the "prod" Environment requires a reviewer (Adam). Also + # makes the OIDC sub …:environment:prod (matches the prod app-role trust). + environment: prod + permissions: + id-token: write + contents: read + env: + ENV: prod + BUCKET: open-swe-prod-assets + DEPLOY_DOC: open-swe-prod-deploy + TARGET_SHA: ${{ inputs.sha }} + steps: + - uses: actions/checkout@v6 + - uses: aws-actions/configure-aws-credentials@v6 + with: + role-to-assume: ${{ vars.AWS_DEPLOY_ROLE_APP_PROD }} + aws-region: us-east-1 + - name: Re-point releases/latest + redeploy + run: bash .github/scripts/rollback.sh diff --git a/infra/lib/constructs/github-deploy-roles.ts b/infra/lib/constructs/github-deploy-roles.ts index a53333f5..fe7533f6 100644 --- a/infra/lib/constructs/github-deploy-roles.ts +++ b/infra/lib/constructs/github-deploy-roles.ts @@ -139,10 +139,14 @@ export class GithubDeployRoles extends Construct { // Object actions are scoped to releases/* (the only prefix CI writes), and to // THIS env's bucket — a dev token can never write the prod bucket. No // bucket-level mutation (no PutBucket*/Delete bucket) — that stays with CDK. + // GetObject + PutObject (S3-to-S3 copy = Get source + Put dest) is all the + // publish/rollback path uses; s3:DeleteObject is deliberately NOT granted so a + // CI token cannot erase an immutable release or the releases/last-good rollback + // fallback (lifecycle expiry handles old-version cleanup, not CI). this.appRole.addToPolicy( new iam.PolicyStatement({ sid: "ReadWriteArtifactObjects", - actions: ["s3:GetObject", "s3:PutObject", "s3:DeleteObject"], + actions: ["s3:GetObject", "s3:PutObject"], resources: [`arn:aws:s3:::open-swe-${envName}-assets/releases/*`], }), );