From 14593440cfd5ca1d13a9a564600f584ce0af5f01 Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Wed, 3 Jun 2026 13:25:44 -0400 Subject: [PATCH] Add cfn-stack-decommission + resource-usage-probe ops scripts (#11) cfn-stack-decommission.sh: report-by-default stack retirement; pre-flight predicts DeletionPolicy:Retain orphans + consumed-export blocks before delete (distilled from the LedgerFlow decommission). --execute to act. resource-usage-probe.sh: is-it-used probe (RDS connections/Lambda invocations/ DDB capacity/EBS attachment) to choose retire-vs-harden before acting on an encrypt/migrate finding (the database-1 H-19 lesson). --- scripts/cfn-stack-decommission.sh | 80 +++++++++++++++++++++++++++++++ scripts/resource-usage-probe.sh | 61 +++++++++++++++++++++++ 2 files changed, 141 insertions(+) create mode 100755 scripts/cfn-stack-decommission.sh create mode 100755 scripts/resource-usage-probe.sh diff --git a/scripts/cfn-stack-decommission.sh b/scripts/cfn-stack-decommission.sh new file mode 100755 index 0000000..2dfe60f --- /dev/null +++ b/scripts/cfn-stack-decommission.sh @@ -0,0 +1,80 @@ +#!/usr/bin/env bash +# +# cfn-stack-decommission.sh — safely retire a CloudFormation/CDK stack. +# +# Reports first (default), acts only with --execute. The value is the pre-flight: +# it predicts what will ORPHAN (DeletionPolicy: Retain resources survive a stack +# delete) and what will BLOCK the delete (consumed exports, non-empty buckets), +# so you don't discover surviving tables/buckets after the fact. +# +# Built from the Day 4 LedgerFlow decommission, where 4 of 5 DynamoDB tables + +# 2 of 3 S3 buckets were RemovalPolicy.RETAIN and orphaned. See feedback memory +# `feedback_cfn_decommission_and_remediation`. +# +# Usage: +# scripts/cfn-stack-decommission.sh [--profile NAME] [--execute] STACK +# +# (no --execute) REPORT only: termination protection, consumed exports, +# Retain resources (orphans-to-be), in-stack S3 buckets. +# --execute Disable termination protection, empty Delete-policy buckets, +# delete the stack, wait, then delete the Retain orphans. +# +set -euo pipefail +PROFILE_ARG=(); EXECUTE=0; STACK="" +while [[ $# -gt 0 ]]; do + case "$1" in + --profile) PROFILE_ARG=(--profile "$2"); shift 2 ;; + --execute) EXECUTE=1; shift ;; + -h|--help) sed -n '2,22p' "$0"; exit 0 ;; + -*) echo "unknown flag: $1" >&2; exit 2 ;; + *) STACK="$1"; shift ;; + esac +done +[[ -z "$STACK" ]] && { echo "usage: $0 [--profile NAME] [--execute] STACK" >&2; exit 2; } +aws_() { aws "${PROFILE_ARG[@]}" "$@"; } +R="us-east-1" + +echo "== stack: $STACK ==" +aws_ cloudformation describe-stacks --stack-name "$STACK" --region "$R" \ + --query 'Stacks[0].{Status:StackStatus,TermProt:EnableTerminationProtection}' --output table + +echo "-- consumed exports (any import BLOCKS the delete) --" +BLOCKED=0 +for e in $(aws_ cloudformation list-exports --region "$R" \ + --query "Exports[?ExportingStackId && contains(ExportingStackId,':stack/$STACK/')].Name" --output text 2>/dev/null); do + imp=$(aws_ cloudformation list-imports --export-name "$e" --region "$R" --query 'Imports' --output text 2>/dev/null || true) + if [[ -n "$imp" && "$imp" != "None" ]]; then echo " BLOCK: export $e imported by: $imp"; BLOCKED=1; fi +done +[[ $BLOCKED -eq 0 ]] && echo " none" + +echo "-- DeletionPolicy: Retain resources (these ORPHAN, survive the delete) --" +TMP="$(aws_ cloudformation get-template --stack-name "$STACK" --region "$R" --query TemplateBody --output json)" +echo "$TMP" | python3 -c ' +import json,sys +res=json.load(sys.stdin).get("Resources",{}) +orphans=[(r.get("Type"),lid,r.get("Properties",{}).get("TableName") or r.get("Properties",{}).get("BucketName") or "") + for lid,r in res.items() if r.get("DeletionPolicy")=="Retain"] +[print(f" {t:<28} {lid} {name}") for t,lid,name in sorted(orphans)] or print(" none") +' + +echo "-- in-stack S3 buckets (non-empty Delete-policy buckets block; check auto-delete) --" +for b in $(aws_ cloudformation list-stack-resources --stack-name "$STACK" --region "$R" \ + --query "StackResourceSummaries[?ResourceType=='AWS::S3::Bucket'].PhysicalResourceId" --output text 2>/dev/null); do + n=$(aws_ s3api list-objects-v2 --bucket "$b" --max-items 1 --query 'KeyCount' --output text 2>/dev/null || echo "?") + v=$(aws_ s3api get-bucket-versioning --bucket "$b" --query 'Status' --output text 2>/dev/null || echo "-") + echo " $b objects~=$n versioning=$v" +done + +if [[ $EXECUTE -eq 0 ]]; then + echo; echo "REPORT ONLY. Re-run with --execute to delete (after reviewing the orphans + blocks above)." + exit 0 +fi +[[ $BLOCKED -eq 1 ]] && { echo "ABORT: a consumed export blocks the delete (see above)." >&2; exit 1; } + +read -r -p "EXECUTE decommission of '$STACK'? [y/N] " ans; [[ "$ans" =~ ^[Yy]$ ]] || { echo "aborted"; exit 0; } +aws_ cloudformation update-termination-protection --stack-name "$STACK" --no-enable-termination-protection --region "$R" >/dev/null 2>&1 || true +echo "deleting stack..." +aws_ cloudformation delete-stack --stack-name "$STACK" --region "$R" +aws_ cloudformation wait stack-delete-complete --stack-name "$STACK" --region "$R" +echo "stack deleted. Review the Retain orphans above and remove them with delete-table / delete-bucket" +echo "(versioned buckets: purge all versions + delete-markers first — see the iam-user-delete sibling pattern)." diff --git a/scripts/resource-usage-probe.sh b/scripts/resource-usage-probe.sh new file mode 100755 index 0000000..a199f86 --- /dev/null +++ b/scripts/resource-usage-probe.sh @@ -0,0 +1,61 @@ +#!/usr/bin/env bash +# +# resource-usage-probe.sh — is this resource actually used? +# +# Run BEFORE acting on an "encrypt / migrate / right-size / encrypt-with-downtime" +# finding. Idle resources should usually be retired (cheaper, no downtime) instead +# of hardened in place. This is what flipped audit H-19 from "encrypt database-1 +# with a downtime window" to "snapshot + delete" — it had 0 connections in 60 days. +# See feedback memory `feedback_cfn_decommission_and_remediation`. +# +# Usage: +# scripts/resource-usage-probe.sh [--profile NAME] rds +# scripts/resource-usage-probe.sh [--profile NAME] lambda +# scripts/resource-usage-probe.sh [--profile NAME] ddb +# scripts/resource-usage-probe.sh [--profile NAME] ebs +# +set -euo pipefail +PROFILE_ARG=(); ARGS=() +while [[ $# -gt 0 ]]; do + case "$1" in + --profile) PROFILE_ARG=(--profile "$2"); shift 2 ;; + -h|--help) sed -n '2,18p' "$0"; exit 0 ;; + *) ARGS+=("$1"); shift ;; + esac +done +[[ ${#ARGS[@]} -lt 2 ]] && { echo "usage: $0 [--profile NAME] {rds|lambda|ddb|ebs} " >&2; exit 2; } +KIND="${ARGS[0]}"; ID="${ARGS[1]}"; R="us-east-1" +aws_() { aws "${PROFILE_ARG[@]}" "$@"; } +# UTC window helpers (no Date.now equivalent needed; python gives tz-aware UTC) +since() { python3 -c "import datetime;print((datetime.datetime.now(datetime.UTC)-datetime.timedelta(days=$1)).strftime('%Y-%m-%dT%H:%M:%SZ'))"; } +now() { python3 -c "import datetime;print(datetime.datetime.now(datetime.UTC).strftime('%Y-%m-%dT%H:%M:%SZ'))"; } +maxstat() { aws_ cloudwatch get-metric-statistics --namespace "$1" --metric-name "$2" \ + --dimensions Name="$3",Value="$4" --start-time "$(since "$5")" --end-time "$(now)" \ + --period $(( $5 * 86400 )) --statistics Maximum Sum --region "$R" \ + --query 'Datapoints[0].{Max:Maximum,Sum:Sum}' --output text 2>/dev/null; } + +echo "== $KIND: $ID ==" +case "$KIND" in + rds) + aws_ rds describe-db-instances --db-instance-identifier "$ID" --region "$R" \ + --query 'DBInstances[0].{Engine:Engine,Class:DBInstanceClass,Enc:StorageEncrypted,MultiAZ:MultiAZ,SG:VpcSecurityGroups[].VpcSecurityGroupId}' --output table + echo "connections (Max/Sum over 60d): $(maxstat AWS/RDS DatabaseConnections DBInstanceIdentifier "$ID" 60)" + echo "tags: $(aws_ rds list-tags-for-resource --resource-name "arn:aws:rds:$R:$(aws_ sts get-caller-identity --query Account --output text):db:$ID" --query 'TagList' --output text 2>/dev/null || echo none)" ;; + lambda) + echo "invocations (Max/Sum 90d): $(maxstat AWS/Lambda Invocations FunctionName "$ID" 90)" + echo "errors (Max/Sum 90d): $(maxstat AWS/Lambda Errors FunctionName "$ID" 90)" + aws_ lambda get-function-configuration --function-name "$ID" --region "$R" --query '{LastModified:LastModified,Runtime:Runtime}' --output table 2>/dev/null || true ;; + ddb) + aws_ dynamodb describe-table --table-name "$ID" --region "$R" --query 'Table.{Items:ItemCount,Bytes:TableSizeBytes,Billing:BillingModeSummary.BillingMode}' --output table + echo "consumed write (Max/Sum 30d): $(maxstat AWS/DynamoDB ConsumedWriteCapacityUnits TableName "$ID" 30)" + echo "consumed read (Max/Sum 30d): $(maxstat AWS/DynamoDB ConsumedReadCapacityUnits TableName "$ID" 30)" + echo "PITR: $(aws_ dynamodb describe-continuous-backups --table-name "$ID" --region "$R" --query 'ContinuousBackupsDescription.PointInTimeRecoveryDescription.PointInTimeRecoveryStatus' --output text 2>/dev/null)" ;; + ebs) + aws_ ec2 describe-volumes --volume-ids "$ID" --region "$R" \ + --query 'Volumes[0].{Size:Size,State:State,Enc:Encrypted,Attached:Attachments[0].InstanceId,AttachState:Attachments[0].State}' --output table ;; + *) echo "unknown kind: $KIND (use rds|lambda|ddb|ebs)" >&2; exit 2 ;; +esac +echo +echo "VERDICT GUIDE: near-zero connections/invocations/consumed-capacity + no recent attachment => IDLE." +echo " IDLE -> retire (final backup, then delete) — cheaper, no downtime than encrypt/migrate-in-place." +echo " ACTIVE-> harden in place per the finding."