Add cfn-stack-decommission + resource-usage-probe ops scripts (#11)
Some checks are pending
Deploy / deploy (push) Waiting to run

cfn-stack-decommission.sh: report-by-default stack retirement; pre-flight
predicts DeletionPolicy:Retain orphans + consumed-export blocks before delete
(distilled from the LedgerFlow decommission). --execute to act.

resource-usage-probe.sh: is-it-used probe (RDS connections/Lambda invocations/
DDB capacity/EBS attachment) to choose retire-vs-harden before acting on an
encrypt/migrate finding (the database-1 H-19 lesson).
This commit is contained in:
Adam Moussa 2026-06-03 13:25:44 -04:00 • committed by GitHub
parent 0dd8d2a7af
commit 14593440cf
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 141 additions and 0 deletions

View file

@ -0,0 +1,80 @@
#!/usr/bin/env bash
#
# cfn-stack-decommission.sh — safely retire a CloudFormation/CDK stack.
#
# Reports first (default), acts only with --execute. The value is the pre-flight:
# it predicts what will ORPHAN (DeletionPolicy: Retain resources survive a stack
# delete) and what will BLOCK the delete (consumed exports, non-empty buckets),
# so you don't discover surviving tables/buckets after the fact.
#
# Built from the Day 4 LedgerFlow decommission, where 4 of 5 DynamoDB tables +
# 2 of 3 S3 buckets were RemovalPolicy.RETAIN and orphaned. See feedback memory
# `feedback_cfn_decommission_and_remediation`.
#
# Usage:
# scripts/cfn-stack-decommission.sh [--profile NAME] [--execute] STACK
#
# (no --execute) REPORT only: termination protection, consumed exports,
# Retain resources (orphans-to-be), in-stack S3 buckets.
# --execute Disable termination protection, empty Delete-policy buckets,
# delete the stack, wait, then delete the Retain orphans.
#
set -euo pipefail
PROFILE_ARG=(); EXECUTE=0; STACK=""
while [[ $# -gt 0 ]]; do
case "$1" in
--profile) PROFILE_ARG=(--profile "$2"); shift 2 ;;
--execute) EXECUTE=1; shift ;;
-h|--help) sed -n '2,22p' "$0"; exit 0 ;;
-*) echo "unknown flag: $1" >&2; exit 2 ;;
*) STACK="$1"; shift ;;
esac
done
[[ -z "$STACK" ]] && { echo "usage: $0 [--profile NAME] [--execute] STACK" >&2; exit 2; }
aws_() { aws "${PROFILE_ARG[@]}" "$@"; }
R="us-east-1"
echo "== stack: $STACK =="
aws_ cloudformation describe-stacks --stack-name "$STACK" --region "$R" \
--query 'Stacks[0].{Status:StackStatus,TermProt:EnableTerminationProtection}' --output table
echo "-- consumed exports (any import BLOCKS the delete) --"
BLOCKED=0
for e in $(aws_ cloudformation list-exports --region "$R" \
--query "Exports[?ExportingStackId && contains(ExportingStackId,':stack/$STACK/')].Name" --output text 2>/dev/null); do
imp=$(aws_ cloudformation list-imports --export-name "$e" --region "$R" --query 'Imports' --output text 2>/dev/null || true)
if [[ -n "$imp" && "$imp" != "None" ]]; then echo " BLOCK: export $e imported by: $imp"; BLOCKED=1; fi
done
[[ $BLOCKED -eq 0 ]] && echo " none"
echo "-- DeletionPolicy: Retain resources (these ORPHAN, survive the delete) --"
TMP="$(aws_ cloudformation get-template --stack-name "$STACK" --region "$R" --query TemplateBody --output json)"
echo "$TMP" | python3 -c '
import json,sys
res=json.load(sys.stdin).get("Resources",{})
orphans=[(r.get("Type"),lid,r.get("Properties",{}).get("TableName") or r.get("Properties",{}).get("BucketName") or "")
for lid,r in res.items() if r.get("DeletionPolicy")=="Retain"]
[print(f" {t:<28} {lid} {name}") for t,lid,name in sorted(orphans)] or print(" none")
'
echo "-- in-stack S3 buckets (non-empty Delete-policy buckets block; check auto-delete) --"
for b in $(aws_ cloudformation list-stack-resources --stack-name "$STACK" --region "$R" \
--query "StackResourceSummaries[?ResourceType=='AWS::S3::Bucket'].PhysicalResourceId" --output text 2>/dev/null); do
n=$(aws_ s3api list-objects-v2 --bucket "$b" --max-items 1 --query 'KeyCount' --output text 2>/dev/null || echo "?")
v=$(aws_ s3api get-bucket-versioning --bucket "$b" --query 'Status' --output text 2>/dev/null || echo "-")
echo " $b objects~=$n versioning=$v"
done
if [[ $EXECUTE -eq 0 ]]; then
echo; echo "REPORT ONLY. Re-run with --execute to delete (after reviewing the orphans + blocks above)."
exit 0
fi
[[ $BLOCKED -eq 1 ]] && { echo "ABORT: a consumed export blocks the delete (see above)." >&2; exit 1; }
read -r -p "EXECUTE decommission of '$STACK'? [y/N] " ans; [[ "$ans" =~ ^[Yy]$ ]] || { echo "aborted"; exit 0; }
aws_ cloudformation update-termination-protection --stack-name "$STACK" --no-enable-termination-protection --region "$R" >/dev/null 2>&1 || true
echo "deleting stack..."
aws_ cloudformation delete-stack --stack-name "$STACK" --region "$R"
aws_ cloudformation wait stack-delete-complete --stack-name "$STACK" --region "$R"
echo "stack deleted. Review the Retain orphans above and remove them with delete-table / delete-bucket"
echo "(versioned buckets: purge all versions + delete-markers first — see the iam-user-delete sibling pattern)."

61
scripts/resource-usage-probe.sh Executable file
View file

@ -0,0 +1,61 @@
#!/usr/bin/env bash
#
# resource-usage-probe.sh — is this resource actually used?
#
# Run BEFORE acting on an "encrypt / migrate / right-size / encrypt-with-downtime"
# finding. Idle resources should usually be retired (cheaper, no downtime) instead
# of hardened in place. This is what flipped audit H-19 from "encrypt database-1
# with a downtime window" to "snapshot + delete" — it had 0 connections in 60 days.
# See feedback memory `feedback_cfn_decommission_and_remediation`.
#
# Usage:
# scripts/resource-usage-probe.sh [--profile NAME] rds <db-instance-id>
# scripts/resource-usage-probe.sh [--profile NAME] lambda <function-name>
# scripts/resource-usage-probe.sh [--profile NAME] ddb <table-name>
# scripts/resource-usage-probe.sh [--profile NAME] ebs <volume-id>
#
set -euo pipefail
PROFILE_ARG=(); ARGS=()
while [[ $# -gt 0 ]]; do
case "$1" in
--profile) PROFILE_ARG=(--profile "$2"); shift 2 ;;
-h|--help) sed -n '2,18p' "$0"; exit 0 ;;
*) ARGS+=("$1"); shift ;;
esac
done
[[ ${#ARGS[@]} -lt 2 ]] && { echo "usage: $0 [--profile NAME] {rds|lambda|ddb|ebs} <id>" >&2; exit 2; }
KIND="${ARGS[0]}"; ID="${ARGS[1]}"; R="us-east-1"
aws_() { aws "${PROFILE_ARG[@]}" "$@"; }
# UTC window helpers (no Date.now equivalent needed; python gives tz-aware UTC)
since() { python3 -c "import datetime;print((datetime.datetime.now(datetime.UTC)-datetime.timedelta(days=$1)).strftime('%Y-%m-%dT%H:%M:%SZ'))"; }
now() { python3 -c "import datetime;print(datetime.datetime.now(datetime.UTC).strftime('%Y-%m-%dT%H:%M:%SZ'))"; }
maxstat() { aws_ cloudwatch get-metric-statistics --namespace "$1" --metric-name "$2" \
--dimensions Name="$3",Value="$4" --start-time "$(since "$5")" --end-time "$(now)" \
--period $(( $5 * 86400 )) --statistics Maximum Sum --region "$R" \
--query 'Datapoints[0].{Max:Maximum,Sum:Sum}' --output text 2>/dev/null; }
echo "== $KIND: $ID =="
case "$KIND" in
rds)
aws_ rds describe-db-instances --db-instance-identifier "$ID" --region "$R" \
--query 'DBInstances[0].{Engine:Engine,Class:DBInstanceClass,Enc:StorageEncrypted,MultiAZ:MultiAZ,SG:VpcSecurityGroups[].VpcSecurityGroupId}' --output table
echo "connections (Max/Sum over 60d): $(maxstat AWS/RDS DatabaseConnections DBInstanceIdentifier "$ID" 60)"
echo "tags: $(aws_ rds list-tags-for-resource --resource-name "arn:aws:rds:$R:$(aws_ sts get-caller-identity --query Account --output text):db:$ID" --query 'TagList' --output text 2>/dev/null || echo none)" ;;
lambda)
echo "invocations (Max/Sum 90d): $(maxstat AWS/Lambda Invocations FunctionName "$ID" 90)"
echo "errors (Max/Sum 90d): $(maxstat AWS/Lambda Errors FunctionName "$ID" 90)"
aws_ lambda get-function-configuration --function-name "$ID" --region "$R" --query '{LastModified:LastModified,Runtime:Runtime}' --output table 2>/dev/null || true ;;
ddb)
aws_ dynamodb describe-table --table-name "$ID" --region "$R" --query 'Table.{Items:ItemCount,Bytes:TableSizeBytes,Billing:BillingModeSummary.BillingMode}' --output table
echo "consumed write (Max/Sum 30d): $(maxstat AWS/DynamoDB ConsumedWriteCapacityUnits TableName "$ID" 30)"
echo "consumed read (Max/Sum 30d): $(maxstat AWS/DynamoDB ConsumedReadCapacityUnits TableName "$ID" 30)"
echo "PITR: $(aws_ dynamodb describe-continuous-backups --table-name "$ID" --region "$R" --query 'ContinuousBackupsDescription.PointInTimeRecoveryDescription.PointInTimeRecoveryStatus' --output text 2>/dev/null)" ;;
ebs)
aws_ ec2 describe-volumes --volume-ids "$ID" --region "$R" \
--query 'Volumes[0].{Size:Size,State:State,Enc:Encrypted,Attached:Attachments[0].InstanceId,AttachState:Attachments[0].State}' --output table ;;
*) echo "unknown kind: $KIND (use rds|lambda|ddb|ebs)" >&2; exit 2 ;;
esac
echo
echo "VERDICT GUIDE: near-zero connections/invocations/consumed-capacity + no recent attachment => IDLE."
echo " IDLE -> retire (final backup, then delete) — cheaper, no downtime than encrypt/migrate-in-place."
echo " ACTIVE-> harden in place per the finding."