From 2a2929b83562dd7ebaf8fbe2b93c7902f3508e32 Mon Sep 17 00:00:00 2001 From: Adam Moussa Date: Fri, 7 Aug 2026 11:31:34 -0400 Subject: [PATCH] chore(cd): hard-cut CDK deploy; HCP is sole mutate path (PLAT-86) Remove push-to-main cd-cdk workflow, document HCP apply + dispose runbook, and park githubdeploy-procurement-ingest for a later IAM cleanup. --- .github/workflows/deploy.yaml | 22 ---- README.md | 60 +++++------ docs/plat-86/cfn-dispose.md | 25 +++++ docs/plat-86/import-map.md | 8 +- infra/deploy-role/README.md | 41 +++----- scripts/plat86-cfn-dispose.sh | 183 ++++++++++++++++++++++++++++++++++ 6 files changed, 252 insertions(+), 87 deletions(-) delete mode 100644 .github/workflows/deploy.yaml create mode 100644 docs/plat-86/cfn-dispose.md create mode 100755 scripts/plat86-cfn-dispose.sh diff --git a/.github/workflows/deploy.yaml b/.github/workflows/deploy.yaml deleted file mode 100644 index 7f206a1..0000000 --- a/.github/workflows/deploy.yaml +++ /dev/null @@ -1,22 +0,0 @@ -name: Deploy -on: - push: - branches: [main] - -permissions: - id-token: write - contents: read - -concurrency: - group: deploy - cancel-in-progress: false - -jobs: - deploy: - uses: Sea-Haven-Industries/.github/.github/workflows/cd-cdk.yaml@3f746774229d41770727e2e4fd63ed5f5555a8b3 # v1.0.3 - with: - python-version: "3.12" - cdk-dir: cdk - post-deploy-script: scripts/post-deploy-smoke.sh - secrets: - deploy-role-arn: ${{ secrets.AWS_DEPLOY_ROLE_ARN }} diff --git a/README.md b/README.md index f5c3457..f646767 100644 --- a/README.md +++ b/README.md @@ -1,10 +1,10 @@ # Procurement Ingest ![Python](https://img.shields.io/badge/Python-3776AB?logo=python&logoColor=white) -![AWS CDK](https://img.shields.io/badge/AWS-CDK-FF9900?logo=amazonaws&logoColor=white) +![Terraform](https://img.shields.io/badge/Terraform-844FBA?logo=terraform&logoColor=white) ![CI](https://github.com/Sea-Haven-Industries/procurement-ingest/actions/workflows/ci.yaml/badge.svg) -Unified email ingestion pipelines for Amazon procurement data. Two independent pipelines — purchase orders (Coupa) and work orders (APM/Hexagon EAM) — share a single repo and CDK app but deploy as separate CloudFormation stacks. +Unified email ingestion pipelines for Amazon procurement data. Two independent pipelines — purchase orders (Coupa) and work orders (APM/Hexagon EAM) — share a single repo and deploy to **seahaven-prod** via HCP Terraform workspace `procurement-ingest-prod` (PLAT-86). ## Pipelines @@ -119,7 +119,7 @@ A read-only REST API (API Gateway + the `procurement-api` Lambda, `lambdas/api/` ## Architecture -**IaC:** AWS CDK (Python), three stacks in one app, region `us-east-1`. The `cdk.Environment` is deliberately **account-agnostic** (region-only, no `account=`): the stacks deploy to whichever account the deploy credentials target. **Live target is only seahaven-prod `011934824531`** (migrated 2026-07; mgmt stacks deleted in PLAT-67 on 2026-08-05). Never deploy from `main` to mgmt `328440206208` — templates would recreate against cold-archive RETAIN leftovers. Every account-derived template value — bucket names, `Lambda::Permission` source account, the site-alerts SNS action ARN, the Bedrock ARN below — renders as the CloudFormation `AWS::AccountId` pseudo-parameter rather than a literal. Pinning `account=` was evaluated in Phase 4 and rejected: it would resolve those tokens to literals, and against the deployed (account-agnostic) templates CloudFormation flags the `RemovalPolicy.RETAIN` email buckets as requiring replacement — a data-loss risk — for no functional gain. +**IaC:** Terraform under `terraform/` (HCP remote apply, Manual until sealed). Workspace `procurement-ingest-prod` in project `seahaven-prod` is the sole mutate path for live resources in seahaven-prod `011934824531` / `us-east-1` (PLAT-86 import-in-place). Former CDK stacks (`po-ingest`, `WorkorderIngestStack`, `procurement-api`) are historical; `cdk/` remains for reference/synth until a follow-up hygiene pass. Never apply against mgmt `328440206208` — PLAT-67 left cold-archive RETAIN leftovers there only. All Lambdas: Python 3.12, ARM64, 60-day log retention. @@ -212,12 +212,12 @@ if isinstance(event, dict) and event.get("healthcheck") is True: This placement is deliberate, not incidental: real mail always arrives as an S3 `ObjectCreated` event whose top-level keys (`Records`) AWS controls, so email content can never set a top-level `healthcheck` key — the branch creates no accept path for forged mail. It also emits **no EMF metric and no log line**, so it can never match the `sender_auth_rejected` log-metric-filter pattern that feeds the `-sender-auth-rejected` alarm (see [CloudWatch alarms](#cloudwatch-alarms)) — that alarm pages at ≥1 match in its window, so repeated healthcheck invokes across deploys (two deploys in ~30 min is routine) must never contribute to it. Unit coverage: `lambdas/po/email_processor/tests/test_po_healthcheck.py` and `lambdas/wo/email_processor/tests/test_healthcheck.py`. -**Post-deploy smoke gate.** `scripts/post-deploy-smoke.sh` is wired into the CD workflow as `cd-cdk.yaml`'s `post-deploy-script` input (see [CI/CD](#cicd)) and runs synchronously after every deploy, before the workflow is considered green. It invokes `po-email-processor`, `workorder-email-processor`, and `procurement-api` with `aws lambda invoke --invocation-type RequestResponse --payload '{"healthcheck": true}'` (region `us-east-1`) and asserts, per function: +**Post-apply smoke gate.** `scripts/post-deploy-smoke.sh` runs after HCP applies (manual or gated) against seahaven-prod. It invokes `po-email-processor`, `workorder-email-processor`, and `procurement-api` with `aws lambda invoke --invocation-type RequestResponse --payload '{"healthcheck": true}'` (region `us-east-1`) and asserts, per function: 1. The invoke response's **`FunctionError` field is absent** — this is the load-bearing check. A broken bundle (e.g. an `ImportError` at module init from a missing sibling module) still returns HTTP 200 from the Lambda Invoke API with `FunctionError=Unhandled`; a bare exit-code check on `aws lambda invoke` would false-pass on exactly the failure this gate exists to catch. 2. The returned payload is **exactly** `{"healthcheck": "ok"}`. -The script runs `set -euo pipefail` and exits non-zero on any invoke failure, any `FunctionError`, or a payload mismatch on either function, failing the deploy job. +The script runs `set -euo pipefail` and exits non-zero on any invoke failure, any `FunctionError`, or a payload mismatch, failing the apply gate. **PO bundling: glob replaces the hand-maintained allowlist.** `cdk/po_stack.py`'s asset bundling command ships PO's Lambda source with a non-recursive glob instead of a hand-maintained list of filenames (`cp handler.py ses_auth.py template_parser.py derived_fields.py /asset-output/`). The glob is functionally identical for today's file set — non-recursive, so `tests/` and other subdirectories are still excluded — but structurally eliminates the failure mode that shipped a broken bundle twice (PR #105 omitted `template_parser.py`; PR #2 nearly omitted `derived_fields.py`): a new sibling module the handler imports now ships automatically instead of requiring someone to remember to add it to the list. @@ -329,63 +329,51 @@ Owned by this repo's `po-ingest` stack (`cdk/po_stack.py`). PK `siteCode` (S); d ## Documentation -The canonical map of Sea Haven's AWS infrastructure lives in Confluence. This project's `po-ingest` and `WorkorderIngestStack` stacks are represented there as Mermaid subgraphs. +The canonical map of Sea Haven's AWS infrastructure lives in Confluence. This project's live path is HCP Terraform workspace `procurement-ingest-prod` (PLAT-86). - **[AWS Architecture Map](https://seahaven.atlassian.net/wiki/spaces/IT/pages/1540098)** (Confluence, IT space, page 1540098) +- Import / dispose notes: `docs/plat-86/` ## CI/CD -> **PLAT-86 (in progress):** CDK CD is soft-frozen (`deploy.yaml` disabled on GitHub) while ownership moves to HCP Terraform workspace `procurement-ingest-prod`. Terraform lives under `terraform/`; import map under `docs/plat-86/`. HCP becomes the sole mutate path after import + green Manual apply; then `deploy.yaml` / `cd-cdk.yaml` are retired. Post-apply smoke still targets `po-email-processor`, `workorder-email-processor`, and `procurement-api` via `scripts/post-deploy-smoke.sh`. - GitHub Actions with reusable workflows from `Sea-Haven-Industries/.github` (all pinned to a commit SHA of `main`): -- **CI** (`ci.yaml`, PR to `main`): linting + `cdk synth` via `ci-python-sam.yaml`. `cdk synth`'s Docker-bundled asset build for `po-email-processor` and `workorder-email-processor` mounts the widened `../lambdas` asset root (Phase 2, see [Deploy-Pipeline Guards](#deploy-pipeline-guards-phase-0)) as build context, not just each function's own subdirectory — the `exclude` list on both `from_asset` calls strips local-only `__pycache__`/`package/` (and `tests/`) cruft from that wider mount's source fingerprint, so CI's asset hash matches a clean local checkout, and each function's scoped `cp` glob copies only its own pipeline's `*.py` into the zip. (The exclude does not, and under `SOURCE` hashing cannot, keep the *other* pipeline's tracked source out of the fingerprint (see the PO bundling note above on `SOURCE` hashing) — but that source is identical in CI and local, so it does not cause hash divergence.) +- **CI** (`ci.yaml`, PR to `main`): linting + `cdk synth` via `ci-python-sam.yaml` (historical CDK tree still synthesizes until a follow-up hygiene pass removes it). - **Terraform CI** (`ci-terraform.yaml`, PR to `main` when `terraform/**` changes): `fmt -check`, `init -backend=false`, `validate`. -- **CD** (`deploy.yaml`, push to `main`): CDK deploy via `cd-cdk.yaml` (OIDC auth), followed by the synchronous `post-deploy-script: scripts/post-deploy-smoke.sh` healthcheck gate (see [Deploy-Pipeline Guards](#deploy-pipeline-guards-phase-0)) — `cd-cdk.yaml`'s `stack-name` input only accepts one stack, so the smoke script itself enumerates `po-email-processor`, `workorder-email-processor`, and `procurement-api`. **Soft-frozen for PLAT-86** (workflow disabled until HCP cutover seals). +- **CD:** HCP Terraform workspace `procurement-ingest-prod` (VCS-bound to `main`, working directory `terraform/`, Manual apply until sealed). There is no push-to-main CDK/SAM deploy workflow. Post-apply smoke: `scripts/post-deploy-smoke.sh`. - Plus dependency review and PR labeler workflows on every PR Branch protection on `main` — all changes through PR. ## Setup -**Account prerequisites** — the stacks import four dependencies by name, so all -of these must exist in the target account BEFORE the first deploy (in -seahaven-prod they are provisioned by the seahaven-org-baseline repo and the -migration Phase 0 runbook): +**Account prerequisites** — Terraform imports these by name/ARN; they must already exist in seahaven-prod (provisioned by seahaven-org-baseline and out-of-band scripts): -- SES receipt rule set `INBOUND_MAIL` (may be inactive; the stacks attach - their receipt rules to it) + a verified `int.seahaven.com` domain identity +- SES receipt rule set `INBOUND_MAIL` + verified `int.seahaven.com` domain identity - SNS topic `site-alerts` (with the `alias/seahaven-alarm-topics` CMK) -- SSM param `/seahaven/dynamodb/cmk-arn` -> KMS `alias/seahaven-dynamodb` -- Secrets Manager secret `procurement-ingest/web-ui-auth-token` (step 3) -- OIDC deploy role `githubdeploy-procurement-ingest` (see `infra/deploy-role/`) - with the `smoke-invoke-lambda` policy, or the post-deploy smoke gate fails - closed +- SSM param `/seahaven/dynamodb/cmk-arn` → KMS `alias/seahaven-dynamodb` +- SSM param `/procurement-api/custom-domain/certificate-arn` (ACM in prod; DNS validation/alias for `procurement-api.seahaven.com` stays in mgmt via `scripts/setup_procurement_api_domain.sh`) +- Secrets Manager secret `procurement-ingest/web-ui-auth-token` +- HCP plan/apply roles `hcptf-procurement-ingest-plan` / `hcptf-procurement-ingest` (org-baseline terraform-substrate) -1. Bootstrap CDK: `cdk bootstrap aws://{AccountId}/us-east-1` -2. Bedrock model access (account first-use). The Bedrock **Model access** console page is retired; serverless foundation models enable on first invoke. Anthropic models may still require a one-time use-case form (`PutUseCaseForModelAccess`) before agreement APIs succeed — that is already done for Sea Haven. Marketplace-served models (including Claude Haiku 4.5) additionally need one account-wide Marketplace subscription: an admin principal with `aws-marketplace:ViewSubscriptions` / `Subscribe` must call `CreateFoundationModelAgreement` (or successfully `InvokeModel` once) for `anthropic.claude-haiku-4-5-20251001-v1:0` in `us-east-1`. After the agreement reaches `AVAILABLE` (often ~2 minutes), processor Lambdas can `InvokeModel` with only their existing `bedrock:InvokeModel` grants — do **not** put `aws-marketplace:*` on `po-email-processor` / `workorder-email-processor` unless that first-use path still AccessDenies after the account agreement is `AVAILABLE`. The CDK grants already cover the `us.*` inference profile plus cross-region foundation-model ARNs (`us-east-1` / `us-east-2` / `us-west-2`). No API key or secret to set — the processors authenticate to Bedrock via their IAM roles. -3. Create the web UI auth-gate shared secret. This secret is **imported by name** - (`Secret.from_secret_name_v2`), not CDK-managed, so it must exist before deploy - or the web-ui Lambdas fail closed: +1. Set workspace variables in HCP (`procurement-ingest-prod`) from `terraform/terraform.tfvars.example` (secret ARNs only; never secret values). +2. Bedrock model access (account first-use). Anthropic Marketplace agreements and inference-profile reach are account-level; processor roles already carry `bedrock:InvokeModel` grants under `/tf-managed/`. +3. Create the web UI auth-gate shared secret if missing (imported by name, not Terraform-managed): ```bash aws secretsmanager create-secret --name procurement-ingest/web-ui-auth-token --secret-string "$(openssl rand -hex 32)" ``` -4. Deploy both stacks: +4. Apply from the HCP workspace (Manual apply until the stack is sealed). Do not use local `terraform apply` against prod. +5. Post-apply smoke: ```bash - cd cdk - pip install -r requirements.txt - cdk deploy --all + AWS_PROFILE=seahaven-prod ./scripts/post-deploy-smoke.sh ``` -5. **Post-deploy cleanup (one-time):** the retired `RETAIN`-policy secrets `po-ingest/anthropic-api-key` and `workorder-ingest/anthropic-api-key` are orphaned by this deploy, not deleted. Remove them and revoke the keys at the provider: - ```bash - aws secretsmanager delete-secret --secret-id po-ingest/anthropic-api-key --force-delete-without-recovery - aws secretsmanager delete-secret --secret-id workorder-ingest/anthropic-api-key --force-delete-without-recovery - ``` -6. Dashboards: `po-web-ui` and `workorder-web-ui` have no public endpoint (the Function URLs were removed 2026-06-08, INFRA-74). A bare `aws lambda invoke --function-name po-web-ui /tmp/out.json` with no headers in the event is guaranteed a `401` — the handler fails closed (see [Web UI auth](#security)). Fetch the token and forward it in the event's `headers`: +6. Dashboards: `po-web-ui` and `workorder-web-ui` have no public endpoint (Function URLs removed 2026-06-08, INFRA-74). Fetch the token and forward it in the event's `headers`: ```bash TOKEN=$(aws secretsmanager get-secret-value --secret-id procurement-ingest/web-ui-auth-token --query SecretString --output text) aws lambda invoke --function-name po-web-ui --payload "{\"headers\":{\"x-auth-token\":\"$TOKEN\"}}" /tmp/out.json ``` +> **Parked:** `githubdeploy-procurement-ingest` (see `infra/deploy-role/`) is unused after the HCP hard-cut. Left in place for a separate IAM-reviewed cleanup. Do not recreate a mgmt twin (deleted in PLAT-67). + ## Tests Offline unit tests (no AWS, no network) run via pytest from the repo root: diff --git a/docs/plat-86/cfn-dispose.md b/docs/plat-86/cfn-dispose.md new file mode 100644 index 0000000..a40a5c7 --- /dev/null +++ b/docs/plat-86/cfn-dispose.md @@ -0,0 +1,25 @@ +# PLAT-86 CFN dispose runbook (import-in-place) + +Prerequisites (already proven 2026-08-07): + +- HCP workspace `procurement-ingest-prod` Manual apply green; verification plan **0/0/0** +- Lambdas on `/tf-managed/` roles +- TF owns `aws_s3_bucket_notification` on both email buckets (`*-inbound` ids) +- Soft-freeze replaced by hard-cut (`.github/workflows/deploy.yaml` removed) +- Smoke green for `po-email-processor`, `workorder-email-processor`, `procurement-api` + +## Rule + +Never run `cfn-stack-decommission.sh --execute` against these stacks. That script purges RETAIN orphans after delete. Here RETAIN orphans are the live TF-owned data plane. + +## Method + +1. **Retain-all update** — for each stack (`po-ingest`, `WorkorderIngestStack`, `procurement-api`), set `DeletionPolicy: Retain` and `UpdateReplacePolicy: Retain` on every resource in the live template, then `update-stack`. This includes `Custom::S3BucketNotifications` so CFN will not invoke the empty `PutBucketNotificationConfiguration` delete handler. +2. **Delete stack** — `delete-stack` after `UPDATE_COMPLETE`. All resources leave CFN without destruction. +3. **Verify immediately** — `GetBucketNotificationConfiguration` still lists inbound Lambda triggers; SES receipt rules for PO/WO still present on `INBOUND_MAIL`; named Lambdas/tables/buckets still exist; smoke script green. +4. **Sweep CDK helpers only** — delete orphaned `*BucketNotificationsHandler*` Lambda functions and their IAM roles/policies (both stacks). Do not delete TF-owned Lambdas, tables, buckets, API GW, SES rules, or SHOC secret/KMS. +5. **Park** `githubdeploy-procurement-ingest` for a separate IAM-reviewed cleanup (do not block dispose). + +## Script + +`scripts/plat86-cfn-dispose.sh` implements steps 1–4 with explicit confirms and post-checks. diff --git a/docs/plat-86/import-map.md b/docs/plat-86/import-map.md index f5d002b..a2cc8d6 100644 --- a/docs/plat-86/import-map.md +++ b/docs/plat-86/import-map.md @@ -17,14 +17,16 @@ Workspace: one `procurement-ingest-prod` covering all three former stacks. ## CFN disposal disposition +Executable runbook: [`cfn-dispose.md`](cfn-dispose.md) + `scripts/plat86-cfn-dispose.sh` (retain-all, then delete-stack; never `cfn-stack-decommission.sh --execute`). + | Class | Disposition on CFN stack delete | |---|---| | DynamoDB tables, email S3 buckets, Lambda log groups | **RETAIN** (must already be TF-owned; never delete) | | Named SQS (`workorder-shoc-emitter-*`), SHOC secret/KMS/**ResourcePolicy**, API GW, Lambdas, alarms, SES receipt rules | TF-owned (import `aws_secretsmanager_secret_policy.shoc_webhook_hmac` before disposal); remove from CFN via retain-on-delete or deletion_policy before stack delete | | CDK `Custom::S3BucketNotifications` | **Do not destroy via the CFN delete-handler.** That handler calls empty `PutBucketNotificationConfiguration` and wipes TF-owned `aws_s3_bucket_notification` on the PO/WO email buckets. After Terraform apply owns notifications: orphan/retain the custom resource (or otherwise skip its delete cleanup), then destroy `BucketNotificationsHandler` Lambda/role with CFN. Immediately verify `GetBucketNotificationConfiguration` still lists the inbound Lambda triggers; if cleared, re-apply Terraform before accepting traffic. | -| `BucketNotificationsHandler` Lambda/role (+ handler IAM policy) | Destroy with CFN only after notifications are TF-owned and the custom resource is orphaned/removed without clearing the bucket config | -| CDK Metadata | Destroy with CFN | -| CDK-generated IAM roles at path `/` | After Lambda repoint to `/tf-managed/`, delete with CFN or sweep | +| `BucketNotificationsHandler` Lambda/role (+ handler IAM policy) | After retain-all stack delete, sweep orphaned handler Lambdas/roles manually (script step 4) | +| CDK Metadata | Orphaned with retain-all; harmless | +| CDK-generated IAM roles at path `/` | After Lambda repoint to `/tf-managed/`, sweep unused leftovers in a follow-up IAM pass | ## Key physical IDs to preserve diff --git a/infra/deploy-role/README.md b/infra/deploy-role/README.md index 0b9ceeb..e51f744 100644 --- a/infra/deploy-role/README.md +++ b/infra/deploy-role/README.md @@ -1,35 +1,24 @@ -# Deploy role: githubdeploy-procurement-ingest (seahaven-prod) +# Deploy role: githubdeploy-procurement-ingest (seahaven-prod) — PARKED -OIDC deploy role for this repo's GitHub Actions pipeline in AWS account -`011934824531` (seahaven-prod), us-east-1. Created during the migration from -the management account (328440206208). The former mgmt twin of this role was -deleted in PLAT-67 (2026-08-05); do not recreate it. +> **PLAT-86 (2026-08-07):** CDK CD is retired. HCP Terraform workspace +> `procurement-ingest-prod` is the sole mutate path. This OIDC role is an +> **unused orphan** retained for a separate IAM-reviewed cleanup (same pattern +> as afi-backup-monitor / front-integrations). Do not use it for deploys. Do +> not recreate a mgmt twin (deleted in PLAT-67). + +OIDC deploy role historically used by this repo's GitHub Actions CDK pipeline in +AWS account `011934824531` (seahaven-prod), us-east-1. ## Files | File | Purpose | |---|---| | `trust-policy.json` | OIDC trust: `repo:Sea-Haven-Industries/procurement-ingest:ref:refs/heads/main` only | -| `permissions-policy.json` | `sts:AssumeRole` on the four `cdk-hnb659fds-*` bootstrap roles, `cloudformation:DescribeStacks` scoped to this repo's stacks + `CDKToolkit` (cd-cdk health check), and `lambda:InvokeFunction` on exactly the three smoke-gated function ARNs (post-deploy smoke gate) | -| `create-deploy-role.sh` | Idempotent create-or-update from the two JSON files, profile `seahaven-prod` | +| `permissions-policy.json` | Historical CDK bootstrap AssumeRole + smoke InvokeFunction grants | +| `create-deploy-role.sh` | Idempotent create-or-update (do not run unless deliberately restoring) | -> **Maintenance note:** `DescribeStacks` is scoped to `stack/po-ingest/*`, `stack/WorkorderIngestStack/*`, `stack/procurement-api/*` (added with the procurement-api stack), and `stack/CDKToolkit/*`. If another stack is ever added to this CDK app, add its ARN pattern here and re-run the review-then-apply flow — otherwise the cd-cdk health check on the new stack will `AccessDenied`. +## Retirement follow-up -## Why the SmokeInvokeLambda statement exists - -`deploy.yaml` runs `scripts/post-deploy-smoke.sh` under the deploy role's own -session, not the assumed `cdk-*` roles. Without `lambda:InvokeFunction` on the -smoke-gated function ARNs (the two email processors + `procurement-api`) the -smoke gate hits AccessDenied and every deploy fails closed. The mgmt-era grant -was applied out-of-band and undocumented; keeping it in these reviewed -artifacts closes that gap. Scope it to exactly the named ARNs, never -`Resource: "*"`. - -## Change process - -1. Edit the JSON artifacts on a branch; both gates must pass on the exact - files before anything is applied: GPT-4.1 cross-family review - (cross_review.py) and /sh-security-review. -2. Run `./create-deploy-role.sh` (idempotent) with the seahaven-prod profile. -3. Verify: `aws iam simulate-principal-policy` for the bootstrap-role - AssumeRole and both InvokeFunction ARNs, then a real pipeline run. +1. Confirm no workflow references `AWS_DEPLOY_ROLE_ARN` / `cd-cdk.yaml`. +2. IAM + security review, then delete the role and drop related substrate entries. +3. Remove this directory in the same cleanup PR. diff --git a/scripts/plat86-cfn-dispose.sh b/scripts/plat86-cfn-dispose.sh new file mode 100755 index 0000000..6b1bde8 --- /dev/null +++ b/scripts/plat86-cfn-dispose.sh @@ -0,0 +1,183 @@ +#!/usr/bin/env bash +# PLAT-86: dispose former CDK CloudFormation stacks after HCP Terraform import. +# +# Import-in-place only. Sets DeletionPolicy=Retain on every resource so stack +# delete orphans CFN ownership without destroying live TF-managed resources. +# Custom::S3BucketNotifications must be retained — its Delete handler would +# empty PutBucketNotificationConfiguration and wipe TF-owned inbound triggers. +# +# Usage: +# AWS_PROFILE=seahaven-prod ./scripts/plat86-cfn-dispose.sh --dry-run +# AWS_PROFILE=seahaven-prod ./scripts/plat86-cfn-dispose.sh --execute +# +# Never pairs with seahaven-org-baseline cfn-stack-decommission.sh --execute +# (that script deletes RETAIN orphans after stack delete). + +set -euo pipefail + +REGION="${AWS_REGION:-us-east-1}" +PROFILE="${AWS_PROFILE:-seahaven-prod}" +STACKS=(po-ingest WorkorderIngestStack procurement-api) +PO_BUCKET="po-ingest-emails-011934824531" +WO_BUCKET="workorder-ingest-emails-011934824531" +MODE="" + +aws_cmd() { + aws --profile "${PROFILE}" --region "${REGION}" "$@" +} + +usage() { + echo "Usage: $0 --dry-run | --execute" >&2 + exit 2 +} + +[[ $# -eq 1 ]] || usage +case "$1" in + --dry-run) MODE=dry-run ;; + --execute) MODE=execute ;; + *) usage ;; +esac + +work_dir="$(mktemp -d)" +trap 'rm -rf "${work_dir}"' EXIT + +echo "==> mode=${MODE} profile=${PROFILE} region=${REGION}" + +verify_notifications() { + local bucket="$1" expect_id="$2" expect_fn="$3" + local id fn + id="$(aws_cmd s3api get-bucket-notification-configuration --bucket "${bucket}" \ + --query 'LambdaFunctionConfigurations[0].Id' --output text)" + fn="$(aws_cmd s3api get-bucket-notification-configuration --bucket "${bucket}" \ + --query 'LambdaFunctionConfigurations[0].LambdaFunctionArn' --output text)" + if [[ "${id}" != "${expect_id}" ]] || [[ "${fn}" != *":function:${expect_fn}" ]]; then + echo "FAIL: ${bucket} notification mismatch id=${id} fn=${fn}" >&2 + return 1 + fi + echo "OK: ${bucket} -> ${id} (${expect_fn})" +} + +verify_core() { + local fn + for fn in po-email-processor workorder-email-processor procurement-api \ + po-ingest-site-extractor workorder-shoc-emitter; do + aws_cmd lambda get-function --function-name "${fn}" --query 'Configuration.FunctionName' --output text >/dev/null + echo "OK: lambda ${fn}" + done + for table in purchase-orders verified-sites pending-site-review WorkOrders WorkOrderComments; do + aws_cmd dynamodb describe-table --table-name "${table}" --query 'Table.TableName' --output text >/dev/null + echo "OK: table ${table}" + done + aws_cmd s3api head-bucket --bucket "${PO_BUCKET}" >/dev/null + aws_cmd s3api head-bucket --bucket "${WO_BUCKET}" >/dev/null + echo "OK: email buckets" + verify_notifications "${PO_BUCKET}" "po-email-processor-inbound" "po-email-processor" + verify_notifications "${WO_BUCKET}" "wo-email-processor-inbound" "workorder-email-processor" +} + +echo "==> pre-checks" +verify_core + +retain_template() { + local stack="$1" + local raw="${work_dir}/${stack}.raw.json" + local out="${work_dir}/${stack}.retain.json" + aws_cmd cloudformation get-template --stack-name "${stack}" --template-stage Original \ + --query TemplateBody --output json >"${raw}" + python3 - "${raw}" "${out}" <<'PY' +import json, sys +raw_path, out_path = sys.argv[1], sys.argv[2] +body = json.load(open(raw_path)) +# get-template may return already-parsed dict or a JSON string +if isinstance(body, str): + body = json.loads(body) +resources = body.get("Resources") or {} +changed = 0 +for name, res in resources.items(): + if not isinstance(res, dict): + continue + before = (res.get("DeletionPolicy"), res.get("UpdateReplacePolicy")) + res["DeletionPolicy"] = "Retain" + res["UpdateReplacePolicy"] = "Retain" + if before != ("Retain", "Retain"): + changed += 1 +print(f"{len(resources)} resources; {changed} policy fields updated") +json.dump(body, open(out_path, "w")) +PY +} + +wait_stack() { + local stack="$1" want="$2" + aws_cmd cloudformation wait "stack-${want}" --stack-name "${stack}" + local status + status="$(aws_cmd cloudformation describe-stacks --stack-name "${stack}" \ + --query 'Stacks[0].StackStatus' --output text 2>/dev/null || echo DELETE_COMPLETE)" + echo "stack ${stack} -> ${status}" + case "${status}" in + *COMPLETE) ;; + *) echo "FAIL: unexpected status ${status}" >&2; exit 1 ;; + esac +} + +for stack in "${STACKS[@]}"; do + echo "==> retain-all template for ${stack}" + retain_template "${stack}" + if [[ "${MODE}" == "dry-run" ]]; then + echo "dry-run: would update-stack ${stack} then delete-stack" + continue + fi + echo "==> update-stack ${stack} (retain-all)" + aws_cmd cloudformation update-stack \ + --stack-name "${stack}" \ + --template-body "file://${work_dir}/${stack}.retain.json" \ + --capabilities CAPABILITY_NAMED_IAM \ + >/dev/null + wait_stack "${stack}" "update-complete" + echo "==> delete-stack ${stack}" + aws_cmd cloudformation delete-stack --stack-name "${stack}" + wait_stack "${stack}" "delete-complete" + echo "==> post-delete verify after ${stack}" + verify_core +done + +if [[ "${MODE}" == "dry-run" ]]; then + echo "dry-run complete; no stacks modified" + exit 0 +fi + +echo "==> sweep BucketNotificationsHandler Lambdas (CDK helpers only)" +mapfile -t handlers < <(aws_cmd lambda list-functions \ + --query "Functions[?contains(FunctionName, 'BucketNotificationsHandler')].FunctionName" \ + --output text | tr '\t' '\n' | grep -E 'po-ingest-|WorkorderIngestStack-' || true) +for h in "${handlers[@]:-}"; do + [[ -n "${h}" ]] || continue + echo "deleting helper lambda ${h}" + aws_cmd lambda delete-function --function-name "${h}" +done + +echo "==> sweep leftover BucketNotificationsHandler IAM roles" +mapfile -t roles < <(aws_cmd iam list-roles \ + --query "Roles[?contains(RoleName, 'BucketNotificationsHandler')].RoleName" \ + --output text | tr '\t' '\n' | grep -E 'po-ingest-|WorkorderIngestStack-' || true) +for role in "${roles[@]:-}"; do + [[ -n "${role}" ]] || continue + echo "deleting helper role ${role}" + # Detach inline + managed then delete + mapfile -t inlines < <(aws_cmd iam list-role-policies --role-name "${role}" --query 'PolicyNames[]' --output text | tr '\t' '\n') + for p in "${inlines[@]:-}"; do + [[ -n "${p}" ]] || continue + aws_cmd iam delete-role-policy --role-name "${role}" --policy-name "${p}" + done + mapfile -t attached < <(aws_cmd iam list-attached-role-policies --role-name "${role}" --query 'AttachedPolicies[].PolicyArn' --output text | tr '\t' '\n') + for a in "${attached[@]:-}"; do + [[ -n "${a}" ]] || continue + aws_cmd iam detach-role-policy --role-name "${role}" --policy-arn "${a}" + done + aws_cmd iam delete-role --role-name "${role}" +done + +echo "==> final verify + smoke" +verify_core +ROOT="$(cd "$(dirname "$0")/.." && pwd)" +AWS_PROFILE="${PROFILE}" AWS_REGION="${REGION}" bash "${ROOT}/scripts/post-deploy-smoke.sh" +echo "PLAT-86 CFN dispose complete"