feat(infra): migrate pipeline and Grafana to HCP Terraform (PLAT-75) (#57)
Some checks are pending
Deploy / Deploy to prod (push) Waiting to run

* feat(infra): migrate pipeline and Grafana to HCP Terraform (PLAT-75)

Move apm-wo-analysis into seahaven-prod under workspace apm-wo-analysis-prod
with in-repo hcptf/githubdeploy IAM, stub Lambdas, and GitHub Actions zip CD.

* chore(iam): add Checkov skip comments for HCP IAM documents

Pre-push HIGH findings are the DLM snapshot describe, tagged EC2 creates,
exec boundary DescribeLogGroups star, and the drop-uploader user policy.
This commit is contained in:
Adam Moussa 2026-09-16 20:33:24 +00:00 • committed by GitHub
parent 022d5e27cc
commit f13d2e3ff1
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
39 changed files with 3693 additions and 533 deletions

View file

@ -1,4 +1,5 @@
name: CI
on:
pull_request:
branches: [main]
@ -8,13 +9,83 @@ permissions:
contents: read
jobs:
pytest:
name: Pytest
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
persist-credentials: false
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
with:
python-version: "3.12"
- name: Install test dependencies
run: |
set -euo pipefail
python -m pip install --upgrade pip
pip install -r tests/requirements.txt
pip install -r lambdas/classifier/requirements.txt
pip install -r lambdas/slack_post/requirements.txt
- name: Pytest
run: pytest
terraform:
name: Terraform
runs-on: ubuntu-latest
timeout-minutes: 15
defaults:
run:
working-directory: terraform
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
persist-credentials: false
- uses: hashicorp/setup-terraform@dfe3c3f87815947d99a8997f908cb6525fc44e9e # v4.0.1
with:
terraform_version: "1.16.0"
terraform_wrapper: false
- name: Terraform fmt
run: terraform fmt -check -recursive
- name: Terraform init
run: terraform init -backend=false
- name: Terraform validate
run: terraform validate
ci:
uses: Sea-Haven-Industries/.github/.github/workflows/ci-python-sam.yaml@e5691d8a7f96ac4d5a841a82975ff0a4354d53ac # v1.0.7
with:
python-version: "3.12"
source-dirs: "cdk lambdas tests"
run-sam-validate: false
run-cdk-synth: true
cdk-dir: cdk
run-tests: true
enable-qemu: true
name: ci / ci
needs: [pytest, terraform]
if: ${{ always() && !cancelled() }}
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- name: Check jobs
env:
PYTEST_RESULT: ${{ needs.pytest.result }}
TERRAFORM_RESULT: ${{ needs.terraform.result }}
run: |
set -euo pipefail
fail=0
check() {
local name="$1"
local result="$2"
case "${result}" in
success)
echo "${name}: ${result}"
;;
*)
echo "${name}: ${result}" >&2
fail=1
;;
esac
}
check pytest "${PYTEST_RESULT}"
check terraform "${TERRAFORM_RESULT}"
exit "${fail}"

View file

@ -1,23 +1,135 @@
name: Deploy
# Terraform owns Lambda skeletons and Grafana infra. This workflow ships zips
# to prod, calls update-function-code, and syncs grafana-config/. It never
# creates an HCP run. No GitHub Releases and no tagging in this workflow.
on:
push:
branches: [main]
paths-ignore:
- "terraform/**"
- "docs/**"
- "README.md"
- "AGENTS.md"
- "CLAUDE.md"
workflow_dispatch:
inputs:
ref:
description: "Git ref to build and deploy (tag, branch, or SHA). Empty means the workflow ref."
required: false
type: string
default: ""
permissions:
id-token: write
contents: read
concurrency:
group: deploy
cancel-in-progress: false
jobs:
deploy:
uses: Sea-Haven-Industries/.github/.github/workflows/cd-cdk.yaml@e5691d8a7f96ac4d5a841a82975ff0a4354d53ac # v1.0.7
with:
python-version: "3.12"
region: us-east-1
cdk-dir: cdk
enable-qemu: true
secrets:
deploy-role-arn: ${{ secrets.AWS_DEPLOY_ROLE_ARN }}
name: Deploy to prod
runs-on: ubuntu-latest
timeout-minutes: 30
environment: prod
concurrency:
group: deploy-apm-wo-analysis-prod
cancel-in-progress: false
permissions:
contents: read
id-token: write
env:
AWS_REGION: us-east-1
DEPLOY_ROLE_ARN: ${{ vars.DEPLOY_ROLE_ARN }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
ref: ${{ github.event_name == 'workflow_dispatch' && inputs.ref || github.sha }}
persist-credentials: false
- name: Resolve commit
id: commit
run: |
set -euo pipefail
sha="$(git rev-parse HEAD)"
echo "sha=${sha}" >> "$GITHUB_OUTPUT"
echo "Building ${sha}"
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
with:
python-version: "3.12"
- name: Build function zips
env:
GIT_SHA: ${{ steps.commit.outputs.sha }}
run: |
set -euo pipefail
python scripts/package_lambdas.py --git-sha "${GIT_SHA}" --out-dir build/packages
python - <<'PY'
import os, zipfile
from pathlib import Path
sha = os.environ["GIT_SHA"]
names = ["classifier", "slack_post", "slack_interactions"]
for name in names:
path = Path("build/packages") / f"{name}.zip"
if not path.is_file():
raise SystemExit(f"missing {path}")
with zipfile.ZipFile(path) as zf:
info = zf.read("build_info.py").decode()
if sha not in info:
raise SystemExit(f"{path} missing GIT_SHA {sha}")
print("zips ok")
PY
- name: Configure AWS credentials using OIDC
uses: aws-actions/configure-aws-credentials@cbe3b392738ccf3f987d68400dafcf4b0624a56c # v6.2.4
with:
role-to-assume: ${{ env.DEPLOY_ROLE_ARN }}
aws-region: us-east-1
audience: sts.amazonaws.com
- name: Get deploy parameters
id: deploy
run: |
set -euo pipefail
prefix=/apm-wo-analysis/deploy
{
echo "artifacts_bucket=$(aws ssm get-parameter --name "${prefix}/artifacts-bucket" --query Parameter.Value --output text)"
echo "exports_bucket=$(aws ssm get-parameter --name "${prefix}/exports-bucket" --query Parameter.Value --output text)"
echo "classifier=$(aws ssm get-parameter --name "${prefix}/classifier-function-name" --query Parameter.Value --output text)"
echo "slack_post=$(aws ssm get-parameter --name "${prefix}/slack_post-function-name" --query Parameter.Value --output text)"
echo "slack_interactions=$(aws ssm get-parameter --name "${prefix}/slack_interactions-function-name" --query Parameter.Value --output text)"
} >> "${GITHUB_OUTPUT}"
- name: Upload zips and update function code
env:
ARTIFACTS_BUCKET: ${{ steps.deploy.outputs.artifacts_bucket }}
GIT_SHA: ${{ steps.commit.outputs.sha }}
CLASSIFIER: ${{ steps.deploy.outputs.classifier }}
SLACK_POST: ${{ steps.deploy.outputs.slack_post }}
SLACK_INTERACTIONS: ${{ steps.deploy.outputs.slack_interactions }}
run: |
set -euo pipefail
keys=(
classifier:"${CLASSIFIER}"
slack_post:"${SLACK_POST}"
slack_interactions:"${SLACK_INTERACTIONS}"
)
for pair in "${keys[@]}"; do
name="${pair%%:*}"
fn="${pair#*:}"
key="functions/${name}/${GIT_SHA}.zip"
aws s3 cp "build/packages/${name}.zip" "s3://${ARTIFACTS_BUCKET}/${key}"
aws lambda update-function-code \
--function-name "${fn}" \
--s3-bucket "${ARTIFACTS_BUCKET}" \
--s3-key "${key}" \
--query '{Function:FunctionName,Sha256:CodeSha256,Updated:LastModified}' \
--output table
aws lambda wait function-updated-v2 --function-name "${fn}"
done
- name: Sync Grafana dashboards-as-code
env:
EXPORTS_BUCKET: ${{ steps.deploy.outputs.exports_bucket }}
run: |
set -euo pipefail
aws s3 sync grafana/ "s3://${EXPORTS_BUCKET}/grafana-config/" --delete --exact-timestamps

8
.gitignore vendored
View file

@ -5,10 +5,16 @@ __pycache__/
.venv/
venv/
# CDK
# CDK (frozen mgmt source until decommission)
cdk.out/
cdk.context.json
# Terraform
terraform/.terraform/
terraform/build/
.terraform.lock.hcl.bak
build/
# Local env / secrets
.env
.env.*

View file

@ -25,8 +25,8 @@ APM export (xlsx/csv)
→ standalone batched 3rd-escalation alert (suppressed if zero)
```
- **Account / region:** 328440206208 / us-east-1
- **IaC:** CDK (Python). `aws-cdk-lib` pinned **exact** (`==`) in `cdk/requirements.txt`, kept current by Dependabot per the handbook Pinning Principle (no blanket ignores) — don't hardcode a version number here, read the requirements file. Lambdas Python 3.12, ARM64.
- **Account / region:** seahaven-prod 011934824531 / us-east-1
- **IaC:** HCP Terraform (Python Lambdas 3.12, ARM64). `cdk/` is frozen leftover from the mgmt stack; do not deploy it.
- **No DynamoDB** — deliberate. This is an analytics workload and Grafana cannot query DynamoDB; S3 + Athena is the store. Do not "helpfully" add a table.
---
@ -90,7 +90,7 @@ The export reaches S3 by **direct upload or a local drop-folder**, never SES/ema
## Repo-specific rules
- Global Sea Haven rules and the engineering handbook apply (naming, secrets placement, cross-review gates, README/Confluence updates, ruff before push).
- Buckets: `apm-wo-analysis-*-328440206208`. Anthropic API key secret: `apm-wo-analysis/anthropic-api-key`.
- Buckets: `apm-wo-analysis-*-011934824531`. Anthropic API key secret: `apm-wo-analysis/anthropic-api-key`.
- Smoke-test the classifier against a **real export** before declaring any classification change done (see the pre-action-smoke-test preference).
- Confluence page for architecture changes: "AWS Architecture Map" (page 1540098). Project memory: `apm-wo-comment-analysis`.

116
README.md
View file

@ -1,7 +1,7 @@
# apm-wo-analysis
![Python](https://img.shields.io/badge/Python-3776AB?logo=python&logoColor=white)
![AWS CDK](https://img.shields.io/badge/AWS-CDK-FF9900?logo=amazonaws&logoColor=white)
![Terraform](https://img.shields.io/badge/Terraform-HCP-844FBA?logo=terraform&logoColor=white)
![Slack](https://img.shields.io/badge/Slack-integration-4A154B?logo=slack&logoColor=white)
![CI](https://github.com/Sea-Haven-Industries/apm-wo-analysis/actions/workflows/ci.yaml/badge.svg)
@ -17,8 +17,8 @@ This is a **sibling concern** to the `apm@` email pipeline in `procurement-inges
tables. There is deliberately **no DynamoDB**: this is an analytics workload
backed by S3 + Athena (Grafana cannot query DynamoDB).
- **Account / region:** 328440206208 / us-east-1
- **IaC:** CDK (Python), `aws-cdk-lib==2.253.1`. Lambdas Python 3.12, ARM64.
- **Account / region:** seahaven-prod `011934824531` / us-east-1 (PLAT-75). Mgmt CDK stacks `apm-wo-analysis-pipeline` and `apm-wo-analysis-grafana` deleted 2026-09-16.
- **IaC:** HCP Terraform workspace `apm-wo-analysis-prod` (working directory `terraform/`, file trigger `terraform/**`). Lambdas Python 3.12, ARM64. GitHub Actions owns zip + Grafana-config content.
## Architecture
@ -35,26 +35,27 @@ APM export (xlsx/csv)
→ drill-down modals via apm-wo.seahaven.com (signature-verified)
```
Two CDK stacks (`cdk/app.py` instantiates both):
One HCP workspace (`apm-wo-analysis-prod`) owns both the pipeline and Grafana:
| Stack | Resources |
| Area | Resources |
|---|---|
| `apm-wo-analysis-pipeline` | S3 exports bucket, drop-uploader IAM user, classifier + slack-post + slack-interactions Lambdas, classifier DLQ, Glue DB + projection table, Athena workgroup, HTTP API (`apm-wo.seahaven.com`), SSM param, scoped IAM |
| `apm-wo-analysis-grafana` | EC2 (Grafana OSS), internet-facing **office-IP-restricted** ALB (`grafana.seahaven.com`), security groups, instance IAM role, Route53 alias, daily DLM snapshot, dashboards-as-code S3 deployment |
| Pipeline | S3 exports + artifacts buckets, drop-uploader IAM user, classifier + slack-post + slack-interactions Lambdas (stub + `ignore_changes`), classifier DLQ, Glue DB + projection table, Athena workgroup, HTTP API (`apm-wo.seahaven.com`), SSM deploy contract, in-repo `hcptf-*` / `githubdeploy-*` IAM |
| Grafana | Dedicated VPC (private instance, public ALB, one NAT), EC2 (Grafana OSS), office-IP-restricted ALB (`grafana.seahaven.com`), instance IAM role, daily DLM snapshot. Dashboards sync from S3 via GitHub Actions. Route53 aliases stay in the mgmt hosted zone and are flipped out of band. |
## AWS Resources
| Resource | Name | Purpose |
|---|---|---|
| S3 bucket | `apm-wo-analysis-exports-328440206208` | Single bucket. Prefixes: `raw/` (incoming, 90-day expiry), `analytics/` (Parquet snapshots, kept), `meta/` (summary/details JSON), `athena-results/` (query output, 30-day expiry), `grafana-config/` (dashboards-as-code). SSE-S3, BPA-all, enforce-SSL, `RETAIN`. |
| S3 bucket | `apm-wo-analysis-exports-011934824531` | Single data bucket. Prefixes: `raw/` (incoming, 90-day expiry), `analytics/` (Parquet snapshots, kept), `meta/` (summary/details JSON), `athena-results/` (query output, 30-day expiry), `grafana-config/` (dashboards-as-code). SSE-S3, BPA-all, enforce-SSL. |
| S3 bucket | `apm-wo-analysis-artifacts-011934824531` | Lambda zip artifacts. GitHub Actions uploads `functions/<name>/<sha>.zip`. |
| IAM user | `apm-wo-drop-uploader` | Drop-folder identity; `s3:PutObject` on `raw/*` only. Access key created out-of-band, stored in local `apm-wo-drop` profile. |
| Glue database | `apm_wo_analysis` | Analytics catalog. |
| Glue table | `apm_wo_snapshots` | External Parquet table over `analytics/`, **partition projection** on `dt` (date, `2026-01-01..NOW`) — no crawler, no `MSCK`. 17-column snapshot schema. |
| Athena workgroup | `apm-wo-analysis` | Enforced result location `athena-results/`, SSE-S3. |
| SQS queue | `apm-wo-analysis-classifier-dlq` | Dead-letter for failed classifier async invocations (14-day retention). |
| SSM parameter | `/apm-wo-analysis/grafana-base-url` | Grafana dashboard URL for the 📊 button / modal links (ops-editable, no redeploy). |
| HTTP API + domain | `apm-wo.seahaven.com` → `POST /slack/interactions` | Slack interactivity endpoint. Stage throttled 10 rps / 20 burst. `*.seahaven.com` ACM cert; Route53 alias. |
| EC2 instance | Grafana (`t4g.small`, AL2023, ARM64) | Self-hosted Grafana OSS in `seahaven-vpc` private subnets, IMDSv2-only, SSM-managed. gp3 20 GB **encrypted**, `DeleteOnTermination=false`, tagged `apm-grafana-backup`. |
| HTTP API + domain | `apm-wo.seahaven.com` → `POST /slack/interactions` | Slack interactivity endpoint. Stage throttled 10 rps / 20 burst. ACM cert in seahaven-prod; Route53 alias in mgmt zone (OOB). |
| EC2 instance | Grafana (`t4g.small`, AL2023, ARM64) | Self-hosted Grafana OSS in a dedicated VPC private subnet, IMDSv2-only, SSM-managed. gp3 20 GB **encrypted**, `DeleteOnTermination=false`, tagged `apm-grafana-backup`. |
| ALB | Grafana ALB (`grafana.seahaven.com`) | Internet-facing, HTTPS 443, SG admits **only office CIDRs** (`47.21.61.4/32`, `96.250.164.146/32`); forwards to instance:3000, health `/api/health`. |
| DLM policy | Grafana volume backup | Daily snapshot (07:00 UTC) of the tagged instance, 7 retained. |
| Route53 | `apm-wo.seahaven.com`, `grafana.seahaven.com` | Aliases in zone `Z06652411XKH89KTZD3XA` (`seahaven.com`). |
@ -71,7 +72,7 @@ All Python 3.12, ARM64, explicit LogGroup (`/aws/lambda/<name>`, 60-day retentio
## Configuration
### Secrets Manager (names only — created out-of-band, never in CloudFormation)
### Secrets Manager (names only — Terraform creates empty shells; values copied out of band)
| Secret | Purpose |
|---|---|
| `apm-wo-analysis/anthropic-api-key` | Claude Haiku fallback for ambiguous free-text comments. |
@ -86,11 +87,11 @@ All Python 3.12, ARM64, explicit LogGroup (`/aws/lambda/<name>`, 60-day retentio
- **classifier:** `APM_HAIKU_FALLBACK` (`on`/`off`), `SLACK_POST_FUNCTION_NAME`.
- **slack-post / slack-interactions:** `SLACK_SECRET_NAME`, `DASHBOARD_URL_PARAM`, `ANALYTICS_BUCKET`.
### GitHub repo secret
- `AWS_DEPLOY_ROLE_ARN` — the OIDC deploy role `githubdeploy-apm-wo-analysis`.
### GitHub Environment `prod`
- `DEPLOY_ROLE_ARN` — the OIDC deploy role `githubdeploy-apm-wo-analysis` (`/tf-managed/`).
### CDK context (`cdk/cdk.json`)
`wildcardCertArn`, `hostedZoneId`/`hostedZoneName`, `slackInteractionsDomain`, `grafanaDomain`, `grafanaVpcId`/`grafanaAzs`/`grafana{Public,Private}SubnetIds`, `officeCidrs`, `athenaPluginVersion` (`3.2.0`).
### HCP workspace
- `apm-wo-analysis-prod` in project `seahaven-prod`. VCS `main`, working directory `terraform`, file trigger `terraform/**`. `TFC_AWS_APPLY_ROLE_ARN` / `TFC_AWS_PLAN_ROLE_ARN` are workspace vars pointing at `hcptf-apm-wo-analysis` / `hcptf-apm-wo-analysis-plan` after the bootstrap window.
## The classification model
@ -119,15 +120,10 @@ Authoritative spec: [`CLAUDE.md`](./CLAUDE.md). Implementation:
## Repository layout
```
cdk/
app.py CDK entry point — instantiates both stacks
cdk.json context: cert, zone, subnets, office CIDRs, plugin version
requirements.txt aws-cdk-lib==2.253.1, constructs>=10.6.0
assets/
grafana_userdata.sh EC2 bootstrap: install Grafana + Athena plugin, S3 config sync
stacks/
pipeline_stack.py S3, Lambdas, DLQ, Glue, Athena, HTTP API, IAM
grafana_stack.py VPC import, EC2, ALB, SG, Route53, instance role, DLM
terraform/ HCP Terraform: IAM, S3, Glue, Athena, Lambdas, API, VPC, Grafana
bootstrap/stub/ committed Lambda stub; GHA replaces code via update-function-code
templates/ Grafana user-data
cdk/ frozen mgmt CDK source until the mgmt stacks are deleted
lambdas/
classifier/ S3-triggered: parse → two-axis classify → Parquet + meta JSON
slack_post/ blockkit.py (builders), handler.py (post), interactions.py
@ -136,16 +132,17 @@ grafana/
provisioning/ Athena datasource + dashboard provider (as code)
dashboards/ apm-work-orders.json (uid apm-wo) — source of truth
slack/manifest.yaml Slack app manifest (interactivity request URL)
scripts/ local drop-folder uploader + launchd plist
tests/ classifier smoke test + offline synth/blockkit assertions
docs/BUILD.md phased, end-to-end build guide
scripts/ drop-folder uploader, launchd plist, package_lambdas.py
tests/ classifier smoke test + Block Kit / handler assertions
docs/BUILD.md original CDK build guide (historical)
docs/RUNBOOK.md operations
```
## Ingestion (no email)
The export reaches S3 by **direct upload or a local drop-folder**, never SES/email.
- **Direct:** `aws s3 cp ./export.xlsx s3://apm-wo-analysis-exports-328440206208/raw/`
- **Direct:** `aws s3 cp ./export.xlsx s3://apm-wo-analysis-exports-011934824531/raw/`
- **Drop-folder (optional zero-touch):** a launchd agent (`scripts/apm-wo-uploader.sh`
+ `scripts/com.seahaven.apm-wo-uploader.plist`) that watches `~/apm-wo-drop/`,
uploads new `.xlsx`/`.csv` files to `raw/`, and archives them locally. Uploads
@ -167,40 +164,35 @@ Any `.xlsx`/`.csv` landing under `raw/` invokes the classifier.
## Local development
- pyenv Python 3.12. `ruff check` + `ruff format --check` before pushing (hook-enforced).
- Tests: `python -m pytest tests/ -q` (~158 tests — classifier rule ladder + handler
transforms + Haiku fallback, Block Kit builders, the signature-verified Slack
interactions endpoint, slack-post orchestration, and offline `cdk.assertions` synth
checks). No AWS needed — external boundaries are monkeypatched. **Tests run in CI**
(`ci.yaml` sets `run-tests: true`); coverage is reported via `pytest-cov`
(`tests/requirements.txt`, ~94%, non-gating).
- Tests: `python -m pytest tests/ -q` (classifier rule ladder + handler
transforms + Haiku fallback, Block Kit builders, Slack interactions, slack-post
orchestration). No AWS needed — external boundaries are monkeypatched. **Tests
run in CI**. Coverage via `pytest-cov` (`tests/requirements.txt`, non-gating).
- The classification quality gate (deterministic "Other" share) runs in CI against a
committed synthetic fixture `tests/fixtures/sample_export.csv`. Additionally,
smoke-test against the **real export** before declaring any classification change
done: `~/Downloads/_documents/Sheet1-1.xlsx` (skips automatically when absent).
- `cdk synth` must pass in CI before merge (**Docker + QEMU** — Lambda deps are
bundled for ARM64; `ci.yaml` sets `enable-qemu: true`).
- `terraform fmt -check -recursive` and `terraform validate` must pass in CI
(`terraform init -backend=false`).
## Deployment
CI/CD via the org reusable workflows (no manual prod deploys in steady state):
Terraform owns containers. GitHub Actions owns Lambda zips and Grafana JSON.
- **CI** (`.github/workflows/ci.yaml`) → `ci-python-sam.yaml@main`: ruff + `pytest` + `cdk synth` (QEMU-enabled). Runs on PRs into `main`.
- **Deploy** (`.github/workflows/deploy.yaml`) → `cd-cdk.yaml@main`: OIDC assume-role, `cdk deploy --all`, single-flight concurrency. Runs on push to `main`.
- **CI** (`.github/workflows/ci.yaml`): pytest + `terraform fmt` + `terraform validate`.
- **Content CD** (`.github/workflows/deploy.yaml`): Environment `prod`, OIDC
`githubdeploy-apm-wo-analysis`, `update-function-code` + `aws s3 sync grafana/`.
`paths-ignore` for `terraform/**`. Never creates an HCP run.
- **Infra:** HCP workspace `apm-wo-analysis-prod`, VCS on `main`, working directory
`terraform`, file trigger `terraform/**`. First apply is Manual via the
`hcptf-bootstrap` window; after seal, `TFC_AWS_*` point at
`hcptf-apm-wo-analysis` / `hcptf-apm-wo-analysis-plan`.
Stack name/region/account: `apm-wo-analysis-{pipeline,grafana}` / us-east-1 / 328440206208.
OIDC deploy role `githubdeploy-apm-wo-analysis` must exist before the first deploy.
Account/region: `011934824531` / us-east-1.
> **Note:** CD is **temporarily disabled** (deploy job gated `if: ${{ false }}` on
> the phase-0 branch) while the Phase 0–5 stack is merged into `main`, to avoid a
> deploy on every merge. **Re-enable as the first Phase 6 step** by reverting that
> commit. Both stacks are already deployed manually and validated in prod.
Manual deploy (emergency/reference; pipeline first so the bucket/table exist):
```bash
cd cdk && pip install -r requirements.txt
cdk deploy apm-wo-analysis-pipeline # S3, Glue, Athena, Lambdas, DLQ, API, IAM
cdk deploy apm-wo-analysis-grafana # EC2, ALB, SG, Route53, DLM, dashboards
```
Lambda code seam: committed stub under `terraform/bootstrap/` plus
`lifecycle.ignore_changes` on filename/s3_key/source_code_hash. A post-deploy
plan must be empty.
## Operations
@ -252,22 +244,24 @@ snapshotted daily by DLM.
by the plugin). Config sync uses `aws s3 sync --exact-timestamps` (plain sync
skips same-size edits). Template vars use `refresh: 1` (on load).
- **No Client VPN exists** — "VPN-only" Grafana is realized as **office-IP SG
restriction**. `seahaven-vpc` has a single NAT (one AZ) for instance egress.
restriction**. Grafana sits in a dedicated VPC with a single NAT (one AZ) for
instance egress.
- **Slack interactions endpoint** is unauthenticated at the gateway **by design**;
the Lambda verifies the Slack signature (replay window + HMAC). Stage-throttled.
- **Route53** for `apm-wo.seahaven.com` and `grafana.seahaven.com` stays in the
mgmt hosted zone `Z06652411XKH89KTZD3XA` and is flipped out of band.
### Known operational debt
- Re-enable CD (revert the phase-0 disable) once the stack is merged.
- One **clean instance replacement** is owed to validate the committed user-data
from a cold boot and to apply root-volume encryption (can't encrypt in place).
- HCP auto-apply stays off until the scoped plan is clean and the bootstrap
trust window is closed.
- Deferred review NITs: `print()`→`logging`, source-IP logging on signature
failure, S3 versioning, the `'${site:raw}'` WO-table SQL tidy, and a
CIDR-maintenance note in the runbook.
## Status
Phases 2–5 implemented, **deployed to prod and validated end-to-end** (classifier,
Slack post + alert + modal, Grafana dashboard). Stacked PRs **#6→#11** are open and
unmerged; cross-review (#2/#4/#5) and `/security-review` of the two public endpoints
are **cleared**. Phase 6 (this docs pass + Confluence + runbook) is in progress on
`feature/phase-6-docs`.
Live in seahaven-prod under HCP workspace `apm-wo-analysis-prod` (PLAT-75, 2026-09-16).
Classification logic is unchanged. Mgmt CDK stacks `apm-wo-analysis-pipeline` and
`apm-wo-analysis-grafana` are deleted; the mgmt exports bucket is RETAIN cold archive.
GitHub Actions `deploy.yaml` on `main` (Environment `prod`) owns Lambda zips and
`grafana-config/` sync. HCP auto-apply stays off until the scoped plan is clean.

View file

@ -1,6 +1,12 @@
# apm-wo-analysis — Step-by-Step Build Guide
End-to-end build instructions. Read `../CLAUDE.md` first for the domain model and locked decisions. Account `328440206208`, region `us-east-1`, all names kebab-case.
Historical CDK build instructions. **PLAT-75 moved this stack to HCP Terraform**
in seahaven-prod (`011934824531`). Authority is `terraform/` plus the README
deployment section. Mgmt CDK stacks were deleted 2026-09-16; do not deploy from
`cdk/`.
End-to-end CDK build instructions below are the original mgmt path. Account
`328440206208`, region `us-east-1`, all names kebab-case.
> Convention gates that apply throughout: OIDC deploy role created **before** any CD; secrets in Secrets Manager; `cross_reviewer` on IAM/handler diffs (orchestrator since archived; use `cross_review.py` in `security-review`); `ruff` clean + `cdk synth` green before push; README + Confluence + memory updated as part of the work, not after.

View file

@ -1,8 +1,8 @@
# apm-wo-analysis — Operational Runbook
Operational procedures and incident response for the daily APM work-order
analysis pipeline. Account **328440206208** / **us-east-1**. Stacks
`apm-wo-analysis-pipeline` and `apm-wo-analysis-grafana`. See [`README.md`](../README.md)
analysis pipeline. Account **011934824531** (seahaven-prod) / **us-east-1**.
HCP workspace `apm-wo-analysis-prod`. See [`README.md`](../README.md)
for architecture and resource detail.
**Admin access:** the Grafana box is **SSM Session Manager only** (no SSH/key pair):
@ -27,9 +27,9 @@ re-processing the export; not permanent data loss.
### Context
| Item | Value |
|---|---|
| Stacks | `apm-wo-analysis-pipeline`, `apm-wo-analysis-grafana` |
| Stacks | HCP workspace `apm-wo-analysis-prod` |
| Lambdas | `apm-wo-analysis-classifier`, `-slack-post`, `-slack-interactions` |
| S3 | `apm-wo-analysis-exports-328440206208` — `raw/`, `analytics/dt=…/`, `meta/dt=…/` |
| S3 | `apm-wo-analysis-exports-011934824531` — `raw/`, `analytics/dt=…/`, `meta/dt=…/` |
| SQS DLQ | `apm-wo-analysis-classifier-dlq` |
| Glue / Athena | db `apm_wo_analysis`, table `apm_wo_snapshots`, workgroup `apm-wo-analysis` |
| External | Slack, Anthropic API (Haiku fallback) |
@ -37,7 +37,7 @@ re-processing the export; not permanent data loss.
| SSM | `/apm-wo-analysis/grafana-base-url` |
### Triage
1. **Export uploaded?** `aws s3 ls s3://apm-wo-analysis-exports-328440206208/raw/` — today's file present? Absent → upstream (§2.1), not the pipeline.
1. **Export uploaded?** `aws s3 ls s3://apm-wo-analysis-exports-011934824531/raw/` — today's file present? Absent → upstream (§2.1), not the pipeline.
2. **Classifier ran/failed?** `/aws/lambda/apm-wo-analysis-classifier` logs; peek the DLQ: `aws sqs receive-message --queue-url <dlq-url> --max-number-of-messages 1`.
3. **Outputs written?** `aws s3 ls .../analytics/dt=<today>/` (Parquet) and `.../meta/dt=<today>/` (`summary.json`, `details.json`).
4. **slack-post ran/failed?** `/aws/lambda/apm-wo-analysis-slack-post` logs — `SlackApiError` (`invalid_auth`, `not_in_channel`, `invalid_blocks`)?
@ -46,7 +46,7 @@ re-processing the export; not permanent data loss.
7. **Recent change?** Any merge/deploy to `main` just before the failure.
### Resolution (by root cause)
1. **Export not uploaded** → `aws s3 cp <export>.xlsx s3://apm-wo-analysis-exports-328440206208/raw/`; then check the drop-folder agent (§2.1).
1. **Export not uploaded** → `aws s3 cp <export>.xlsx s3://apm-wo-analysis-exports-011934824531/raw/`; then check the drop-folder agent (§2.1).
2. **Classifier failed (DLQ)** → read the DLQ message; fix; **reprocess by re-uploading the export to `raw/`** (`overwrite_partitions` makes same-`dt` idempotent).
3. **Outputs present, no Slack post** → re-invoke:
```bash
@ -73,7 +73,7 @@ The curated daily APM filter-view export (~350 WOs, `.xlsx`/`.csv`) reaches S3 b
**direct upload or a local drop-folder** — never SES/email. Any object under
`raw/` with a `.xlsx`/`.csv` suffix triggers the classifier.
- **Direct:** `aws s3 cp ./export.xlsx s3://apm-wo-analysis-exports-328440206208/raw/`
- **Direct:** `aws s3 cp ./export.xlsx s3://apm-wo-analysis-exports-011934824531/raw/`
- **Drop-folder (zero-touch):** a launchd agent (`com.seahaven.apm-wo-uploader`)
watches `~/apm-wo-drop/`, uploads new files to `raw/` using the scoped
`apm-wo-drop` profile (IAM user **`apm-wo-drop-uploader`** — `s3:PutObject` on
@ -94,7 +94,7 @@ access key expired/rotated (`aws configure --profile apm-wo-drop`).
The Grafana EC2 box (`t4g.small`, **Amazon Linux 2023**, ARM64) is the only
patch-bearing piece — everything else is serverless. It's **reproducible from
`cdk/assets/grafana_userdata.sh`**, so the preferred patch path is a **clean
`terraform/templates/grafana_userdata.sh.tftpl`, so the preferred patch path is a **clean
instance replacement** rather than long-lived in-place drift.
- **OS (recommended monthly + on critical CVEs):** via SSM —
@ -102,16 +102,14 @@ instance replacement** rather than long-lived in-place drift.
Run Command / Patch Manager maintenance window — **TBD: not yet automated**).
- **Grafana OSS:** `sudo dnf upgrade grafana -y && sudo systemctl restart grafana-server`
(installed from the pinned `rpm.grafana.com` repo).
- **Athena datasource plugin:** pinned to **`3.2.0`** in `cdk/cdk.json`
(`athenaPluginVersion`). Bump there, then redeploy/replace the instance.
- **Preferred = clean replacement:** terminate the instance; `cdk deploy
apm-wo-analysis-grafana` relaunches it from the latest AL2023 AMI and re-runs
user-data (fresh Grafana + plugin + config sync). The root volume is
`DeleteOnTermination=false`, so detach/reuse or restore `grafana.db` (§2.4) if
local settings must persist. Validates the committed user-data from a cold boot.
> **Outstanding:** one clean instance replacement is owed to validate cold-boot
> user-data and apply root-volume encryption (encryption can't be added in place).
- **Athena datasource plugin:** pinned to **`3.2.0`** in `terraform/variables.tf`
(`athena_plugin_version`). Bump there, then replace the instance (AMI/user-data
are `ignore_changes`; taint/replace to pick up user-data edits).
- **Preferred = clean replacement:** terminate the instance; HCP apply relaunches
it from the latest AL2023 AMI and re-runs user-data only if `ami` /
`user_data` ignore_changes is lifted or the instance is replaced. The root
volume is `DeleteOnTermination=false`, so detach/reuse or restore `grafana.db`
(§2.4) if local settings must persist.
### 2.3 Dashboard-JSON redeploy
@ -119,8 +117,8 @@ Source of truth is **`grafana/dashboards/apm-work-orders.json`** in this repo
(uid **`apm-wo`**); the running instance is never the source of truth
(`allowUiUpdates: false` — UI edits are reverted on the next sync).
**Flow:** edit JSON in repo → `cdk deploy apm-wo-analysis-grafana` (the
`BucketDeployment` uploads `grafana/` to `s3://…/grafana-config/`) → the instance
**Flow:** edit JSON in repo → GitHub Actions `deploy.yaml` syncs `grafana/` to
`s3://…/grafana-config/` → the instance
syncs S3 → `/var/lib/grafana/dashboards/` (on boot + a **15-min systemd timer**)
→ Grafana's file provider polls every **60 s** and reloads.
@ -143,8 +141,8 @@ sudo /usr/local/bin/grafana-config-sync.sh # pulls grafana-config/ from S3
Two distinct layers:
**Config (dashboards, datasources, provisioning)** — fully **reproducible from
git** (`grafana/` → S3 `grafana-config/`). *Restore:* `cdk deploy
apm-wo-analysis-grafana` (or `grafana-config-sync.sh` on the box). No snapshot needed.
git** (`grafana/` → S3 `grafana-config/`). *Restore:* re-run `deploy.yaml` Grafana
sync (or `grafana-config-sync.sh` on the box). No snapshot needed.
**Local state (`/var/lib/grafana/grafana.db`)** — Grafana's SQLite (admin user,
any API keys, org prefs). Lives on the **gp3 root volume** (encrypted,
@ -180,7 +178,7 @@ fresh instance + provisioning recovers everything else.
| Grafana unreachable | instance down / ALB unhealthy / office IP changed (`officeCidrs`) | SSM triage; `aws elbv2 describe-target-health` |
| Slack modal click does nothing / error | `apm-wo-analysis-slack-interactions`, API Gateway, or signing-secret mismatch | `/aws/lambda/apm-wo-analysis-slack-interactions` logs |
| Exports never arrive in `raw/` | drop-folder agent unloaded / TCC / expired key | §2.1 |
| Deploy not applying | OIDC role, CloudFormation rollback, Docker bundling | CloudFormation events; CI logs |
| Deploy not applying | OIDC role, empty `DEPLOY_ROLE_ARN`, or HCP apply role | GitHub Environment prod; HCP run |
---

View file

@ -15,4 +15,4 @@ datasources:
catalog: AwsDataCatalog
database: apm_wo_analysis
workgroup: apm-wo-analysis
outputLocation: s3://apm-wo-analysis-exports-328440206208/athena-results/
outputLocation: s3://apm-wo-analysis-exports-011934824531/athena-results/

View file

@ -1,4 +1,4 @@
[tool.pytest.ini_options]
testpaths = ["tests"]
pythonpath = ["lambdas/classifier", "lambdas/slack_post", "cdk"]
pythonpath = ["lambdas/classifier", "lambdas/slack_post"]
addopts = "--cov=lambdas --cov-report=term-missing"

4
renovate.json Normal file
View file

@ -0,0 +1,4 @@
{
"$schema": "https://docs.renovatebot.com/renovate-schema.json",
"extends": ["local>Sea-Haven-Industries/.github"]
}

View file

@ -20,7 +20,7 @@ UPLOADED_DIR="$DROP_DIR/uploaded"
STATE_DIR="$HOME/.local/log"
LOG_FILE="$STATE_DIR/apm-wo-uploader.log"
LOCK_DIR="$STATE_DIR/apm-wo-uploader.lock"
BUCKET="apm-wo-analysis-exports-328440206208"
BUCKET="apm-wo-analysis-exports-011934824531"
PROFILE="${APM_WO_AWS_PROFILE:-apm-wo-drop}"
mkdir -p "$UPLOADED_DIR" "$STATE_DIR"

131
scripts/package_lambdas.py Normal file
View file

@ -0,0 +1,131 @@
#!/usr/bin/env python3
"""Build Lambda zips for deploy.yaml.
Classifier zips only openpyxl (pandas comes from the AWS-managed layer).
Slack zips include slack_sdk. Both are built for manylinux2014_aarch64.
"""
from __future__ import annotations
import argparse
import os
import shutil
import subprocess
import sys
import tempfile
import zipfile
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
# Keys match terraform/locals.tf local.functions.
FUNCTIONS = {
"classifier": ROOT / "lambdas" / "classifier",
"slack_post": ROOT / "lambdas" / "slack_post",
"slack_interactions": ROOT / "lambdas" / "slack_post",
}
SKIP_INSTALL_PREFIXES = ("boto3", "botocore")
SKIP_COPY_NAMES = {"requirements.txt", "__pycache__"}
def _req_lines(path: Path) -> list[str]:
lines: list[str] = []
if not path.is_file():
return lines
for raw in path.read_text().splitlines():
line = raw.strip()
if not line or line.startswith("#"):
continue
lower = line.lower()
if any(lower.startswith(prefix) for prefix in SKIP_INSTALL_PREFIXES):
continue
lines.append(line)
return lines
def _copy_tree(src: Path, dest: Path) -> None:
dest.mkdir(parents=True, exist_ok=True)
for item in src.iterdir():
if item.name in SKIP_COPY_NAMES or item.name.endswith(".pyc"):
continue
target = dest / item.name
if item.is_dir():
if item.name == "__pycache__":
continue
shutil.copytree(
item, target, ignore=shutil.ignore_patterns("__pycache__", "*.pyc")
)
else:
shutil.copy2(item, target)
def build_function(name: str, src: Path, git_sha: str, out_dir: Path) -> Path:
with tempfile.TemporaryDirectory(prefix=f"apm-wo-{name}-") as tmp:
dest = Path(tmp)
_copy_tree(src, dest)
(dest / "build_info.py").write_text(
f'"""Pinned at zip time by scripts/package_lambdas.py."""\n\nGIT_SHA = "{git_sha}"\n',
encoding="utf-8",
)
unique: list[str] = []
seen: set[str] = set()
for line in _req_lines(src / "requirements.txt"):
if line not in seen:
seen.add(line)
unique.append(line)
if unique:
cmd = [
sys.executable,
"-m",
"pip",
"install",
"--disable-pip-version-check",
"--no-compile",
"--python-version",
"3.12",
"--platform",
"manylinux2014_aarch64",
"--only-binary=:all:",
"--target",
str(dest),
*unique,
]
subprocess.run(cmd, check=True)
out_dir.mkdir(parents=True, exist_ok=True)
zip_path = out_dir / f"{name}.zip"
if zip_path.exists():
zip_path.unlink()
with zipfile.ZipFile(zip_path, "w", compression=zipfile.ZIP_DEFLATED) as zf:
for dirpath, dirnames, filenames in os.walk(dest):
dirnames[:] = [d for d in dirnames if d != "__pycache__"]
for filename in filenames:
if filename.endswith(".pyc"):
continue
full = Path(dirpath) / filename
rel = full.relative_to(dest)
zf.write(full, rel.as_posix())
return zip_path
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--git-sha", required=True)
parser.add_argument("--out-dir", type=Path, default=ROOT / "build" / "packages")
parser.add_argument("--only", nargs="*", default=())
args = parser.parse_args()
selected = args.only or list(FUNCTIONS)
missing = [name for name in selected if name not in FUNCTIONS]
if missing:
print(f"unknown function keys: {missing}", file=sys.stderr)
return 2
for name in selected:
path = build_function(name, FUNCTIONS[name], args.git_sha, args.out_dir)
print(path)
return 0
if __name__ == "__main__":
raise SystemExit(main())

47
terraform/.terraform.lock.hcl generated Normal file
View file

@ -0,0 +1,47 @@
# This file is maintained automatically by "terraform init".
# Manual edits may be lost in future updates.
provider "registry.terraform.io/hashicorp/archive" {
version = "2.8.1"
constraints = "~> 2.8"
hashes = [
"h1:aLNmq6dc3cDcqZc8s/8eKtn0I+UQXyJMGrmo4rRFtNw=",
"zh:03de290604114a89fcd45c2e5bc7787d5a1ebfc5f964fb5989306bea7a4c79ec",
"zh:0a7d69dc9fbbc48960bc2f04588c8fb1bd92c78a8f306566b7fb17fc4a4f2058",
"zh:4df1f3981379c35f1757da957470f7f7724496b57219da3485177cf1647bdf59",
"zh:50f0e72ba53bfe6e11b03b7fc899e1f72536354381d302a743a274ae2a3f45f4",
"zh:5c4e15a04c98e2a8cafb1cd9632b48ad318051e8462d105b1153006277985c35",
"zh:66069e604bcf5c4af0278e15997d9e6bd755c54fb3801d78885838b889729c5f",
"zh:78d5eefdd9e494defcb3c68d282b8f96630502cac21d1ea161f53cfe9bb483b3",
"zh:979765db3f42601870ab377104ba70befb029547b278337ec4ada980e3582de6",
"zh:b165254da774f49945a73fbccc3ef1b63d70ea00a98fa9e14716665ba80ecaad",
"zh:c0bb2697b525da9fec4511f569ed2bd2b42f25d5c1f9fa00f8f3645b50cfc2e9",
"zh:c48f6695d12df0d0fa5231f7c1b8d518a4feae91a733fdd35a5804af3a303831",
"zh:d589954c93f075180c9f4e2ce91780d5edcb56dfd0d3cda8ba08e13c10b52249",
"zh:f5792ed06da65d0daf7ca3711f5399ff78c7cb4400afe53fbce9a926fbde4477",
]
}
provider "registry.terraform.io/hashicorp/aws" {
version = "6.64.0"
constraints = "~> 6.64"
hashes = [
"h1:wXARLY+IeQ7ufYxCLTPCwToWGMRvOpiOTfJS97iwUzI=",
"zh:07172315d67bc9781240272759cdfc7bd32b7e72384a56862c2c1da3cca99a81",
"zh:154ce7d2659de9a59ddfe96d7cab41a9ddc2cb267a7d4bcdf4e737ff2ffdec06",
"zh:17324d4335a7a7ac01cc23eded530775606680ff53b47cb74a3cb95d1121f836",
"zh:307ab92324ec5a61b124881ab8cac1d9e316f4527dfd0e1b59794c229407eb4e",
"zh:31e25f1903661332e36a95283042dd3ec50b47c186db00663fbd976a11e6a6b2",
"zh:3311d9f3bd12a24886027dbe73859dcd1e67bd0e3046227a338cf2c7ca04d18e",
"zh:37916156a3aac3b29be3acebd15d53145ea4ab5d4aaa825eaebe75481fa00500",
"zh:4158cb8c38b3ac6aa98eb15935ec6bd7c30838d85d2b00acc9812df8382ae908",
"zh:5bfb9499c66d9db5b34dc5c60f426a1ab1baa5457ce2aefebca826a9c3f92fb0",
"zh:6eb29ead5a4aca3b1f35812e7e8c75419180e1928e479b458f206861277736db",
"zh:7a82b6dd0c0cdef8045a4adfbddd36acb86b6b23fcbed8e189c2d71f7dc4a502",
"zh:9556bd792032c3f7e73ea4dd08cec88dc1327f5a4a57d79c30ba844ae2b9a3c0",
"zh:9b12af85486a96aedd8d7984b0ff811a4b42e3d88dad1a3fb4c0b580d04fa425",
"zh:c5234180464cb800c83a41f57462742b802c150ad7d4417626fcd9cb511c01d2",
"zh:cd776b83b1f7b36635957350afe7ce28ba4e4ea3a5e2deb00d13dbd3b35d9d40",
"zh:fb583a7b791c6f915b86573d04f05ddbf7f1a5e4120c5d8a7450a3086c1225c4",
]
}

123
terraform/alarms.tf Normal file
View file

@ -0,0 +1,123 @@
locals {
lambda_alarm_matrix = {
errors = {
metric_name = "Errors"
statistic = "Sum"
evaluation_periods = 1
datapoints_to_alarm = 1
threshold = 1
comparison = "GreaterThanOrEqualToThreshold"
period = 300
}
throttles = {
metric_name = "Throttles"
statistic = "Sum"
evaluation_periods = 1
datapoints_to_alarm = 1
threshold = 1
comparison = "GreaterThanOrEqualToThreshold"
period = 300
}
}
lambda_alarms = {
for pair in flatten([
for fn_key, fn in local.functions : [
for metric_key, metric in local.lambda_alarm_matrix : {
key = "${fn_key}-${metric_key}"
function = fn.function_name
metric_key = metric_key
metric_name = metric.metric_name
statistic = metric.statistic
evaluation = metric.evaluation_periods
datapoints = metric.datapoints_to_alarm
threshold = metric.threshold
comparison = metric.comparison
period = metric.period
description = metric_key == "errors" ? "${fn.function_name} reported one or more errors" : "${fn.function_name} was throttled (concurrency limit hit)"
}
]
]) : pair.key => pair
}
}
resource "aws_cloudwatch_metric_alarm" "lambda_errors_throttles" {
for_each = local.lambda_alarms
alarm_name = "Lambda-${title(each.value.metric_key)}-${each.value.function}"
alarm_description = each.value.description
namespace = "AWS/Lambda"
metric_name = each.value.metric_name
dimensions = { FunctionName = each.value.function }
statistic = each.value.statistic
period = each.value.period
evaluation_periods = each.value.evaluation
datapoints_to_alarm = each.value.datapoints
threshold = each.value.threshold
comparison_operator = each.value.comparison
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "lambda_duration" {
for_each = local.functions
alarm_name = "Lambda-Duration-${each.value.function_name}"
alarm_description = "${each.value.function_name} duration approaching its ${each.value.timeout}s timeout (>=${each.value.duration_ms}ms)"
namespace = "AWS/Lambda"
metric_name = "Duration"
dimensions = { FunctionName = each.value.function_name }
statistic = "Maximum"
period = 300
evaluation_periods = 3
datapoints_to_alarm = 2
threshold = each.value.duration_ms
comparison_operator = "GreaterThanOrEqualToThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "classifier_dlq" {
alarm_name = "apm-wo-analysis-classifier-dlq-messages"
alarm_description = "apm-wo-analysis-classifier DLQ has visible messages (dropped classifier run)"
namespace = "AWS/SQS"
metric_name = "ApproximateNumberOfMessagesVisible"
dimensions = { QueueName = aws_sqs_queue.classifier_dlq.name }
statistic = "Maximum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "classifier_invocations" {
alarm_name = "apm-wo-analysis-classifier-invocations"
alarm_description = "apm-wo-analysis-classifier had fewer than 1 invocation in 24h (daily export missing or S3 trigger broken)"
namespace = "AWS/Lambda"
metric_name = "Invocations"
dimensions = { FunctionName = local.functions.classifier.function_name }
statistic = "Sum"
period = 86400
evaluation_periods = 1
threshold = 1
comparison_operator = "LessThanThreshold"
treat_missing_data = "breaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "api_5xx" {
alarm_name = "ApiGateway-5xx-${aws_apigatewayv2_api.http.id}"
alarm_description = "5xx responses on the apm-wo-analysis HTTP API"
namespace = "AWS/ApiGateway"
metric_name = "5xx"
dimensions = { ApiId = aws_apigatewayv2_api.http.id }
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 1
comparison_operator = "GreaterThanOrEqualToThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}

77
terraform/apigateway.tf Normal file
View file

@ -0,0 +1,77 @@
resource "aws_apigatewayv2_api" "http" {
name = local.project
protocol_type = "HTTP"
description = "apm-wo-analysis Slack interactivity API"
}
resource "aws_apigatewayv2_integration" "interactions" {
api_id = aws_apigatewayv2_api.http.id
integration_type = "AWS_PROXY"
integration_method = "POST"
integration_uri = aws_lambda_function.this["slack_interactions"].invoke_arn
payload_format_version = "2.0"
timeout_milliseconds = 30000
}
resource "aws_apigatewayv2_route" "interactions" {
api_id = aws_apigatewayv2_api.http.id
route_key = "POST /slack/interactions"
target = "integrations/${aws_apigatewayv2_integration.interactions.id}"
}
resource "aws_apigatewayv2_stage" "default" {
api_id = aws_apigatewayv2_api.http.id
name = "$default"
auto_deploy = true
access_log_settings {
destination_arn = aws_cloudwatch_log_group.api_access.arn
format = "{\"requestId\":\"$context.requestId\",\"ip\":\"$context.identity.sourceIp\",\"requestTime\":\"$context.requestTime\",\"method\":\"$context.httpMethod\",\"routeKey\":\"$context.routeKey\",\"status\":\"$context.status\",\"protocol\":\"$context.protocol\",\"responseLength\":\"$context.responseLength\",\"integrationError\":\"$context.integrationErrorMessage\"}"
}
default_route_settings {
throttling_burst_limit = 20
throttling_rate_limit = 10
}
depends_on = [
aws_apigatewayv2_route.interactions,
aws_iam_role_policy_attachment.hcptf_apply_services,
]
}
resource "aws_lambda_permission" "api_interactions" {
statement_id = "AllowApiGatewayInvokeInteractions"
action = "lambda:InvokeFunction"
function_name = aws_lambda_function.this["slack_interactions"].function_name
principal = "apigateway.amazonaws.com"
source_arn = "${aws_apigatewayv2_api.http.execution_arn}/*/*"
}
data "aws_acm_certificate" "slack" {
count = var.attach_custom_domains ? 1 : 0
domain = var.slack_interactions_domain
statuses = ["ISSUED"]
most_recent = true
}
resource "aws_apigatewayv2_domain_name" "slack" {
count = var.attach_custom_domains ? 1 : 0
domain_name = var.slack_interactions_domain
domain_name_configuration {
certificate_arn = data.aws_acm_certificate.slack[0].arn
endpoint_type = "REGIONAL"
security_policy = "TLS_1_2"
}
}
resource "aws_apigatewayv2_api_mapping" "slack" {
count = var.attach_custom_domains ? 1 : 0
api_id = aws_apigatewayv2_api.http.id
domain_name = aws_apigatewayv2_domain_name.slack[0].id
stage = aws_apigatewayv2_stage.default.id
}

118
terraform/artifacts.tf Normal file
View file

@ -0,0 +1,118 @@
# Lambda artifacts bucket. Terraform ships only the bootstrap stub.
# .github/workflows/deploy.yaml uploads functions/<name>/<sha>.zip and calls
# update-function-code. Functions ignore code attributes afterwards.
resource "aws_s3_bucket" "artifacts" {
bucket = local.artifacts_bucket_name
tags = {
Purpose = "Lambda deployment packages for apm-wo-analysis"
}
}
resource "aws_s3_bucket_public_access_block" "artifacts" {
bucket = aws_s3_bucket.artifacts.id
block_public_acls = true
block_public_policy = true
ignore_public_acls = true
restrict_public_buckets = true
}
resource "aws_s3_bucket_ownership_controls" "artifacts" {
bucket = aws_s3_bucket.artifacts.id
rule {
object_ownership = "BucketOwnerEnforced"
}
}
resource "aws_s3_bucket_server_side_encryption_configuration" "artifacts" {
bucket = aws_s3_bucket.artifacts.id
rule {
apply_server_side_encryption_by_default {
sse_algorithm = "AES256"
}
}
}
resource "aws_s3_bucket_versioning" "artifacts" {
bucket = aws_s3_bucket.artifacts.id
versioning_configuration {
status = "Enabled"
}
}
resource "aws_s3_bucket_lifecycle_configuration" "artifacts" {
bucket = aws_s3_bucket.artifacts.id
rule {
id = "expire-noncurrent-packages"
status = "Enabled"
filter {}
noncurrent_version_expiration {
noncurrent_days = 180
}
}
rule {
id = "abort-incomplete-multipart"
status = "Enabled"
filter {}
abort_incomplete_multipart_upload {
days_after_initiation = 7
}
}
depends_on = [aws_s3_bucket_versioning.artifacts]
}
data "aws_iam_policy_document" "artifacts" {
statement {
sid = "DenyInsecureTransport"
effect = "Deny"
principals {
type = "*"
identifiers = ["*"]
}
actions = ["s3:*"]
resources = [
aws_s3_bucket.artifacts.arn,
"${aws_s3_bucket.artifacts.arn}/*",
]
condition {
test = "Bool"
variable = "aws:SecureTransport"
values = ["false"]
}
}
}
resource "aws_s3_bucket_policy" "artifacts" {
bucket = aws_s3_bucket.artifacts.id
policy = data.aws_iam_policy_document.artifacts.json
depends_on = [aws_s3_bucket_public_access_block.artifacts]
}
data "archive_file" "bootstrap_stub" {
type = "zip"
source_dir = "${path.module}/bootstrap/stub"
output_path = "${path.module}/build/packages/bootstrap-stub.zip"
}
resource "aws_s3_object" "bootstrap_stub" {
bucket = aws_s3_bucket.artifacts.id
key = "functions/bootstrap-stub.zip"
content_base64 = filebase64(data.archive_file.bootstrap_stub.output_path)
source_hash = data.archive_file.bootstrap_stub.output_base64sha256
}

17
terraform/athena.tf Normal file
View file

@ -0,0 +1,17 @@
resource "aws_athena_workgroup" "this" {
name = local.athena_workgroup
force_destroy = true
configuration {
enforce_workgroup_configuration = true
publish_cloudwatch_metrics_enabled = true
result_configuration {
output_location = "s3://${aws_s3_bucket.exports.bucket}/athena-results/"
encryption_configuration {
encryption_option = "SSE_S3"
}
}
}
}

View file

@ -0,0 +1,9 @@
"""Bootstrap stub. GitHub Actions replaces this zip via update-function-code."""
def handler(event, context):
return {
"statusCode": 503,
"headers": {"content-type": "application/json"},
"body": '{"error":{"code":"NOT_DEPLOYED","message":"Function code has not been deployed yet."}}',
}

89
terraform/dlm.tf Normal file
View file

@ -0,0 +1,89 @@
data "aws_iam_policy_document" "dlm_assume" {
statement {
effect = "Allow"
actions = ["sts:AssumeRole"]
principals {
type = "Service"
identifiers = ["dlm.amazonaws.com"]
}
}
}
resource "aws_iam_role" "dlm" {
name = "apm-wo-analysis-grafana-dlm"
path = "/tf-managed/"
description = "DLM snapshot role for the Grafana instance volume"
assume_role_policy = data.aws_iam_policy_document.dlm_assume.json
permissions_boundary = aws_iam_policy.exec_boundary.arn
}
data "aws_iam_policy_document" "dlm" {
# checkov:skip=CKV_AWS_111: DLM CreateSnapshot/Describe* require Resource=*. Role is boundary-attached and limited to the tagged Grafana volume.
statement {
sid = "DlmSnapshots"
effect = "Allow"
actions = [
"ec2:CreateSnapshot",
"ec2:CreateSnapshots",
"ec2:DeleteSnapshot",
"ec2:DescribeInstances",
"ec2:DescribeVolumes",
"ec2:DescribeSnapshots",
"ec2:DescribeTags",
"ec2:CreateTags",
"ec2:DeleteTags",
]
resources = ["*"]
}
statement {
sid = "DlmKms"
effect = "Allow"
actions = [
"kms:CreateGrant",
"kms:DescribeKey",
"kms:GenerateDataKeyWithoutPlaintext",
"kms:ReEncryptFrom",
"kms:ReEncryptTo",
"kms:ListGrants",
]
resources = [
"arn:aws:kms:${var.aws_region}:${local.account_id}:key/*",
]
}
}
resource "aws_iam_role_policy" "dlm" {
name = "grafana-dlm-snapshots"
role = aws_iam_role.dlm.id
policy = data.aws_iam_policy_document.dlm.json
}
resource "aws_dlm_lifecycle_policy" "grafana" {
description = "Daily snapshot of the apm-wo grafana volume"
execution_role_arn = aws_iam_role.dlm.arn
state = "ENABLED"
policy_details {
resource_types = ["INSTANCE"]
target_tags = {
(local.grafana_backup_tag) = "true"
}
schedule {
name = "daily"
create_rule {
interval = 24
interval_unit = "HOURS"
times = ["07:00"]
}
retain_rule {
count = 7
}
}
}
}

43
terraform/glue.tf Normal file
View file

@ -0,0 +1,43 @@
resource "aws_glue_catalog_database" "analytics" {
name = local.glue_database
}
resource "aws_glue_catalog_table" "snapshots" {
name = local.glue_table
database_name = aws_glue_catalog_database.analytics.name
table_type = "EXTERNAL_TABLE"
parameters = {
classification = "parquet"
EXTERNAL = "TRUE"
"projection.enabled" = "true"
"projection.dt.type" = "date"
"projection.dt.format" = "yyyy-MM-dd"
"projection.dt.range" = "2026-01-01,NOW"
"storage.location.template" = "s3://${aws_s3_bucket.exports.bucket}/analytics/dt=$${dt}/"
}
partition_keys {
name = "dt"
type = "string"
}
storage_descriptor {
location = "s3://${aws_s3_bucket.exports.bucket}/analytics/"
input_format = "org.apache.hadoop.hive.ql.io.parquet.MapredParquetInputFormat"
output_format = "org.apache.hadoop.hive.ql.io.parquet.MapredParquetOutputFormat"
ser_de_info {
serialization_library = "org.apache.hadoop.hive.ql.io.parquet.serde.ParquetHiveSerDe"
}
dynamic "columns" {
for_each = local.snapshot_columns
content {
name = columns.value.name
type = columns.value.type
}
}
}
}

251
terraform/grafana.tf Normal file
View file

@ -0,0 +1,251 @@
data "aws_ssm_parameter" "al2023_arm" {
name = "/aws/service/ami-amazon-linux-latest/al2023-ami-kernel-default-arm64"
}
data "aws_acm_certificate" "grafana" {
domain = var.grafana_domain
statuses = ["ISSUED"]
most_recent = true
}
resource "aws_security_group" "alb" {
name = "apm-wo-analysis-grafana-alb"
description = "apm-wo grafana ALB"
vpc_id = aws_vpc.grafana.id
dynamic "ingress" {
for_each = var.office_cidrs
content {
description = "HTTPS from office ${ingress.value}"
from_port = 443
to_port = 443
protocol = "tcp"
cidr_blocks = [ingress.value]
}
}
egress {
from_port = 0
to_port = 0
protocol = "-1"
cidr_blocks = ["0.0.0.0/0"]
}
tags = {
Name = "apm-wo-analysis-grafana-alb"
}
}
resource "aws_security_group" "instance" {
name = "apm-wo-analysis-grafana-instance"
description = "apm-wo grafana instance"
vpc_id = aws_vpc.grafana.id
ingress {
description = "Grafana HTTP from the ALB only"
from_port = 3000
to_port = 3000
protocol = "tcp"
security_groups = [aws_security_group.alb.id]
}
egress {
from_port = 0
to_port = 0
protocol = "-1"
cidr_blocks = ["0.0.0.0/0"]
}
tags = {
Name = "apm-wo-analysis-grafana-instance"
}
}
data "aws_iam_policy_document" "grafana_assume" {
statement {
effect = "Allow"
actions = ["sts:AssumeRole"]
principals {
type = "Service"
identifiers = ["ec2.amazonaws.com"]
}
}
}
resource "aws_iam_role" "grafana" {
name = "apm-wo-analysis-grafana"
path = "/tf-managed/"
description = "Grafana instance role: Athena, Glue, S3, SSM"
assume_role_policy = data.aws_iam_policy_document.grafana_assume.json
permissions_boundary = aws_iam_policy.exec_boundary.arn
}
data "aws_iam_policy_document" "grafana" {
statement {
sid = "AthenaQuery"
effect = "Allow"
actions = [
"athena:StartQueryExecution",
"athena:StopQueryExecution",
"athena:GetQueryExecution",
"athena:GetQueryResults",
"athena:GetWorkGroup",
]
resources = [
"arn:aws:athena:${var.aws_region}:${local.account_id}:workgroup/${local.athena_workgroup}",
]
}
statement {
sid = "AthenaList"
effect = "Allow"
actions = ["athena:ListWorkGroups"]
resources = ["*"]
}
statement {
sid = "GlueReadOnly"
effect = "Allow"
actions = [
"glue:GetDatabase",
"glue:GetDatabases",
"glue:GetTable",
"glue:GetTables",
"glue:GetPartition",
"glue:GetPartitions",
]
resources = [
"arn:aws:glue:${var.aws_region}:${local.account_id}:catalog",
"arn:aws:glue:${var.aws_region}:${local.account_id}:database/${local.glue_database}",
"arn:aws:glue:${var.aws_region}:${local.account_id}:table/${local.glue_database}/*",
]
}
statement {
sid = "ReadAnalytics"
effect = "Allow"
actions = [
"s3:GetObject",
]
resources = [
"${aws_s3_bucket.exports.arn}/analytics/*",
"${aws_s3_bucket.exports.arn}/${local.grafana_config_prefix}/*",
]
}
statement {
sid = "AthenaResults"
effect = "Allow"
actions = [
"s3:GetObject",
"s3:PutObject",
"s3:AbortMultipartUpload",
]
resources = [
"${aws_s3_bucket.exports.arn}/athena-results/*",
]
}
statement {
sid = "ListExports"
effect = "Allow"
actions = ["s3:ListBucket", "s3:GetBucketLocation"]
resources = [aws_s3_bucket.exports.arn]
}
}
resource "aws_iam_role_policy" "grafana" {
name = "grafana-athena-s3"
role = aws_iam_role.grafana.id
policy = data.aws_iam_policy_document.grafana.json
}
resource "aws_iam_role_policy_attachment" "grafana_ssm" {
role = aws_iam_role.grafana.name
policy_arn = "arn:aws:iam::aws:policy/AmazonSSMManagedInstanceCore"
}
resource "aws_iam_instance_profile" "grafana" {
name = "apm-wo-analysis-grafana"
path = "/tf-managed/"
role = aws_iam_role.grafana.name
}
resource "aws_instance" "grafana" {
ami = data.aws_ssm_parameter.al2023_arm.value
instance_type = "t4g.small"
subnet_id = aws_subnet.private["${var.aws_region}a"].id
vpc_security_group_ids = [aws_security_group.instance.id]
iam_instance_profile = aws_iam_instance_profile.grafana.name
user_data = templatefile("${path.module}/templates/grafana_userdata.sh.tftpl", {
config_bucket = aws_s3_bucket.exports.bucket
config_prefix = local.grafana_config_prefix
plugin_version = var.athena_plugin_version
grafana_domain = var.grafana_domain
})
metadata_options {
http_endpoint = "enabled"
http_tokens = "required"
}
root_block_device {
volume_type = "gp3"
volume_size = 20
encrypted = true
delete_on_termination = false
}
tags = {
Name = "apm-wo-analysis-grafana"
(local.grafana_backup_tag) = "true"
}
lifecycle {
ignore_changes = [ami, user_data]
}
}
resource "aws_lb" "grafana" {
name = "apm-wo-analysis-grafana"
internal = false
load_balancer_type = "application"
security_groups = [aws_security_group.alb.id]
subnets = [for s in aws_subnet.public : s.id]
}
resource "aws_lb_target_group" "grafana" {
name = "apm-wo-analysis-grafana"
port = 3000
protocol = "HTTP"
vpc_id = aws_vpc.grafana.id
target_type = "instance"
health_check {
path = "/api/health"
matcher = "200"
healthy_threshold = 2
unhealthy_threshold = 3
}
}
resource "aws_lb_target_group_attachment" "grafana" {
target_group_arn = aws_lb_target_group.grafana.arn
target_id = aws_instance.grafana.id
port = 3000
}
resource "aws_lb_listener" "grafana_https" {
load_balancer_arn = aws_lb.grafana.arn
port = 443
protocol = "HTTPS"
ssl_policy = "ELBSecurityPolicy-TLS13-1-2-2021-06"
certificate_arn = data.aws_acm_certificate.grafana.arn
default_action {
type = "forward"
target_group_arn = aws_lb_target_group.grafana.arn
}
}

1270
terraform/hcp_iam.tf Normal file

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,125 @@
# GitHub Actions OIDC role for .github/workflows/deploy.yaml.
#
# Trust is pinned three ways: aud, sub to Environment prod (immutable and
# classic subject forms), and job_workflow_ref to deploy.yaml at
# refs/heads/main only. Live GitHub Actions presented the classic sub; both
# forms are listed. No v* tags until a later release ticket.
#
# Not a Lambda execution role: no permissions_boundary. Path /tf-managed/ so
# seahaven-hcptf-iam-management DenySelfMutation (role/githubdeploy-*) does not
# match.
data "aws_iam_policy_document" "github_deploy_assume" {
statement {
sid = "GithubDeployOidc"
effect = "Allow"
actions = ["sts:AssumeRoleWithWebIdentity"]
principals {
type = "Federated"
identifiers = [local.github_oidc_provider_arn]
}
condition {
test = "StringEquals"
variable = "token.actions.githubusercontent.com:aud"
values = ["sts.amazonaws.com"]
}
condition {
test = "StringEquals"
variable = "token.actions.githubusercontent.com:sub"
values = [
local.github_oidc_sub,
"repo:${var.github_repo}:environment:prod",
]
}
condition {
test = "StringEquals"
variable = "token.actions.githubusercontent.com:job_workflow_ref"
values = [
"${var.github_repo}/.github/workflows/deploy.yaml@refs/heads/${var.github_deploy_branch}",
]
}
}
}
resource "aws_iam_role" "github_deploy" {
name = local.deploy_role
path = "/tf-managed/"
description = "GitHub Actions deploy role for ${var.github_repo} Environment prod"
assume_role_policy = data.aws_iam_policy_document.github_deploy_assume.json
max_session_duration = 3600
}
data "aws_iam_policy_document" "github_deploy" {
statement {
sid = "ListArtifactsBucket"
effect = "Allow"
actions = [
"s3:GetBucketLocation",
"s3:ListBucket",
]
resources = [aws_s3_bucket.artifacts.arn]
}
statement {
sid = "UploadFunctionArtifacts"
effect = "Allow"
actions = [
"s3:GetObject",
"s3:PutObject",
]
resources = ["${aws_s3_bucket.artifacts.arn}/functions/*"]
}
statement {
sid = "ListExportsBucket"
effect = "Allow"
actions = [
"s3:GetBucketLocation",
"s3:ListBucket",
]
resources = [aws_s3_bucket.exports.arn]
}
statement {
sid = "SyncGrafanaConfig"
effect = "Allow"
actions = [
"s3:GetObject",
"s3:PutObject",
"s3:DeleteObject",
]
resources = ["${aws_s3_bucket.exports.arn}/${local.grafana_config_prefix}/*"]
}
statement {
sid = "UpdateFunctionCode"
effect = "Allow"
actions = [
"lambda:GetFunction",
"lambda:GetFunctionConfiguration",
"lambda:UpdateFunctionCode",
]
resources = [for fn in local.functions : "arn:aws:lambda:${var.aws_region}:${local.account_id}:function:${fn.function_name}"]
}
statement {
sid = "DeployParams"
effect = "Allow"
actions = [
"ssm:GetParameter",
]
resources = [
"arn:aws:ssm:${var.aws_region}:${local.account_id}:parameter${local.ssm_prefix}/deploy/*",
]
}
}
resource "aws_iam_role_policy" "github_deploy" {
name = "apm-wo-analysis-deploy"
role = aws_iam_role.github_deploy.id
policy = data.aws_iam_policy_document.github_deploy.json
}

227
terraform/lambda.tf Normal file
View file

@ -0,0 +1,227 @@
# Terraform owns the function skeletons (role, runtime, memory, environment).
# Code is owned by .github/workflows/deploy.yaml, which uploads
# functions/<name>/<sha>.zip and calls update-function-code. The lifecycle
# block is the seam: an app deploy is not drift, and a Terraform apply never
# rolls the code back to the bootstrap stub.
data "aws_iam_policy_document" "lambda_assume" {
statement {
effect = "Allow"
actions = ["sts:AssumeRole"]
principals {
type = "Service"
identifiers = ["lambda.amazonaws.com"]
}
}
}
locals {
lambda_identity = {
classifier = [
{
sid = "ReadRaw"
actions = ["s3:GetObject"]
resources = ["${aws_s3_bucket.exports.arn}/raw/*"]
},
{
sid = "WriteAnalytics"
actions = ["s3:GetObject", "s3:PutObject", "s3:DeleteObject", "s3:AbortMultipartUpload"]
resources = ["${aws_s3_bucket.exports.arn}/analytics/*"]
},
{
sid = "WriteMeta"
actions = ["s3:GetObject", "s3:PutObject", "s3:DeleteObject"]
resources = ["${aws_s3_bucket.exports.arn}/meta/*"]
},
{
sid = "ListExports"
actions = ["s3:ListBucket"]
resources = [aws_s3_bucket.exports.arn]
},
{
sid = "AnthropicSecret"
actions = ["secretsmanager:GetSecretValue"]
resources = ["arn:aws:secretsmanager:${var.aws_region}:${local.account_id}:secret:apm-wo-analysis/anthropic-api-key*"]
},
{
sid = "InvokeSlackPost"
actions = ["lambda:InvokeFunction"]
resources = ["arn:aws:lambda:${var.aws_region}:${local.account_id}:function:apm-wo-analysis-slack-post"]
},
{
sid = "DlqSend"
actions = ["sqs:SendMessage"]
resources = [aws_sqs_queue.classifier_dlq.arn]
},
]
slack_post = [
{
sid = "ReadMeta"
actions = ["s3:GetObject"]
resources = ["${aws_s3_bucket.exports.arn}/meta/*"]
},
{
sid = "ListExports"
actions = ["s3:ListBucket"]
resources = [aws_s3_bucket.exports.arn]
},
{
sid = "SlackSecret"
actions = ["secretsmanager:GetSecretValue"]
resources = ["arn:aws:secretsmanager:${var.aws_region}:${local.account_id}:secret:apm-wo-analysis/slack-credentials*"]
},
{
sid = "DashboardParam"
actions = ["ssm:GetParameter"]
resources = ["arn:aws:ssm:${var.aws_region}:${local.account_id}:parameter${local.grafana_url_param}"]
},
]
slack_interactions = [
{
sid = "ReadMeta"
actions = ["s3:GetObject"]
resources = ["${aws_s3_bucket.exports.arn}/meta/*"]
},
{
sid = "ListExports"
actions = ["s3:ListBucket"]
resources = [aws_s3_bucket.exports.arn]
},
{
sid = "SlackSecret"
actions = ["secretsmanager:GetSecretValue"]
resources = ["arn:aws:secretsmanager:${var.aws_region}:${local.account_id}:secret:apm-wo-analysis/slack-credentials*"]
},
{
sid = "DashboardParam"
actions = ["ssm:GetParameter"]
resources = ["arn:aws:ssm:${var.aws_region}:${local.account_id}:parameter${local.grafana_url_param}"]
},
]
}
lambda_env = {
classifier = {
APM_HAIKU_FALLBACK = "on"
SLACK_POST_FUNCTION_NAME = local.functions.slack_post.function_name
}
slack_post = {
SLACK_SECRET_NAME = "apm-wo-analysis/slack-credentials"
DASHBOARD_URL_PARAM = local.grafana_url_param
ANALYTICS_BUCKET = aws_s3_bucket.exports.bucket
}
slack_interactions = {
SLACK_SECRET_NAME = "apm-wo-analysis/slack-credentials"
DASHBOARD_URL_PARAM = local.grafana_url_param
ANALYTICS_BUCKET = aws_s3_bucket.exports.bucket
}
}
}
resource "aws_iam_role" "lambda" {
for_each = local.functions
name = each.value.role_name
path = "/tf-managed/"
description = "Lambda execution role for ${each.value.function_name}"
assume_role_policy = data.aws_iam_policy_document.lambda_assume.json
permissions_boundary = aws_iam_policy.exec_boundary.arn
}
data "aws_iam_policy_document" "lambda" {
for_each = local.functions
dynamic "statement" {
for_each = local.lambda_identity[each.key]
content {
sid = statement.value.sid
effect = "Allow"
actions = statement.value.actions
resources = statement.value.resources
}
}
}
resource "aws_iam_role_policy" "lambda" {
for_each = local.functions
name = each.key
role = aws_iam_role.lambda[each.key].id
policy = data.aws_iam_policy_document.lambda[each.key].json
}
resource "aws_iam_role_policy_attachment" "lambda_basic" {
for_each = local.functions
role = aws_iam_role.lambda[each.key].name
policy_arn = "arn:aws:iam::aws:policy/service-role/AWSLambdaBasicExecutionRole"
}
resource "aws_lambda_function" "this" {
for_each = local.functions
function_name = each.value.function_name
role = aws_iam_role.lambda[each.key].arn
handler = each.value.handler
runtime = "python3.12"
architectures = ["arm64"]
memory_size = each.value.memory_size
timeout = each.value.timeout
layers = each.value.layers
s3_bucket = aws_s3_bucket.artifacts.id
s3_key = aws_s3_object.bootstrap_stub.key
source_code_hash = data.archive_file.bootstrap_stub.output_base64sha256
environment {
variables = local.lambda_env[each.key]
}
dynamic "dead_letter_config" {
for_each = each.key == "classifier" ? [aws_sqs_queue.classifier_dlq.arn] : []
content {
target_arn = dead_letter_config.value
}
}
lifecycle {
ignore_changes = [filename, s3_bucket, s3_key, s3_object_version, source_code_hash]
}
depends_on = [
aws_cloudwatch_log_group.lambda,
aws_iam_role_policy.lambda,
aws_iam_role_policy_attachment.lambda_basic,
]
}
resource "aws_lambda_permission" "s3_classifier" {
statement_id = "AllowS3InvokeClassifier"
action = "lambda:InvokeFunction"
function_name = aws_lambda_function.this["classifier"].function_name
principal = "s3.amazonaws.com"
source_arn = aws_s3_bucket.exports.arn
}
resource "aws_s3_bucket_notification" "raw_exports" {
bucket = aws_s3_bucket.exports.id
lambda_function {
lambda_function_arn = aws_lambda_function.this["classifier"].arn
events = ["s3:ObjectCreated:*"]
filter_prefix = "raw/"
filter_suffix = ".xlsx"
}
lambda_function {
lambda_function_arn = aws_lambda_function.this["classifier"].arn
events = ["s3:ObjectCreated:*"]
filter_prefix = "raw/"
filter_suffix = ".csv"
}
depends_on = [aws_lambda_permission.s3_classifier]
}

View file

@ -0,0 +1,224 @@
# Per-workload permissions boundary. Created on the first (bootstrap) apply.
# The scoped apply role denies iam:CreatePolicy / CreatePolicyVersion, so later
# edits to this document need the hcptf-bootstrap window.
data "aws_iam_policy_document" "exec_boundary" {
# checkov:skip=CKV_AWS_108: Boundary is an upper bound, not a grant. DescribeLogGroups requires Resource=*. Bucket, secret, DLQ, and Athena are ARN-prefixed.
# checkov:skip=CKV_AWS_109: Boundary is an upper bound, not a grant. No IAM permission-management actions.
# checkov:skip=CKV_AWS_111: AWS requires Resource=* for logs:DescribeLogGroups. Exports, secrets, DLQ, and Athena are ARN-pinned.
statement {
sid = "CloudWatchLogsWrite"
effect = "Allow"
actions = [
"logs:CreateLogGroup",
"logs:CreateLogStream",
"logs:PutLogEvents",
"logs:DescribeLogStreams",
]
resources = [
"arn:aws:logs:${var.aws_region}:${local.account_id}:log-group:/aws/lambda*",
]
}
statement {
sid = "CloudWatchLogsDescribe"
effect = "Allow"
actions = ["logs:DescribeLogGroups"]
resources = ["*"]
}
statement {
sid = "ExportsBucket"
effect = "Allow"
actions = [
"s3:GetObject",
"s3:PutObject",
"s3:DeleteObject",
"s3:AbortMultipartUpload",
"s3:ListBucket",
"s3:GetBucketLocation",
]
resources = [
"arn:aws:s3:::${local.exports_bucket_name}",
"arn:aws:s3:::${local.exports_bucket_name}/*",
]
}
statement {
sid = "Secrets"
effect = "Allow"
actions = [
"secretsmanager:GetSecretValue",
"secretsmanager:DescribeSecret",
]
resources = [
"arn:aws:secretsmanager:${var.aws_region}:${local.account_id}:secret:apm-wo-analysis/*",
]
}
statement {
sid = "SsmParams"
effect = "Allow"
actions = [
"ssm:GetParameter",
"ssm:GetParameters",
]
resources = [
"arn:aws:ssm:${var.aws_region}:${local.account_id}:parameter${local.ssm_prefix}/*",
]
}
statement {
sid = "InvokeSlackPost"
effect = "Allow"
actions = [
"lambda:InvokeFunction",
]
resources = [
"arn:aws:lambda:${var.aws_region}:${local.account_id}:function:apm-wo-analysis-slack-post",
]
}
statement {
sid = "ClassifierDlq"
effect = "Allow"
actions = [
"sqs:SendMessage",
]
resources = [
"arn:aws:sqs:${var.aws_region}:${local.account_id}:apm-wo-analysis-classifier-dlq",
]
}
statement {
sid = "AthenaQuery"
effect = "Allow"
actions = [
"athena:StartQueryExecution",
"athena:StopQueryExecution",
"athena:GetQueryExecution",
"athena:GetQueryResults",
"athena:GetWorkGroup",
]
resources = [
"arn:aws:athena:${var.aws_region}:${local.account_id}:workgroup/${local.athena_workgroup}",
]
}
statement {
sid = "AthenaList"
effect = "Allow"
actions = ["athena:ListWorkGroups"]
resources = ["*"]
}
statement {
sid = "GlueRead"
effect = "Allow"
actions = [
"glue:GetDatabase",
"glue:GetDatabases",
"glue:GetTable",
"glue:GetTables",
"glue:GetPartition",
"glue:GetPartitions",
]
resources = [
"arn:aws:glue:${var.aws_region}:${local.account_id}:catalog",
"arn:aws:glue:${var.aws_region}:${local.account_id}:database/${local.glue_database}",
"arn:aws:glue:${var.aws_region}:${local.account_id}:table/${local.glue_database}/*",
]
}
statement {
sid = "SsmManagedInstance"
effect = "Allow"
actions = [
"ssm:DescribeAssociation",
"ssm:GetDeployablePatchSnapshotForInstance",
"ssm:GetDocument",
"ssm:DescribeDocument",
"ssm:GetManifest",
"ssm:GetParameter",
"ssm:GetParameters",
"ssm:GetParametersByPath",
"ssm:ListAssociations",
"ssm:ListInstanceAssociations",
"ssm:UpdateAssociationStatus",
"ssm:UpdateInstanceAssociationStatus",
"ssm:UpdateInstanceInformation",
"ssmmessages:CreateControlChannel",
"ssmmessages:CreateDataChannel",
"ssmmessages:OpenControlChannel",
"ssmmessages:OpenDataChannel",
"ec2messages:AcknowledgeMessage",
"ec2messages:DeleteMessage",
"ec2messages:FailMessage",
"ec2messages:GetEndpoint",
"ec2messages:GetMessages",
"ec2messages:SendReply",
"ec2:DescribeInstanceStatus",
]
resources = ["*"]
}
statement {
sid = "SsmAgentS3"
effect = "Allow"
actions = [
"s3:GetObject",
]
resources = [
"arn:aws:s3:::aws-ssm-*/*",
"arn:aws:s3:::amazon-ssm-*/*",
"arn:aws:s3:::amazon-ssm-packages-*/*",
"arn:aws:s3:::patch-baseline-snapshot-*/*",
]
}
statement {
sid = "DlmSnapshots"
effect = "Allow"
actions = [
"ec2:CreateSnapshot",
"ec2:CreateSnapshots",
"ec2:DeleteSnapshot",
"ec2:DescribeInstances",
"ec2:DescribeVolumes",
"ec2:DescribeSnapshots",
"ec2:EnableFastSnapshotRestores",
"ec2:DescribeFastSnapshotRestores",
"ec2:DisableFastSnapshotRestores",
"ec2:CopySnapshot",
"ec2:ModifySnapshotAttribute",
"ec2:DescribeSnapshotAttribute",
"ec2:DescribeTags",
"ec2:CreateTags",
"ec2:DeleteTags",
]
resources = ["*"]
}
statement {
sid = "DlmKms"
effect = "Allow"
actions = [
"kms:CreateGrant",
"kms:DescribeKey",
"kms:GenerateDataKeyWithoutPlaintext",
"kms:ReEncryptFrom",
"kms:ReEncryptTo",
"kms:ListGrants",
]
resources = [
"arn:aws:kms:${var.aws_region}:${local.account_id}:key/*",
]
}
}
resource "aws_iam_policy" "exec_boundary" {
name = "apm-wo-analysis-exec-boundary"
path = "/tf-managed/"
description = "Per-workload permissions boundary for apm-wo-analysis (PLAT-75)."
policy = data.aws_iam_policy_document.exec_boundary.json
}

97
terraform/locals.tf Normal file
View file

@ -0,0 +1,97 @@
locals {
project = "apm-wo-analysis"
account_id = "011934824531"
environment = "prod"
hcp_project = "seahaven-prod"
hcp_workspace = "apm-wo-analysis-prod"
apply_role = "hcptf-apm-wo-analysis"
plan_role = "hcptf-apm-wo-analysis-plan"
deploy_role = "githubdeploy-apm-wo-analysis"
stack_name = local.project
stack_prefix = "apm-wo-analysis-"
artifacts_bucket_name = "apm-wo-analysis-artifacts-${local.account_id}"
exports_bucket_name = "apm-wo-analysis-exports-${local.account_id}"
ssm_prefix = "/apm-wo-analysis"
site_alerts_arn = "arn:aws:sns:${var.aws_region}:${local.account_id}:site-alerts"
github_oidc_provider_arn = "arn:aws:iam::${local.account_id}:oidc-provider/token.actions.githubusercontent.com"
# Org has Actions OIDC use_immutable_subject=true.
github_oidc_sub = "repo:Sea-Haven-Industries@183236204/apm-wo-analysis@1252708919:environment:prod"
glue_database = "apm_wo_analysis"
glue_table = "apm_wo_snapshots"
athena_workgroup = "apm-wo-analysis"
pandas_layer_arn = "arn:aws:lambda:us-east-1:336392948345:layer:AWSSDKPandas-Python312-Arm64:27"
secret_names = [
"apm-wo-analysis/slack-credentials",
"apm-wo-analysis/anthropic-api-key",
]
grafana_url_param = "${local.ssm_prefix}/grafana-base-url"
grafana_dashboard_url = "https://${var.grafana_domain}/d/apm-wo/apm-work-orders?from=now-30d&to=now"
grafana_config_prefix = "grafana-config"
grafana_backup_tag = "apm-grafana-backup"
azs = ["${var.aws_region}a", "${var.aws_region}b"]
public_subnet_cidrs = {
"${var.aws_region}a" = cidrsubnet(var.vpc_cidr, 8, 0)
"${var.aws_region}b" = cidrsubnet(var.vpc_cidr, 8, 1)
}
private_subnet_cidrs = {
"${var.aws_region}a" = cidrsubnet(var.vpc_cidr, 8, 10)
"${var.aws_region}b" = cidrsubnet(var.vpc_cidr, 8, 11)
}
snapshot_columns = [
{ name = "wo_number", type = "string" },
{ name = "wo_description", type = "string" },
{ name = "equipment_code", type = "string" },
{ name = "site", type = "string" },
{ name = "due_date", type = "string" },
{ name = "department", type = "string" },
{ name = "wo_status", type = "string" },
{ name = "hold_reason", type = "string" },
{ name = "last_comment", type = "string" },
{ name = "last_comment_by", type = "string" },
{ name = "last_comment_date", type = "string" },
{ name = "contractor", type = "string" },
{ name = "contractor_description", type = "string" },
{ name = "category", type = "string" },
{ name = "is_escalation", type = "boolean" },
{ name = "is_action", type = "boolean" },
{ name = "mismatch", type = "string" },
]
functions = {
classifier = {
function_name = "apm-wo-analysis-classifier"
role_name = "apm-wo-analysis-classifier"
handler = "handler.handler"
timeout = 120
memory_size = 512
duration_ms = 96000
layers = [local.pandas_layer_arn]
}
slack_post = {
function_name = "apm-wo-analysis-slack-post"
role_name = "apm-wo-analysis-slack-post"
handler = "handler.handler"
timeout = 30
memory_size = 256
duration_ms = 24000
layers = []
}
slack_interactions = {
function_name = "apm-wo-analysis-slack-interactions"
role_name = "apm-wo-analysis-slack-interactions"
handler = "interactions.handler"
timeout = 30
memory_size = 256
duration_ms = 24000
layers = []
}
}
}

11
terraform/logs.tf Normal file
View file

@ -0,0 +1,11 @@
resource "aws_cloudwatch_log_group" "lambda" {
for_each = local.functions
name = "/aws/lambda/${each.value.function_name}"
retention_in_days = 60
}
resource "aws_cloudwatch_log_group" "api_access" {
name = "/aws/apigateway/${local.project}"
retention_in_days = 60
}

49
terraform/outputs.tf Normal file
View file

@ -0,0 +1,49 @@
output "slack_execute_api_url" {
description = "Slack interactivity URL on the execute-api origin. Use until DNS is flipped."
value = "${aws_apigatewayv2_api.http.api_endpoint}/slack/interactions"
}
output "slack_custom_domain_target" {
description = "API Gateway regional domain to point apm-wo.seahaven.com at (OOB Route53)."
value = try(aws_apigatewayv2_domain_name.slack[0].domain_name_configuration[0].target_domain_name, null)
}
output "grafana_alb_dns_name" {
description = "ALB DNS name to point grafana.seahaven.com at (OOB Route53)."
value = aws_lb.grafana.dns_name
}
output "grafana_alb_zone_id" {
description = "ALB hosted zone id for the grafana.seahaven.com alias."
value = aws_lb.grafana.zone_id
}
output "exports_bucket_name" {
description = "Exports bucket. Uploader and Athena read this."
value = aws_s3_bucket.exports.id
}
output "github_deploy_role_arn" {
description = "OIDC role ARN for .github/workflows/deploy.yaml (GitHub Environment prod variable DEPLOY_ROLE_ARN)."
value = aws_iam_role.github_deploy.arn
}
output "artifacts_bucket_name" {
description = "Lambda artifacts bucket. deploy.yaml uploads functions/<name>/<sha>.zip."
value = aws_s3_bucket.artifacts.id
}
output "hcptf_apply_role_arn" {
description = "HCP apply role ARN. Set TFC_AWS_APPLY_ROLE_ARN after the bootstrap window."
value = aws_iam_role.hcptf_apply.arn
}
output "hcptf_plan_role_arn" {
description = "HCP plan role ARN. Set TFC_AWS_PLAN_ROLE_ARN after the bootstrap window."
value = aws_iam_role.hcptf_plan.arn
}
output "grafana_instance_id" {
description = "Grafana EC2 instance id for SSM Session Manager."
value = aws_instance.grafana.id
}

12
terraform/providers.tf Normal file
View file

@ -0,0 +1,12 @@
provider "aws" {
region = var.aws_region
default_tags {
tags = {
Project = local.project
Environment = "prod"
ManagedBy = "terraform"
Workspace = local.hcp_workspace
}
}
}

132
terraform/s3.tf Normal file
View file

@ -0,0 +1,132 @@
# Exports bucket: raw/ (incoming), analytics/ (Parquet), meta/ (JSON),
# athena-results/, grafana-config/. RETAIN equivalent: no force_destroy.
resource "aws_s3_bucket" "exports" {
bucket = local.exports_bucket_name
tags = {
Purpose = "APM WO export snapshots and Grafana config"
}
}
resource "aws_s3_bucket_public_access_block" "exports" {
bucket = aws_s3_bucket.exports.id
block_public_acls = true
block_public_policy = true
ignore_public_acls = true
restrict_public_buckets = true
}
resource "aws_s3_bucket_ownership_controls" "exports" {
bucket = aws_s3_bucket.exports.id
rule {
object_ownership = "BucketOwnerEnforced"
}
}
resource "aws_s3_bucket_server_side_encryption_configuration" "exports" {
bucket = aws_s3_bucket.exports.id
rule {
apply_server_side_encryption_by_default {
sse_algorithm = "AES256"
}
}
}
resource "aws_s3_bucket_lifecycle_configuration" "exports" {
bucket = aws_s3_bucket.exports.id
rule {
id = "expire-raw-exports"
status = "Enabled"
filter {
prefix = "raw/"
}
expiration {
days = 90
}
}
rule {
id = "expire-athena-results"
status = "Enabled"
filter {
prefix = "athena-results/"
}
expiration {
days = 30
}
}
rule {
id = "abort-incomplete-multipart"
status = "Enabled"
filter {}
abort_incomplete_multipart_upload {
days_after_initiation = 7
}
}
}
data "aws_iam_policy_document" "exports" {
statement {
sid = "DenyInsecureTransport"
effect = "Deny"
principals {
type = "*"
identifiers = ["*"]
}
actions = ["s3:*"]
resources = [
aws_s3_bucket.exports.arn,
"${aws_s3_bucket.exports.arn}/*",
]
condition {
test = "Bool"
variable = "aws:SecureTransport"
values = ["false"]
}
}
}
resource "aws_s3_bucket_policy" "exports" {
bucket = aws_s3_bucket.exports.id
policy = data.aws_iam_policy_document.exports.json
depends_on = [aws_s3_bucket_public_access_block.exports]
}
# Drop-folder identity. Access keys are created out of band
# (aws iam create-access-key) and stored in the local apm-wo-drop profile.
resource "aws_iam_user" "drop_uploader" {
name = "apm-wo-drop-uploader"
path = "/tf-managed/"
}
data "aws_iam_policy_document" "drop_uploader" {
statement {
sid = "PutRawExportsOnly"
effect = "Allow"
actions = ["s3:PutObject"]
resources = ["${aws_s3_bucket.exports.arn}/raw/*"]
}
}
resource "aws_iam_user_policy" "drop_uploader" {
# checkov:skip=CKV_AWS_40: Drop-folder identity is an IAM user by design (local launchd profile). Policy is s3:PutObject on raw/* only.
name = "PutRawExportsOnly"
user = aws_iam_user.drop_uploader.name
policy = data.aws_iam_policy_document.drop_uploader.json
}

12
terraform/secrets.tf Normal file
View file

@ -0,0 +1,12 @@
# Secret shells only. Values are copied from mgmt out of band after create.
resource "aws_secretsmanager_secret" "this" {
for_each = toset(local.secret_names)
name = each.value
recovery_window_in_days = 30
tags = {
Purpose = "apm-wo-analysis secret shell"
}
}

31
terraform/sqs.tf Normal file
View file

@ -0,0 +1,31 @@
resource "aws_sqs_queue" "classifier_dlq" {
name = "apm-wo-analysis-classifier-dlq"
message_retention_seconds = 1209600
sqs_managed_sse_enabled = true
}
data "aws_iam_policy_document" "classifier_dlq" {
statement {
sid = "DenyInsecureTransport"
effect = "Deny"
principals {
type = "*"
identifiers = ["*"]
}
actions = ["sqs:*"]
resources = [aws_sqs_queue.classifier_dlq.arn]
condition {
test = "Bool"
variable = "aws:SecureTransport"
values = ["false"]
}
}
}
resource "aws_sqs_queue_policy" "classifier_dlq" {
queue_url = aws_sqs_queue.classifier_dlq.id
policy = data.aws_iam_policy_document.classifier_dlq.json
}

29
terraform/ssm.tf Normal file
View file

@ -0,0 +1,29 @@
resource "aws_ssm_parameter" "deploy_artifacts_bucket" {
name = "${local.ssm_prefix}/deploy/artifacts-bucket"
type = "String"
value = aws_s3_bucket.artifacts.id
description = "Lambda artifacts bucket; deploy.yaml uploads functions/<name>/<sha>.zip"
}
resource "aws_ssm_parameter" "deploy_exports_bucket" {
name = "${local.ssm_prefix}/deploy/exports-bucket"
type = "String"
value = aws_s3_bucket.exports.id
description = "Exports bucket; deploy.yaml syncs grafana/ to grafana-config/"
}
resource "aws_ssm_parameter" "deploy_function_name" {
for_each = local.functions
name = "${local.ssm_prefix}/deploy/${each.key}-function-name"
type = "String"
value = each.value.function_name
description = "Lambda function name for ${each.key}; deploy.yaml calls update-function-code"
}
resource "aws_ssm_parameter" "grafana_url" {
name = local.grafana_url_param
type = "String"
value = local.grafana_dashboard_url
description = "Grafana dashboard URL for the Slack Open dashboard button"
}

View file

@ -0,0 +1,86 @@
#!/bin/bash
# Grafana OSS bootstrap for the apm-wo-analysis dashboard host (Amazon Linux 2023,
# ARM64). Idempotent enough to re-run. Config + dashboards are pulled from S3
# (the repo is the source of truth); a systemd timer re-syncs dashboards so panel
# updates ship by re-uploading to S3 — no instance rebuild.
set -euxo pipefail
CONFIG_BUCKET="${config_bucket}"
CONFIG_PREFIX="${config_prefix}"
PLUGIN_VERSION="${plugin_version}"
GRAFANA_DOMAIN="${grafana_domain}"
cat >/etc/yum.repos.d/grafana.repo <<'REPO'
[grafana]
name=grafana
baseurl=https://rpm.grafana.com
repo_gpgcheck=1
enabled=1
gpgcheck=1
gpgkey=https://rpm.grafana.com/gpg.key
sslverify=1
REPO
dnf install -y grafana
grafana-cli --homepath=/usr/share/grafana --pluginsDir=/var/lib/grafana/plugins \
plugins install grafana-athena-datasource "$${PLUGIN_VERSION}"
cat >/etc/grafana/grafana.ini <<INI
[server]
protocol = http
http_port = 3000
root_url = https://$${GRAFANA_DOMAIN}/
enforce_domain = false
[security]
allow_embedding = true
cookie_secure = true
[users]
default_theme = dark
[analytics]
reporting_enabled = false
check_for_updates = false
INI
sync_config() {
aws s3 sync "s3://$${CONFIG_BUCKET}/$${CONFIG_PREFIX}/provisioning/" /etc/grafana/provisioning/ --delete --exact-timestamps
aws s3 sync "s3://$${CONFIG_BUCKET}/$${CONFIG_PREFIX}/dashboards/" /var/lib/grafana/dashboards/ --delete --exact-timestamps
chown -R grafana:grafana /etc/grafana/provisioning /var/lib/grafana/dashboards
}
mkdir -p /var/lib/grafana/dashboards
sync_config
systemctl daemon-reload
systemctl enable --now grafana-server
cat >/usr/local/bin/grafana-config-sync.sh <<SYNC
#!/bin/bash
set -euo pipefail
aws s3 sync "s3://$${CONFIG_BUCKET}/$${CONFIG_PREFIX}/provisioning/" /etc/grafana/provisioning/ --delete --exact-timestamps
aws s3 sync "s3://$${CONFIG_BUCKET}/$${CONFIG_PREFIX}/dashboards/" /var/lib/grafana/dashboards/ --delete --exact-timestamps
chown -R grafana:grafana /etc/grafana/provisioning /var/lib/grafana/dashboards
SYNC
chmod +x /usr/local/bin/grafana-config-sync.sh
cat >/etc/systemd/system/grafana-config-sync.service <<'SVC'
[Unit]
Description=Sync apm-wo Grafana config/dashboards from S3
[Service]
Type=oneshot
ExecStart=/usr/local/bin/grafana-config-sync.sh
SVC
cat >/etc/systemd/system/grafana-config-sync.timer <<'TIMER'
[Unit]
Description=Periodic apm-wo Grafana config sync
[Timer]
OnBootSec=5min
OnUnitActiveSec=15min
[Install]
WantedBy=timers.target
TIMER
systemctl daemon-reload
systemctl enable --now grafana-config-sync.timer

53
terraform/variables.tf Normal file
View file

@ -0,0 +1,53 @@
variable "aws_region" {
description = "Region every resource in this configuration is created in."
type = string
default = "us-east-1"
}
variable "github_repo" {
description = "GitHub owner/name for the deploy OIDC trust."
type = string
default = "Sea-Haven-Industries/apm-wo-analysis"
}
variable "github_deploy_branch" {
description = "Git branch pinned in job_workflow_ref for the deploy role."
type = string
default = "main"
}
variable "office_cidrs" {
description = "Office CIDRs allowed to reach the Grafana ALB on 443."
type = list(string)
default = ["47.21.61.4/32", "96.250.164.146/32"]
}
variable "slack_interactions_domain" {
description = "Custom domain for the Slack interactivity HTTP API."
type = string
default = "apm-wo.seahaven.com"
}
variable "grafana_domain" {
description = "Public hostname for the Grafana ALB."
type = string
default = "grafana.seahaven.com"
}
variable "athena_plugin_version" {
description = "Pinned grafana-athena-datasource plugin version installed by user-data."
type = string
default = "3.2.0"
}
variable "attach_custom_domains" {
description = "Attach the Slack HTTP API custom domain. False until mgmt releases apm-wo.seahaven.com at cutover. Grafana HTTPS does not use this flag."
type = bool
default = true
}
variable "vpc_cidr" {
description = "CIDR for the dedicated Grafana VPC."
type = string
default = "10.80.0.0/16"
}

22
terraform/versions.tf Normal file
View file

@ -0,0 +1,22 @@
terraform {
required_version = ">= 1.14.0"
required_providers {
aws = {
source = "hashicorp/aws"
version = "~> 6.64"
}
archive = {
source = "hashicorp/archive"
version = "~> 2.8"
}
}
cloud {
organization = "seahaven"
workspaces {
name = "apm-wo-analysis-prod"
}
}
}

103
terraform/vpc.tf Normal file
View file

@ -0,0 +1,103 @@
resource "aws_vpc" "grafana" {
cidr_block = var.vpc_cidr
enable_dns_support = true
enable_dns_hostnames = true
tags = {
Name = "apm-wo-analysis-grafana"
}
}
resource "aws_internet_gateway" "grafana" {
vpc_id = aws_vpc.grafana.id
tags = {
Name = "apm-wo-analysis-grafana"
}
}
resource "aws_subnet" "public" {
for_each = local.public_subnet_cidrs
vpc_id = aws_vpc.grafana.id
cidr_block = each.value
availability_zone = each.key
map_public_ip_on_launch = true
tags = {
Name = "apm-wo-analysis-grafana-public-${each.key}"
}
}
resource "aws_subnet" "private" {
for_each = local.private_subnet_cidrs
vpc_id = aws_vpc.grafana.id
cidr_block = each.value
availability_zone = each.key
tags = {
Name = "apm-wo-analysis-grafana-private-${each.key}"
}
}
resource "aws_eip" "nat" {
domain = "vpc"
tags = {
Name = "apm-wo-analysis-grafana-nat"
}
depends_on = [aws_internet_gateway.grafana]
}
resource "aws_nat_gateway" "grafana" {
allocation_id = aws_eip.nat.id
subnet_id = aws_subnet.public["${var.aws_region}a"].id
tags = {
Name = "apm-wo-analysis-grafana"
}
}
resource "aws_route_table" "public" {
vpc_id = aws_vpc.grafana.id
tags = {
Name = "apm-wo-analysis-grafana-public"
}
}
resource "aws_route" "public_internet" {
route_table_id = aws_route_table.public.id
destination_cidr_block = "0.0.0.0/0"
gateway_id = aws_internet_gateway.grafana.id
}
resource "aws_route_table_association" "public" {
for_each = aws_subnet.public
subnet_id = each.value.id
route_table_id = aws_route_table.public.id
}
resource "aws_route_table" "private" {
vpc_id = aws_vpc.grafana.id
tags = {
Name = "apm-wo-analysis-grafana-private"
}
}
resource "aws_route" "private_nat" {
route_table_id = aws_route_table.private.id
destination_cidr_block = "0.0.0.0/0"
nat_gateway_id = aws_nat_gateway.grafana.id
}
resource "aws_route_table_association" "private" {
for_each = aws_subnet.private
subnet_id = each.value.id
route_table_id = aws_route_table.private.id
}

View file

@ -1,181 +0,0 @@
"""Synth-level assertions for the Grafana stack (Phase 5).
Synthesizes ``apm-wo-analysis-grafana`` and asserts the security posture that
can't be eyeballed: the ALB only admits the office CIDRs on 443 (never
0.0.0.0/0), the instance only takes traffic from the ALB SG, the instance role
carries no static keys and only scoped Athena/Glue-read/S3 access, the root
volume is gp3 + retained, a daily DLM backup exists, and grafana.seahaven.com
aliases the ALB. No AWS, no Docker (bundling skipped).
Run with the repo venv:
python -m pytest tests/test_grafana_synth.py -q
"""
import json
import sys
from pathlib import Path
import aws_cdk as cdk
from aws_cdk.assertions import Match, Template
CDK_DIR = Path(__file__).resolve().parents[1] / "cdk"
sys.path.insert(0, str(CDK_DIR))
from stacks.grafana_stack import GrafanaStack # noqa: E402
OFFICE_CIDRS = {"47.21.61.4/32", "96.250.164.146/32"}
def _cdk_context() -> dict:
ctx = json.loads((CDK_DIR / "cdk.json").read_text())["context"]
ctx["aws:cdk:bundling-stacks"] = []
return ctx
def _template() -> Template:
app = cdk.App(context=_cdk_context())
stack = GrafanaStack(
app,
"apm-wo-analysis-grafana",
env=cdk.Environment(account="328440206208", region="us-east-1"),
)
return Template.from_stack(stack)
def _all_cidr_ingress(t: Template):
"""Every CIDR-based ingress rule, inline on SGs and standalone, as
(cidr, from_port, to_port) tuples."""
rules = []
for sg in t.find_resources("AWS::EC2::SecurityGroup").values():
for r in sg["Properties"].get("SecurityGroupIngress", []):
if "CidrIp" in r:
rules.append((r["CidrIp"], r.get("FromPort"), r.get("ToPort")))
for ing in t.find_resources("AWS::EC2::SecurityGroupIngress").values():
p = ing["Properties"]
if "CidrIp" in p:
rules.append((p["CidrIp"], p.get("FromPort"), p.get("ToPort")))
return rules
def test_alb_only_admits_office_cidrs_on_443():
rules = _all_cidr_ingress(_template())
cidrs_443 = {c for c, fp, tp in rules if fp == 443 and tp == 443}
assert cidrs_443 == OFFICE_CIDRS, (
f"443 ingress should be office-only, got {cidrs_443}"
)
# Nothing anywhere may be open to the world.
assert all(c != "0.0.0.0/0" for c, _, _ in rules), "found a 0.0.0.0/0 ingress"
def test_instance_only_reachable_from_alb_on_3000():
# The instance SG ingress on 3000 is a SourceSecurityGroup rule, not a CIDR.
_template().has_resource_properties(
"AWS::EC2::SecurityGroupIngress",
Match.object_like(
{
"FromPort": 3000,
"ToPort": 3000,
"SourceSecurityGroupId": Match.any_value(),
}
),
)
def test_alb_internet_facing_https_listener():
t = _template()
t.has_resource_properties(
"AWS::ElasticLoadBalancingV2::LoadBalancer", {"Scheme": "internet-facing"}
)
t.has_resource_properties(
"AWS::ElasticLoadBalancingV2::Listener",
Match.object_like(
{"Port": 443, "Protocol": "HTTPS", "Certificates": Match.any_value()}
),
)
def test_no_static_keys_in_stack():
t = _template()
t.resource_count_is("AWS::IAM::User", 0)
t.resource_count_is("AWS::IAM::AccessKey", 0)
def test_instance_role_scoped_and_uses_ssm():
t = _template()
# Session Manager (no SSH) — the SSM managed policy is attached.
t.has_resource_properties(
"AWS::IAM::Role",
Match.object_like(
{
"ManagedPolicyArns": Match.array_with(
[
{
"Fn::Join": [
"",
Match.array_with(
[":iam::aws:policy/AmazonSSMManagedInstanceCore"]
),
]
}
]
)
}
),
)
# The instance role must not be able to write the catalog or run wide Athena.
for policy in t.find_resources("AWS::IAM::Policy").values():
for stmt in policy["Properties"]["PolicyDocument"]["Statement"]:
actions = stmt.get("Action", [])
actions = [actions] if isinstance(actions, str) else actions
for a in actions:
if isinstance(a, str):
assert a not in ("glue:*", "athena:*", "s3:*", "*"), (
f"too broad: {a}"
)
assert not a.startswith("glue:Create"), f"no Glue writes: {a}"
assert not a.startswith("glue:Update"), f"no Glue writes: {a}"
def test_root_volume_gp3_and_retained():
_template().has_resource_properties(
"AWS::EC2::Instance",
Match.object_like(
{
"BlockDeviceMappings": Match.array_with(
[
Match.object_like(
{
"Ebs": Match.object_like(
{
"VolumeType": "gp3",
"DeleteOnTermination": False,
"Encrypted": True,
}
)
}
)
]
)
}
),
)
def test_daily_dlm_backup_enabled():
_template().has_resource_properties(
"AWS::DLM::LifecyclePolicy",
Match.object_like(
{
"State": "ENABLED",
"PolicyDetails": Match.object_like({"ResourceTypes": ["INSTANCE"]}),
}
),
)
def test_route53_alias_for_grafana():
_template().has_resource_properties(
"AWS::Route53::RecordSet",
Match.object_like({"Type": "A", "Name": "grafana.seahaven.com."}),
)

View file

@ -1,238 +0,0 @@
"""Synth-level assertions for the analytics dataset (Phase 3) and Slack surfaces
(Phase 4).
Synthesizes ``apm-wo-analysis-pipeline`` and asserts: the Glue table carries
partition projection with the classifier's column schema; the Athena workgroup
enforces its result location; the classifier role has zero Glue access; and the
Phase 4 Slack post + interactions Lambdas exist behind an HTTP API with only the
scoped S3/Secrets/SSM permissions. No AWS, no Docker: ``aws:cdk:bundling-stacks=[]``
skips asset bundling so this is a fast offline gate.
Run with the repo venv:
python -m pytest tests/test_pipeline_synth.py -q
"""
import json
import sys
from pathlib import Path
import aws_cdk as cdk
from aws_cdk.assertions import Match, Template
CDK_DIR = Path(__file__).resolve().parents[1] / "cdk"
sys.path.insert(0, str(CDK_DIR))
from stacks.pipeline_stack import PipelineStack # noqa: E402
def _s3_path_ending(suffix: str):
"""Match an ``Fn::Join`` S3 path (bucket name is a Ref) ending in ``suffix``."""
return {"Fn::Join": ["", Match.array_with([suffix])]}
def _cdk_context() -> dict:
"""The real cdk.json context (cert ARN, hosted zone, Slack domain) — the
hand-built App below doesn't auto-load it the way `cdk synth` does."""
ctx = json.loads((CDK_DIR / "cdk.json").read_text())["context"]
ctx["aws:cdk:bundling-stacks"] = []
return ctx
def _template() -> Template:
app = cdk.App(context=_cdk_context())
stack = PipelineStack(
app,
"apm-wo-analysis-pipeline",
env=cdk.Environment(account="328440206208", region="us-east-1"),
)
return Template.from_stack(stack)
def test_snapshots_table_uses_partition_projection():
_template().has_resource_properties(
"AWS::Glue::Table",
{
"TableInput": {
"Name": "apm_wo_snapshots",
"PartitionKeys": [{"Name": "dt", "Type": "string"}],
"Parameters": Match.object_like(
{
"projection.enabled": "true",
"projection.dt.type": "date",
"projection.dt.format": "yyyy-MM-dd",
"projection.dt.range": "2026-01-01,NOW",
"storage.location.template": _s3_path_ending(
"/analytics/dt=${dt}/"
),
}
),
}
},
)
def test_snapshots_table_column_schema_matches_classifier():
# Booleans typed correctly and the columns the BUILD.md prose omitted
# (contractor_description) are present — schema mirrors the writer.
_template().has_resource_properties(
"AWS::Glue::Table",
{
"TableInput": {
"StorageDescriptor": Match.object_like(
{
# array_with is order-sensitive: list patterns in the
# same order the classifier writes them.
"Columns": Match.array_with(
[
{"Name": "contractor_description", "Type": "string"},
{"Name": "is_escalation", "Type": "boolean"},
{"Name": "is_action", "Type": "boolean"},
{"Name": "mismatch", "Type": "string"},
]
),
}
),
}
},
)
def test_athena_workgroup_enforces_results_location():
_template().has_resource_properties(
"AWS::Athena::WorkGroup",
{
"Name": "apm-wo-analysis",
"WorkGroupConfiguration": Match.object_like(
{
"EnforceWorkGroupConfiguration": True,
"ResultConfiguration": {
"OutputLocation": _s3_path_ending("/athena-results/"),
"EncryptionConfiguration": {"EncryptionOption": "SSE_S3"},
},
}
),
},
)
def test_classifier_role_has_no_glue_access():
# Partition projection => the classifier only writes Parquet to S3. No IAM
# policy in the stack should grant any glue:* action.
policies = _template().find_resources("AWS::IAM::Policy")
for policy in policies.values():
for stmt in policy["Properties"]["PolicyDocument"]["Statement"]:
actions = stmt.get("Action", [])
actions = [actions] if isinstance(actions, str) else actions
offending = [
a for a in actions if isinstance(a, str) and a.startswith("glue:")
]
assert not offending, f"unexpected Glue access: {offending}"
# ----- Phase 4 — Slack surfaces -----
def test_slack_lambdas_exist():
t = _template()
for fn_name, handler in (
("apm-wo-analysis-slack-post", "handler.handler"),
("apm-wo-analysis-slack-interactions", "interactions.handler"),
):
t.has_resource_properties(
"AWS::Lambda::Function",
{
"FunctionName": fn_name,
"Handler": handler,
"Runtime": "python3.12",
"Architectures": ["arm64"],
},
)
def test_classifier_has_dlq():
# Failed async invocations must surface, not silently drop a day's data.
t = _template()
t.resource_count_is("AWS::SQS::Queue", 1)
t.has_resource_properties(
"AWS::Lambda::Function",
Match.object_like(
{
"FunctionName": "apm-wo-analysis-classifier",
"DeadLetterConfig": Match.any_value(),
}
),
)
def test_classifier_daily_invocation_alarm_treats_missing_as_breaching():
# A silent day publishes no Invocations datapoint. NOT_BREACHING would
# hide the outage; BREACHING is the page. Dimension Value is a Ref to the
# function (function_name is set, but CDK still Refs the resource).
_template().has_resource_properties(
"AWS::CloudWatch::Alarm",
Match.object_like(
{
"AlarmName": "apm-wo-analysis-classifier-invocations",
"Namespace": "AWS/Lambda",
"MetricName": "Invocations",
"Statistic": "Sum",
"Period": 86400,
"Threshold": 1,
"ComparisonOperator": "LessThanThreshold",
"EvaluationPeriods": 1,
"TreatMissingData": "breaching",
}
),
)
def test_interactions_stage_is_throttled():
# The public Slack interactions endpoint caps rate/burst.
_template().has_resource_properties(
"AWS::ApiGatewayV2::Stage",
Match.object_like(
{
"DefaultRouteSettings": {
"ThrottlingRateLimit": 10,
"ThrottlingBurstLimit": 20,
}
}
),
)
def test_interactions_api_routes_post_to_slack_endpoint():
t = _template()
t.resource_count_is("AWS::ApiGatewayV2::Api", 1)
t.has_resource_properties(
"AWS::ApiGatewayV2::Route", {"RouteKey": "POST /slack/interactions"}
)
# Custom domain on apm-wo.seahaven.com + a Route53 alias for it.
t.has_resource_properties(
"AWS::ApiGatewayV2::DomainName", {"DomainName": "apm-wo.seahaven.com"}
)
t.has_resource_properties("AWS::Route53::RecordSet", {"Type": "A"})
def test_slack_roles_have_no_broad_or_write_access():
# The Slack Lambdas should only read analytics/, the Slack secret, and the
# dashboard SSM param — never write S3, never s3:*/secretsmanager:* wildcards.
# Scope to the Slack policies by logical ID (the classifier/drop-uploader
# legitimately hold s3:PutObject).
t = _template()
slack_policies = {
lid: p
for lid, p in t.find_resources("AWS::IAM::Policy").items()
if lid.startswith(("SlackPost", "SlackInteractions"))
}
assert slack_policies, "expected scoped policies for the Slack roles"
for policy in slack_policies.values():
for stmt in policy["Properties"]["PolicyDocument"]["Statement"]:
actions = stmt.get("Action", [])
actions = [actions] if isinstance(actions, str) else actions
for a in actions:
if not isinstance(a, str):
continue
assert a not in ("s3:*", "secretsmanager:*", "*"), f"too broad: {a}"
assert a != "s3:PutObject", "Slack roles must not write S3"