feat(infra): add lightweight meal-order-manager-dev (PLAT-210) (#195)

* feat(infra): add lightweight meal-order-manager-dev (PLAT-210)

Parameterize the HCP root for seahaven-dev with schedules, PITR, alarms, and Paychex gated off so a second env does not clone production cost or side effects.

* fix(infra): drop prod-only authorizer import so dev can create it (PLAT-210)

The PLAT-102 import is already in meal-order-manager-prod state. A shared import block fails in seahaven-dev because the permission does not exist there.

* fix(iam): allow creating the weekly-menu githubdeploy role in seahaven-dev (PLAT-210)

Prod imported that role. A new account needs CreateRole on tf-managed/githubdeploy-meal-order-manager-weekly-menu.
This commit is contained in:
Adam Moussa 2026-09-18 18:38:45 +00:00 • committed by GitHub
parent 12f1eb881f
commit 44c79fdefd
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
15 changed files with 150 additions and 48 deletions

View file

@ -14,7 +14,13 @@
# 3. Run terraform apply. Until step 2 completes, this data source finds no
# ISSUED certificate and the plan fails closed.
data "aws_acm_certificate" "orders" {
count = var.attach_custom_domain ? 1 : 0
domain = var.domain_name
statuses = ["ISSUED"]
most_recent = true
}
moved {
from = data.aws_acm_certificate.orders
to = data.aws_acm_certificate.orders[0]
}

View file

@ -12,7 +12,7 @@
# here for throttle coverage.
locals {
alarm_functions = {
alarm_functions = local.is_prod ? {
"submit-order" = {
function_name = aws_lambda_function.submit_order.function_name
duration_threshold = 8000
@ -49,7 +49,7 @@ locals {
duration_timeout = "60s"
duration_datapoints = 1
}
}
} : {}
}
resource "aws_cloudwatch_metric_alarm" "lambda_errors" {
@ -65,7 +65,7 @@ resource "aws_cloudwatch_metric_alarm" "lambda_errors" {
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
alarm_actions = [data.aws_sns_topic.site_alerts[0].arn]
dimensions = {
FunctionName = each.value.function_name
@ -85,7 +85,7 @@ resource "aws_cloudwatch_metric_alarm" "lambda_throttles" {
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
alarm_actions = [data.aws_sns_topic.site_alerts[0].arn]
dimensions = {
FunctionName = each.value.function_name
@ -106,7 +106,7 @@ resource "aws_cloudwatch_metric_alarm" "lambda_duration" {
threshold = each.value.duration_threshold
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
alarm_actions = [data.aws_sns_topic.site_alerts[0].arn]
dimensions = {
FunctionName = each.value.function_name
@ -118,6 +118,8 @@ resource "aws_cloudwatch_metric_alarm" "lambda_duration" {
# ---------------------------------------------------------------------------
resource "aws_cloudwatch_metric_alarm" "orders_read_throttle" {
count = local.is_prod ? 1 : 0
alarm_name = "${local.project}-orders-read-throttle"
alarm_description = "orders table read requests were throttled in the last 5 minutes."
namespace = "AWS/DynamoDB"
@ -128,7 +130,7 @@ resource "aws_cloudwatch_metric_alarm" "orders_read_throttle" {
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
alarm_actions = [data.aws_sns_topic.site_alerts[0].arn]
dimensions = {
TableName = aws_dynamodb_table.orders.name
@ -136,6 +138,8 @@ resource "aws_cloudwatch_metric_alarm" "orders_read_throttle" {
}
resource "aws_cloudwatch_metric_alarm" "orders_write_throttle" {
count = local.is_prod ? 1 : 0
alarm_name = "${local.project}-orders-write-throttle"
alarm_description = "orders table write requests were throttled in the last 5 minutes."
namespace = "AWS/DynamoDB"
@ -146,7 +150,7 @@ resource "aws_cloudwatch_metric_alarm" "orders_write_throttle" {
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
alarm_actions = [data.aws_sns_topic.site_alerts[0].arn]
dimensions = {
TableName = aws_dynamodb_table.orders.name
@ -158,6 +162,8 @@ resource "aws_cloudwatch_metric_alarm" "orders_write_throttle" {
# ---------------------------------------------------------------------------
resource "aws_cloudwatch_metric_alarm" "api_5xx" {
count = local.is_prod ? 1 : 0
alarm_name = "${local.project}-order-api-5xx"
alarm_description = "OrderApi returned one or more 5xx responses in 5 minutes."
namespace = "AWS/ApiGateway"
@ -168,7 +174,7 @@ resource "aws_cloudwatch_metric_alarm" "api_5xx" {
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
alarm_actions = [data.aws_sns_topic.site_alerts[0].arn]
dimensions = {
ApiId = aws_apigatewayv2_api.order_api.id
@ -178,6 +184,8 @@ resource "aws_cloudwatch_metric_alarm" "api_5xx" {
# Threshold raised and 2-of-3 datapoints, to absorb the routine 401s the
# token-based admin authorizer produces without paging.
resource "aws_cloudwatch_metric_alarm" "api_4xx" {
count = local.is_prod ? 1 : 0
alarm_name = "${local.project}-order-api-4xx"
alarm_description = "OrderApi 4xx responses exceeded 20 in 5 minutes (beyond routine auth noise)."
namespace = "AWS/ApiGateway"
@ -189,7 +197,7 @@ resource "aws_cloudwatch_metric_alarm" "api_4xx" {
threshold = 20
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
alarm_actions = [data.aws_sns_topic.site_alerts[0].arn]
dimensions = {
ApiId = aws_apigatewayv2_api.order_api.id
@ -197,6 +205,8 @@ resource "aws_cloudwatch_metric_alarm" "api_4xx" {
}
resource "aws_cloudwatch_metric_alarm" "api_latency" {
count = local.is_prod ? 1 : 0
alarm_name = "${local.project}-order-api-latency"
alarm_description = "OrderApi p99 latency exceeded 3000ms."
namespace = "AWS/ApiGateway"
@ -208,9 +218,34 @@ resource "aws_cloudwatch_metric_alarm" "api_latency" {
threshold = 3000
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
alarm_actions = [data.aws_sns_topic.site_alerts[0].arn]
dimensions = {
ApiId = aws_apigatewayv2_api.order_api.id
}
}
moved {
from = aws_cloudwatch_metric_alarm.orders_read_throttle
to = aws_cloudwatch_metric_alarm.orders_read_throttle[0]
}
moved {
from = aws_cloudwatch_metric_alarm.orders_write_throttle
to = aws_cloudwatch_metric_alarm.orders_write_throttle[0]
}
moved {
from = aws_cloudwatch_metric_alarm.api_5xx
to = aws_cloudwatch_metric_alarm.api_5xx[0]
}
moved {
from = aws_cloudwatch_metric_alarm.api_4xx
to = aws_cloudwatch_metric_alarm.api_4xx[0]
}
moved {
from = aws_cloudwatch_metric_alarm.api_latency
to = aws_cloudwatch_metric_alarm.api_latency[0]
}

View file

@ -114,13 +114,9 @@ resource "aws_lambda_permission" "api_route" {
# Grant API Gateway permission to invoke the admin authorizer Lambda. Source ARN
# is the authorizer itself (not a route), matching the AWS HTTP API docs.
# Import adopts the PLAT-102 live hotfix statement so the first apply does not
# attempt a duplicate AddPermission.
import {
to = aws_lambda_permission.admin_authorizer
id = "meal-order-manager-admin-authorizer/AllowApiGatewayInvokeAuthorizer"
}
# The PLAT-102 live hotfix was imported into meal-order-manager-prod state; do
# not keep a workspace-global import block or seahaven-dev cannot create this
# permission.
resource "aws_lambda_permission" "admin_authorizer" {
statement_id = "AllowApiGatewayInvokeAuthorizer"
action = "lambda:InvokeFunction"

View file

@ -191,12 +191,15 @@ resource "aws_cloudfront_distribution" "form" {
viewer_certificate {
cloudfront_default_certificate = !var.attach_custom_domain
acm_certificate_arn = var.attach_custom_domain ? data.aws_acm_certificate.orders.arn : null
acm_certificate_arn = var.attach_custom_domain ? data.aws_acm_certificate.orders[0].arn : null
ssl_support_method = var.attach_custom_domain ? "sni-only" : null
minimum_protocol_version = var.attach_custom_domain ? "TLSv1.2_2021" : null
}
lifecycle {
# prevent_destroy cannot interpolate. Keep it on so a tagged apply cannot
# destroy the live distribution; tear down a leftover dev distribution
# from the AWS console.
prevent_destroy = true
}
}

View file

@ -6,13 +6,20 @@ data "aws_caller_identity" "current" {}
check "correct_account" {
assert {
condition = data.aws_caller_identity.current.account_id == local.account_id
error_message = "This configuration targets account ${local.account_id}, but the credentials resolve to ${data.aws_caller_identity.current.account_id}."
error_message = "This configuration targets account ${local.account_id} (${var.environment}), but the credentials resolve to ${data.aws_caller_identity.current.account_id}."
}
}
# Alarm sink owned by seahaven-org-baseline, not by this configuration.
# Looked up only in prod because CloudWatch alarms are skipped in dev.
data "aws_sns_topic" "site_alerts" {
name = "site-alerts"
count = local.is_prod ? 1 : 0
name = "site-alerts"
}
moved {
from = data.aws_sns_topic.site_alerts
to = data.aws_sns_topic.site_alerts[0]
}
# Shared CloudFront WAF WebACL (audit M-17), published to Parameter Store by the
@ -40,3 +47,24 @@ check "google_client_id_present" {
error_message = "SSM parameter ${local.google_client_id_param} is missing or empty; create it out of band before apply."
}
}
check "dev_has_no_paychex" {
assert {
condition = local.is_prod || var.checkcomponents_queue_url == ""
error_message = "checkcomponents_queue_url must be empty in non-prod so aggregate-orders cannot send to the prod Paychex queue."
}
}
check "checkcomponents_pair" {
assert {
condition = (var.checkcomponents_queue_url == "") == (var.checkcomponents_queue_arn == "")
error_message = "checkcomponents_queue_url and checkcomponents_queue_arn must both be set or both be empty."
}
}
check "dev_has_no_custom_domain" {
assert {
condition = local.is_prod || !var.attach_custom_domain
error_message = "attach_custom_domain must be false in non-prod; use the CloudFront distribution domain."
}
}

View file

@ -1,14 +1,18 @@
# Single-table store for orders, roster entries and weekly settings.
#
# Deletion protection and point-in-time recovery are both on: this table holds
# the only copy of submitted orders, and a rebuild would lose payroll history.
# Deletion protection and point-in-time recovery are on in prod: this table
# holds the only copy of submitted orders, and a rebuild would lose payroll
# history. Dev is disposable (PITR and deletion protection off).
# lifecycle.prevent_destroy cannot interpolate, so it stays true in both
# environments. Destroy a leftover dev table from the AWS console after
# setting deletion_protection_enabled = false in a follow-up apply.
resource "aws_dynamodb_table" "orders" {
name = local.table_name
billing_mode = "PAY_PER_REQUEST"
hash_key = "PK"
range_key = "SK"
deletion_protection_enabled = true
deletion_protection_enabled = local.is_prod
attribute {
name = "PK"
@ -26,7 +30,7 @@ resource "aws_dynamodb_table" "orders" {
}
point_in_time_recovery {
enabled = true
enabled = local.is_prod
}
lifecycle {

View file

@ -62,7 +62,7 @@ resource "aws_cloudwatch_event_rule" "schedule" {
name = "${local.project}-${each.key}"
description = each.value.description
schedule_expression = each.value.schedule
state = "ENABLED"
state = local.is_prod ? "ENABLED" : "DISABLED"
}
resource "aws_cloudwatch_event_target" "schedule" {

View file

@ -4,14 +4,14 @@
# Live seahaven-hcptf-iam-management DenySelfMutation blocks DetachRolePolicy
# and PutRolePolicy on hcptf-* (including this role). Import apply sequence:
# 1. seahaven-org-baseline scripts/create-hcptf-bootstrap-roles.sh
# --account prod --allow-workspace meal-order-manager-prod
# --account prod|dev --allow-workspace meal-order-manager-<env>
# 2. Point this workspace's TFC_AWS_* at hcptf-bootstrap /
# hcptf-bootstrap-plan (workspace vars, never a project set).
# 3. One Manual apply (import + detach seahaven-hcptf-iam-management +
# put scoped inline).
# 4. Point TFC_AWS_* back at hcptf-meal-order-manager / hcptf-meal-order-manager-plan.
# 5. Re-run the script without --allow-workspace to pin trust back to
# iam-bootstrap-prod only.
# iam-bootstrap-<env> only.
# seahaven-lambda-execution-boundary remains seahaven-lambda-execution-boundary-meal-order-manager.
import {
@ -70,7 +70,7 @@ data "aws_iam_policy_document" "hcptf_apply_trust" {
test = "StringEquals"
variable = "app.terraform.io:sub"
values = [
"organization:seahaven:project:seahaven-prod:workspace:meal-order-manager-prod:run_phase:apply",
"organization:seahaven:project:${local.hcp_project}:workspace:${local.hcp_workspace}:run_phase:apply",
]
}
}
@ -97,7 +97,7 @@ data "aws_iam_policy_document" "hcptf_plan_trust" {
test = "StringEquals"
variable = "app.terraform.io:sub"
values = [
"organization:seahaven:project:seahaven-prod:workspace:meal-order-manager-prod:run_phase:plan",
"organization:seahaven:project:${local.hcp_project}:workspace:${local.hcp_workspace}:run_phase:plan",
]
}
}
@ -181,6 +181,8 @@ data "aws_iam_policy_document" "hcptf_scoped_iam" {
effect = "Allow"
actions = [
"iam:AttachRolePolicy",
"iam:CreateRole",
"iam:DeleteRole",
"iam:DeleteRolePolicy",
"iam:DetachRolePolicy",
"iam:PutRolePolicy",

View file

@ -252,11 +252,14 @@ data "aws_iam_policy_document" "aggregate_orders" {
resources = [aws_lambda_function.slack_notifier.arn]
}
statement {
sid = "CheckcomponentsSend"
effect = "Allow"
actions = ["sqs:SendMessage"]
resources = [var.checkcomponents_queue_arn]
dynamic "statement" {
for_each = var.checkcomponents_queue_arn != "" ? [var.checkcomponents_queue_arn] : []
content {
sid = "CheckcomponentsSend"
effect = "Allow"
actions = ["sqs:SendMessage"]
resources = [statement.value]
}
}
}

View file

@ -1,6 +1,9 @@
locals {
project = "meal-order-manager"
account_id = "011934824531"
project = "meal-order-manager"
is_prod = var.environment == "prod"
account_id = local.is_prod ? "011934824531" : "710827005802"
hcp_project = "seahaven-${var.environment}"
hcp_workspace = "${local.project}-${var.environment}"
# Lambda execution roles under /tf-managed/ carry the per-workload ceiling
# (PLAT-52). githubdeploy-meal-order-manager-weekly-menu must not.

View file

@ -7,11 +7,11 @@ resource "aws_cloudwatch_log_group" "function" {
for_each = local.function_packages
name = "/aws/lambda/${local.project}-${replace(each.key, "_", "-")}"
retention_in_days = 60
retention_in_days = local.is_prod ? 60 : 14
}
# Longer than the Lambda groups on purpose, for request-level forensics.
resource "aws_cloudwatch_log_group" "api_access" {
name = "/aws/apigateway/${local.project}"
retention_in_days = 90
retention_in_days = local.is_prod ? 90 : 14
}

View file

@ -3,9 +3,10 @@ provider "aws" {
default_tags {
tags = {
Project = "meal-order-manager"
ManagedBy = "terraform"
Workspace = "meal-order-manager-prod"
Project = "meal-order-manager"
Environment = var.environment
ManagedBy = "terraform"
Workspace = "meal-order-manager-${var.environment}"
}
}
}

View file

@ -1,25 +1,30 @@
# Values for the meal-order-manager-prod HCP Terraform workspace.
# Workspace variable reference for meal-order-manager HCP workspaces.
# Set these as workspace variables; this file is a reference, not an input.
# HCP workspace stage. Selects account 011934824531 (prod) or 710827005802 (dev).
# Required. meal-order-manager-prod must set "prod" in the same cut as this
# variable is added so a prod plan does not target seahaven-dev.
environment = "prod"
# Region every resource is created in.
aws_region = "us-east-1"
# Custom domain for the order form. An ISSUED ACM certificate for this domain
# must already exist in us-east-1 (see acm.tf). Attach only at DNS cutover.
domain_name = "orders.seahaven.com"
attach_custom_domain = false
# must already exist in us-east-1 (see acm.tf). Keep false in dev.
domain_name = "orders.seahaven.com"
attach_custom_domain = false
# ARN of the out-of-band Secrets Manager secret holding the Slack bot token.
# Only the ARN is used; the value never enters Terraform state.
slack_bot_secret_arn = "arn:aws:secretsmanager:us-east-1:011934824531:secret:meal-order-manager/slack-bot-token-XXXXXX"
# Slack channel that receives meal order notifications. Written to
# /meal-order-manager/slack-channel-id.
# /meal-order-manager/slack-channel-id. Use a sandbox channel in dev.
slack_channel_id = "C00000000000"
# Portal Cognito pools and public app clients accepted by the meals API.
# The primary pair is required. extra_trust lists additional pools so
# portal-dev and portal-prod tokens can both call this prod meals stack.
# portal-dev and portal-prod tokens can both call this meals stack.
portal_cognito_issuer = "https://cognito-idp.us-east-1.amazonaws.com/us-east-1_EXAMPLE"
portal_cognito_audience = "examplepublicappclientid"
portal_cognito_extra_trust = [
@ -28,3 +33,9 @@ portal_cognito_extra_trust = [
audience = "exampleprodappclientid"
},
]
# paychex-checkcomponents SQS. Defaults in variables.tf are the prod queue.
# meal-order-manager-dev must set both to empty so aggregate-orders cannot
# SendMessage to prod Paychex.
# checkcomponents_queue_url = ""
# checkcomponents_queue_arn = ""

View file

@ -4,6 +4,16 @@ variable "aws_region" {
default = "us-east-1"
}
variable "environment" {
description = "HCP workspace stage. Selects account and workspace name."
type = string
validation {
condition = contains(["dev", "prod"], var.environment)
error_message = "environment must be \"dev\" or \"prod\"."
}
}
variable "domain_name" {
description = "Custom domain served by the CloudFront distribution when attach_custom_domain is true. An ISSUED ACM certificate for this domain must already exist in us-east-1 (see acm.tf)."
type = string

View file

@ -20,7 +20,7 @@ terraform {
organization = "seahaven"
workspaces {
name = "meal-order-manager-prod"
tags = ["app:meal-order-manager"]
}
}
}