meal-order-manager/terraform/alarms.tf
Adam Moussa 40ea4ed898
feat(infra): migrate meal-order-manager to HCP Terraform
Freeze SAM CD and add greenfield Terraform for seahaven-prod so HCP is the sole stack deploy path.
2026-08-07 19:19:51 -04:00

222 lines
7.6 KiB
HCL

# CloudWatch alarms. All notify the shared site-alerts topic. No OK actions (no
# recovery spam), and treat_missing_data = notBreaching so cron functions do not
# sit in ALARM between invocations.
#
# Duration alarms use the p99 extended statistic at ~80% of each function's
# timeout. API-fronted functions evaluate 3 datapoints; cron and async-invoked
# functions evaluate one, because they fire too rarely to fill a longer window.
#
# DynamoDB note: the table does not publish ThrottledRequests or SystemErrors at
# the TableName-only dimension, so no alarm on those would ever evaluate.
# ReadThrottleEvents and WriteThrottleEvents do carry TableName and are used
# here for throttle coverage.
locals {
alarm_functions = {
"submit-order" = {
function_name = aws_lambda_function.submit_order.function_name
duration_threshold = 8000
duration_timeout = "10s"
duration_datapoints = 3
}
"admin-authorizer" = {
function_name = aws_lambda_function.admin_authorizer.function_name
duration_threshold = 8000
duration_timeout = "10s"
duration_datapoints = 3
}
"close-form" = {
function_name = aws_lambda_function.close_form.function_name
duration_threshold = 24000
duration_timeout = "30s"
duration_datapoints = 1
}
"aggregate-orders" = {
function_name = aws_lambda_function.aggregate_orders.function_name
duration_threshold = 48000
duration_timeout = "60s"
duration_datapoints = 1
}
"slack-notifier" = {
function_name = aws_lambda_function.slack_notifier.function_name
duration_threshold = 24000
duration_timeout = "30s"
duration_datapoints = 1
}
"sync-roster" = {
function_name = aws_lambda_function.sync_roster.function_name
duration_threshold = 48000
duration_timeout = "60s"
duration_datapoints = 1
}
"email-report" = {
function_name = aws_lambda_function.email_report.function_name
duration_threshold = 24000
duration_timeout = "30s"
duration_datapoints = 1
}
}
}
resource "aws_cloudwatch_metric_alarm" "lambda_errors" {
for_each = local.alarm_functions
alarm_name = "${local.project}-${each.key}-errors"
alarm_description = "${each.key} Lambda reported one or more errors in 5 minutes."
namespace = "AWS/Lambda"
metric_name = "Errors"
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
dimensions = {
FunctionName = each.value.function_name
}
}
resource "aws_cloudwatch_metric_alarm" "lambda_throttles" {
for_each = local.alarm_functions
alarm_name = "${local.project}-${each.key}-throttles"
alarm_description = "${each.key} Lambda was throttled in the last 5 minutes."
namespace = "AWS/Lambda"
metric_name = "Throttles"
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
dimensions = {
FunctionName = each.value.function_name
}
}
resource "aws_cloudwatch_metric_alarm" "lambda_duration" {
for_each = local.alarm_functions
alarm_name = "${local.project}-${each.key}-duration"
alarm_description = "${each.key} p99 duration exceeded ${each.value.duration_threshold}ms (80% of its ${each.value.duration_timeout} timeout)."
namespace = "AWS/Lambda"
metric_name = "Duration"
extended_statistic = "p99"
period = 300
evaluation_periods = each.value.duration_datapoints
datapoints_to_alarm = each.value.duration_datapoints
threshold = each.value.duration_threshold
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
dimensions = {
FunctionName = each.value.function_name
}
}
# ---------------------------------------------------------------------------
# DynamoDB
# ---------------------------------------------------------------------------
resource "aws_cloudwatch_metric_alarm" "orders_read_throttle" {
alarm_name = "${local.project}-orders-read-throttle"
alarm_description = "orders table read requests were throttled in the last 5 minutes."
namespace = "AWS/DynamoDB"
metric_name = "ReadThrottleEvents"
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
dimensions = {
TableName = aws_dynamodb_table.orders.name
}
}
resource "aws_cloudwatch_metric_alarm" "orders_write_throttle" {
alarm_name = "${local.project}-orders-write-throttle"
alarm_description = "orders table write requests were throttled in the last 5 minutes."
namespace = "AWS/DynamoDB"
metric_name = "WriteThrottleEvents"
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
dimensions = {
TableName = aws_dynamodb_table.orders.name
}
}
# ---------------------------------------------------------------------------
# HTTP API
# ---------------------------------------------------------------------------
resource "aws_cloudwatch_metric_alarm" "api_5xx" {
alarm_name = "${local.project}-order-api-5xx"
alarm_description = "OrderApi returned one or more 5xx responses in 5 minutes."
namespace = "AWS/ApiGateway"
metric_name = "5xx"
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
dimensions = {
ApiId = aws_apigatewayv2_api.order_api.id
}
}
# Threshold raised and 2-of-3 datapoints, to absorb the routine 401s the
# token-based admin authorizer produces without paging.
resource "aws_cloudwatch_metric_alarm" "api_4xx" {
alarm_name = "${local.project}-order-api-4xx"
alarm_description = "OrderApi 4xx responses exceeded 20 in 5 minutes (beyond routine auth noise)."
namespace = "AWS/ApiGateway"
metric_name = "4xx"
statistic = "Sum"
period = 300
evaluation_periods = 3
datapoints_to_alarm = 2
threshold = 20
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
dimensions = {
ApiId = aws_apigatewayv2_api.order_api.id
}
}
resource "aws_cloudwatch_metric_alarm" "api_latency" {
alarm_name = "${local.project}-order-api-latency"
alarm_description = "OrderApi p99 latency exceeded 3000ms."
namespace = "AWS/ApiGateway"
metric_name = "Latency"
extended_statistic = "p99"
period = 300
evaluation_periods = 3
datapoints_to_alarm = 3
threshold = 3000
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [data.aws_sns_topic.site_alerts.arn]
dimensions = {
ApiId = aws_apigatewayv2_api.order_api.id
}
}