# CloudWatch alarms. All notify the shared site-alerts topic. No OK actions (no # recovery spam), and treat_missing_data = notBreaching so cron functions do not # sit in ALARM between invocations. # # Duration alarms use the p99 extended statistic at ~80% of each function's # timeout. API-fronted functions evaluate 3 datapoints; cron and async-invoked # functions evaluate one, because they fire too rarely to fill a longer window. # # DynamoDB note: the table does not publish ThrottledRequests or SystemErrors at # the TableName-only dimension, so no alarm on those would ever evaluate. # ReadThrottleEvents and WriteThrottleEvents do carry TableName and are used # here for throttle coverage. locals { alarm_functions = { "submit-order" = { function_name = aws_lambda_function.submit_order.function_name duration_threshold = 8000 duration_timeout = "10s" duration_datapoints = 3 } "admin-authorizer" = { function_name = aws_lambda_function.admin_authorizer.function_name duration_threshold = 8000 duration_timeout = "10s" duration_datapoints = 3 } "close-form" = { function_name = aws_lambda_function.close_form.function_name duration_threshold = 24000 duration_timeout = "30s" duration_datapoints = 1 } "aggregate-orders" = { function_name = aws_lambda_function.aggregate_orders.function_name duration_threshold = 48000 duration_timeout = "60s" duration_datapoints = 1 } "slack-notifier" = { function_name = aws_lambda_function.slack_notifier.function_name duration_threshold = 24000 duration_timeout = "30s" duration_datapoints = 1 } "sync-roster" = { function_name = aws_lambda_function.sync_roster.function_name duration_threshold = 48000 duration_timeout = "60s" duration_datapoints = 1 } } } resource "aws_cloudwatch_metric_alarm" "lambda_errors" { for_each = local.alarm_functions alarm_name = "${local.project}-${each.key}-errors" alarm_description = "${each.key} Lambda reported one or more errors in 5 minutes." namespace = "AWS/Lambda" metric_name = "Errors" statistic = "Sum" period = 300 evaluation_periods = 1 threshold = 0 comparison_operator = "GreaterThanThreshold" treat_missing_data = "notBreaching" alarm_actions = [data.aws_sns_topic.site_alerts.arn] dimensions = { FunctionName = each.value.function_name } } resource "aws_cloudwatch_metric_alarm" "lambda_throttles" { for_each = local.alarm_functions alarm_name = "${local.project}-${each.key}-throttles" alarm_description = "${each.key} Lambda was throttled in the last 5 minutes." namespace = "AWS/Lambda" metric_name = "Throttles" statistic = "Sum" period = 300 evaluation_periods = 1 threshold = 0 comparison_operator = "GreaterThanThreshold" treat_missing_data = "notBreaching" alarm_actions = [data.aws_sns_topic.site_alerts.arn] dimensions = { FunctionName = each.value.function_name } } resource "aws_cloudwatch_metric_alarm" "lambda_duration" { for_each = local.alarm_functions alarm_name = "${local.project}-${each.key}-duration" alarm_description = "${each.key} p99 duration exceeded ${each.value.duration_threshold}ms (80% of its ${each.value.duration_timeout} timeout)." namespace = "AWS/Lambda" metric_name = "Duration" extended_statistic = "p99" period = 300 evaluation_periods = each.value.duration_datapoints datapoints_to_alarm = each.value.duration_datapoints threshold = each.value.duration_threshold comparison_operator = "GreaterThanThreshold" treat_missing_data = "notBreaching" alarm_actions = [data.aws_sns_topic.site_alerts.arn] dimensions = { FunctionName = each.value.function_name } } # --------------------------------------------------------------------------- # DynamoDB # --------------------------------------------------------------------------- resource "aws_cloudwatch_metric_alarm" "orders_read_throttle" { alarm_name = "${local.project}-orders-read-throttle" alarm_description = "orders table read requests were throttled in the last 5 minutes." namespace = "AWS/DynamoDB" metric_name = "ReadThrottleEvents" statistic = "Sum" period = 300 evaluation_periods = 1 threshold = 0 comparison_operator = "GreaterThanThreshold" treat_missing_data = "notBreaching" alarm_actions = [data.aws_sns_topic.site_alerts.arn] dimensions = { TableName = aws_dynamodb_table.orders.name } } resource "aws_cloudwatch_metric_alarm" "orders_write_throttle" { alarm_name = "${local.project}-orders-write-throttle" alarm_description = "orders table write requests were throttled in the last 5 minutes." namespace = "AWS/DynamoDB" metric_name = "WriteThrottleEvents" statistic = "Sum" period = 300 evaluation_periods = 1 threshold = 0 comparison_operator = "GreaterThanThreshold" treat_missing_data = "notBreaching" alarm_actions = [data.aws_sns_topic.site_alerts.arn] dimensions = { TableName = aws_dynamodb_table.orders.name } } # --------------------------------------------------------------------------- # HTTP API # --------------------------------------------------------------------------- resource "aws_cloudwatch_metric_alarm" "api_5xx" { alarm_name = "${local.project}-order-api-5xx" alarm_description = "OrderApi returned one or more 5xx responses in 5 minutes." namespace = "AWS/ApiGateway" metric_name = "5xx" statistic = "Sum" period = 300 evaluation_periods = 1 threshold = 0 comparison_operator = "GreaterThanThreshold" treat_missing_data = "notBreaching" alarm_actions = [data.aws_sns_topic.site_alerts.arn] dimensions = { ApiId = aws_apigatewayv2_api.order_api.id } } # Threshold raised and 2-of-3 datapoints, to absorb the routine 401s the # token-based admin authorizer produces without paging. resource "aws_cloudwatch_metric_alarm" "api_4xx" { alarm_name = "${local.project}-order-api-4xx" alarm_description = "OrderApi 4xx responses exceeded 20 in 5 minutes (beyond routine auth noise)." namespace = "AWS/ApiGateway" metric_name = "4xx" statistic = "Sum" period = 300 evaluation_periods = 3 datapoints_to_alarm = 2 threshold = 20 comparison_operator = "GreaterThanThreshold" treat_missing_data = "notBreaching" alarm_actions = [data.aws_sns_topic.site_alerts.arn] dimensions = { ApiId = aws_apigatewayv2_api.order_api.id } } resource "aws_cloudwatch_metric_alarm" "api_latency" { alarm_name = "${local.project}-order-api-latency" alarm_description = "OrderApi p99 latency exceeded 3000ms." namespace = "AWS/ApiGateway" metric_name = "Latency" extended_statistic = "p99" period = 300 evaluation_periods = 3 datapoints_to_alarm = 3 threshold = 3000 comparison_operator = "GreaterThanThreshold" treat_missing_data = "notBreaching" alarm_actions = [data.aws_sns_topic.site_alerts.arn] dimensions = { ApiId = aws_apigatewayv2_api.order_api.id } }