locals { lambda_alarm_matrix = { errors = { metric_name = "Errors" statistic = "Sum" evaluation_periods = 1 datapoints_to_alarm = 1 threshold = 1 comparison = "GreaterThanOrEqualToThreshold" period = 300 } throttles = { metric_name = "Throttles" statistic = "Sum" evaluation_periods = 1 datapoints_to_alarm = 1 threshold = 1 comparison = "GreaterThanOrEqualToThreshold" period = 300 } } lambda_alarms = { for pair in flatten([ for fn_key, fn in local.functions : [ for metric_key, metric in local.lambda_alarm_matrix : { key = "${fn_key}-${metric_key}" fn_key = fn_key function = fn.function_name metric_key = metric_key metric_name = metric.metric_name statistic = metric.statistic evaluation = metric.evaluation_periods datapoints = metric.datapoints_to_alarm threshold = metric.threshold comparison = metric.comparison period = metric.period description = metric_key == "errors" ? "${fn.function_name} reported one or more errors" : "${fn.function_name} was throttled (concurrency limit hit)" } ] ]) : pair.key => pair } } resource "aws_cloudwatch_metric_alarm" "lambda_errors_throttles" { for_each = local.lambda_alarms alarm_name = "Lambda-${title(each.value.metric_key)}-${each.value.function}" alarm_description = each.value.description namespace = "AWS/Lambda" metric_name = each.value.metric_name dimensions = { FunctionName = each.value.function } statistic = each.value.statistic period = each.value.period evaluation_periods = each.value.evaluation datapoints_to_alarm = each.value.datapoints threshold = each.value.threshold comparison_operator = each.value.comparison treat_missing_data = "notBreaching" alarm_actions = [local.site_alerts_arn] } resource "aws_cloudwatch_metric_alarm" "lambda_duration" { for_each = local.functions alarm_name = "Lambda-Duration-${each.value.function_name}" alarm_description = "${each.value.function_name} duration approaching its ${each.value.timeout}s timeout (>=${each.value.duration_ms}ms)" namespace = "AWS/Lambda" metric_name = "Duration" dimensions = { FunctionName = each.value.function_name } statistic = "Maximum" period = 300 evaluation_periods = 3 datapoints_to_alarm = 2 threshold = each.value.duration_ms comparison_operator = "GreaterThanOrEqualToThreshold" treat_missing_data = "notBreaching" alarm_actions = [local.site_alerts_arn] } resource "aws_cloudwatch_metric_alarm" "ddb_read_throttle" { alarm_name = "DDB-ReadThrottle-${local.table_name}" alarm_description = "afterhours-shifts table had one or more read throttle events" namespace = "AWS/DynamoDB" metric_name = "ReadThrottleEvents" dimensions = { TableName = aws_dynamodb_table.shifts.name } statistic = "Sum" period = 300 evaluation_periods = 1 threshold = 0 comparison_operator = "GreaterThanThreshold" treat_missing_data = "notBreaching" alarm_actions = [local.site_alerts_arn] } resource "aws_cloudwatch_metric_alarm" "ddb_write_throttle" { alarm_name = "DDB-WriteThrottle-${local.table_name}" alarm_description = "afterhours-shifts table had one or more write throttle events" namespace = "AWS/DynamoDB" metric_name = "WriteThrottleEvents" dimensions = { TableName = aws_dynamodb_table.shifts.name } statistic = "Sum" period = 300 evaluation_periods = 1 threshold = 0 comparison_operator = "GreaterThanThreshold" treat_missing_data = "notBreaching" alarm_actions = [local.site_alerts_arn] } resource "aws_cloudwatch_metric_alarm" "api_4xx" { alarm_name = "ApiGateway-4xx-${aws_apigatewayv2_api.http.id}" alarm_description = "Elevated 4xx responses on the afterhours HTTP API" namespace = "AWS/ApiGateway" metric_name = "4xx" dimensions = { ApiId = aws_apigatewayv2_api.http.id } statistic = "Sum" period = 300 evaluation_periods = 1 threshold = 5 comparison_operator = "GreaterThanOrEqualToThreshold" treat_missing_data = "notBreaching" alarm_actions = [local.site_alerts_arn] } resource "aws_cloudwatch_metric_alarm" "api_5xx" { alarm_name = "ApiGateway-5xx-${aws_apigatewayv2_api.http.id}" alarm_description = "5xx responses on the afterhours HTTP API" namespace = "AWS/ApiGateway" metric_name = "5xx" dimensions = { ApiId = aws_apigatewayv2_api.http.id } statistic = "Sum" period = 300 evaluation_periods = 1 threshold = 1 comparison_operator = "GreaterThanOrEqualToThreshold" treat_missing_data = "notBreaching" alarm_actions = [local.site_alerts_arn] } resource "aws_cloudwatch_metric_alarm" "api_latency" { alarm_name = "ApiGateway-Latency-${aws_apigatewayv2_api.http.id}" alarm_description = "p99 latency on the afterhours HTTP API exceeded 3s" namespace = "AWS/ApiGateway" metric_name = "Latency" dimensions = { ApiId = aws_apigatewayv2_api.http.id } extended_statistic = "p99" period = 300 evaluation_periods = 3 datapoints_to_alarm = 2 threshold = 3000 comparison_operator = "GreaterThanOrEqualToThreshold" treat_missing_data = "notBreaching" alarm_actions = [local.site_alerts_arn] } resource "aws_cloudwatch_metric_alarm" "alb_5xx" { alarm_name = "ALB-5xx-${local.project}" alarm_description = "ALB 5xx from afterhours-shift-manager" namespace = "AWS/ApplicationELB" metric_name = "HTTPCode_Target_5XX_Count" dimensions = { LoadBalancer = aws_lb.api.arn_suffix } statistic = "Sum" period = 300 evaluation_periods = 1 threshold = 0 comparison_operator = "GreaterThanThreshold" treat_missing_data = "notBreaching" alarm_actions = [local.site_alerts_arn] } resource "aws_cloudwatch_metric_alarm" "alb_latency" { alarm_name = "ALB-Latency-${local.project}" alarm_description = "p99 target response time on the afterhours ALB exceeded 3s" namespace = "AWS/ApplicationELB" metric_name = "TargetResponseTime" dimensions = { LoadBalancer = aws_lb.api.arn_suffix } extended_statistic = "p99" period = 300 evaluation_periods = 3 datapoints_to_alarm = 2 threshold = 3 comparison_operator = "GreaterThanOrEqualToThreshold" treat_missing_data = "notBreaching" alarm_actions = [local.site_alerts_arn] } resource "aws_cloudwatch_metric_alarm" "alb_unhealthy_hosts" { alarm_name = "ALB-UnhealthyHost-${local.project}" alarm_description = "Unhealthy Fargate targets on the afterhours ALB" namespace = "AWS/ApplicationELB" metric_name = "UnHealthyHostCount" dimensions = { LoadBalancer = aws_lb.api.arn_suffix TargetGroup = aws_lb_target_group.api.arn_suffix } statistic = "Maximum" period = 60 evaluation_periods = 3 datapoints_to_alarm = 3 threshold = 0 comparison_operator = "GreaterThanThreshold" treat_missing_data = "notBreaching" alarm_actions = [local.site_alerts_arn] }