afterhours-shift-manager/terraform/alarms.tf
Adam Moussa 26adb8e6c0
feat(api): collapse Slack, portal, and jobs onto Fargate (PLAT-216) (#259)
* feat(api): collapse Slack, portal, and jobs onto Fargate (PLAT-216)

Move HTTP and scheduled work onto one always-on Flask task so after-hours
loses Lambda cold start without changing the Cognito or roster contracts.

* fix(portal-api): keep CORS headers on unexpected 500s

Portal SPA error handling needs Access-Control-Allow-Origin even when
DynamoDB or other internals fail, otherwise the browser hides the 500.

* fix(api): retarget holidays per account and ship App Home changelog (PLAT-216)

* fix(iam): list ECS tasks and fail closed on non-prod Paychex (PLAT-216)

* fix(portal-api): serve portal JSON with an explicit JSON content type
2026-09-21 19:13:30 +00:00

205 lines
7.6 KiB
HCL

locals {
lambda_alarm_matrix = {
errors = {
metric_name = "Errors"
statistic = "Sum"
evaluation_periods = 1
datapoints_to_alarm = 1
threshold = 1
comparison = "GreaterThanOrEqualToThreshold"
period = 300
}
throttles = {
metric_name = "Throttles"
statistic = "Sum"
evaluation_periods = 1
datapoints_to_alarm = 1
threshold = 1
comparison = "GreaterThanOrEqualToThreshold"
period = 300
}
}
lambda_alarms = {
for pair in flatten([
for fn_key, fn in local.functions : [
for metric_key, metric in local.lambda_alarm_matrix : {
key = "${fn_key}-${metric_key}"
fn_key = fn_key
function = fn.function_name
metric_key = metric_key
metric_name = metric.metric_name
statistic = metric.statistic
evaluation = metric.evaluation_periods
datapoints = metric.datapoints_to_alarm
threshold = metric.threshold
comparison = metric.comparison
period = metric.period
description = metric_key == "errors" ? "${fn.function_name} reported one or more errors" : "${fn.function_name} was throttled (concurrency limit hit)"
}
]
]) : pair.key => pair
}
}
resource "aws_cloudwatch_metric_alarm" "lambda_errors_throttles" {
for_each = local.lambda_alarms
alarm_name = "Lambda-${title(each.value.metric_key)}-${each.value.function}"
alarm_description = each.value.description
namespace = "AWS/Lambda"
metric_name = each.value.metric_name
dimensions = { FunctionName = each.value.function }
statistic = each.value.statistic
period = each.value.period
evaluation_periods = each.value.evaluation
datapoints_to_alarm = each.value.datapoints
threshold = each.value.threshold
comparison_operator = each.value.comparison
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "lambda_duration" {
for_each = local.functions
alarm_name = "Lambda-Duration-${each.value.function_name}"
alarm_description = "${each.value.function_name} duration approaching its ${each.value.timeout}s timeout (>=${each.value.duration_ms}ms)"
namespace = "AWS/Lambda"
metric_name = "Duration"
dimensions = { FunctionName = each.value.function_name }
statistic = "Maximum"
period = 300
evaluation_periods = 3
datapoints_to_alarm = 2
threshold = each.value.duration_ms
comparison_operator = "GreaterThanOrEqualToThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "ddb_read_throttle" {
alarm_name = "DDB-ReadThrottle-${local.table_name}"
alarm_description = "afterhours-shifts table had one or more read throttle events"
namespace = "AWS/DynamoDB"
metric_name = "ReadThrottleEvents"
dimensions = { TableName = aws_dynamodb_table.shifts.name }
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "ddb_write_throttle" {
alarm_name = "DDB-WriteThrottle-${local.table_name}"
alarm_description = "afterhours-shifts table had one or more write throttle events"
namespace = "AWS/DynamoDB"
metric_name = "WriteThrottleEvents"
dimensions = { TableName = aws_dynamodb_table.shifts.name }
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "api_4xx" {
alarm_name = "ApiGateway-4xx-${aws_apigatewayv2_api.http.id}"
alarm_description = "Elevated 4xx responses on the afterhours HTTP API"
namespace = "AWS/ApiGateway"
metric_name = "4xx"
dimensions = { ApiId = aws_apigatewayv2_api.http.id }
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 5
comparison_operator = "GreaterThanOrEqualToThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "api_5xx" {
alarm_name = "ApiGateway-5xx-${aws_apigatewayv2_api.http.id}"
alarm_description = "5xx responses on the afterhours HTTP API"
namespace = "AWS/ApiGateway"
metric_name = "5xx"
dimensions = { ApiId = aws_apigatewayv2_api.http.id }
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 1
comparison_operator = "GreaterThanOrEqualToThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "api_latency" {
alarm_name = "ApiGateway-Latency-${aws_apigatewayv2_api.http.id}"
alarm_description = "p99 latency on the afterhours HTTP API exceeded 3s"
namespace = "AWS/ApiGateway"
metric_name = "Latency"
dimensions = { ApiId = aws_apigatewayv2_api.http.id }
extended_statistic = "p99"
period = 300
evaluation_periods = 3
datapoints_to_alarm = 2
threshold = 3000
comparison_operator = "GreaterThanOrEqualToThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "alb_5xx" {
alarm_name = "ALB-5xx-${local.project}"
alarm_description = "ALB 5xx from afterhours-shift-manager"
namespace = "AWS/ApplicationELB"
metric_name = "HTTPCode_Target_5XX_Count"
dimensions = { LoadBalancer = aws_lb.api.arn_suffix }
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "alb_latency" {
alarm_name = "ALB-Latency-${local.project}"
alarm_description = "p99 target response time on the afterhours ALB exceeded 3s"
namespace = "AWS/ApplicationELB"
metric_name = "TargetResponseTime"
dimensions = { LoadBalancer = aws_lb.api.arn_suffix }
extended_statistic = "p99"
period = 300
evaluation_periods = 3
datapoints_to_alarm = 2
threshold = 3
comparison_operator = "GreaterThanOrEqualToThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "alb_unhealthy_hosts" {
alarm_name = "ALB-UnhealthyHost-${local.project}"
alarm_description = "Unhealthy Fargate targets on the afterhours ALB"
namespace = "AWS/ApplicationELB"
metric_name = "UnHealthyHostCount"
dimensions = {
LoadBalancer = aws_lb.api.arn_suffix
TargetGroup = aws_lb_target_group.api.arn_suffix
}
statistic = "Maximum"
period = 60
evaluation_periods = 3
datapoints_to_alarm = 3
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}