afterhours-shift-manager/terraform/alarms.tf
Adam Moussa cbda46ba87
chore(infra): remove API Gateway and Lambda dual-run (PLAT-216) (#264)
Origins already point at the Fargate hostnames. Drop the HTTP API, eight
functions, zip CD, and Lambda/API Gateway alarms while keeping leftover
Lambda IAM so Paychex can still name weekly-post.
2026-09-21 22:37:13 +00:00

79 lines
3 KiB
HCL

resource "aws_cloudwatch_metric_alarm" "ddb_read_throttle" {
alarm_name = "DDB-ReadThrottle-${local.table_name}"
alarm_description = "afterhours-shifts table had one or more read throttle events"
namespace = "AWS/DynamoDB"
metric_name = "ReadThrottleEvents"
dimensions = { TableName = aws_dynamodb_table.shifts.name }
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "ddb_write_throttle" {
alarm_name = "DDB-WriteThrottle-${local.table_name}"
alarm_description = "afterhours-shifts table had one or more write throttle events"
namespace = "AWS/DynamoDB"
metric_name = "WriteThrottleEvents"
dimensions = { TableName = aws_dynamodb_table.shifts.name }
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "alb_5xx" {
alarm_name = "ALB-5xx-${local.project}"
alarm_description = "ALB 5xx from afterhours-shift-manager"
namespace = "AWS/ApplicationELB"
metric_name = "HTTPCode_Target_5XX_Count"
dimensions = { LoadBalancer = aws_lb.api.arn_suffix }
statistic = "Sum"
period = 300
evaluation_periods = 1
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "alb_latency" {
alarm_name = "ALB-Latency-${local.project}"
alarm_description = "p99 target response time on the afterhours ALB exceeded 3s"
namespace = "AWS/ApplicationELB"
metric_name = "TargetResponseTime"
dimensions = { LoadBalancer = aws_lb.api.arn_suffix }
extended_statistic = "p99"
period = 300
evaluation_periods = 3
datapoints_to_alarm = 2
threshold = 3
comparison_operator = "GreaterThanOrEqualToThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}
resource "aws_cloudwatch_metric_alarm" "alb_unhealthy_hosts" {
alarm_name = "ALB-UnhealthyHost-${local.project}"
alarm_description = "Unhealthy Fargate targets on the afterhours ALB"
namespace = "AWS/ApplicationELB"
metric_name = "UnHealthyHostCount"
dimensions = {
LoadBalancer = aws_lb.api.arn_suffix
TargetGroup = aws_lb_target_group.api.arn_suffix
}
statistic = "Maximum"
period = 60
evaluation_periods = 3
datapoints_to_alarm = 3
threshold = 0
comparison_operator = "GreaterThanThreshold"
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
}