syslog-server/terraform/alarms.tf
Adam Moussa ace0a29e25
fix(infra): keep no-logs alarm quiet until UniFi cutover
The new prod unifi-syslog group is empty until devices are re-pointed, so treat_missing_data=breaching would page site-alerts on first apply.
2026-09-16 17:25:53 -04:00

56 lines
1.9 KiB
HCL

resource "aws_cloudwatch_metric_alarm" "no_incoming_logs" {
alarm_name = "Syslog-NoIncomingLogs"
alarm_description = "No log events delivered to unifi-syslog for 2 days — syslog pipeline may be down."
comparison_operator = "LessThanThreshold"
evaluation_periods = 2
metric_name = "IncomingLogEvents"
namespace = "AWS/Logs"
period = 86400
statistic = "Sum"
threshold = 1
treat_missing_data = var.no_logs_treat_missing_data
alarm_actions = [local.site_alerts_arn]
dimensions = {
LogGroupName = local.log_group_name
}
}
resource "aws_cloudwatch_metric_alarm" "status_check" {
alarm_name = "EC2-StatusCheck-syslog-server"
alarm_description = "syslog-server EC2 status check failed (instance and/or system) for 10 min — host may be hung or unreachable."
comparison_operator = "GreaterThanOrEqualToThreshold"
evaluation_periods = 2
metric_name = "StatusCheckFailed"
namespace = "AWS/EC2"
period = 300
statistic = "Maximum"
threshold = 1
treat_missing_data = "breaching"
alarm_actions = [local.site_alerts_arn]
dimensions = {
InstanceId = aws_instance.this.id
}
}
resource "aws_cloudwatch_metric_alarm" "system_recover" {
alarm_name = "EC2-StatusCheckSystem-syslog-server-recover"
alarm_description = "syslog-server EC2 system status check failed — underlying host impaired; auto-recovering onto new hardware."
comparison_operator = "GreaterThanOrEqualToThreshold"
evaluation_periods = 2
metric_name = "StatusCheckFailed_System"
namespace = "AWS/EC2"
period = 300
statistic = "Maximum"
threshold = 1
treat_missing_data = "notBreaching"
alarm_actions = [
local.site_alerts_arn,
"arn:aws:automate:${var.aws_region}:ec2:recover",
]
dimensions = {
InstanceId = aws_instance.this.id
}
}