syslog-server/terraform/alarms.tf

75 lines
2.6 KiB
Terraform
Raw Normal View History

resource "aws_cloudwatch_metric_alarm" "no_incoming_records" {
alarm_name = "Syslog-NoIncomingRecords"
alarm_description = "No Firehose IncomingRecords for 2 days. UniFi pipeline may be down."
comparison_operator = "LessThanThreshold"
evaluation_periods = 2
metric_name = "IncomingRecords"
namespace = "AWS/Firehose"
period = 86400
statistic = "Sum"
threshold = 1
treat_missing_data = var.no_logs_treat_missing_data
alarm_actions = [local.site_alerts_arn]
dimensions = {
DeliveryStreamName = aws_kinesis_firehose_delivery_stream.unifi.name
}
}
resource "aws_cloudwatch_metric_alarm" "firehose_delivery" {
alarm_name = "Syslog-FirehoseDeliveryFailed"
alarm_description = "Firehose DeliveryToS3.Success average below 1 for 10 min. S3 PUTs are failing."
comparison_operator = "LessThanThreshold"
evaluation_periods = 2
metric_name = "DeliveryToS3.Success"
namespace = "AWS/Firehose"
period = 300
statistic = "Average"
threshold = 1
treat_missing_data = "notBreaching"
alarm_actions = [local.site_alerts_arn]
dimensions = {
DeliveryStreamName = aws_kinesis_firehose_delivery_stream.unifi.name
}
}
resource "aws_cloudwatch_metric_alarm" "status_check" {
alarm_name = "EC2-StatusCheck-syslog-server"
alarm_description = "syslog-server EC2 status check failed (instance and/or system) for 10 min. Host may be hung or unreachable."
comparison_operator = "GreaterThanOrEqualToThreshold"
evaluation_periods = 2
metric_name = "StatusCheckFailed"
namespace = "AWS/EC2"
period = 300
statistic = "Maximum"
threshold = 1
treat_missing_data = "breaching"
alarm_actions = [local.site_alerts_arn]
dimensions = {
InstanceId = aws_instance.this.id
}
}
resource "aws_cloudwatch_metric_alarm" "system_recover" {
alarm_name = "EC2-StatusCheckSystem-syslog-server-recover"
alarm_description = "syslog-server EC2 system status check failed. Underlying host impaired; auto-recovering onto new hardware."
comparison_operator = "GreaterThanOrEqualToThreshold"
evaluation_periods = 2
metric_name = "StatusCheckFailed_System"
namespace = "AWS/EC2"
period = 300
statistic = "Maximum"
threshold = 1
treat_missing_data = "notBreaching"
alarm_actions = [
local.site_alerts_arn,
"arn:aws:automate:${var.aws_region}:ec2:recover",
]
dimensions = {
InstanceId = aws_instance.this.id
}
}