mirror of
https://github.com/Sea-Haven-Industries/syslog-server.git
synced 2026-09-30 03:03:14 +00:00
* feat(infra): migrate syslog-server to HCP Terraform Replace the mgmt CDK stack with a seahaven-prod HCP workspace so the collector is owned by Terraform before UniFi cutover. * fix(infra): keep no-logs alarm quiet until UniFi cutover The new prod unifi-syslog group is empty until devices are re-pointed, so treat_missing_data=breaching would page site-alerts on first apply. * fix(infra): allow scoped apply to modify SG rules in place Authorize/Revoke plus description updates are not enough for aws_vpc_security_group_*_rule in-place changes after the bootstrap window.
56 lines
1.9 KiB
HCL
56 lines
1.9 KiB
HCL
resource "aws_cloudwatch_metric_alarm" "no_incoming_logs" {
|
|
alarm_name = "Syslog-NoIncomingLogs"
|
|
alarm_description = "No log events delivered to unifi-syslog for 2 days — syslog pipeline may be down."
|
|
comparison_operator = "LessThanThreshold"
|
|
evaluation_periods = 2
|
|
metric_name = "IncomingLogEvents"
|
|
namespace = "AWS/Logs"
|
|
period = 86400
|
|
statistic = "Sum"
|
|
threshold = 1
|
|
treat_missing_data = var.no_logs_treat_missing_data
|
|
alarm_actions = [local.site_alerts_arn]
|
|
|
|
dimensions = {
|
|
LogGroupName = local.log_group_name
|
|
}
|
|
}
|
|
|
|
resource "aws_cloudwatch_metric_alarm" "status_check" {
|
|
alarm_name = "EC2-StatusCheck-syslog-server"
|
|
alarm_description = "syslog-server EC2 status check failed (instance and/or system) for 10 min — host may be hung or unreachable."
|
|
comparison_operator = "GreaterThanOrEqualToThreshold"
|
|
evaluation_periods = 2
|
|
metric_name = "StatusCheckFailed"
|
|
namespace = "AWS/EC2"
|
|
period = 300
|
|
statistic = "Maximum"
|
|
threshold = 1
|
|
treat_missing_data = "breaching"
|
|
alarm_actions = [local.site_alerts_arn]
|
|
|
|
dimensions = {
|
|
InstanceId = aws_instance.this.id
|
|
}
|
|
}
|
|
|
|
resource "aws_cloudwatch_metric_alarm" "system_recover" {
|
|
alarm_name = "EC2-StatusCheckSystem-syslog-server-recover"
|
|
alarm_description = "syslog-server EC2 system status check failed — underlying host impaired; auto-recovering onto new hardware."
|
|
comparison_operator = "GreaterThanOrEqualToThreshold"
|
|
evaluation_periods = 2
|
|
metric_name = "StatusCheckFailed_System"
|
|
namespace = "AWS/EC2"
|
|
period = 300
|
|
statistic = "Maximum"
|
|
threshold = 1
|
|
treat_missing_data = "notBreaching"
|
|
alarm_actions = [
|
|
local.site_alerts_arn,
|
|
"arn:aws:automate:${var.aws_region}:ec2:recover",
|
|
]
|
|
|
|
dimensions = {
|
|
InstanceId = aws_instance.this.id
|
|
}
|
|
}
|