From 6c34bf7e66baa1816cba8949eead1a29df426071 Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Wed, 17 Jun 2026 15:02:04 -0400 Subject: [PATCH] feat(syslog-server): IaC-managed EC2 StatusCheckFailed alarm + auto-recovery (INFRA-58) (#4) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add a CloudFormation-managed EC2 status-check alarm for the live syslog instance, replacing the orphaned EC2-StatusCheck-syslog-server alarm that still points at the terminated i-0a7470914f0151b97. The instance is a CFN resource in this stack, so both alarms dimension on instance.instanceId (Ref) rather than a literal id — they follow the instance across future replacements (e.g. userDataCausesReplacement). - EC2-StatusCheck-syslog-server: combined StatusCheckFailed, Maximum >= 1, 300s period, 2 eval periods, treatMissingData=breaching, SNS -> site-alerts. Mirrors the existing Syslog-NoIncomingLogs SNS reference. - EC2-StatusCheckSystem-syslog-server-recover: StatusCheckFailed_System with an EC2 recover action (+ SNS). AWS only allows RECOVER on the _System metric, not the combined metric, so it is a separate alarm; treatMissingData=notBreaching per AWS recovery-alarm guidance. Validated with tsc, cdk synth, cfn-lint (W2001 bootstrap warning only). The pre-existing orphaned alarm is deleted separately, not here. --- lib/syslog-server-stack.ts | 53 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 53 insertions(+) diff --git a/lib/syslog-server-stack.ts b/lib/syslog-server-stack.ts index b6a23e8..b28f7a3 100644 --- a/lib/syslog-server-stack.ts +++ b/lib/syslog-server-stack.ts @@ -289,6 +289,59 @@ export class SyslogServerStack extends cdk.Stack { }); noLogsAlarm.addAlarmAction(new cwactions.SnsAction(alarmTopic)); + // Primary EC2 status-check alarm — pages on-call when the box is hung or + // unreachable. Uses the combined StatusCheckFailed metric so it covers BOTH + // instance and system failures. Dimension is instance.instanceId (Ref), not + // a literal id, so the alarm tracks the CFN-managed instance across future + // replacements (e.g. userDataCausesReplacement above) — this is the durable + // fix for the orphaned EC2-StatusCheck-syslog-server alarm that pointed at a + // since-terminated instance. Maximum>=1 over two 5-min periods; + // treatMissingData=breaching so a metric gap (instance gone/not reporting) + // also fires. + const statusCheckAlarm = new cloudwatch.Alarm(this, "StatusCheckFailedAlarm", { + alarmName: "EC2-StatusCheck-syslog-server", + alarmDescription: + "syslog-server EC2 status check failed (instance and/or system) for 10 min — host may be hung or unreachable.", + metric: new cloudwatch.Metric({ + namespace: "AWS/EC2", + metricName: "StatusCheckFailed", + dimensionsMap: { InstanceId: instance.instanceId }, + statistic: "Maximum", + period: cdk.Duration.seconds(300), + }), + threshold: 1, + comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, + evaluationPeriods: 2, + treatMissingData: cloudwatch.TreatMissingData.BREACHING, + }); + statusCheckAlarm.addAlarmAction(new cwactions.SnsAction(alarmTopic)); + + // Auto-recovery alarm — migrates the instance onto healthy hardware on an + // underlying host failure (instance id, EIP association and EBS volume are + // preserved). AWS only permits the RECOVER action on StatusCheckFailed_System + // (NOT the combined StatusCheckFailed / _Instance), so this is a separate + // alarm. Per AWS guidance for recovery alarms, missing data is treated as + // NOT breaching to avoid a spurious recover on transient INSUFFICIENT_DATA, + // and evaluation periods differ from any reboot alarm to avoid a race. + const systemRecoverAlarm = new cloudwatch.Alarm(this, "StatusCheckSystemRecoverAlarm", { + alarmName: "EC2-StatusCheckSystem-syslog-server-recover", + alarmDescription: + "syslog-server EC2 system status check failed — underlying host impaired; auto-recovering onto new hardware.", + metric: new cloudwatch.Metric({ + namespace: "AWS/EC2", + metricName: "StatusCheckFailed_System", + dimensionsMap: { InstanceId: instance.instanceId }, + statistic: "Maximum", + period: cdk.Duration.seconds(300), + }), + threshold: 1, + comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, + evaluationPeriods: 2, + treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING, + }); + systemRecoverAlarm.addAlarmAction(new cwactions.SnsAction(alarmTopic)); + systemRecoverAlarm.addAlarmAction(new cwactions.Ec2Action(cwactions.Ec2InstanceAction.RECOVER)); + new cdk.CfnOutput(this, "InstanceId", { value: instance.instanceId }); new cdk.CfnOutput(this, "PublicIp", { value: "184.72.154.32",