import * as cdk from 'aws-cdk-lib'; import { Construct } from 'constructs'; import * as cloudwatch from 'aws-cdk-lib/aws-cloudwatch'; import * as cwActions from 'aws-cdk-lib/aws-cloudwatch-actions'; import * as sns from 'aws-cdk-lib/aws-sns'; import * as lambda from 'aws-cdk-lib/aws-lambda'; import * as dynamodb from 'aws-cdk-lib/aws-dynamodb'; import * as apigatewayv2 from 'aws-cdk-lib/aws-apigatewayv2'; import * as ecs from 'aws-cdk-lib/aws-ecs'; /** * A Lambda function plus the short name used to label its alarms. * `name` becomes the `seahaven--` alarm-name prefix and must * match the function's kebab-case short name (e.g. 'slack-processor'). */ export interface MonitoredLambda { name: string; fn: lambda.IFunction; /** Function timeout — used to derive the p99 Duration threshold (~80% of timeout). */ timeout: cdk.Duration; } /** A DynamoDB table plus the short name used to label its alarms. */ export interface MonitoredTable { /** kebab-case short name, e.g. 'ddb-conversations'. */ name: string; table: dynamodb.Table; } export interface MonitoringConstructProps { /** Lambdas to cover with Errors + Throttles + Duration alarms. */ lambdas: MonitoredLambda[]; /** In-stack DynamoDB tables to cover with throttle + system-error alarms. */ tables: MonitoredTable[]; /** HTTP API (API Gateway v2) to cover with 5xx/4xx/Latency alarms. */ httpApi: apigatewayv2.HttpApi; /** ECS Fargate service to cover with CPU/Memory utilization alarms. */ ecsService: ecs.FargateService; /** * ECS cluster — required for the RunningTaskCount alarm, which depends on * Container Insights being enabled (gated behind `enableRunningTaskAlarm`). */ ecsCluster: ecs.ICluster; /** * When true, add the RunningTaskCount alarm. The metric only emits when * Container Insights is enabled on the cluster — enabling it is a separate, * cost-bearing config change that must be made on the cluster itself. * Defaults to false so the alarm is opt-in. */ enableRunningTaskAlarm?: boolean; } /** * Centralised CloudWatch alarm coverage for the seahaven-slack-bot stack. * * Every alarm: * - notifies the shared `site-alerts` SNS topic (alarm action only, no OK action) * - treats missing data as NOT_BREACHING * - is named `seahaven--` (repo-namespaced kebab-case) */ export class MonitoringConstruct extends Construct { private readonly alertsTopic: sns.ITopic; constructor(scope: Construct, id: string, props: MonitoringConstructProps) { super(scope, id); // Shared site-wide alerts topic — imported ONCE, reused for every alarm. this.alertsTopic = sns.Topic.fromTopicArn( this, 'SiteAlerts', 'arn:aws:sns:us-east-1:328440206208:site-alerts', ); for (const ml of props.lambdas) { this.addLambdaAlarms(ml); } for (const mt of props.tables) { this.addDynamoAlarms(mt); } this.addApiGatewayAlarms(props.httpApi); this.addEcsServiceAlarms(props.ecsService); if (props.enableRunningTaskAlarm) { this.addEcsRunningTaskAlarm(props.ecsService, props.ecsCluster); } } /** Attach the SNS alarm action (no OK action) and return the alarm. */ private wire(alarm: cloudwatch.Alarm): cloudwatch.Alarm { alarm.addAlarmAction(new cwActions.SnsAction(this.alertsTopic)); return alarm; } // ── Lambda: Errors + Throttles + Duration ────────────────────────────────── private addLambdaAlarms(ml: MonitoredLambda): void { const { name, fn, timeout } = ml; // Errors — any invocation error over a 5-min window. this.wire( new cloudwatch.Alarm(this, `${name}-errors`, { alarmName: `seahaven-${name}-errors`, alarmDescription: `seahaven-${name} Lambda invocation errors`, metric: fn.metricErrors({ period: cdk.Duration.minutes(5), statistic: 'Sum' }), threshold: 1, evaluationPeriods: 1, comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING, }), ); // Throttles — concurrency exhaustion. this.wire( new cloudwatch.Alarm(this, `${name}-throttles`, { alarmName: `seahaven-${name}-throttles`, alarmDescription: `seahaven-${name} Lambda throttles`, metric: fn.metricThrottles({ period: cdk.Duration.minutes(5), statistic: 'Sum' }), threshold: 1, evaluationPeriods: 1, comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING, }), ); // Duration — p99 approaching the timeout (~80%). eval3/datapoints2 to ride // out single slow invocations while still catching sustained latency. const thresholdMs = Math.round(timeout.toMilliseconds() * 0.8); this.wire( new cloudwatch.Alarm(this, `${name}-duration`, { alarmName: `seahaven-${name}-duration`, alarmDescription: `seahaven-${name} Lambda p99 duration ≥ 80% of ${timeout.toSeconds()}s timeout`, metric: fn.metricDuration({ period: cdk.Duration.minutes(5), statistic: 'p99' }), threshold: thresholdMs, evaluationPeriods: 3, datapointsToAlarm: 2, comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING, }), ); } // ── DynamoDB: ThrottledRequests + SystemErrors ───────────────────────────── // ThrottledRequests/SystemErrors are emitted per TableName+Operation (there is // no valid TableName-only aggregate — the bare metricThrottledRequests / // metricSystemErrors helpers are deprecated). The per-operations helpers build // metric-math summing across operations; CloudWatch caps an alarm math // expression at 10 metrics, and DynamoDB defines 14 operations — so we scope to // the operations these tables actually use (read/write CRUD paths). private static readonly DDB_OPERATIONS: dynamodb.Operation[] = [ dynamodb.Operation.GET_ITEM, dynamodb.Operation.PUT_ITEM, dynamodb.Operation.UPDATE_ITEM, dynamodb.Operation.DELETE_ITEM, dynamodb.Operation.QUERY, dynamodb.Operation.BATCH_WRITE_ITEM, ]; private addDynamoAlarms(mt: MonitoredTable): void { const { name: shortName, table } = mt; this.wire( new cloudwatch.Alarm(this, `${shortName}-throttles`, { alarmName: `seahaven-${shortName}-throttles`, alarmDescription: `${table.tableName} DynamoDB throttled requests`, metric: table.metricThrottledRequestsForOperations({ operations: MonitoringConstruct.DDB_OPERATIONS, period: cdk.Duration.minutes(5), statistic: 'Sum', }), threshold: 1, evaluationPeriods: 1, comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING, }), ); this.wire( new cloudwatch.Alarm(this, `${shortName}-system-errors`, { alarmName: `seahaven-${shortName}-system-errors`, alarmDescription: `${table.tableName} DynamoDB system errors (5xx)`, metric: table.metricSystemErrorsForOperations({ operations: MonitoringConstruct.DDB_OPERATIONS, period: cdk.Duration.minutes(5), statistic: 'Sum', }), threshold: 1, evaluationPeriods: 1, comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING, }), ); } // ── API Gateway v2 (HTTP API): 5xx + 4xx + Latency ───────────────────────── private addApiGatewayAlarms(api: apigatewayv2.HttpApi): void { // metricServerError/metricClientError/metricLatency resolve to the v2 // metric names (5xx/4xx/Latency) under the ApiId dimension automatically. this.wire( new cloudwatch.Alarm(this, 'slack-webhook-5xx', { alarmName: 'seahaven-slack-webhook-5xx', alarmDescription: 'seahaven-slack-webhook API Gateway 5xx errors', metric: api.metricServerError({ period: cdk.Duration.minutes(5), statistic: 'Sum' }), threshold: 1, evaluationPeriods: 1, comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING, }), ); // 4xx is noisier (bad OAuth callbacks, scanners) — require a sustained // burst rather than a single request. eval3/datapoints2 over 5-min periods. this.wire( new cloudwatch.Alarm(this, 'slack-webhook-4xx', { alarmName: 'seahaven-slack-webhook-4xx', alarmDescription: 'seahaven-slack-webhook API Gateway sustained 4xx errors', metric: api.metricClientError({ period: cdk.Duration.minutes(5), statistic: 'Sum' }), threshold: 10, evaluationPeriods: 3, datapointsToAlarm: 2, comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING, }), ); // Latency p99 ≥ 3s — OAuth routes call Intuit; allow headroom. this.wire( new cloudwatch.Alarm(this, 'slack-webhook-latency', { alarmName: 'seahaven-slack-webhook-latency', alarmDescription: 'seahaven-slack-webhook API Gateway p99 latency ≥ 3s', metric: api.metricLatency({ period: cdk.Duration.minutes(5), statistic: 'p99' }), threshold: 3000, evaluationPeriods: 3, datapointsToAlarm: 2, comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING, }), ); } // ── ECS Fargate: CPU + Memory utilization (no Container Insights needed) ──── private addEcsServiceAlarms(service: ecs.FargateService): void { this.wire( new cloudwatch.Alarm(this, 'socket-mode-cpu', { alarmName: 'seahaven-socket-mode-cpu', alarmDescription: 'seahaven-socket-mode ECS service CPU utilization ≥ 85%', metric: service.metricCpuUtilization({ period: cdk.Duration.minutes(5), statistic: 'Average' }), threshold: 85, evaluationPeriods: 3, datapointsToAlarm: 2, comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING, }), ); this.wire( new cloudwatch.Alarm(this, 'socket-mode-memory', { alarmName: 'seahaven-socket-mode-memory', alarmDescription: 'seahaven-socket-mode ECS service memory utilization ≥ 85%', metric: service.metricMemoryUtilization({ period: cdk.Duration.minutes(5), statistic: 'Average' }), threshold: 85, evaluationPeriods: 3, datapointsToAlarm: 2, comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING, }), ); } // ── ECS RunningTaskCount (REQUIRES Container Insights) ────────────────────── // RunningTaskCount is published only when Container Insights is enabled on the // cluster. desiredCount is 1, so alarm when running tasks drop below 1. private addEcsRunningTaskAlarm(service: ecs.FargateService, cluster: ecs.ICluster): void { const runningTasks = new cloudwatch.Metric({ namespace: 'ECS/ContainerInsights', metricName: 'RunningTaskCount', dimensionsMap: { ClusterName: cluster.clusterName, ServiceName: service.serviceName, }, period: cdk.Duration.minutes(1), statistic: 'Average', }); this.wire( new cloudwatch.Alarm(this, 'socket-mode-running-tasks', { alarmName: 'seahaven-socket-mode-running-tasks', alarmDescription: 'seahaven-socket-mode ECS running task count < 1 (Container Insights required)', metric: runningTasks, threshold: 1, evaluationPeriods: 3, datapointsToAlarm: 2, comparisonOperator: cloudwatch.ComparisonOperator.LESS_THAN_THRESHOLD, treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING, }), ); } }