This repository has been archived on 2026-08-04. You can view files and clone it, but cannot push or open issues or pull requests.
seahaven-slack-bot/lib/constructs/monitoring.ts
Adam Moussa c28d5b1170 Add CloudWatch alarm coverage via MonitoringConstruct
Add a MonitoringConstruct (lib/constructs/monitoring.ts) wiring CloudWatch
alarms to the shared site-alerts SNS topic for the seahaven-slack-bot stack.
Every alarm uses an SNS alarm action only (no OK action) and treats missing
data as NOT_BREACHING; alarm names are repo-namespaced kebab-case.

Coverage:
- Lambda (9 fns): Errors, Throttles, Duration (p99, eval3/dp2, ~80% of timeout)
- DynamoDB (seahaven-conversations, seahaven-unanswered-questions):
  ThrottledRequests + SystemErrors via the per-operations metric-math helpers
  (the bare TableName-only helpers are deprecated/invalid); operations scoped
  to 6 CRUD ops to stay under the 10-metric alarm-math cap
- API Gateway v2 (seahaven-slack-webhook): 5xx, 4xx, Latency (ApiId dimension)
- ECS Fargate (seahaven-socket-mode): CPU + Memory utilization (AWS/ECS)

Expose qbo-oauth Lambda and the ECS cluster/service as public readonly handles
without changing logical IDs. RunningTaskCount alarm is gated off pending
Container Insights sign-off (separate commit).
2026-06-17 13:57:07 -04:00

299 lines
12 KiB
TypeScript

import * as cdk from 'aws-cdk-lib';
import { Construct } from 'constructs';
import * as cloudwatch from 'aws-cdk-lib/aws-cloudwatch';
import * as cwActions from 'aws-cdk-lib/aws-cloudwatch-actions';
import * as sns from 'aws-cdk-lib/aws-sns';
import * as lambda from 'aws-cdk-lib/aws-lambda';
import * as dynamodb from 'aws-cdk-lib/aws-dynamodb';
import * as apigatewayv2 from 'aws-cdk-lib/aws-apigatewayv2';
import * as ecs from 'aws-cdk-lib/aws-ecs';
/**
* A Lambda function plus the short name used to label its alarms.
* `name` becomes the `seahaven-<name>-<signal>` alarm-name prefix and must
* match the function's kebab-case short name (e.g. 'slack-processor').
*/
export interface MonitoredLambda {
name: string;
fn: lambda.IFunction;
/** Function timeout — used to derive the p99 Duration threshold (~80% of timeout). */
timeout: cdk.Duration;
}
/** A DynamoDB table plus the short name used to label its alarms. */
export interface MonitoredTable {
/** kebab-case short name, e.g. 'ddb-conversations'. */
name: string;
table: dynamodb.Table;
}
export interface MonitoringConstructProps {
/** Lambdas to cover with Errors + Throttles + Duration alarms. */
lambdas: MonitoredLambda[];
/** In-stack DynamoDB tables to cover with throttle + system-error alarms. */
tables: MonitoredTable[];
/** HTTP API (API Gateway v2) to cover with 5xx/4xx/Latency alarms. */
httpApi: apigatewayv2.HttpApi;
/** ECS Fargate service to cover with CPU/Memory utilization alarms. */
ecsService: ecs.FargateService;
/**
* ECS cluster — required for the RunningTaskCount alarm, which depends on
* Container Insights being enabled (gated behind `enableRunningTaskAlarm`).
*/
ecsCluster: ecs.ICluster;
/**
* When true, add the RunningTaskCount alarm. The metric only emits when
* Container Insights is enabled on the cluster — enabling it is a separate,
* cost-bearing config change that must be made on the cluster itself.
* Defaults to false so the alarm is opt-in.
*/
enableRunningTaskAlarm?: boolean;
}
/**
* Centralised CloudWatch alarm coverage for the seahaven-slack-bot stack.
*
* Every alarm:
* - notifies the shared `site-alerts` SNS topic (alarm action only, no OK action)
* - treats missing data as NOT_BREACHING
* - is named `seahaven-<fn>-<signal>` (repo-namespaced kebab-case)
*/
export class MonitoringConstruct extends Construct {
private readonly alertsTopic: sns.ITopic;
constructor(scope: Construct, id: string, props: MonitoringConstructProps) {
super(scope, id);
// Shared site-wide alerts topic — imported ONCE, reused for every alarm.
this.alertsTopic = sns.Topic.fromTopicArn(
this,
'SiteAlerts',
'arn:aws:sns:us-east-1:328440206208:site-alerts',
);
for (const ml of props.lambdas) {
this.addLambdaAlarms(ml);
}
for (const mt of props.tables) {
this.addDynamoAlarms(mt);
}
this.addApiGatewayAlarms(props.httpApi);
this.addEcsServiceAlarms(props.ecsService);
if (props.enableRunningTaskAlarm) {
this.addEcsRunningTaskAlarm(props.ecsService, props.ecsCluster);
}
}
/** Attach the SNS alarm action (no OK action) and return the alarm. */
private wire(alarm: cloudwatch.Alarm): cloudwatch.Alarm {
alarm.addAlarmAction(new cwActions.SnsAction(this.alertsTopic));
return alarm;
}
// ── Lambda: Errors + Throttles + Duration ──────────────────────────────────
private addLambdaAlarms(ml: MonitoredLambda): void {
const { name, fn, timeout } = ml;
// Errors — any invocation error over a 5-min window.
this.wire(
new cloudwatch.Alarm(this, `${name}-errors`, {
alarmName: `seahaven-${name}-errors`,
alarmDescription: `seahaven-${name} Lambda invocation errors`,
metric: fn.metricErrors({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
// Throttles — concurrency exhaustion.
this.wire(
new cloudwatch.Alarm(this, `${name}-throttles`, {
alarmName: `seahaven-${name}-throttles`,
alarmDescription: `seahaven-${name} Lambda throttles`,
metric: fn.metricThrottles({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
// Duration — p99 approaching the timeout (~80%). eval3/datapoints2 to ride
// out single slow invocations while still catching sustained latency.
const thresholdMs = Math.round(timeout.toMilliseconds() * 0.8);
this.wire(
new cloudwatch.Alarm(this, `${name}-duration`, {
alarmName: `seahaven-${name}-duration`,
alarmDescription: `seahaven-${name} Lambda p99 duration ≥ 80% of ${timeout.toSeconds()}s timeout`,
metric: fn.metricDuration({ period: cdk.Duration.minutes(5), statistic: 'p99' }),
threshold: thresholdMs,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
// ── DynamoDB: ThrottledRequests + SystemErrors ─────────────────────────────
// ThrottledRequests/SystemErrors are emitted per TableName+Operation (there is
// no valid TableName-only aggregate — the bare metricThrottledRequests /
// metricSystemErrors helpers are deprecated). The per-operations helpers build
// metric-math summing across operations; CloudWatch caps an alarm math
// expression at 10 metrics, and DynamoDB defines 14 operations — so we scope to
// the operations these tables actually use (read/write CRUD paths).
private static readonly DDB_OPERATIONS: dynamodb.Operation[] = [
dynamodb.Operation.GET_ITEM,
dynamodb.Operation.PUT_ITEM,
dynamodb.Operation.UPDATE_ITEM,
dynamodb.Operation.DELETE_ITEM,
dynamodb.Operation.QUERY,
dynamodb.Operation.BATCH_WRITE_ITEM,
];
private addDynamoAlarms(mt: MonitoredTable): void {
const { name: shortName, table } = mt;
this.wire(
new cloudwatch.Alarm(this, `${shortName}-throttles`, {
alarmName: `seahaven-${shortName}-throttles`,
alarmDescription: `${table.tableName} DynamoDB throttled requests`,
metric: table.metricThrottledRequestsForOperations({
operations: MonitoringConstruct.DDB_OPERATIONS,
period: cdk.Duration.minutes(5),
statistic: 'Sum',
}),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
this.wire(
new cloudwatch.Alarm(this, `${shortName}-system-errors`, {
alarmName: `seahaven-${shortName}-system-errors`,
alarmDescription: `${table.tableName} DynamoDB system errors (5xx)`,
metric: table.metricSystemErrorsForOperations({
operations: MonitoringConstruct.DDB_OPERATIONS,
period: cdk.Duration.minutes(5),
statistic: 'Sum',
}),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
// ── API Gateway v2 (HTTP API): 5xx + 4xx + Latency ─────────────────────────
private addApiGatewayAlarms(api: apigatewayv2.HttpApi): void {
// metricServerError/metricClientError/metricLatency resolve to the v2
// metric names (5xx/4xx/Latency) under the ApiId dimension automatically.
this.wire(
new cloudwatch.Alarm(this, 'slack-webhook-5xx', {
alarmName: 'seahaven-slack-webhook-5xx',
alarmDescription: 'seahaven-slack-webhook API Gateway 5xx errors',
metric: api.metricServerError({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
// 4xx is noisier (bad OAuth callbacks, scanners) — require a sustained
// burst rather than a single request. eval3/datapoints2 over 5-min periods.
this.wire(
new cloudwatch.Alarm(this, 'slack-webhook-4xx', {
alarmName: 'seahaven-slack-webhook-4xx',
alarmDescription: 'seahaven-slack-webhook API Gateway sustained 4xx errors',
metric: api.metricClientError({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
threshold: 10,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
// Latency p99 ≥ 3s — OAuth routes call Intuit; allow headroom.
this.wire(
new cloudwatch.Alarm(this, 'slack-webhook-latency', {
alarmName: 'seahaven-slack-webhook-latency',
alarmDescription: 'seahaven-slack-webhook API Gateway p99 latency ≥ 3s',
metric: api.metricLatency({ period: cdk.Duration.minutes(5), statistic: 'p99' }),
threshold: 3000,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
// ── ECS Fargate: CPU + Memory utilization (no Container Insights needed) ────
private addEcsServiceAlarms(service: ecs.FargateService): void {
this.wire(
new cloudwatch.Alarm(this, 'socket-mode-cpu', {
alarmName: 'seahaven-socket-mode-cpu',
alarmDescription: 'seahaven-socket-mode ECS service CPU utilization ≥ 85%',
metric: service.metricCpuUtilization({ period: cdk.Duration.minutes(5), statistic: 'Average' }),
threshold: 85,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
this.wire(
new cloudwatch.Alarm(this, 'socket-mode-memory', {
alarmName: 'seahaven-socket-mode-memory',
alarmDescription: 'seahaven-socket-mode ECS service memory utilization ≥ 85%',
metric: service.metricMemoryUtilization({ period: cdk.Duration.minutes(5), statistic: 'Average' }),
threshold: 85,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
// ── ECS RunningTaskCount (REQUIRES Container Insights) ──────────────────────
// RunningTaskCount is published only when Container Insights is enabled on the
// cluster. desiredCount is 1, so alarm when running tasks drop below 1.
private addEcsRunningTaskAlarm(service: ecs.FargateService, cluster: ecs.ICluster): void {
const runningTasks = new cloudwatch.Metric({
namespace: 'ECS/ContainerInsights',
metricName: 'RunningTaskCount',
dimensionsMap: {
ClusterName: cluster.clusterName,
ServiceName: service.serviceName,
},
period: cdk.Duration.minutes(1),
statistic: 'Average',
});
this.wire(
new cloudwatch.Alarm(this, 'socket-mode-running-tasks', {
alarmName: 'seahaven-socket-mode-running-tasks',
alarmDescription:
'seahaven-socket-mode ECS running task count < 1 (Container Insights required)',
metric: runningTasks,
threshold: 1,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.LESS_THAN_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
}