Add a MonitoringConstruct (lib/constructs/monitoring.ts) wiring CloudWatch alarms to the shared site-alerts SNS topic for the seahaven-slack-bot stack. Every alarm uses an SNS alarm action only (no OK action) and treats missing data as NOT_BREACHING; alarm names are repo-namespaced kebab-case. Coverage: - Lambda (9 fns): Errors, Throttles, Duration (p99, eval3/dp2, ~80% of timeout) - DynamoDB (seahaven-conversations, seahaven-unanswered-questions): ThrottledRequests + SystemErrors via the per-operations metric-math helpers (the bare TableName-only helpers are deprecated/invalid); operations scoped to 6 CRUD ops to stay under the 10-metric alarm-math cap - API Gateway v2 (seahaven-slack-webhook): 5xx, 4xx, Latency (ApiId dimension) - ECS Fargate (seahaven-socket-mode): CPU + Memory utilization (AWS/ECS) Expose qbo-oauth Lambda and the ECS cluster/service as public readonly handles without changing logical IDs. RunningTaskCount alarm is gated off pending Container Insights sign-off (separate commit).
299 lines
12 KiB
TypeScript
299 lines
12 KiB
TypeScript
import * as cdk from 'aws-cdk-lib';
|
|
import { Construct } from 'constructs';
|
|
import * as cloudwatch from 'aws-cdk-lib/aws-cloudwatch';
|
|
import * as cwActions from 'aws-cdk-lib/aws-cloudwatch-actions';
|
|
import * as sns from 'aws-cdk-lib/aws-sns';
|
|
import * as lambda from 'aws-cdk-lib/aws-lambda';
|
|
import * as dynamodb from 'aws-cdk-lib/aws-dynamodb';
|
|
import * as apigatewayv2 from 'aws-cdk-lib/aws-apigatewayv2';
|
|
import * as ecs from 'aws-cdk-lib/aws-ecs';
|
|
|
|
/**
|
|
* A Lambda function plus the short name used to label its alarms.
|
|
* `name` becomes the `seahaven-<name>-<signal>` alarm-name prefix and must
|
|
* match the function's kebab-case short name (e.g. 'slack-processor').
|
|
*/
|
|
export interface MonitoredLambda {
|
|
name: string;
|
|
fn: lambda.IFunction;
|
|
/** Function timeout — used to derive the p99 Duration threshold (~80% of timeout). */
|
|
timeout: cdk.Duration;
|
|
}
|
|
|
|
/** A DynamoDB table plus the short name used to label its alarms. */
|
|
export interface MonitoredTable {
|
|
/** kebab-case short name, e.g. 'ddb-conversations'. */
|
|
name: string;
|
|
table: dynamodb.Table;
|
|
}
|
|
|
|
export interface MonitoringConstructProps {
|
|
/** Lambdas to cover with Errors + Throttles + Duration alarms. */
|
|
lambdas: MonitoredLambda[];
|
|
/** In-stack DynamoDB tables to cover with throttle + system-error alarms. */
|
|
tables: MonitoredTable[];
|
|
/** HTTP API (API Gateway v2) to cover with 5xx/4xx/Latency alarms. */
|
|
httpApi: apigatewayv2.HttpApi;
|
|
/** ECS Fargate service to cover with CPU/Memory utilization alarms. */
|
|
ecsService: ecs.FargateService;
|
|
/**
|
|
* ECS cluster — required for the RunningTaskCount alarm, which depends on
|
|
* Container Insights being enabled (gated behind `enableRunningTaskAlarm`).
|
|
*/
|
|
ecsCluster: ecs.ICluster;
|
|
/**
|
|
* When true, add the RunningTaskCount alarm. The metric only emits when
|
|
* Container Insights is enabled on the cluster — enabling it is a separate,
|
|
* cost-bearing config change that must be made on the cluster itself.
|
|
* Defaults to false so the alarm is opt-in.
|
|
*/
|
|
enableRunningTaskAlarm?: boolean;
|
|
}
|
|
|
|
/**
|
|
* Centralised CloudWatch alarm coverage for the seahaven-slack-bot stack.
|
|
*
|
|
* Every alarm:
|
|
* - notifies the shared `site-alerts` SNS topic (alarm action only, no OK action)
|
|
* - treats missing data as NOT_BREACHING
|
|
* - is named `seahaven-<fn>-<signal>` (repo-namespaced kebab-case)
|
|
*/
|
|
export class MonitoringConstruct extends Construct {
|
|
private readonly alertsTopic: sns.ITopic;
|
|
|
|
constructor(scope: Construct, id: string, props: MonitoringConstructProps) {
|
|
super(scope, id);
|
|
|
|
// Shared site-wide alerts topic — imported ONCE, reused for every alarm.
|
|
this.alertsTopic = sns.Topic.fromTopicArn(
|
|
this,
|
|
'SiteAlerts',
|
|
'arn:aws:sns:us-east-1:328440206208:site-alerts',
|
|
);
|
|
|
|
for (const ml of props.lambdas) {
|
|
this.addLambdaAlarms(ml);
|
|
}
|
|
|
|
for (const mt of props.tables) {
|
|
this.addDynamoAlarms(mt);
|
|
}
|
|
|
|
this.addApiGatewayAlarms(props.httpApi);
|
|
this.addEcsServiceAlarms(props.ecsService);
|
|
|
|
if (props.enableRunningTaskAlarm) {
|
|
this.addEcsRunningTaskAlarm(props.ecsService, props.ecsCluster);
|
|
}
|
|
}
|
|
|
|
/** Attach the SNS alarm action (no OK action) and return the alarm. */
|
|
private wire(alarm: cloudwatch.Alarm): cloudwatch.Alarm {
|
|
alarm.addAlarmAction(new cwActions.SnsAction(this.alertsTopic));
|
|
return alarm;
|
|
}
|
|
|
|
// ── Lambda: Errors + Throttles + Duration ──────────────────────────────────
|
|
private addLambdaAlarms(ml: MonitoredLambda): void {
|
|
const { name, fn, timeout } = ml;
|
|
|
|
// Errors — any invocation error over a 5-min window.
|
|
this.wire(
|
|
new cloudwatch.Alarm(this, `${name}-errors`, {
|
|
alarmName: `seahaven-${name}-errors`,
|
|
alarmDescription: `seahaven-${name} Lambda invocation errors`,
|
|
metric: fn.metricErrors({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
|
|
threshold: 1,
|
|
evaluationPeriods: 1,
|
|
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
}),
|
|
);
|
|
|
|
// Throttles — concurrency exhaustion.
|
|
this.wire(
|
|
new cloudwatch.Alarm(this, `${name}-throttles`, {
|
|
alarmName: `seahaven-${name}-throttles`,
|
|
alarmDescription: `seahaven-${name} Lambda throttles`,
|
|
metric: fn.metricThrottles({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
|
|
threshold: 1,
|
|
evaluationPeriods: 1,
|
|
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
}),
|
|
);
|
|
|
|
// Duration — p99 approaching the timeout (~80%). eval3/datapoints2 to ride
|
|
// out single slow invocations while still catching sustained latency.
|
|
const thresholdMs = Math.round(timeout.toMilliseconds() * 0.8);
|
|
this.wire(
|
|
new cloudwatch.Alarm(this, `${name}-duration`, {
|
|
alarmName: `seahaven-${name}-duration`,
|
|
alarmDescription: `seahaven-${name} Lambda p99 duration ≥ 80% of ${timeout.toSeconds()}s timeout`,
|
|
metric: fn.metricDuration({ period: cdk.Duration.minutes(5), statistic: 'p99' }),
|
|
threshold: thresholdMs,
|
|
evaluationPeriods: 3,
|
|
datapointsToAlarm: 2,
|
|
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
}),
|
|
);
|
|
}
|
|
|
|
// ── DynamoDB: ThrottledRequests + SystemErrors ─────────────────────────────
|
|
// ThrottledRequests/SystemErrors are emitted per TableName+Operation (there is
|
|
// no valid TableName-only aggregate — the bare metricThrottledRequests /
|
|
// metricSystemErrors helpers are deprecated). The per-operations helpers build
|
|
// metric-math summing across operations; CloudWatch caps an alarm math
|
|
// expression at 10 metrics, and DynamoDB defines 14 operations — so we scope to
|
|
// the operations these tables actually use (read/write CRUD paths).
|
|
private static readonly DDB_OPERATIONS: dynamodb.Operation[] = [
|
|
dynamodb.Operation.GET_ITEM,
|
|
dynamodb.Operation.PUT_ITEM,
|
|
dynamodb.Operation.UPDATE_ITEM,
|
|
dynamodb.Operation.DELETE_ITEM,
|
|
dynamodb.Operation.QUERY,
|
|
dynamodb.Operation.BATCH_WRITE_ITEM,
|
|
];
|
|
|
|
private addDynamoAlarms(mt: MonitoredTable): void {
|
|
const { name: shortName, table } = mt;
|
|
|
|
this.wire(
|
|
new cloudwatch.Alarm(this, `${shortName}-throttles`, {
|
|
alarmName: `seahaven-${shortName}-throttles`,
|
|
alarmDescription: `${table.tableName} DynamoDB throttled requests`,
|
|
metric: table.metricThrottledRequestsForOperations({
|
|
operations: MonitoringConstruct.DDB_OPERATIONS,
|
|
period: cdk.Duration.minutes(5),
|
|
statistic: 'Sum',
|
|
}),
|
|
threshold: 1,
|
|
evaluationPeriods: 1,
|
|
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
}),
|
|
);
|
|
|
|
this.wire(
|
|
new cloudwatch.Alarm(this, `${shortName}-system-errors`, {
|
|
alarmName: `seahaven-${shortName}-system-errors`,
|
|
alarmDescription: `${table.tableName} DynamoDB system errors (5xx)`,
|
|
metric: table.metricSystemErrorsForOperations({
|
|
operations: MonitoringConstruct.DDB_OPERATIONS,
|
|
period: cdk.Duration.minutes(5),
|
|
statistic: 'Sum',
|
|
}),
|
|
threshold: 1,
|
|
evaluationPeriods: 1,
|
|
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
}),
|
|
);
|
|
}
|
|
|
|
// ── API Gateway v2 (HTTP API): 5xx + 4xx + Latency ─────────────────────────
|
|
private addApiGatewayAlarms(api: apigatewayv2.HttpApi): void {
|
|
// metricServerError/metricClientError/metricLatency resolve to the v2
|
|
// metric names (5xx/4xx/Latency) under the ApiId dimension automatically.
|
|
this.wire(
|
|
new cloudwatch.Alarm(this, 'slack-webhook-5xx', {
|
|
alarmName: 'seahaven-slack-webhook-5xx',
|
|
alarmDescription: 'seahaven-slack-webhook API Gateway 5xx errors',
|
|
metric: api.metricServerError({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
|
|
threshold: 1,
|
|
evaluationPeriods: 1,
|
|
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
}),
|
|
);
|
|
|
|
// 4xx is noisier (bad OAuth callbacks, scanners) — require a sustained
|
|
// burst rather than a single request. eval3/datapoints2 over 5-min periods.
|
|
this.wire(
|
|
new cloudwatch.Alarm(this, 'slack-webhook-4xx', {
|
|
alarmName: 'seahaven-slack-webhook-4xx',
|
|
alarmDescription: 'seahaven-slack-webhook API Gateway sustained 4xx errors',
|
|
metric: api.metricClientError({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
|
|
threshold: 10,
|
|
evaluationPeriods: 3,
|
|
datapointsToAlarm: 2,
|
|
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
}),
|
|
);
|
|
|
|
// Latency p99 ≥ 3s — OAuth routes call Intuit; allow headroom.
|
|
this.wire(
|
|
new cloudwatch.Alarm(this, 'slack-webhook-latency', {
|
|
alarmName: 'seahaven-slack-webhook-latency',
|
|
alarmDescription: 'seahaven-slack-webhook API Gateway p99 latency ≥ 3s',
|
|
metric: api.metricLatency({ period: cdk.Duration.minutes(5), statistic: 'p99' }),
|
|
threshold: 3000,
|
|
evaluationPeriods: 3,
|
|
datapointsToAlarm: 2,
|
|
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
}),
|
|
);
|
|
}
|
|
|
|
// ── ECS Fargate: CPU + Memory utilization (no Container Insights needed) ────
|
|
private addEcsServiceAlarms(service: ecs.FargateService): void {
|
|
this.wire(
|
|
new cloudwatch.Alarm(this, 'socket-mode-cpu', {
|
|
alarmName: 'seahaven-socket-mode-cpu',
|
|
alarmDescription: 'seahaven-socket-mode ECS service CPU utilization ≥ 85%',
|
|
metric: service.metricCpuUtilization({ period: cdk.Duration.minutes(5), statistic: 'Average' }),
|
|
threshold: 85,
|
|
evaluationPeriods: 3,
|
|
datapointsToAlarm: 2,
|
|
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
}),
|
|
);
|
|
|
|
this.wire(
|
|
new cloudwatch.Alarm(this, 'socket-mode-memory', {
|
|
alarmName: 'seahaven-socket-mode-memory',
|
|
alarmDescription: 'seahaven-socket-mode ECS service memory utilization ≥ 85%',
|
|
metric: service.metricMemoryUtilization({ period: cdk.Duration.minutes(5), statistic: 'Average' }),
|
|
threshold: 85,
|
|
evaluationPeriods: 3,
|
|
datapointsToAlarm: 2,
|
|
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
|
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
}),
|
|
);
|
|
}
|
|
|
|
// ── ECS RunningTaskCount (REQUIRES Container Insights) ──────────────────────
|
|
// RunningTaskCount is published only when Container Insights is enabled on the
|
|
// cluster. desiredCount is 1, so alarm when running tasks drop below 1.
|
|
private addEcsRunningTaskAlarm(service: ecs.FargateService, cluster: ecs.ICluster): void {
|
|
const runningTasks = new cloudwatch.Metric({
|
|
namespace: 'ECS/ContainerInsights',
|
|
metricName: 'RunningTaskCount',
|
|
dimensionsMap: {
|
|
ClusterName: cluster.clusterName,
|
|
ServiceName: service.serviceName,
|
|
},
|
|
period: cdk.Duration.minutes(1),
|
|
statistic: 'Average',
|
|
});
|
|
|
|
this.wire(
|
|
new cloudwatch.Alarm(this, 'socket-mode-running-tasks', {
|
|
alarmName: 'seahaven-socket-mode-running-tasks',
|
|
alarmDescription:
|
|
'seahaven-socket-mode ECS running task count < 1 (Container Insights required)',
|
|
metric: runningTasks,
|
|
threshold: 1,
|
|
evaluationPeriods: 3,
|
|
datapointsToAlarm: 2,
|
|
comparisonOperator: cloudwatch.ComparisonOperator.LESS_THAN_THRESHOLD,
|
|
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
|
}),
|
|
);
|
|
}
|
|
}
|