Add CloudWatch alarm coverage via MonitoringConstruct

Add a MonitoringConstruct (lib/constructs/monitoring.ts) wiring CloudWatch
alarms to the shared site-alerts SNS topic for the seahaven-slack-bot stack.
Every alarm uses an SNS alarm action only (no OK action) and treats missing
data as NOT_BREACHING; alarm names are repo-namespaced kebab-case.

Coverage:
- Lambda (9 fns): Errors, Throttles, Duration (p99, eval3/dp2, ~80% of timeout)
- DynamoDB (seahaven-conversations, seahaven-unanswered-questions):
  ThrottledRequests + SystemErrors via the per-operations metric-math helpers
  (the bare TableName-only helpers are deprecated/invalid); operations scoped
  to 6 CRUD ops to stay under the 10-metric alarm-math cap
- API Gateway v2 (seahaven-slack-webhook): 5xx, 4xx, Latency (ApiId dimension)
- ECS Fargate (seahaven-socket-mode): CPU + Memory utilization (AWS/ECS)

Expose qbo-oauth Lambda and the ECS cluster/service as public readonly handles
without changing logical IDs. RunningTaskCount alarm is gated off pending
Container Insights sign-off (separate commit).
This commit is contained in:
Adam Moussa 2026-06-17 13:57:07 -04:00
parent 58be1563e5
commit c28d5b1170
4 changed files with 339 additions and 9 deletions

View file

@ -0,0 +1,299 @@
import * as cdk from 'aws-cdk-lib';
import { Construct } from 'constructs';
import * as cloudwatch from 'aws-cdk-lib/aws-cloudwatch';
import * as cwActions from 'aws-cdk-lib/aws-cloudwatch-actions';
import * as sns from 'aws-cdk-lib/aws-sns';
import * as lambda from 'aws-cdk-lib/aws-lambda';
import * as dynamodb from 'aws-cdk-lib/aws-dynamodb';
import * as apigatewayv2 from 'aws-cdk-lib/aws-apigatewayv2';
import * as ecs from 'aws-cdk-lib/aws-ecs';
/**
* A Lambda function plus the short name used to label its alarms.
* `name` becomes the `seahaven-<name>-<signal>` alarm-name prefix and must
* match the function's kebab-case short name (e.g. 'slack-processor').
*/
export interface MonitoredLambda {
name: string;
fn: lambda.IFunction;
/** Function timeout — used to derive the p99 Duration threshold (~80% of timeout). */
timeout: cdk.Duration;
}
/** A DynamoDB table plus the short name used to label its alarms. */
export interface MonitoredTable {
/** kebab-case short name, e.g. 'ddb-conversations'. */
name: string;
table: dynamodb.Table;
}
export interface MonitoringConstructProps {
/** Lambdas to cover with Errors + Throttles + Duration alarms. */
lambdas: MonitoredLambda[];
/** In-stack DynamoDB tables to cover with throttle + system-error alarms. */
tables: MonitoredTable[];
/** HTTP API (API Gateway v2) to cover with 5xx/4xx/Latency alarms. */
httpApi: apigatewayv2.HttpApi;
/** ECS Fargate service to cover with CPU/Memory utilization alarms. */
ecsService: ecs.FargateService;
/**
* ECS cluster — required for the RunningTaskCount alarm, which depends on
* Container Insights being enabled (gated behind `enableRunningTaskAlarm`).
*/
ecsCluster: ecs.ICluster;
/**
* When true, add the RunningTaskCount alarm. The metric only emits when
* Container Insights is enabled on the cluster — enabling it is a separate,
* cost-bearing config change that must be made on the cluster itself.
* Defaults to false so the alarm is opt-in.
*/
enableRunningTaskAlarm?: boolean;
}
/**
* Centralised CloudWatch alarm coverage for the seahaven-slack-bot stack.
*
* Every alarm:
* - notifies the shared `site-alerts` SNS topic (alarm action only, no OK action)
* - treats missing data as NOT_BREACHING
* - is named `seahaven-<fn>-<signal>` (repo-namespaced kebab-case)
*/
export class MonitoringConstruct extends Construct {
private readonly alertsTopic: sns.ITopic;
constructor(scope: Construct, id: string, props: MonitoringConstructProps) {
super(scope, id);
// Shared site-wide alerts topic — imported ONCE, reused for every alarm.
this.alertsTopic = sns.Topic.fromTopicArn(
this,
'SiteAlerts',
'arn:aws:sns:us-east-1:328440206208:site-alerts',
);
for (const ml of props.lambdas) {
this.addLambdaAlarms(ml);
}
for (const mt of props.tables) {
this.addDynamoAlarms(mt);
}
this.addApiGatewayAlarms(props.httpApi);
this.addEcsServiceAlarms(props.ecsService);
if (props.enableRunningTaskAlarm) {
this.addEcsRunningTaskAlarm(props.ecsService, props.ecsCluster);
}
}
/** Attach the SNS alarm action (no OK action) and return the alarm. */
private wire(alarm: cloudwatch.Alarm): cloudwatch.Alarm {
alarm.addAlarmAction(new cwActions.SnsAction(this.alertsTopic));
return alarm;
}
// ── Lambda: Errors + Throttles + Duration ──────────────────────────────────
private addLambdaAlarms(ml: MonitoredLambda): void {
const { name, fn, timeout } = ml;
// Errors — any invocation error over a 5-min window.
this.wire(
new cloudwatch.Alarm(this, `${name}-errors`, {
alarmName: `seahaven-${name}-errors`,
alarmDescription: `seahaven-${name} Lambda invocation errors`,
metric: fn.metricErrors({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
// Throttles — concurrency exhaustion.
this.wire(
new cloudwatch.Alarm(this, `${name}-throttles`, {
alarmName: `seahaven-${name}-throttles`,
alarmDescription: `seahaven-${name} Lambda throttles`,
metric: fn.metricThrottles({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
// Duration — p99 approaching the timeout (~80%). eval3/datapoints2 to ride
// out single slow invocations while still catching sustained latency.
const thresholdMs = Math.round(timeout.toMilliseconds() * 0.8);
this.wire(
new cloudwatch.Alarm(this, `${name}-duration`, {
alarmName: `seahaven-${name}-duration`,
alarmDescription: `seahaven-${name} Lambda p99 duration ≥ 80% of ${timeout.toSeconds()}s timeout`,
metric: fn.metricDuration({ period: cdk.Duration.minutes(5), statistic: 'p99' }),
threshold: thresholdMs,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
// ── DynamoDB: ThrottledRequests + SystemErrors ─────────────────────────────
// ThrottledRequests/SystemErrors are emitted per TableName+Operation (there is
// no valid TableName-only aggregate — the bare metricThrottledRequests /
// metricSystemErrors helpers are deprecated). The per-operations helpers build
// metric-math summing across operations; CloudWatch caps an alarm math
// expression at 10 metrics, and DynamoDB defines 14 operations — so we scope to
// the operations these tables actually use (read/write CRUD paths).
private static readonly DDB_OPERATIONS: dynamodb.Operation[] = [
dynamodb.Operation.GET_ITEM,
dynamodb.Operation.PUT_ITEM,
dynamodb.Operation.UPDATE_ITEM,
dynamodb.Operation.DELETE_ITEM,
dynamodb.Operation.QUERY,
dynamodb.Operation.BATCH_WRITE_ITEM,
];
private addDynamoAlarms(mt: MonitoredTable): void {
const { name: shortName, table } = mt;
this.wire(
new cloudwatch.Alarm(this, `${shortName}-throttles`, {
alarmName: `seahaven-${shortName}-throttles`,
alarmDescription: `${table.tableName} DynamoDB throttled requests`,
metric: table.metricThrottledRequestsForOperations({
operations: MonitoringConstruct.DDB_OPERATIONS,
period: cdk.Duration.minutes(5),
statistic: 'Sum',
}),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
this.wire(
new cloudwatch.Alarm(this, `${shortName}-system-errors`, {
alarmName: `seahaven-${shortName}-system-errors`,
alarmDescription: `${table.tableName} DynamoDB system errors (5xx)`,
metric: table.metricSystemErrorsForOperations({
operations: MonitoringConstruct.DDB_OPERATIONS,
period: cdk.Duration.minutes(5),
statistic: 'Sum',
}),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
// ── API Gateway v2 (HTTP API): 5xx + 4xx + Latency ─────────────────────────
private addApiGatewayAlarms(api: apigatewayv2.HttpApi): void {
// metricServerError/metricClientError/metricLatency resolve to the v2
// metric names (5xx/4xx/Latency) under the ApiId dimension automatically.
this.wire(
new cloudwatch.Alarm(this, 'slack-webhook-5xx', {
alarmName: 'seahaven-slack-webhook-5xx',
alarmDescription: 'seahaven-slack-webhook API Gateway 5xx errors',
metric: api.metricServerError({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
// 4xx is noisier (bad OAuth callbacks, scanners) — require a sustained
// burst rather than a single request. eval3/datapoints2 over 5-min periods.
this.wire(
new cloudwatch.Alarm(this, 'slack-webhook-4xx', {
alarmName: 'seahaven-slack-webhook-4xx',
alarmDescription: 'seahaven-slack-webhook API Gateway sustained 4xx errors',
metric: api.metricClientError({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
threshold: 10,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
// Latency p99 ≥ 3s — OAuth routes call Intuit; allow headroom.
this.wire(
new cloudwatch.Alarm(this, 'slack-webhook-latency', {
alarmName: 'seahaven-slack-webhook-latency',
alarmDescription: 'seahaven-slack-webhook API Gateway p99 latency ≥ 3s',
metric: api.metricLatency({ period: cdk.Duration.minutes(5), statistic: 'p99' }),
threshold: 3000,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
// ── ECS Fargate: CPU + Memory utilization (no Container Insights needed) ────
private addEcsServiceAlarms(service: ecs.FargateService): void {
this.wire(
new cloudwatch.Alarm(this, 'socket-mode-cpu', {
alarmName: 'seahaven-socket-mode-cpu',
alarmDescription: 'seahaven-socket-mode ECS service CPU utilization ≥ 85%',
metric: service.metricCpuUtilization({ period: cdk.Duration.minutes(5), statistic: 'Average' }),
threshold: 85,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
this.wire(
new cloudwatch.Alarm(this, 'socket-mode-memory', {
alarmName: 'seahaven-socket-mode-memory',
alarmDescription: 'seahaven-socket-mode ECS service memory utilization ≥ 85%',
metric: service.metricMemoryUtilization({ period: cdk.Duration.minutes(5), statistic: 'Average' }),
threshold: 85,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
// ── ECS RunningTaskCount (REQUIRES Container Insights) ──────────────────────
// RunningTaskCount is published only when Container Insights is enabled on the
// cluster. desiredCount is 1, so alarm when running tasks drop below 1.
private addEcsRunningTaskAlarm(service: ecs.FargateService, cluster: ecs.ICluster): void {
const runningTasks = new cloudwatch.Metric({
namespace: 'ECS/ContainerInsights',
metricName: 'RunningTaskCount',
dimensionsMap: {
ClusterName: cluster.clusterName,
ServiceName: service.serviceName,
},
period: cdk.Duration.minutes(1),
statistic: 'Average',
});
this.wire(
new cloudwatch.Alarm(this, 'socket-mode-running-tasks', {
alarmName: 'seahaven-socket-mode-running-tasks',
alarmDescription:
'seahaven-socket-mode ECS running task count < 1 (Container Insights required)',
metric: runningTasks,
threshold: 1,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.LESS_THAN_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
}

View file

@ -29,6 +29,7 @@ export interface SlackHandlerProps {
export class SlackHandlerConstruct extends Construct {
public readonly processorLambda: lambdaNodejs.NodejsFunction;
public readonly appHomeLambda: lambdaNodejs.NodejsFunction;
public readonly qboOAuthLambda: lambdaNodejs.NodejsFunction;
public readonly api: apigatewayv2.HttpApi;
constructor(scope: Construct, id: string, props: SlackHandlerProps) {
@ -106,7 +107,7 @@ export class SlackHandlerConstruct extends Construct {
this, 'QBOSecret', 'seahaven/qbo/oauth',
);
const qboOAuthLambda = new lambdaNodejs.NodejsFunction(this, 'QBOOAuthFn', {
this.qboOAuthLambda = new lambdaNodejs.NodejsFunction(this, 'QBOOAuthFn', {
functionName: 'seahaven-qbo-oauth',
entry: path.join(__dirname, '../../lambda/qbo-oauth/index.ts'),
handler: 'handler',
@ -125,10 +126,10 @@ export class SlackHandlerConstruct extends Construct {
bundling,
});
qboSecret.grantRead(qboOAuthLambda);
qboSecret.grantWrite(qboOAuthLambda);
qboSecret.grantRead(this.qboOAuthLambda);
qboSecret.grantWrite(this.qboOAuthLambda);
const qboOAuthIntegration = new HttpLambdaIntegration('QBOOAuthIntegration', qboOAuthLambda);
const qboOAuthIntegration = new HttpLambdaIntegration('QBOOAuthIntegration', this.qboOAuthLambda);
for (const qboPath of ['/qbo/connect', '/qbo/callback', '/qbo/disconnect', '/qbo/launch']) {
this.api.addRoutes({

View file

@ -14,6 +14,9 @@ export interface SocketModeProps {
}
export class SocketModeConstruct extends Construct {
public readonly cluster: ecs.Cluster;
public readonly service: ecs.FargateService;
constructor(scope: Construct, id: string, props: SocketModeProps) {
super(scope, id);
@ -25,6 +28,7 @@ export class SocketModeConstruct extends Construct {
clusterName: 'seahaven-socket-mode',
vpc: props.vpc,
});
this.cluster = cluster;
const taskDef = new ecs.FargateTaskDefinition(this, 'TaskDef', {
memoryLimitMiB: 512,
@ -55,7 +59,7 @@ export class SocketModeConstruct extends Construct {
props.processorLambda.grantInvoke(taskDef.taskRole);
props.appHomeLambda.grantInvoke(taskDef.taskRole);
new ecs.FargateService(this, 'Service', {
this.service = new ecs.FargateService(this, 'Service', {
serviceName: 'seahaven-socket-mode',
cluster,
taskDefinition: taskDef,

View file

@ -9,6 +9,7 @@ import { SocketModeConstruct } from './constructs/socket-mode';
import { NotionSyncConstruct } from './constructs/notion-sync';
import { PoSyncConstruct } from './constructs/po-sync';
import { WorkorderSyncConstruct } from './constructs/workorder-sync';
import { MonitoringConstruct } from './constructs/monitoring';
export class SeahavenSlackBotStack extends cdk.Stack {
constructor(scope: Construct, id: string, props?: cdk.StackProps) {
@ -52,7 +53,7 @@ export class SeahavenSlackBotStack extends cdk.Stack {
});
// ── Notion → KB daily sync (EventBridge + Lambda) ────────────────────────
new NotionSyncConstruct(this, 'NotionSync', {
const notionSync = new NotionSyncConstruct(this, 'NotionSync', {
region: this.region,
kbDocsBucket: knowledgeBase.docsBucket,
knowledgeBaseId: knowledgeBase.knowledgeBase.knowledgeBaseId,
@ -60,7 +61,7 @@ export class SeahavenSlackBotStack extends cdk.Stack {
});
// ── Purchase Orders → KB daily sync (EventBridge + Lambda) ────────────────
new PoSyncConstruct(this, 'PoSync', {
const poSync = new PoSyncConstruct(this, 'PoSync', {
region: this.region,
kbDocsBucket: knowledgeBase.docsBucket,
knowledgeBaseId: knowledgeBase.knowledgeBase.knowledgeBaseId,
@ -68,7 +69,7 @@ export class SeahavenSlackBotStack extends cdk.Stack {
});
// ── Work Orders → KB daily sync (EventBridge + Lambda) ────────────────────
new WorkorderSyncConstruct(this, 'WorkorderSync', {
const workorderSync = new WorkorderSyncConstruct(this, 'WorkorderSync', {
region: this.region,
kbDocsBucket: knowledgeBase.docsBucket,
knowledgeBaseId: knowledgeBase.knowledgeBase.knowledgeBaseId,
@ -89,12 +90,37 @@ export class SeahavenSlackBotStack extends cdk.Stack {
});
// ── Socket Mode (ECS Fargate — replaces webhook Lambda) ──────────────────
new SocketModeConstruct(this, 'SocketMode', {
const socketMode = new SocketModeConstruct(this, 'SocketMode', {
vpc,
processorLambda: slackHandler.processorLambda,
appHomeLambda: slackHandler.appHomeLambda,
});
// ── CloudWatch alarm coverage → site-alerts SNS ───────────────────────────
new MonitoringConstruct(this, 'Monitoring', {
lambdas: [
{ name: 'slack-processor', fn: slackHandler.processorLambda, timeout: cdk.Duration.minutes(5) },
{ name: 'app-home', fn: slackHandler.appHomeLambda, timeout: cdk.Duration.seconds(10) },
{ name: 'qbo-oauth', fn: slackHandler.qboOAuthLambda, timeout: cdk.Duration.seconds(15) },
{ name: 'qbo-lookup', fn: bedrockAgent.qboLambda, timeout: cdk.Duration.seconds(30) },
{ name: 'maps-lookup', fn: bedrockAgent.mapsLambda, timeout: cdk.Duration.seconds(30) },
{ name: 'wo-po-lookup', fn: bedrockAgent.woPoLambda, timeout: cdk.Duration.seconds(30) },
{ name: 'po-sync', fn: poSync.syncLambda, timeout: cdk.Duration.minutes(15) },
{ name: 'workorder-sync', fn: workorderSync.syncLambda, timeout: cdk.Duration.minutes(5) },
{ name: 'notion-sync', fn: notionSync.syncLambda, timeout: cdk.Duration.minutes(5) },
],
tables: [
{ name: 'ddb-conversations', table: conversationLog.table },
{ name: 'ddb-unanswered-questions', table: conversationLog.unansweredTable },
],
httpApi: slackHandler.api,
ecsService: socketMode.service,
ecsCluster: socketMode.cluster,
// RunningTaskCount alarm requires Container Insights (cost/config change) —
// gated off by default; enabled in a separate, sign-off-gated commit.
enableRunningTaskAlarm: false,
});
// ── Stack outputs ─────────────────────────────────────────────────────────
new cdk.CfnOutput(this, 'KBDocsBucketName', {
value: knowledgeBase.docsBucket.bucketName,