Add CloudWatch alarm coverage (MonitoringConstruct) #61

Closed
amoussa1229 wants to merge 3 commits from infra-cw-alarm-coverage into main
5 changed files with 375 additions and 9 deletions

View file

@ -82,6 +82,35 @@ EventBridge (daily 02:00 UTC)
This bot reads the `purchase-orders` DynamoDB table **read-only** (via the `po-sync` and `wo-po-lookup` Lambdas, both granted `grantReadData`). The table is owned by the `procurement-ingest` repo (`po-ingest` stack), which is the sole authoritative writer. Any change to the `purchase-orders` schema must be coordinated with `procurement-ingest` (owner) and `payments-dashboard` (the other read-only consumer).
## Monitoring & Alarms
CloudWatch alarm coverage is defined in `lib/constructs/monitoring.ts` (`MonitoringConstruct`). Every alarm sends an **alarm action only** (no OK action) to the shared `site-alerts` SNS topic (`arn:aws:sns:us-east-1:328440206208:site-alerts`) and treats missing data as `NOT_BREACHING`. Alarm names are repo-namespaced kebab-case: `seahaven-<fn>-<signal>`.
| Target | Alarms |
|---|---|
| **Lambda** (all 9 functions) | `errors` (≥1 / 5 min), `throttles` (≥1 / 5 min), `duration` (p99 ≥ ~80% of each function's timeout, 3 eval / 2 datapoints) |
| **DynamoDB** `seahaven-conversations`, `seahaven-unanswered-questions` | `throttles` (ThrottledRequests), `system-errors` (SystemErrors) |
| **API Gateway v2** `seahaven-slack-webhook` | `5xx` (≥1), `4xx` (≥10 sustained), `latency` (p99 ≥ 3s) |
| **ECS Fargate** `seahaven-socket-mode` | `cpu` (≥85%), `memory` (≥85%), `running-tasks` (< 1)¹ |
Per-function Duration thresholds (p99, ~80% of timeout):
| Function | Timeout | Threshold |
|---|---|---|
| `seahaven-slack-processor` | 5 min | 240 s |
| `seahaven-app-home` | 10 s | 8 s |
| `seahaven-qbo-oauth` | 15 s | 12 s |
| `seahaven-qbo-lookup` | 30 s | 24 s |
| `seahaven-maps-lookup` | 30 s | 24 s |
| `seahaven-wo-po-lookup` | 30 s | 24 s |
| `seahaven-po-sync` | 15 min | 720 s |
| `seahaven-workorder-sync` | 5 min | 240 s |
| `seahaven-notion-sync` | 5 min | 240 s |
**DynamoDB metric note:** `ThrottledRequests` and `SystemErrors` have no valid `TableName`-only aggregate (the bare CDK helpers are deprecated). The alarms use the per-operations metric-math helpers, scoped to the six CRUD operations these tables use (GetItem, PutItem, UpdateItem, DeleteItem, Query, BatchWriteItem) to stay under CloudWatch's 10-metric alarm-math cap.
¹ **`running-tasks` requires Container Insights** on the `seahaven-socket-mode` cluster (`containerInsightsV2: ENABLED`), which adds CloudWatch metric + log-ingestion cost. The CPU and Memory alarms use the standard `AWS/ECS` namespace and need no Container Insights.
## Prerequisites
- Node.js 22+
@ -170,6 +199,7 @@ lib/
notion-sync.ts EventBridge daily cron + notion-sync Lambda
po-sync.ts EventBridge daily cron + po-sync Lambda
workorder-sync.ts EventBridge daily cron + workorder-sync Lambda
monitoring.ts CloudWatch alarm coverage → site-alerts SNS topic
services/
socket-mode/ Socket Mode ECS service (Dockerfile, TypeScript, @slack/socket-mode)
lambda/

View file

@ -0,0 +1,299 @@
import * as cdk from 'aws-cdk-lib';
import { Construct } from 'constructs';
import * as cloudwatch from 'aws-cdk-lib/aws-cloudwatch';
import * as cwActions from 'aws-cdk-lib/aws-cloudwatch-actions';
import * as sns from 'aws-cdk-lib/aws-sns';
import * as lambda from 'aws-cdk-lib/aws-lambda';
import * as dynamodb from 'aws-cdk-lib/aws-dynamodb';
import * as apigatewayv2 from 'aws-cdk-lib/aws-apigatewayv2';
import * as ecs from 'aws-cdk-lib/aws-ecs';
/**
* A Lambda function plus the short name used to label its alarms.
* `name` becomes the `seahaven-<name>-<signal>` alarm-name prefix and must
* match the function's kebab-case short name (e.g. 'slack-processor').
*/
export interface MonitoredLambda {
name: string;
fn: lambda.IFunction;
/** Function timeout — used to derive the p99 Duration threshold (~80% of timeout). */
timeout: cdk.Duration;
}
/** A DynamoDB table plus the short name used to label its alarms. */
export interface MonitoredTable {
/** kebab-case short name, e.g. 'ddb-conversations'. */
name: string;
table: dynamodb.Table;
}
export interface MonitoringConstructProps {
/** Lambdas to cover with Errors + Throttles + Duration alarms. */
lambdas: MonitoredLambda[];
/** In-stack DynamoDB tables to cover with throttle + system-error alarms. */
tables: MonitoredTable[];
/** HTTP API (API Gateway v2) to cover with 5xx/4xx/Latency alarms. */
httpApi: apigatewayv2.HttpApi;
/** ECS Fargate service to cover with CPU/Memory utilization alarms. */
ecsService: ecs.FargateService;
/**
* ECS cluster — required for the RunningTaskCount alarm, which depends on
* Container Insights being enabled (gated behind `enableRunningTaskAlarm`).
*/
ecsCluster: ecs.ICluster;
/**
* When true, add the RunningTaskCount alarm. The metric only emits when
* Container Insights is enabled on the cluster — enabling it is a separate,
* cost-bearing config change that must be made on the cluster itself.
* Defaults to false so the alarm is opt-in.
*/
enableRunningTaskAlarm?: boolean;
}
/**
* Centralised CloudWatch alarm coverage for the seahaven-slack-bot stack.
*
* Every alarm:
* - notifies the shared `site-alerts` SNS topic (alarm action only, no OK action)
* - treats missing data as NOT_BREACHING
* - is named `seahaven-<fn>-<signal>` (repo-namespaced kebab-case)
*/
export class MonitoringConstruct extends Construct {
private readonly alertsTopic: sns.ITopic;
constructor(scope: Construct, id: string, props: MonitoringConstructProps) {
super(scope, id);
// Shared site-wide alerts topic — imported ONCE, reused for every alarm.
this.alertsTopic = sns.Topic.fromTopicArn(
this,
'SiteAlerts',
'arn:aws:sns:us-east-1:328440206208:site-alerts',
);
for (const ml of props.lambdas) {
this.addLambdaAlarms(ml);
}
for (const mt of props.tables) {
this.addDynamoAlarms(mt);
}
this.addApiGatewayAlarms(props.httpApi);
this.addEcsServiceAlarms(props.ecsService);
if (props.enableRunningTaskAlarm) {
this.addEcsRunningTaskAlarm(props.ecsService, props.ecsCluster);
}
}
/** Attach the SNS alarm action (no OK action) and return the alarm. */
private wire(alarm: cloudwatch.Alarm): cloudwatch.Alarm {
alarm.addAlarmAction(new cwActions.SnsAction(this.alertsTopic));
return alarm;
}
// ── Lambda: Errors + Throttles + Duration ──────────────────────────────────
private addLambdaAlarms(ml: MonitoredLambda): void {
const { name, fn, timeout } = ml;
// Errors — any invocation error over a 5-min window.
this.wire(
new cloudwatch.Alarm(this, `${name}-errors`, {
alarmName: `seahaven-${name}-errors`,
alarmDescription: `seahaven-${name} Lambda invocation errors`,
metric: fn.metricErrors({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
// Throttles — concurrency exhaustion.
this.wire(
new cloudwatch.Alarm(this, `${name}-throttles`, {
alarmName: `seahaven-${name}-throttles`,
alarmDescription: `seahaven-${name} Lambda throttles`,
metric: fn.metricThrottles({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
// Duration — p99 approaching the timeout (~80%). eval3/datapoints2 to ride
// out single slow invocations while still catching sustained latency.
const thresholdMs = Math.round(timeout.toMilliseconds() * 0.8);
this.wire(
new cloudwatch.Alarm(this, `${name}-duration`, {
alarmName: `seahaven-${name}-duration`,
alarmDescription: `seahaven-${name} Lambda p99 duration ≥ 80% of ${timeout.toSeconds()}s timeout`,
metric: fn.metricDuration({ period: cdk.Duration.minutes(5), statistic: 'p99' }),
threshold: thresholdMs,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
// ── DynamoDB: ThrottledRequests + SystemErrors ─────────────────────────────
// ThrottledRequests/SystemErrors are emitted per TableName+Operation (there is
// no valid TableName-only aggregate — the bare metricThrottledRequests /
// metricSystemErrors helpers are deprecated). The per-operations helpers build
// metric-math summing across operations; CloudWatch caps an alarm math
// expression at 10 metrics, and DynamoDB defines 14 operations — so we scope to
// the operations these tables actually use (read/write CRUD paths).
private static readonly DDB_OPERATIONS: dynamodb.Operation[] = [
dynamodb.Operation.GET_ITEM,
dynamodb.Operation.PUT_ITEM,
dynamodb.Operation.UPDATE_ITEM,
dynamodb.Operation.DELETE_ITEM,
dynamodb.Operation.QUERY,
dynamodb.Operation.BATCH_WRITE_ITEM,
];
private addDynamoAlarms(mt: MonitoredTable): void {
const { name: shortName, table } = mt;
this.wire(
new cloudwatch.Alarm(this, `${shortName}-throttles`, {
alarmName: `seahaven-${shortName}-throttles`,
alarmDescription: `${table.tableName} DynamoDB throttled requests`,
metric: table.metricThrottledRequestsForOperations({
operations: MonitoringConstruct.DDB_OPERATIONS,
period: cdk.Duration.minutes(5),
statistic: 'Sum',
}),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
this.wire(
new cloudwatch.Alarm(this, `${shortName}-system-errors`, {
alarmName: `seahaven-${shortName}-system-errors`,
alarmDescription: `${table.tableName} DynamoDB system errors (5xx)`,
metric: table.metricSystemErrorsForOperations({
operations: MonitoringConstruct.DDB_OPERATIONS,
period: cdk.Duration.minutes(5),
statistic: 'Sum',
}),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
// ── API Gateway v2 (HTTP API): 5xx + 4xx + Latency ─────────────────────────
private addApiGatewayAlarms(api: apigatewayv2.HttpApi): void {
// metricServerError/metricClientError/metricLatency resolve to the v2
// metric names (5xx/4xx/Latency) under the ApiId dimension automatically.
this.wire(
new cloudwatch.Alarm(this, 'slack-webhook-5xx', {
alarmName: 'seahaven-slack-webhook-5xx',
alarmDescription: 'seahaven-slack-webhook API Gateway 5xx errors',
metric: api.metricServerError({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
threshold: 1,
evaluationPeriods: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
// 4xx is noisier (bad OAuth callbacks, scanners) — require a sustained
// burst rather than a single request. eval3/datapoints2 over 5-min periods.
this.wire(
new cloudwatch.Alarm(this, 'slack-webhook-4xx', {
alarmName: 'seahaven-slack-webhook-4xx',
alarmDescription: 'seahaven-slack-webhook API Gateway sustained 4xx errors',
metric: api.metricClientError({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
threshold: 10,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
// Latency p99 ≥ 3s — OAuth routes call Intuit; allow headroom.
this.wire(
new cloudwatch.Alarm(this, 'slack-webhook-latency', {
alarmName: 'seahaven-slack-webhook-latency',
alarmDescription: 'seahaven-slack-webhook API Gateway p99 latency ≥ 3s',
metric: api.metricLatency({ period: cdk.Duration.minutes(5), statistic: 'p99' }),
threshold: 3000,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
// ── ECS Fargate: CPU + Memory utilization (no Container Insights needed) ────
private addEcsServiceAlarms(service: ecs.FargateService): void {
this.wire(
new cloudwatch.Alarm(this, 'socket-mode-cpu', {
alarmName: 'seahaven-socket-mode-cpu',
alarmDescription: 'seahaven-socket-mode ECS service CPU utilization ≥ 85%',
metric: service.metricCpuUtilization({ period: cdk.Duration.minutes(5), statistic: 'Average' }),
threshold: 85,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
this.wire(
new cloudwatch.Alarm(this, 'socket-mode-memory', {
alarmName: 'seahaven-socket-mode-memory',
alarmDescription: 'seahaven-socket-mode ECS service memory utilization ≥ 85%',
metric: service.metricMemoryUtilization({ period: cdk.Duration.minutes(5), statistic: 'Average' }),
threshold: 85,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
// ── ECS RunningTaskCount (REQUIRES Container Insights) ──────────────────────
// RunningTaskCount is published only when Container Insights is enabled on the
// cluster. desiredCount is 1, so alarm when running tasks drop below 1.
private addEcsRunningTaskAlarm(service: ecs.FargateService, cluster: ecs.ICluster): void {
const runningTasks = new cloudwatch.Metric({
namespace: 'ECS/ContainerInsights',
metricName: 'RunningTaskCount',
dimensionsMap: {
ClusterName: cluster.clusterName,
ServiceName: service.serviceName,
},
period: cdk.Duration.minutes(1),
statistic: 'Average',
});
this.wire(
new cloudwatch.Alarm(this, 'socket-mode-running-tasks', {
alarmName: 'seahaven-socket-mode-running-tasks',
alarmDescription:
'seahaven-socket-mode ECS running task count < 1 (Container Insights required)',
metric: runningTasks,
threshold: 1,
evaluationPeriods: 3,
datapointsToAlarm: 2,
comparisonOperator: cloudwatch.ComparisonOperator.LESS_THAN_THRESHOLD,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}),
);
}
}

View file

@ -29,6 +29,7 @@ export interface SlackHandlerProps {
export class SlackHandlerConstruct extends Construct {
public readonly processorLambda: lambdaNodejs.NodejsFunction;
public readonly appHomeLambda: lambdaNodejs.NodejsFunction;
public readonly qboOAuthLambda: lambdaNodejs.NodejsFunction;
public readonly api: apigatewayv2.HttpApi;
constructor(scope: Construct, id: string, props: SlackHandlerProps) {
@ -106,7 +107,7 @@ export class SlackHandlerConstruct extends Construct {
this, 'QBOSecret', 'seahaven/qbo/oauth',
);
const qboOAuthLambda = new lambdaNodejs.NodejsFunction(this, 'QBOOAuthFn', {
this.qboOAuthLambda = new lambdaNodejs.NodejsFunction(this, 'QBOOAuthFn', {
functionName: 'seahaven-qbo-oauth',
entry: path.join(__dirname, '../../lambda/qbo-oauth/index.ts'),
handler: 'handler',
@ -125,10 +126,10 @@ export class SlackHandlerConstruct extends Construct {
bundling,
});
qboSecret.grantRead(qboOAuthLambda);
qboSecret.grantWrite(qboOAuthLambda);
qboSecret.grantRead(this.qboOAuthLambda);
qboSecret.grantWrite(this.qboOAuthLambda);
const qboOAuthIntegration = new HttpLambdaIntegration('QBOOAuthIntegration', qboOAuthLambda);
const qboOAuthIntegration = new HttpLambdaIntegration('QBOOAuthIntegration', this.qboOAuthLambda);
for (const qboPath of ['/qbo/connect', '/qbo/callback', '/qbo/disconnect', '/qbo/launch']) {
this.api.addRoutes({

View file

@ -14,6 +14,9 @@ export interface SocketModeProps {
}
export class SocketModeConstruct extends Construct {
public readonly cluster: ecs.Cluster;
public readonly service: ecs.FargateService;
constructor(scope: Construct, id: string, props: SocketModeProps) {
super(scope, id);
@ -24,7 +27,12 @@ export class SocketModeConstruct extends Construct {
const cluster = new ecs.Cluster(this, 'Cluster', {
clusterName: 'seahaven-socket-mode',
vpc: props.vpc,
// Container Insights publishes ECS/ContainerInsights metrics (incl.
// RunningTaskCount) needed for the task-count alarm. This adds CloudWatch
// metric + log cost — see PR body "NEEDS ADAM SIGN-OFF (cost/config)".
containerInsightsV2: ecs.ContainerInsights.ENABLED,
});
this.cluster = cluster;
const taskDef = new ecs.FargateTaskDefinition(this, 'TaskDef', {
memoryLimitMiB: 512,
@ -55,7 +63,7 @@ export class SocketModeConstruct extends Construct {
props.processorLambda.grantInvoke(taskDef.taskRole);
props.appHomeLambda.grantInvoke(taskDef.taskRole);
new ecs.FargateService(this, 'Service', {
this.service = new ecs.FargateService(this, 'Service', {
serviceName: 'seahaven-socket-mode',
cluster,
taskDefinition: taskDef,

View file

@ -9,6 +9,7 @@ import { SocketModeConstruct } from './constructs/socket-mode';
import { NotionSyncConstruct } from './constructs/notion-sync';
import { PoSyncConstruct } from './constructs/po-sync';
import { WorkorderSyncConstruct } from './constructs/workorder-sync';
import { MonitoringConstruct } from './constructs/monitoring';
export class SeahavenSlackBotStack extends cdk.Stack {
constructor(scope: Construct, id: string, props?: cdk.StackProps) {
@ -52,7 +53,7 @@ export class SeahavenSlackBotStack extends cdk.Stack {
});
// ── Notion → KB daily sync (EventBridge + Lambda) ────────────────────────
new NotionSyncConstruct(this, 'NotionSync', {
const notionSync = new NotionSyncConstruct(this, 'NotionSync', {
region: this.region,
kbDocsBucket: knowledgeBase.docsBucket,
knowledgeBaseId: knowledgeBase.knowledgeBase.knowledgeBaseId,
@ -60,7 +61,7 @@ export class SeahavenSlackBotStack extends cdk.Stack {
});
// ── Purchase Orders → KB daily sync (EventBridge + Lambda) ────────────────
new PoSyncConstruct(this, 'PoSync', {
const poSync = new PoSyncConstruct(this, 'PoSync', {
region: this.region,
kbDocsBucket: knowledgeBase.docsBucket,
knowledgeBaseId: knowledgeBase.knowledgeBase.knowledgeBaseId,
@ -68,7 +69,7 @@ export class SeahavenSlackBotStack extends cdk.Stack {
});
// ── Work Orders → KB daily sync (EventBridge + Lambda) ────────────────────
new WorkorderSyncConstruct(this, 'WorkorderSync', {
const workorderSync = new WorkorderSyncConstruct(this, 'WorkorderSync', {
region: this.region,
kbDocsBucket: knowledgeBase.docsBucket,
knowledgeBaseId: knowledgeBase.knowledgeBase.knowledgeBaseId,
@ -89,12 +90,39 @@ export class SeahavenSlackBotStack extends cdk.Stack {
});
// ── Socket Mode (ECS Fargate — replaces webhook Lambda) ──────────────────
new SocketModeConstruct(this, 'SocketMode', {
const socketMode = new SocketModeConstruct(this, 'SocketMode', {
vpc,
processorLambda: slackHandler.processorLambda,
appHomeLambda: slackHandler.appHomeLambda,
});
// ── CloudWatch alarm coverage → site-alerts SNS ───────────────────────────
new MonitoringConstruct(this, 'Monitoring', {
lambdas: [
{ name: 'slack-processor', fn: slackHandler.processorLambda, timeout: cdk.Duration.minutes(5) },
{ name: 'app-home', fn: slackHandler.appHomeLambda, timeout: cdk.Duration.seconds(10) },
{ name: 'qbo-oauth', fn: slackHandler.qboOAuthLambda, timeout: cdk.Duration.seconds(15) },
{ name: 'qbo-lookup', fn: bedrockAgent.qboLambda, timeout: cdk.Duration.seconds(30) },
{ name: 'maps-lookup', fn: bedrockAgent.mapsLambda, timeout: cdk.Duration.seconds(30) },
{ name: 'wo-po-lookup', fn: bedrockAgent.woPoLambda, timeout: cdk.Duration.seconds(30) },
{ name: 'po-sync', fn: poSync.syncLambda, timeout: cdk.Duration.minutes(15) },
{ name: 'workorder-sync', fn: workorderSync.syncLambda, timeout: cdk.Duration.minutes(5) },
{ name: 'notion-sync', fn: notionSync.syncLambda, timeout: cdk.Duration.minutes(5) },
],
tables: [
{ name: 'ddb-conversations', table: conversationLog.table },
{ name: 'ddb-unanswered-questions', table: conversationLog.unansweredTable },
],
httpApi: slackHandler.api,
ecsService: socketMode.service,
ecsCluster: socketMode.cluster,
// RunningTaskCount alarm requires Container Insights, enabled on the
// socket-mode cluster (see socket-mode.ts). This adds CloudWatch metric +
// log cost — flagged in the PR as "NEEDS ADAM SIGN-OFF (cost/config)".
// Drop this commit (and the cluster's containerInsightsV2 line) to decline.
enableRunningTaskAlarm: true,
});
// ── Stack outputs ─────────────────────────────────────────────────────────
new cdk.CfnOutput(this, 'KBDocsBucketName', {
value: knowledgeBase.docsBucket.bucketName,