feat(exec-aide): CloudWatch alarm coverage (Lambda + DynamoDB + ECS) #59

Closed
amoussa1229 wants to merge 2 commits from infra-cloudwatch-alarm-coverage into main
4 changed files with 244 additions and 6 deletions

View file

@ -52,8 +52,29 @@ The conversation Lambda has access to these tools:
### CDK Constructs
- `lib/constructs/email-pipeline.ts` — DynamoDB table, all four Lambda functions (fetch-classify, daily-digest, conversation, reminder), EventBridge schedules
- `lib/constructs/socket-mode.ts` — VPC, ECS cluster, Fargate service, ECR repo (image built automatically via `ContainerImage.fromAsset()`)
- `lib/constructs/email-pipeline.ts` — DynamoDB table, the three pipeline Lambda functions (fetch-classify, daily-digest, conversation), EventBridge schedules, and their CloudWatch alarms (Lambda + DynamoDB)
- `lib/constructs/socket-mode.ts` — VPC, ECS cluster, Fargate service, ECR repo (image built automatically via `ContainerImage.fromAsset()`), and the listener's CloudWatch alarms (CPU, memory, running-task count)
## Monitoring & Alarms
All CloudWatch alarms are ALARM-only (no OK / InsufficientData actions), use `treatMissingData: NOT_BREACHING`, and publish to the shared cross-stack `site-alerts` SNS topic (`arn:aws:sns:us-east-1:328440206208:site-alerts`, managed by `seahaven-account-baseline`, encrypted with `alias/seahaven-alarm-topics`). The topic is imported once at the stack level and injected into both constructs via props. Alarm names follow `exec-aide-<resource>-<signal>`.
| Alarm | Resource | Condition |
|---|---|---|
| `exec-aide-<fn>-errors` | each Lambda | Errors Sum ≥ 1 over 5 min |
| `exec-aide-<fn>-throttles` | each Lambda | Throttles Sum ≥ 1 over 5 min |
| `exec-aide-<fn>-duration` | each Lambda | p99 Duration > ~80% of timeout (96000 ms for the 120 s fns, 144000 ms for conversation's 180 s), eval 3 / datapoints 2 |
| `exec-aide-table-throttles` | DynamoDB `exec-aide` | ThrottledRequests Sum ≥ 1 (summed across the operations the app issues) |
| `exec-aide-table-system-errors` | DynamoDB `exec-aide` | SystemErrors Sum ≥ 1 (summed across the operations the app issues) |
| `exec-aide-listener-cpu-high` | Fargate service | CPU Average > 80%, eval 3 / datapoints 2 |
| `exec-aide-listener-memory-high` | Fargate service | Memory Average > 80%, eval 3 / datapoints 2 |
| `exec-aide-listener-no-running-tasks` | Fargate service | RunningTaskCount Minimum < 1, eval 3 / datapoints 2 |
`<fn>` ∈ {`fetch-classify`, `daily-digest`, `conversation`}.
Notes:
- DynamoDB `ThrottledRequests` / `SystemErrors` are not published at the bare `TableName` dimension — AWS keys them by `Operation`. The alarms use CDK's `*ForOperations` math helpers scoped to the six operations this single-table app issues (GetItem, PutItem, Query, Scan, UpdateItem, DeleteItem) to stay within CloudWatch's 10-metric math-expression limit.
- `exec-aide-listener-no-running-tasks` reads `RunningTaskCount` from the `ECS/ContainerInsights` namespace, which requires **Container Insights** to be enabled on the `exec-aide` cluster (a cost/config change).
## Classification Rules

View file

@ -9,14 +9,26 @@ import * as targets from 'aws-cdk-lib/aws-events-targets';
import * as scheduler from 'aws-cdk-lib/aws-scheduler';
import * as iam from 'aws-cdk-lib/aws-iam';
import * as logs from 'aws-cdk-lib/aws-logs';
import * as cloudwatch from 'aws-cdk-lib/aws-cloudwatch';
import * as cloudwatchActions from 'aws-cdk-lib/aws-cloudwatch-actions';
import * as sns from 'aws-cdk-lib/aws-sns';
import { PythonFunction } from '@aws-cdk/aws-lambda-python-alpha';
import * as path from 'path';
export interface EmailPipelineProps {
/**
* Shared site-alerts SNS topic for CloudWatch ALARM actions. Imported once at
* the stack level (sns.Topic.fromTopicArn) and injected here, mirroring how
* the table is wired into the socket-mode construct.
*/
alarmTopic: sns.ITopic;
}
export class EmailPipelineConstruct extends Construct {
public readonly table: dynamodb.Table;
public readonly conversationFn: lambda.IFunction;
constructor(scope: Construct, id: string) {
constructor(scope: Construct, id: string, props: EmailPipelineProps) {
super(scope, id);
const account = cdk.Stack.of(this).account;
@ -198,5 +210,118 @@ export class EmailPipelineConstruct extends Construct {
dailyDigest.grantInvoke(conversation);
this.conversationFn = conversation;
// ── CloudWatch Alarms ─────────────────────────────────────
// ALARM-only (no OK / InsufficientData action); treatMissingData
// NOT_BREACHING. Single shared site-alerts action, mirroring the
// proposal-system alarm construct. Names are repo-namespaced kebab-case:
// exec-aide-<fn>-<signal>.
const alarmAction = new cloudwatchActions.SnsAction(props.alarmTopic);
// Per-function timeout (ms). Duration alarms fire at ~80% of timeout so a
// function trending toward a hard timeout pages before requests start
// failing outright. These thresholds need Adam sign-off.
const lambdaAlarmSpecs: {
readonly fn: lambda.Function;
readonly name: string;
readonly durationThresholdMs: number;
}[] = [
{ fn: fetchClassify, name: 'fetch-classify', durationThresholdMs: 96000 }, // 80% of 120s
{ fn: dailyDigest, name: 'daily-digest', durationThresholdMs: 96000 }, // 80% of 120s
{ fn: conversation, name: 'conversation', durationThresholdMs: 144000 }, // 80% of 180s
];
for (const { fn, name, durationThresholdMs } of lambdaAlarmSpecs) {
// Errors: any invocation error pages.
new cloudwatch.Alarm(this, `Errors-${name}`, {
alarmName: `exec-aide-${name}-errors`,
alarmDescription: `exec-aide-${name} Lambda is erroring`,
metric: fn.metricErrors({
period: cdk.Duration.minutes(5),
statistic: 'Sum',
}),
threshold: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
evaluationPeriods: 1,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}).addAlarmAction(alarmAction);
// Throttles: concurrency exhaustion / account limits.
new cloudwatch.Alarm(this, `Throttles-${name}`, {
alarmName: `exec-aide-${name}-throttles`,
alarmDescription: `exec-aide-${name} Lambda is being throttled`,
metric: fn.metricThrottles({
period: cdk.Duration.minutes(5),
statistic: 'Sum',
}),
threshold: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
evaluationPeriods: 1,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}).addAlarmAction(alarmAction);
// Duration: p99 trending toward the timeout. eval3 / datapoints2 to ride
// out a single slow Bedrock/Gmail call without paging.
new cloudwatch.Alarm(this, `Duration-${name}`, {
alarmName: `exec-aide-${name}-duration`,
alarmDescription: `exec-aide-${name} Lambda p99 duration approaching its timeout (~80%)`,
metric: fn.metricDuration({
period: cdk.Duration.minutes(5),
statistic: 'p99',
}),
threshold: durationThresholdMs,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
evaluationPeriods: 3,
datapointsToAlarm: 2,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}).addAlarmAction(alarmAction);
}
// DynamoDB alarms. NOTE on dimensions: the raw `ThrottledRequests` and
// `SystemErrors` metrics are NOT published at the bare TableName dimension —
// AWS emits them keyed by the `Operation` dimension. CDK's
// metricThrottledRequests / metricSystemErrors helpers are deprecated for
// exactly this reason ("returns an invalid metric"). We therefore use the
// *ForOperations math helpers, which build a per-operation math expression
// scoped to TableName=exec-aide. An alarm math expression caps at 10
// metrics, and the default "all operations" set exceeds that, so we scope to
// the operations this single-table app actually issues (GetItem, PutItem,
// Query, Scan, UpdateItem, DeleteItem — confirmed from src/).
const tableOperations = [
dynamodb.Operation.GET_ITEM,
dynamodb.Operation.PUT_ITEM,
dynamodb.Operation.QUERY,
dynamodb.Operation.SCAN,
dynamodb.Operation.UPDATE_ITEM,
dynamodb.Operation.DELETE_ITEM,
];
new cloudwatch.Alarm(this, 'DdbThrottles', {
alarmName: 'exec-aide-table-throttles',
alarmDescription: 'exec-aide DynamoDB table is throttling requests (summed across the operations the app uses)',
metric: this.table.metricThrottledRequestsForOperations({
operations: tableOperations,
period: cdk.Duration.minutes(5),
statistic: 'Sum',
}),
threshold: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
evaluationPeriods: 1,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}).addAlarmAction(alarmAction);
new cloudwatch.Alarm(this, 'DdbSystemErrors', {
alarmName: 'exec-aide-table-system-errors',
alarmDescription: 'exec-aide DynamoDB table is returning SystemErrors (summed across the operations the app uses)',
metric: this.table.metricSystemErrorsForOperations({
operations: tableOperations,
period: cdk.Duration.minutes(5),
statistic: 'Sum',
}),
threshold: 1,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
evaluationPeriods: 1,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}).addAlarmAction(alarmAction);
}
}

View file

@ -7,11 +7,20 @@ import * as ecrAssets from 'aws-cdk-lib/aws-ecr-assets';
import * as logs from 'aws-cdk-lib/aws-logs';
import * as iam from 'aws-cdk-lib/aws-iam';
import * as dynamodb from 'aws-cdk-lib/aws-dynamodb';
import * as cloudwatch from 'aws-cdk-lib/aws-cloudwatch';
import * as cloudwatchActions from 'aws-cdk-lib/aws-cloudwatch-actions';
import * as sns from 'aws-cdk-lib/aws-sns';
import * as path from 'path';
export interface SocketModeProps {
table: dynamodb.ITable;
conversationFnArn: string;
/**
* Shared site-alerts SNS topic for CloudWatch ALARM actions. Imported once at
* the stack level (sns.Topic.fromTopicArn) and injected here, mirroring how
* the table is wired into this construct.
*/
alarmTopic: sns.ITopic;
}
export class SocketModeConstruct extends Construct {
@ -62,6 +71,11 @@ export class SocketModeConstruct extends Construct {
const cluster = new ecs.Cluster(this, 'Cluster', {
clusterName: 'exec-aide',
vpc,
// NEEDS ADAM SIGN-OFF (cost/config): Container Insights is required for
// the RunningTaskCount alarm below (that metric lives in the
// ECS/ContainerInsights namespace; AWS/ECS does not publish it). Enabling
// it incurs additional CloudWatch ingestion/storage cost for this cluster.
containerInsightsV2: ecs.ContainerInsights.ENABLED,
});
const taskDef = new ecs.FargateTaskDefinition(this, 'TaskDef', {
@ -123,8 +137,10 @@ export class SocketModeConstruct extends Construct {
}));
// ── Fargate Service ────────────────────────────────────────
new ecs.FargateService(this, 'Service', {
// Assigned to a const (no logical-id change vs. the prior anonymous
// construct — id 'Service' is unchanged) so the service-level CloudWatch
// metrics below can reference it.
const service = new ecs.FargateService(this, 'Service', {
serviceName: 'exec-aide-listener',
cluster,
taskDefinition: taskDef,
@ -133,5 +149,66 @@ export class SocketModeConstruct extends Construct {
securityGroups: [sg],
vpcSubnets: { subnetType: ec2.SubnetType.PUBLIC },
});
// ── CloudWatch Alarms ──────────────────────────────────────
// ALARM-only (no OK / InsufficientData action); treatMissingData
// NOT_BREACHING. Shared site-alerts action injected via props.
// CPU and Memory come from AWS/ECS service metrics (no Container Insights
// required). The RunningTaskCount alarm — which DOES require Container
// Insights — is added in a separate, sign-off-gated commit.
const alarmAction = new cloudwatchActions.SnsAction(props.alarmTopic);
new cloudwatch.Alarm(this, 'ListenerCpuHigh', {
alarmName: 'exec-aide-listener-cpu-high',
alarmDescription: 'exec-aide-listener Fargate service CPU utilization is high',
metric: service.metricCpuUtilization({
period: cdk.Duration.minutes(5),
statistic: 'Average',
}),
threshold: 80,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
evaluationPeriods: 3,
datapointsToAlarm: 2,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}).addAlarmAction(alarmAction);
new cloudwatch.Alarm(this, 'ListenerMemoryHigh', {
alarmName: 'exec-aide-listener-memory-high',
alarmDescription: 'exec-aide-listener Fargate service memory utilization is high',
metric: service.metricMemoryUtilization({
period: cdk.Duration.minutes(5),
statistic: 'Average',
}),
threshold: 80,
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
evaluationPeriods: 3,
datapointsToAlarm: 2,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}).addAlarmAction(alarmAction);
// NEEDS ADAM SIGN-OFF (cost/config): RunningTaskCount. The desiredCount is
// 1; this fires if the listener has no running task (Slack Socket Mode goes
// dark — DMs and @mentions stop being handled). RunningTaskCount is only
// emitted in the ECS/ContainerInsights namespace, which requires Container
// Insights to be enabled on the cluster (done above).
new cloudwatch.Alarm(this, 'ListenerNoRunningTasks', {
alarmName: 'exec-aide-listener-no-running-tasks',
alarmDescription: 'exec-aide-listener Fargate service has no running tasks — Slack Socket Mode is down',
metric: new cloudwatch.Metric({
namespace: 'ECS/ContainerInsights',
metricName: 'RunningTaskCount',
dimensionsMap: {
ClusterName: cluster.clusterName,
ServiceName: service.serviceName,
},
statistic: 'Minimum',
period: cdk.Duration.minutes(5),
}),
threshold: 1,
comparisonOperator: cloudwatch.ComparisonOperator.LESS_THAN_THRESHOLD,
evaluationPeriods: 3,
datapointsToAlarm: 2,
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
}).addAlarmAction(alarmAction);
}
}

View file

@ -1,4 +1,5 @@
import * as cdk from 'aws-cdk-lib';
import * as sns from 'aws-cdk-lib/aws-sns';
import { Construct } from 'constructs';
import { EmailPipelineConstruct } from './constructs/email-pipeline';
import { SocketModeConstruct } from './constructs/socket-mode';
@ -7,11 +8,25 @@ export class ExecAideStack extends cdk.Stack {
constructor(scope: Construct, id: string, props?: cdk.StackProps) {
super(scope, id, props);
const emailPipeline = new EmailPipelineConstruct(this, 'EmailPipeline');
// Shared cross-stack site-alerts SNS topic for all CloudWatch ALARM actions.
// Managed by seahaven-account-baseline and encrypted with the
// alias/seahaven-alarm-topics CMK (CloudWatch cannot publish to topics using
// alias/aws/sns — silent failure). Imported ONCE here and injected into both
// constructs via props, the same way the DynamoDB table is wired through.
const alarmTopic = sns.Topic.fromTopicArn(
this,
'SiteAlerts',
`arn:aws:sns:${this.region}:${this.account}:site-alerts`,
);
const emailPipeline = new EmailPipelineConstruct(this, 'EmailPipeline', {
alarmTopic,
});
new SocketModeConstruct(this, 'SocketMode', {
table: emailPipeline.table,
conversationFnArn: emailPipeline.conversationFn.functionArn,
alarmTopic,
});
new cdk.CfnOutput(this, 'TableName', {