Add CloudWatch alarm coverage (MonitoringConstruct) #61
5 changed files with 375 additions and 9 deletions
30
README.md
30
README.md
|
|
@ -82,6 +82,35 @@ EventBridge (daily 02:00 UTC)
|
|||
|
||||
This bot reads the `purchase-orders` DynamoDB table **read-only** (via the `po-sync` and `wo-po-lookup` Lambdas, both granted `grantReadData`). The table is owned by the `procurement-ingest` repo (`po-ingest` stack), which is the sole authoritative writer. Any change to the `purchase-orders` schema must be coordinated with `procurement-ingest` (owner) and `payments-dashboard` (the other read-only consumer).
|
||||
|
||||
## Monitoring & Alarms
|
||||
|
||||
CloudWatch alarm coverage is defined in `lib/constructs/monitoring.ts` (`MonitoringConstruct`). Every alarm sends an **alarm action only** (no OK action) to the shared `site-alerts` SNS topic (`arn:aws:sns:us-east-1:328440206208:site-alerts`) and treats missing data as `NOT_BREACHING`. Alarm names are repo-namespaced kebab-case: `seahaven-<fn>-<signal>`.
|
||||
|
||||
| Target | Alarms |
|
||||
|---|---|
|
||||
| **Lambda** (all 9 functions) | `errors` (≥1 / 5 min), `throttles` (≥1 / 5 min), `duration` (p99 ≥ ~80% of each function's timeout, 3 eval / 2 datapoints) |
|
||||
| **DynamoDB** `seahaven-conversations`, `seahaven-unanswered-questions` | `throttles` (ThrottledRequests), `system-errors` (SystemErrors) |
|
||||
| **API Gateway v2** `seahaven-slack-webhook` | `5xx` (≥1), `4xx` (≥10 sustained), `latency` (p99 ≥ 3s) |
|
||||
| **ECS Fargate** `seahaven-socket-mode` | `cpu` (≥85%), `memory` (≥85%), `running-tasks` (< 1)¹ |
|
||||
|
||||
Per-function Duration thresholds (p99, ~80% of timeout):
|
||||
|
||||
| Function | Timeout | Threshold |
|
||||
|---|---|---|
|
||||
| `seahaven-slack-processor` | 5 min | 240 s |
|
||||
| `seahaven-app-home` | 10 s | 8 s |
|
||||
| `seahaven-qbo-oauth` | 15 s | 12 s |
|
||||
| `seahaven-qbo-lookup` | 30 s | 24 s |
|
||||
| `seahaven-maps-lookup` | 30 s | 24 s |
|
||||
| `seahaven-wo-po-lookup` | 30 s | 24 s |
|
||||
| `seahaven-po-sync` | 15 min | 720 s |
|
||||
| `seahaven-workorder-sync` | 5 min | 240 s |
|
||||
| `seahaven-notion-sync` | 5 min | 240 s |
|
||||
|
||||
**DynamoDB metric note:** `ThrottledRequests` and `SystemErrors` have no valid `TableName`-only aggregate (the bare CDK helpers are deprecated). The alarms use the per-operations metric-math helpers, scoped to the six CRUD operations these tables use (GetItem, PutItem, UpdateItem, DeleteItem, Query, BatchWriteItem) to stay under CloudWatch's 10-metric alarm-math cap.
|
||||
|
||||
¹ **`running-tasks` requires Container Insights** on the `seahaven-socket-mode` cluster (`containerInsightsV2: ENABLED`), which adds CloudWatch metric + log-ingestion cost. The CPU and Memory alarms use the standard `AWS/ECS` namespace and need no Container Insights.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Node.js 22+
|
||||
|
|
@ -170,6 +199,7 @@ lib/
|
|||
notion-sync.ts EventBridge daily cron + notion-sync Lambda
|
||||
po-sync.ts EventBridge daily cron + po-sync Lambda
|
||||
workorder-sync.ts EventBridge daily cron + workorder-sync Lambda
|
||||
monitoring.ts CloudWatch alarm coverage → site-alerts SNS topic
|
||||
services/
|
||||
socket-mode/ Socket Mode ECS service (Dockerfile, TypeScript, @slack/socket-mode)
|
||||
lambda/
|
||||
|
|
|
|||
299
lib/constructs/monitoring.ts
Normal file
299
lib/constructs/monitoring.ts
Normal file
|
|
@ -0,0 +1,299 @@
|
|||
import * as cdk from 'aws-cdk-lib';
|
||||
import { Construct } from 'constructs';
|
||||
import * as cloudwatch from 'aws-cdk-lib/aws-cloudwatch';
|
||||
import * as cwActions from 'aws-cdk-lib/aws-cloudwatch-actions';
|
||||
import * as sns from 'aws-cdk-lib/aws-sns';
|
||||
import * as lambda from 'aws-cdk-lib/aws-lambda';
|
||||
import * as dynamodb from 'aws-cdk-lib/aws-dynamodb';
|
||||
import * as apigatewayv2 from 'aws-cdk-lib/aws-apigatewayv2';
|
||||
import * as ecs from 'aws-cdk-lib/aws-ecs';
|
||||
|
||||
/**
|
||||
* A Lambda function plus the short name used to label its alarms.
|
||||
* `name` becomes the `seahaven-<name>-<signal>` alarm-name prefix and must
|
||||
* match the function's kebab-case short name (e.g. 'slack-processor').
|
||||
*/
|
||||
export interface MonitoredLambda {
|
||||
name: string;
|
||||
fn: lambda.IFunction;
|
||||
/** Function timeout — used to derive the p99 Duration threshold (~80% of timeout). */
|
||||
timeout: cdk.Duration;
|
||||
}
|
||||
|
||||
/** A DynamoDB table plus the short name used to label its alarms. */
|
||||
export interface MonitoredTable {
|
||||
/** kebab-case short name, e.g. 'ddb-conversations'. */
|
||||
name: string;
|
||||
table: dynamodb.Table;
|
||||
}
|
||||
|
||||
export interface MonitoringConstructProps {
|
||||
/** Lambdas to cover with Errors + Throttles + Duration alarms. */
|
||||
lambdas: MonitoredLambda[];
|
||||
/** In-stack DynamoDB tables to cover with throttle + system-error alarms. */
|
||||
tables: MonitoredTable[];
|
||||
/** HTTP API (API Gateway v2) to cover with 5xx/4xx/Latency alarms. */
|
||||
httpApi: apigatewayv2.HttpApi;
|
||||
/** ECS Fargate service to cover with CPU/Memory utilization alarms. */
|
||||
ecsService: ecs.FargateService;
|
||||
/**
|
||||
* ECS cluster — required for the RunningTaskCount alarm, which depends on
|
||||
* Container Insights being enabled (gated behind `enableRunningTaskAlarm`).
|
||||
*/
|
||||
ecsCluster: ecs.ICluster;
|
||||
/**
|
||||
* When true, add the RunningTaskCount alarm. The metric only emits when
|
||||
* Container Insights is enabled on the cluster — enabling it is a separate,
|
||||
* cost-bearing config change that must be made on the cluster itself.
|
||||
* Defaults to false so the alarm is opt-in.
|
||||
*/
|
||||
enableRunningTaskAlarm?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Centralised CloudWatch alarm coverage for the seahaven-slack-bot stack.
|
||||
*
|
||||
* Every alarm:
|
||||
* - notifies the shared `site-alerts` SNS topic (alarm action only, no OK action)
|
||||
* - treats missing data as NOT_BREACHING
|
||||
* - is named `seahaven-<fn>-<signal>` (repo-namespaced kebab-case)
|
||||
*/
|
||||
export class MonitoringConstruct extends Construct {
|
||||
private readonly alertsTopic: sns.ITopic;
|
||||
|
||||
constructor(scope: Construct, id: string, props: MonitoringConstructProps) {
|
||||
super(scope, id);
|
||||
|
||||
// Shared site-wide alerts topic — imported ONCE, reused for every alarm.
|
||||
this.alertsTopic = sns.Topic.fromTopicArn(
|
||||
this,
|
||||
'SiteAlerts',
|
||||
'arn:aws:sns:us-east-1:328440206208:site-alerts',
|
||||
);
|
||||
|
||||
for (const ml of props.lambdas) {
|
||||
this.addLambdaAlarms(ml);
|
||||
}
|
||||
|
||||
for (const mt of props.tables) {
|
||||
this.addDynamoAlarms(mt);
|
||||
}
|
||||
|
||||
this.addApiGatewayAlarms(props.httpApi);
|
||||
this.addEcsServiceAlarms(props.ecsService);
|
||||
|
||||
if (props.enableRunningTaskAlarm) {
|
||||
this.addEcsRunningTaskAlarm(props.ecsService, props.ecsCluster);
|
||||
}
|
||||
}
|
||||
|
||||
/** Attach the SNS alarm action (no OK action) and return the alarm. */
|
||||
private wire(alarm: cloudwatch.Alarm): cloudwatch.Alarm {
|
||||
alarm.addAlarmAction(new cwActions.SnsAction(this.alertsTopic));
|
||||
return alarm;
|
||||
}
|
||||
|
||||
// ── Lambda: Errors + Throttles + Duration ──────────────────────────────────
|
||||
private addLambdaAlarms(ml: MonitoredLambda): void {
|
||||
const { name, fn, timeout } = ml;
|
||||
|
||||
// Errors — any invocation error over a 5-min window.
|
||||
this.wire(
|
||||
new cloudwatch.Alarm(this, `${name}-errors`, {
|
||||
alarmName: `seahaven-${name}-errors`,
|
||||
alarmDescription: `seahaven-${name} Lambda invocation errors`,
|
||||
metric: fn.metricErrors({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
|
||||
threshold: 1,
|
||||
evaluationPeriods: 1,
|
||||
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
||||
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
||||
}),
|
||||
);
|
||||
|
||||
// Throttles — concurrency exhaustion.
|
||||
this.wire(
|
||||
new cloudwatch.Alarm(this, `${name}-throttles`, {
|
||||
alarmName: `seahaven-${name}-throttles`,
|
||||
alarmDescription: `seahaven-${name} Lambda throttles`,
|
||||
metric: fn.metricThrottles({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
|
||||
threshold: 1,
|
||||
evaluationPeriods: 1,
|
||||
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
||||
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
||||
}),
|
||||
);
|
||||
|
||||
// Duration — p99 approaching the timeout (~80%). eval3/datapoints2 to ride
|
||||
// out single slow invocations while still catching sustained latency.
|
||||
const thresholdMs = Math.round(timeout.toMilliseconds() * 0.8);
|
||||
this.wire(
|
||||
new cloudwatch.Alarm(this, `${name}-duration`, {
|
||||
alarmName: `seahaven-${name}-duration`,
|
||||
alarmDescription: `seahaven-${name} Lambda p99 duration ≥ 80% of ${timeout.toSeconds()}s timeout`,
|
||||
metric: fn.metricDuration({ period: cdk.Duration.minutes(5), statistic: 'p99' }),
|
||||
threshold: thresholdMs,
|
||||
evaluationPeriods: 3,
|
||||
datapointsToAlarm: 2,
|
||||
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
||||
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
||||
}),
|
||||
);
|
||||
}
|
||||
|
||||
// ── DynamoDB: ThrottledRequests + SystemErrors ─────────────────────────────
|
||||
// ThrottledRequests/SystemErrors are emitted per TableName+Operation (there is
|
||||
// no valid TableName-only aggregate — the bare metricThrottledRequests /
|
||||
// metricSystemErrors helpers are deprecated). The per-operations helpers build
|
||||
// metric-math summing across operations; CloudWatch caps an alarm math
|
||||
// expression at 10 metrics, and DynamoDB defines 14 operations — so we scope to
|
||||
// the operations these tables actually use (read/write CRUD paths).
|
||||
private static readonly DDB_OPERATIONS: dynamodb.Operation[] = [
|
||||
dynamodb.Operation.GET_ITEM,
|
||||
dynamodb.Operation.PUT_ITEM,
|
||||
dynamodb.Operation.UPDATE_ITEM,
|
||||
dynamodb.Operation.DELETE_ITEM,
|
||||
dynamodb.Operation.QUERY,
|
||||
dynamodb.Operation.BATCH_WRITE_ITEM,
|
||||
];
|
||||
|
||||
private addDynamoAlarms(mt: MonitoredTable): void {
|
||||
const { name: shortName, table } = mt;
|
||||
|
||||
this.wire(
|
||||
new cloudwatch.Alarm(this, `${shortName}-throttles`, {
|
||||
alarmName: `seahaven-${shortName}-throttles`,
|
||||
alarmDescription: `${table.tableName} DynamoDB throttled requests`,
|
||||
metric: table.metricThrottledRequestsForOperations({
|
||||
operations: MonitoringConstruct.DDB_OPERATIONS,
|
||||
period: cdk.Duration.minutes(5),
|
||||
statistic: 'Sum',
|
||||
}),
|
||||
threshold: 1,
|
||||
evaluationPeriods: 1,
|
||||
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
||||
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
||||
}),
|
||||
);
|
||||
|
||||
this.wire(
|
||||
new cloudwatch.Alarm(this, `${shortName}-system-errors`, {
|
||||
alarmName: `seahaven-${shortName}-system-errors`,
|
||||
alarmDescription: `${table.tableName} DynamoDB system errors (5xx)`,
|
||||
metric: table.metricSystemErrorsForOperations({
|
||||
operations: MonitoringConstruct.DDB_OPERATIONS,
|
||||
period: cdk.Duration.minutes(5),
|
||||
statistic: 'Sum',
|
||||
}),
|
||||
threshold: 1,
|
||||
evaluationPeriods: 1,
|
||||
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
||||
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
||||
}),
|
||||
);
|
||||
}
|
||||
|
||||
// ── API Gateway v2 (HTTP API): 5xx + 4xx + Latency ─────────────────────────
|
||||
private addApiGatewayAlarms(api: apigatewayv2.HttpApi): void {
|
||||
// metricServerError/metricClientError/metricLatency resolve to the v2
|
||||
// metric names (5xx/4xx/Latency) under the ApiId dimension automatically.
|
||||
this.wire(
|
||||
new cloudwatch.Alarm(this, 'slack-webhook-5xx', {
|
||||
alarmName: 'seahaven-slack-webhook-5xx',
|
||||
alarmDescription: 'seahaven-slack-webhook API Gateway 5xx errors',
|
||||
metric: api.metricServerError({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
|
||||
threshold: 1,
|
||||
evaluationPeriods: 1,
|
||||
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
||||
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
||||
}),
|
||||
);
|
||||
|
||||
// 4xx is noisier (bad OAuth callbacks, scanners) — require a sustained
|
||||
// burst rather than a single request. eval3/datapoints2 over 5-min periods.
|
||||
this.wire(
|
||||
new cloudwatch.Alarm(this, 'slack-webhook-4xx', {
|
||||
alarmName: 'seahaven-slack-webhook-4xx',
|
||||
alarmDescription: 'seahaven-slack-webhook API Gateway sustained 4xx errors',
|
||||
metric: api.metricClientError({ period: cdk.Duration.minutes(5), statistic: 'Sum' }),
|
||||
threshold: 10,
|
||||
evaluationPeriods: 3,
|
||||
datapointsToAlarm: 2,
|
||||
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
||||
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
||||
}),
|
||||
);
|
||||
|
||||
// Latency p99 ≥ 3s — OAuth routes call Intuit; allow headroom.
|
||||
this.wire(
|
||||
new cloudwatch.Alarm(this, 'slack-webhook-latency', {
|
||||
alarmName: 'seahaven-slack-webhook-latency',
|
||||
alarmDescription: 'seahaven-slack-webhook API Gateway p99 latency ≥ 3s',
|
||||
metric: api.metricLatency({ period: cdk.Duration.minutes(5), statistic: 'p99' }),
|
||||
threshold: 3000,
|
||||
evaluationPeriods: 3,
|
||||
datapointsToAlarm: 2,
|
||||
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
||||
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
||||
}),
|
||||
);
|
||||
}
|
||||
|
||||
// ── ECS Fargate: CPU + Memory utilization (no Container Insights needed) ────
|
||||
private addEcsServiceAlarms(service: ecs.FargateService): void {
|
||||
this.wire(
|
||||
new cloudwatch.Alarm(this, 'socket-mode-cpu', {
|
||||
alarmName: 'seahaven-socket-mode-cpu',
|
||||
alarmDescription: 'seahaven-socket-mode ECS service CPU utilization ≥ 85%',
|
||||
metric: service.metricCpuUtilization({ period: cdk.Duration.minutes(5), statistic: 'Average' }),
|
||||
threshold: 85,
|
||||
evaluationPeriods: 3,
|
||||
datapointsToAlarm: 2,
|
||||
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
||||
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
||||
}),
|
||||
);
|
||||
|
||||
this.wire(
|
||||
new cloudwatch.Alarm(this, 'socket-mode-memory', {
|
||||
alarmName: 'seahaven-socket-mode-memory',
|
||||
alarmDescription: 'seahaven-socket-mode ECS service memory utilization ≥ 85%',
|
||||
metric: service.metricMemoryUtilization({ period: cdk.Duration.minutes(5), statistic: 'Average' }),
|
||||
threshold: 85,
|
||||
evaluationPeriods: 3,
|
||||
datapointsToAlarm: 2,
|
||||
comparisonOperator: cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
|
||||
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
||||
}),
|
||||
);
|
||||
}
|
||||
|
||||
// ── ECS RunningTaskCount (REQUIRES Container Insights) ──────────────────────
|
||||
// RunningTaskCount is published only when Container Insights is enabled on the
|
||||
// cluster. desiredCount is 1, so alarm when running tasks drop below 1.
|
||||
private addEcsRunningTaskAlarm(service: ecs.FargateService, cluster: ecs.ICluster): void {
|
||||
const runningTasks = new cloudwatch.Metric({
|
||||
namespace: 'ECS/ContainerInsights',
|
||||
metricName: 'RunningTaskCount',
|
||||
dimensionsMap: {
|
||||
ClusterName: cluster.clusterName,
|
||||
ServiceName: service.serviceName,
|
||||
},
|
||||
period: cdk.Duration.minutes(1),
|
||||
statistic: 'Average',
|
||||
});
|
||||
|
||||
this.wire(
|
||||
new cloudwatch.Alarm(this, 'socket-mode-running-tasks', {
|
||||
alarmName: 'seahaven-socket-mode-running-tasks',
|
||||
alarmDescription:
|
||||
'seahaven-socket-mode ECS running task count < 1 (Container Insights required)',
|
||||
metric: runningTasks,
|
||||
threshold: 1,
|
||||
evaluationPeriods: 3,
|
||||
datapointsToAlarm: 2,
|
||||
comparisonOperator: cloudwatch.ComparisonOperator.LESS_THAN_THRESHOLD,
|
||||
treatMissingData: cloudwatch.TreatMissingData.NOT_BREACHING,
|
||||
}),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
|
@ -29,6 +29,7 @@ export interface SlackHandlerProps {
|
|||
export class SlackHandlerConstruct extends Construct {
|
||||
public readonly processorLambda: lambdaNodejs.NodejsFunction;
|
||||
public readonly appHomeLambda: lambdaNodejs.NodejsFunction;
|
||||
public readonly qboOAuthLambda: lambdaNodejs.NodejsFunction;
|
||||
public readonly api: apigatewayv2.HttpApi;
|
||||
|
||||
constructor(scope: Construct, id: string, props: SlackHandlerProps) {
|
||||
|
|
@ -106,7 +107,7 @@ export class SlackHandlerConstruct extends Construct {
|
|||
this, 'QBOSecret', 'seahaven/qbo/oauth',
|
||||
);
|
||||
|
||||
const qboOAuthLambda = new lambdaNodejs.NodejsFunction(this, 'QBOOAuthFn', {
|
||||
this.qboOAuthLambda = new lambdaNodejs.NodejsFunction(this, 'QBOOAuthFn', {
|
||||
functionName: 'seahaven-qbo-oauth',
|
||||
entry: path.join(__dirname, '../../lambda/qbo-oauth/index.ts'),
|
||||
handler: 'handler',
|
||||
|
|
@ -125,10 +126,10 @@ export class SlackHandlerConstruct extends Construct {
|
|||
bundling,
|
||||
});
|
||||
|
||||
qboSecret.grantRead(qboOAuthLambda);
|
||||
qboSecret.grantWrite(qboOAuthLambda);
|
||||
qboSecret.grantRead(this.qboOAuthLambda);
|
||||
qboSecret.grantWrite(this.qboOAuthLambda);
|
||||
|
||||
const qboOAuthIntegration = new HttpLambdaIntegration('QBOOAuthIntegration', qboOAuthLambda);
|
||||
const qboOAuthIntegration = new HttpLambdaIntegration('QBOOAuthIntegration', this.qboOAuthLambda);
|
||||
|
||||
for (const qboPath of ['/qbo/connect', '/qbo/callback', '/qbo/disconnect', '/qbo/launch']) {
|
||||
this.api.addRoutes({
|
||||
|
|
|
|||
|
|
@ -14,6 +14,9 @@ export interface SocketModeProps {
|
|||
}
|
||||
|
||||
export class SocketModeConstruct extends Construct {
|
||||
public readonly cluster: ecs.Cluster;
|
||||
public readonly service: ecs.FargateService;
|
||||
|
||||
constructor(scope: Construct, id: string, props: SocketModeProps) {
|
||||
super(scope, id);
|
||||
|
||||
|
|
@ -24,7 +27,12 @@ export class SocketModeConstruct extends Construct {
|
|||
const cluster = new ecs.Cluster(this, 'Cluster', {
|
||||
clusterName: 'seahaven-socket-mode',
|
||||
vpc: props.vpc,
|
||||
// Container Insights publishes ECS/ContainerInsights metrics (incl.
|
||||
// RunningTaskCount) needed for the task-count alarm. This adds CloudWatch
|
||||
// metric + log cost — see PR body "NEEDS ADAM SIGN-OFF (cost/config)".
|
||||
containerInsightsV2: ecs.ContainerInsights.ENABLED,
|
||||
});
|
||||
this.cluster = cluster;
|
||||
|
||||
const taskDef = new ecs.FargateTaskDefinition(this, 'TaskDef', {
|
||||
memoryLimitMiB: 512,
|
||||
|
|
@ -55,7 +63,7 @@ export class SocketModeConstruct extends Construct {
|
|||
props.processorLambda.grantInvoke(taskDef.taskRole);
|
||||
props.appHomeLambda.grantInvoke(taskDef.taskRole);
|
||||
|
||||
new ecs.FargateService(this, 'Service', {
|
||||
this.service = new ecs.FargateService(this, 'Service', {
|
||||
serviceName: 'seahaven-socket-mode',
|
||||
cluster,
|
||||
taskDefinition: taskDef,
|
||||
|
|
|
|||
|
|
@ -9,6 +9,7 @@ import { SocketModeConstruct } from './constructs/socket-mode';
|
|||
import { NotionSyncConstruct } from './constructs/notion-sync';
|
||||
import { PoSyncConstruct } from './constructs/po-sync';
|
||||
import { WorkorderSyncConstruct } from './constructs/workorder-sync';
|
||||
import { MonitoringConstruct } from './constructs/monitoring';
|
||||
|
||||
export class SeahavenSlackBotStack extends cdk.Stack {
|
||||
constructor(scope: Construct, id: string, props?: cdk.StackProps) {
|
||||
|
|
@ -52,7 +53,7 @@ export class SeahavenSlackBotStack extends cdk.Stack {
|
|||
});
|
||||
|
||||
// ── Notion → KB daily sync (EventBridge + Lambda) ────────────────────────
|
||||
new NotionSyncConstruct(this, 'NotionSync', {
|
||||
const notionSync = new NotionSyncConstruct(this, 'NotionSync', {
|
||||
region: this.region,
|
||||
kbDocsBucket: knowledgeBase.docsBucket,
|
||||
knowledgeBaseId: knowledgeBase.knowledgeBase.knowledgeBaseId,
|
||||
|
|
@ -60,7 +61,7 @@ export class SeahavenSlackBotStack extends cdk.Stack {
|
|||
});
|
||||
|
||||
// ── Purchase Orders → KB daily sync (EventBridge + Lambda) ────────────────
|
||||
new PoSyncConstruct(this, 'PoSync', {
|
||||
const poSync = new PoSyncConstruct(this, 'PoSync', {
|
||||
region: this.region,
|
||||
kbDocsBucket: knowledgeBase.docsBucket,
|
||||
knowledgeBaseId: knowledgeBase.knowledgeBase.knowledgeBaseId,
|
||||
|
|
@ -68,7 +69,7 @@ export class SeahavenSlackBotStack extends cdk.Stack {
|
|||
});
|
||||
|
||||
// ── Work Orders → KB daily sync (EventBridge + Lambda) ────────────────────
|
||||
new WorkorderSyncConstruct(this, 'WorkorderSync', {
|
||||
const workorderSync = new WorkorderSyncConstruct(this, 'WorkorderSync', {
|
||||
region: this.region,
|
||||
kbDocsBucket: knowledgeBase.docsBucket,
|
||||
knowledgeBaseId: knowledgeBase.knowledgeBase.knowledgeBaseId,
|
||||
|
|
@ -89,12 +90,39 @@ export class SeahavenSlackBotStack extends cdk.Stack {
|
|||
});
|
||||
|
||||
// ── Socket Mode (ECS Fargate — replaces webhook Lambda) ──────────────────
|
||||
new SocketModeConstruct(this, 'SocketMode', {
|
||||
const socketMode = new SocketModeConstruct(this, 'SocketMode', {
|
||||
vpc,
|
||||
processorLambda: slackHandler.processorLambda,
|
||||
appHomeLambda: slackHandler.appHomeLambda,
|
||||
});
|
||||
|
||||
// ── CloudWatch alarm coverage → site-alerts SNS ───────────────────────────
|
||||
new MonitoringConstruct(this, 'Monitoring', {
|
||||
lambdas: [
|
||||
{ name: 'slack-processor', fn: slackHandler.processorLambda, timeout: cdk.Duration.minutes(5) },
|
||||
{ name: 'app-home', fn: slackHandler.appHomeLambda, timeout: cdk.Duration.seconds(10) },
|
||||
{ name: 'qbo-oauth', fn: slackHandler.qboOAuthLambda, timeout: cdk.Duration.seconds(15) },
|
||||
{ name: 'qbo-lookup', fn: bedrockAgent.qboLambda, timeout: cdk.Duration.seconds(30) },
|
||||
{ name: 'maps-lookup', fn: bedrockAgent.mapsLambda, timeout: cdk.Duration.seconds(30) },
|
||||
{ name: 'wo-po-lookup', fn: bedrockAgent.woPoLambda, timeout: cdk.Duration.seconds(30) },
|
||||
{ name: 'po-sync', fn: poSync.syncLambda, timeout: cdk.Duration.minutes(15) },
|
||||
{ name: 'workorder-sync', fn: workorderSync.syncLambda, timeout: cdk.Duration.minutes(5) },
|
||||
{ name: 'notion-sync', fn: notionSync.syncLambda, timeout: cdk.Duration.minutes(5) },
|
||||
],
|
||||
tables: [
|
||||
{ name: 'ddb-conversations', table: conversationLog.table },
|
||||
{ name: 'ddb-unanswered-questions', table: conversationLog.unansweredTable },
|
||||
],
|
||||
httpApi: slackHandler.api,
|
||||
ecsService: socketMode.service,
|
||||
ecsCluster: socketMode.cluster,
|
||||
// RunningTaskCount alarm requires Container Insights, enabled on the
|
||||
// socket-mode cluster (see socket-mode.ts). This adds CloudWatch metric +
|
||||
// log cost — flagged in the PR as "NEEDS ADAM SIGN-OFF (cost/config)".
|
||||
// Drop this commit (and the cluster's containerInsightsV2 line) to decline.
|
||||
enableRunningTaskAlarm: true,
|
||||
});
|
||||
|
||||
// ── Stack outputs ─────────────────────────────────────────────────────────
|
||||
new cdk.CfnOutput(this, 'KBDocsBucketName', {
|
||||
value: knowledgeBase.docsBucket.bucketName,
|
||||
|
|
|
|||
Reference in a new issue