From 0fdf407e1d0c9cc536ae95ae360cc205f84ff62e Mon Sep 17 00:00:00 2001 From: Adam Moussa <166072409+amoussa1229@users.noreply.github.com> Date: Wed, 17 Jun 2026 14:46:03 -0400 Subject: [PATCH] Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70) * Add CloudWatch alarm coverage for po-ingest and workorder-ingest Expands alarm coverage across both CDK stacks. All alarms are ALARM-only (no OK action) to the shared site-alerts SNS topic, with TreatMissingData NOT_BREACHING. The site-alerts topic is now imported once near the top of each stack so every alarm reuses one Topic instance. po-ingest (cdk/po_stack.py): - Errors: po-ingest-site-extractor - Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui - Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption), po-ingest-site-extractor, po-web-ui - DynamoDB throttle + system-error: purchase-orders, verified-sites, pending-site-review workorder-ingest (cdk/wo_stack.py): - Throttles: workorder-email-processor - Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption) - DynamoDB throttle + system-error: WorkOrders, WorkOrderComments DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation dimension set, so each table alarm is a Sum math expression across operations via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0). Refs INFRA-41 / audit H-8. * Drop NEEDS ADAM SIGN-OFF wording from alarm comments Duration alarm thresholds are owner-approved; remove the sign-off flag from po_stack.py and wo_stack.py comments. Threshold values, eval config, and orphan-delete notes are unchanged. --- README.md | 18 +++- cdk/po_stack.py | 219 ++++++++++++++++++++++++++++++++++++++++++++++-- cdk/wo_stack.py | 124 +++++++++++++++++++++++++-- 3 files changed, 346 insertions(+), 15 deletions(-) diff --git a/README.md b/README.md index 28a8b80..851e903 100644 --- a/README.md +++ b/README.md @@ -71,7 +71,23 @@ All Lambdas: Python 3.12, ARM64, 60-day log retention. **SES:** Both stacks add rules to the shared `INBOUND_MAIL` receipt rule set on `int.seahaven.com`. -**Failure handling (INFRA-41):** Each email-processor is async-invoked (S3 → Lambda). Both have a CDK-managed SQS dead-letter queue (`dead_letter_queue=`, 14-day retention, SSL-enforced) so a failed parse is captured rather than silently dropped after Lambda's retries, plus an ALARM-only CloudWatch `Errors` alarm (Sum, threshold > 0) wired to the shared `site-alerts` SNS topic. +**Failure handling (INFRA-41):** Each email-processor is async-invoked (S3 → Lambda). Both have a CDK-managed SQS dead-letter queue (`dead_letter_queue=`, 14-day retention, SSL-enforced) so a failed parse is captured rather than silently dropped after Lambda's retries. + +### CloudWatch alarms + +Every alarm is **ALARM-only** (no OK action), sends to the shared `site-alerts` SNS topic (imported once per stack via `Topic.from_topic_arn`), and uses `TreatMissingData.NOT_BREACHING`. + +**Lambda alarms** (`AWS/Lambda`, `FunctionName` dimension): + +| Alarm | Functions | Metric / config | +|---|---|---| +| `-errors` | `po-email-processor`, `po-ingest-site-extractor`, `workorder-email-processor` | `Errors` Sum, 5 min, `> 0`, eval 1 | +| `-throttles` | `po-email-processor`, `po-ingest-site-extractor`, `po-web-ui`, `workorder-email-processor` | `Throttles` Sum, 5 min, `> 0`, eval 1 | +| `-duration` | `po-email-processor`, `po-ingest-site-extractor`, `po-web-ui` (p99); `workorder-email-processor` (p95) | `Duration` percentile, 5 min, `>= 45000` ms (75% of the 60s timeout), eval 3 / datapoints 2 | + +The `-duration` and `-throttles` alarms for `po-email-processor` and `workorder-email-processor` supersede the orphaned, CLI-created `Lambda-Duration-*` / `Lambda-Throttles-*` alarms (deleted post-deploy). + +**DynamoDB alarms** (`AWS/DynamoDB`): each owned table gets `-throttles` (`ThrottledRequests`) and `
-system-errors` (`SystemErrors`). These metrics emit only at the `TableName` + `Operation` dimension set, so each alarm is a `Sum` math expression across the operations the table uses (Get/BatchGet/Query/Scan/Put/Update/Delete/BatchWrite). Tables covered: `purchase-orders`, `verified-sites`, `pending-site-review` (po-ingest); `WorkOrders`, `WorkOrderComments` (workorder-ingest). ## Shared Resources diff --git a/cdk/po_stack.py b/cdk/po_stack.py index 82a6bd6..3e2036e 100644 --- a/cdk/po_stack.py +++ b/cdk/po_stack.py @@ -23,11 +23,75 @@ from aws_cdk import ( ) from constructs import Construct +# Operations these tables actually issue (PutItem/UpdateItem/DeleteItem writes, +# GetItem/Query/BatchGetItem reads). DynamoDB emits ThrottledRequests/SystemErrors +# keyed by TableName + Operation only, so the CDK *_for_operations helpers (which +# render a SUM MathExpression across these per-operation metrics) are the correct, +# non-deprecated way to roll a table up to a single alarmable series. +_DDB_ALARM_OPERATIONS = [ + dynamodb.Operation.GET_ITEM, + dynamodb.Operation.BATCH_GET_ITEM, + dynamodb.Operation.QUERY, + dynamodb.Operation.SCAN, + dynamodb.Operation.PUT_ITEM, + dynamodb.Operation.UPDATE_ITEM, + dynamodb.Operation.DELETE_ITEM, + dynamodb.Operation.BATCH_WRITE_ITEM, +] + + +def _add_ddb_alarms(scope, id_prefix, table, alarm_name_prefix, alarm_topic): + """Add throttle + system-error alarms for a DynamoDB table. + + Both fire on any non-zero datapoint in a 5-min window. ALARM-only SnsAction + to site-alerts (no OK action); TreatMissingData NOT_BREACHING. + """ + table.metric_throttled_requests_for_operations( + operations=_DDB_ALARM_OPERATIONS, + period=Duration.minutes(5), + statistic="Sum", + ).create_alarm( + scope, + f"{id_prefix}ThrottlesAlarm", + alarm_name=f"{alarm_name_prefix}-throttles", + alarm_description=f"{alarm_name_prefix} DynamoDB throttled requests", + threshold=0, + evaluation_periods=1, + comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD, + treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, + ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + + table.metric_system_errors_for_operations( + operations=_DDB_ALARM_OPERATIONS, + period=Duration.minutes(5), + statistic="Sum", + ).create_alarm( + scope, + f"{id_prefix}SystemErrorsAlarm", + alarm_name=f"{alarm_name_prefix}-system-errors", + alarm_description=f"{alarm_name_prefix} DynamoDB server-side (5xx) errors", + threshold=0, + evaluation_periods=1, + comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD, + treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, + ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + class PoIngestStack(Stack): def __init__(self, scope: Construct, construct_id: str, **kwargs): super().__init__(scope, construct_id, **kwargs) + # --- Shared alarm SNS topic (site-alerts) --- + # Imported once near the top so every alarm in this stack reuses the same + # Topic construct instance (avoids duplicate logical IDs). ALARM-only + # SnsAction; no OK action, per the CloudWatch-alarm preference. The + # topic's CMK (alias/seahaven-alarm-topics) lives on the topic itself. + alarm_topic = sns.Topic.from_topic_arn( + self, + "SiteAlertsTopic", + f"arn:aws:sns:{self.region}:{self.account}:site-alerts", + ) + # --- S3 bucket for raw emails --- email_bucket = s3.Bucket( self, @@ -130,13 +194,7 @@ class PoIngestStack(Stack): # --- Errors alarm (INFRA-41 / audit H-8) --- # ALARM-only (no OK action, per the CloudWatch-alarm preference) to the - # shared site-alerts topic (CMK alias/seahaven-alarm-topics lives on the - # topic). Any errored invocation in a 5-min window pages. - alarm_topic = sns.Topic.from_topic_arn( - self, - "SiteAlertsTopic", - f"arn:aws:sns:{self.region}:{self.account}:site-alerts", - ) + # shared site-alerts topic. Any errored invocation in a 5-min window pages. email_processor.metric_errors( period=Duration.minutes(5), statistic="Sum", @@ -151,6 +209,44 @@ class PoIngestStack(Stack): treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + # --- Throttles alarm: po-email-processor --- + # Any throttled invocation (concurrency cap hit) in a 5-min window pages. + # ALARM-only to site-alerts; no OK action; NOT_BREACHING when no data. + email_processor.metric_throttles( + period=Duration.minutes(5), + statistic="Sum", + ).create_alarm( + self, + "EmailProcessorThrottlesAlarm", + alarm_name="po-email-processor-throttles", + alarm_description="po-email-processor invocation throttles", + threshold=0, + evaluation_periods=1, + comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD, + treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, + ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + + # --- Duration alarm: po-email-processor (orphan adoption) --- + # Adopts the orphaned CLI alarm Lambda-Duration-po-email-processor under + # the repo's -duration naming (NEW logical name → no deploy collision; + # delete the orphan post-deploy). p99 / 45000 ms + # (75% of the 60s timeout) / eval 3 of 3 — tighter than the orphan's + # Maximum>=48000 / 1-of-1. + email_processor.metric_duration( + period=Duration.minutes(5), + statistic="p99", + ).create_alarm( + self, + "EmailProcessorDurationAlarm", + alarm_name="po-email-processor-duration", + alarm_description="po-email-processor p99 duration approaching the 60s timeout", + threshold=45000, + evaluation_periods=3, + datapoints_to_alarm=2, + comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, + treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, + ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + # S3 event notification → Lambda email_bucket.add_event_notification( s3.EventType.OBJECT_CREATED, @@ -196,6 +292,39 @@ class PoIngestStack(Stack): po_table.grant_read_data(web_ui) + # --- Throttles alarm: po-web-ui --- + web_ui.metric_throttles( + period=Duration.minutes(5), + statistic="Sum", + ).create_alarm( + self, + "WebUiThrottlesAlarm", + alarm_name="po-web-ui-throttles", + alarm_description="po-web-ui invocation throttles", + threshold=0, + evaluation_periods=1, + comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD, + treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, + ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + + # --- Duration alarm: po-web-ui --- + # Net-new (no orphan exists for this function). + # p99 / 45000 ms (75% of the 60s timeout) / eval 3, datapoints 2. + web_ui.metric_duration( + period=Duration.minutes(5), + statistic="p99", + ).create_alarm( + self, + "WebUiDurationAlarm", + alarm_name="po-web-ui-duration", + alarm_description="po-web-ui p99 duration approaching the 60s timeout", + threshold=45000, + evaluation_periods=3, + datapoints_to_alarm=2, + comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, + treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, + ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + # Public Function URL removed 2026-06-08 (INFRA-74 / audit C-5): the # unauthenticated FunctionUrlAuthType.NONE URL was deleted out-of-band # via CLI. Removing the construct (and its auto-generated Principal:* @@ -247,6 +376,57 @@ class PoIngestStack(Stack): ) ) + # --- Errors alarm: po-ingest-site-extractor --- + # Stream-consumer errors retry per the event-source config, but a + # persistent failure stalls the verified-sites pipeline. ALARM-only to + # site-alerts; no OK action; NOT_BREACHING when no data. + site_extractor.metric_errors( + period=Duration.minutes(5), + statistic="Sum", + ).create_alarm( + self, + "SiteExtractorErrorsAlarm", + alarm_name="po-ingest-site-extractor-errors", + alarm_description="po-ingest-site-extractor invocation errors", + threshold=0, + evaluation_periods=1, + comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD, + treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, + ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + + # --- Throttles alarm: po-ingest-site-extractor --- + site_extractor.metric_throttles( + period=Duration.minutes(5), + statistic="Sum", + ).create_alarm( + self, + "SiteExtractorThrottlesAlarm", + alarm_name="po-ingest-site-extractor-throttles", + alarm_description="po-ingest-site-extractor invocation throttles", + threshold=0, + evaluation_periods=1, + comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD, + treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, + ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + + # --- Duration alarm: po-ingest-site-extractor --- + # Net-new (no orphan exists for this function). + # p99 / 45000 ms (75% of the 60s timeout) / eval 3, datapoints 2. + site_extractor.metric_duration( + period=Duration.minutes(5), + statistic="p99", + ).create_alarm( + self, + "SiteExtractorDurationAlarm", + alarm_name="po-ingest-site-extractor-duration", + alarm_description="po-ingest-site-extractor p99 duration approaching the 60s timeout", + threshold=45000, + evaluation_periods=3, + datapoints_to_alarm=2, + comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, + treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, + ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + cdk.CfnOutput( self, "VerifiedSitesTableName", @@ -269,3 +449,28 @@ class PoIngestStack(Stack): pending_review_table.grant_read_write_data(site_extractor) verified_sites_table.grant_read_data(site_extractor) + + # --- DynamoDB throttle + system-error alarms --- + # ThrottledRequests / SystemErrors emit at TableName + Operation only + # (verified against live CloudWatch: no TableName-only rollup exists, and + # metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0). + # Each table currently has zero throttle/error datapoints, so the series + # only materialise on first occurrence — NOT_BREACHING keeps them OK until + # then. + _add_ddb_alarms( + self, "PurchaseOrdersTable", po_table, "purchase-orders", alarm_topic + ) + _add_ddb_alarms( + self, + "VerifiedSitesTable", + verified_sites_table, + "verified-sites", + alarm_topic, + ) + _add_ddb_alarms( + self, + "PendingSiteReviewTable", + pending_review_table, + "pending-site-review", + alarm_topic, + ) diff --git a/cdk/wo_stack.py b/cdk/wo_stack.py index c26313c..8aa766d 100644 --- a/cdk/wo_stack.py +++ b/cdk/wo_stack.py @@ -22,11 +22,75 @@ from aws_cdk import ( ) from constructs import Construct +# Operations these tables actually issue (PutItem/UpdateItem/DeleteItem writes, +# GetItem/Query/BatchGetItem reads). DynamoDB emits ThrottledRequests/SystemErrors +# keyed by TableName + Operation only, so the CDK *_for_operations helpers (which +# render a SUM MathExpression across these per-operation metrics) are the correct, +# non-deprecated way to roll a table up to a single alarmable series. +_DDB_ALARM_OPERATIONS = [ + dynamodb.Operation.GET_ITEM, + dynamodb.Operation.BATCH_GET_ITEM, + dynamodb.Operation.QUERY, + dynamodb.Operation.SCAN, + dynamodb.Operation.PUT_ITEM, + dynamodb.Operation.UPDATE_ITEM, + dynamodb.Operation.DELETE_ITEM, + dynamodb.Operation.BATCH_WRITE_ITEM, +] + + +def _add_ddb_alarms(scope, id_prefix, table, alarm_name_prefix, alarm_topic): + """Add throttle + system-error alarms for a DynamoDB table. + + Both fire on any non-zero datapoint in a 5-min window. ALARM-only SnsAction + to site-alerts (no OK action); TreatMissingData NOT_BREACHING. + """ + table.metric_throttled_requests_for_operations( + operations=_DDB_ALARM_OPERATIONS, + period=Duration.minutes(5), + statistic="Sum", + ).create_alarm( + scope, + f"{id_prefix}ThrottlesAlarm", + alarm_name=f"{alarm_name_prefix}-throttles", + alarm_description=f"{alarm_name_prefix} DynamoDB throttled requests", + threshold=0, + evaluation_periods=1, + comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD, + treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, + ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + + table.metric_system_errors_for_operations( + operations=_DDB_ALARM_OPERATIONS, + period=Duration.minutes(5), + statistic="Sum", + ).create_alarm( + scope, + f"{id_prefix}SystemErrorsAlarm", + alarm_name=f"{alarm_name_prefix}-system-errors", + alarm_description=f"{alarm_name_prefix} DynamoDB server-side (5xx) errors", + threshold=0, + evaluation_periods=1, + comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD, + treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, + ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + class WorkorderIngestStack(Stack): def __init__(self, scope: Construct, construct_id: str, **kwargs): super().__init__(scope, construct_id, **kwargs) + # --- Shared alarm SNS topic (site-alerts) --- + # Imported once near the top so every alarm in this stack reuses the same + # Topic construct instance (avoids duplicate logical IDs). ALARM-only + # SnsAction; no OK action, per the CloudWatch-alarm preference. The + # topic's CMK (alias/seahaven-alarm-topics) lives on the topic itself. + alarm_topic = sns.Topic.from_topic_arn( + self, + "SiteAlertsTopic", + f"arn:aws:sns:{self.region}:{self.account}:site-alerts", + ) + # --- S3 bucket for raw emails --- email_bucket = s3.Bucket( self, @@ -147,13 +211,7 @@ class WorkorderIngestStack(Stack): # --- Errors alarm (INFRA-41 / audit H-8) --- # ALARM-only (no OK action, per the CloudWatch-alarm preference) to the - # shared site-alerts topic (CMK alias/seahaven-alarm-topics lives on the - # topic). Any errored invocation in a 5-min window pages. - alarm_topic = sns.Topic.from_topic_arn( - self, - "SiteAlertsTopic", - f"arn:aws:sns:{self.region}:{self.account}:site-alerts", - ) + # shared site-alerts topic. Any errored invocation in a 5-min window pages. email_processor.metric_errors( period=Duration.minutes(5), statistic="Sum", @@ -168,6 +226,44 @@ class WorkorderIngestStack(Stack): treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + # --- Throttles alarm: workorder-email-processor --- + # Any throttled invocation (concurrency cap hit) in a 5-min window pages. + # ALARM-only to site-alerts; no OK action; NOT_BREACHING when no data. + email_processor.metric_throttles( + period=Duration.minutes(5), + statistic="Sum", + ).create_alarm( + self, + "EmailProcessorThrottlesAlarm", + alarm_name="workorder-email-processor-throttles", + alarm_description="workorder-email-processor invocation throttles", + threshold=0, + evaluation_periods=1, + comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD, + treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, + ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + + # --- Duration alarm: workorder-email-processor (orphan adoption) --- + # Adopts the orphaned CLI alarm Lambda-Duration-workorder-email-processor + # under the repo's -duration naming (NEW logical name → no deploy + # collision; delete the orphan post-deploy). p95 / + # 45000 ms (75% of the 60s timeout) / eval 3 of which 2 datapoints — + # tighter than the orphan's Maximum>=48000 / 1-of-1. + email_processor.metric_duration( + period=Duration.minutes(5), + statistic="p95", + ).create_alarm( + self, + "EmailProcessorDurationAlarm", + alarm_name="workorder-email-processor-duration", + alarm_description="workorder-email-processor p95 duration approaching the 60s timeout", + threshold=45000, + evaluation_periods=3, + datapoints_to_alarm=2, + comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD, + treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING, + ).add_alarm_action(cw_actions.SnsAction(alarm_topic)) + # S3 event notification -> Lambda email_bucket.add_event_notification( s3.EventType.OBJECT_CREATED, @@ -214,6 +310,20 @@ class WorkorderIngestStack(Stack): work_orders_table.grant_read_data(web_ui) comments_table.grant_read_data(web_ui) + # --- DynamoDB throttle + system-error alarms --- + # ThrottledRequests / SystemErrors emit at TableName + Operation only + # (verified against live CloudWatch: no TableName-only rollup exists, and + # metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0). + # Each table currently has zero throttle/error datapoints, so the series + # only materialise on first occurrence — NOT_BREACHING keeps them OK until + # then. + _add_ddb_alarms( + self, "WorkOrdersTable", work_orders_table, "WorkOrders", alarm_topic + ) + _add_ddb_alarms( + self, "WorkOrderComments", comments_table, "WorkOrderComments", alarm_topic + ) + # Public Function URL removed 2026-06-08 (INFRA-74 / audit C-5): the # unauthenticated FunctionUrlAuthType.NONE URL was deleted out-of-band # via CLI. Removing the construct (and its auto-generated Principal:*