procurement-ingest/cdk/po_stack.py

518 lines
22 KiB
Python
Raw Normal View History

"""CDK stack for the Coupa PO email ingestion pipeline."""
import aws_cdk as cdk
from aws_cdk import (
Duration,
RemovalPolicy,
Stack,
aws_cloudwatch as cloudwatch,
aws_cloudwatch_actions as cw_actions,
aws_dynamodb as dynamodb,
aws_kms as kms,
aws_lambda as lambda_,
aws_lambda_event_sources as lambda_event_sources,
aws_logs as logs,
aws_s3 as s3,
aws_s3_notifications as s3n,
aws_ses as ses,
aws_ses_actions as ses_actions,
aws_secretsmanager as secretsmanager,
aws_sns as sns,
aws_sqs as sqs,
aws_ssm as ssm,
)
from constructs import Construct
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70) * Add CloudWatch alarm coverage for po-ingest and workorder-ingest Expands alarm coverage across both CDK stacks. All alarms are ALARM-only (no OK action) to the shared site-alerts SNS topic, with TreatMissingData NOT_BREACHING. The site-alerts topic is now imported once near the top of each stack so every alarm reuses one Topic instance. po-ingest (cdk/po_stack.py): - Errors: po-ingest-site-extractor - Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui - Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption), po-ingest-site-extractor, po-web-ui - DynamoDB throttle + system-error: purchase-orders, verified-sites, pending-site-review workorder-ingest (cdk/wo_stack.py): - Throttles: workorder-email-processor - Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption) - DynamoDB throttle + system-error: WorkOrders, WorkOrderComments DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation dimension set, so each table alarm is a Sum math expression across operations via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0). Refs INFRA-41 / audit H-8. * Drop NEEDS ADAM SIGN-OFF wording from alarm comments Duration alarm thresholds are owner-approved; remove the sign-off flag from po_stack.py and wo_stack.py comments. Threshold values, eval config, and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
# Operations these tables actually issue (PutItem/UpdateItem/DeleteItem writes,
# GetItem/Query/BatchGetItem reads). DynamoDB emits ThrottledRequests/SystemErrors
# keyed by TableName + Operation only, so the CDK *_for_operations helpers (which
# render a SUM MathExpression across these per-operation metrics) are the correct,
# non-deprecated way to roll a table up to a single alarmable series.
_DDB_ALARM_OPERATIONS = [
dynamodb.Operation.GET_ITEM,
dynamodb.Operation.BATCH_GET_ITEM,
dynamodb.Operation.QUERY,
dynamodb.Operation.SCAN,
dynamodb.Operation.PUT_ITEM,
dynamodb.Operation.UPDATE_ITEM,
dynamodb.Operation.DELETE_ITEM,
dynamodb.Operation.BATCH_WRITE_ITEM,
]
def _add_ddb_alarms(scope, id_prefix, table, alarm_name_prefix, alarm_topic):
"""Add throttle + system-error alarms for a DynamoDB table.
Both fire on any non-zero datapoint in a 5-min window. ALARM-only SnsAction
to site-alerts (no OK action); TreatMissingData NOT_BREACHING.
"""
table.metric_throttled_requests_for_operations(
operations=_DDB_ALARM_OPERATIONS,
period=Duration.minutes(5),
statistic="Sum",
).create_alarm(
scope,
f"{id_prefix}ThrottlesAlarm",
alarm_name=f"{alarm_name_prefix}-throttles",
alarm_description=f"{alarm_name_prefix} DynamoDB throttled requests",
threshold=0,
evaluation_periods=1,
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
table.metric_system_errors_for_operations(
operations=_DDB_ALARM_OPERATIONS,
period=Duration.minutes(5),
statistic="Sum",
).create_alarm(
scope,
f"{id_prefix}SystemErrorsAlarm",
alarm_name=f"{alarm_name_prefix}-system-errors",
alarm_description=f"{alarm_name_prefix} DynamoDB server-side (5xx) errors",
threshold=0,
evaluation_periods=1,
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
class PoIngestStack(Stack):
def __init__(self, scope: Construct, construct_id: str, **kwargs):
super().__init__(scope, construct_id, **kwargs)
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70) * Add CloudWatch alarm coverage for po-ingest and workorder-ingest Expands alarm coverage across both CDK stacks. All alarms are ALARM-only (no OK action) to the shared site-alerts SNS topic, with TreatMissingData NOT_BREACHING. The site-alerts topic is now imported once near the top of each stack so every alarm reuses one Topic instance. po-ingest (cdk/po_stack.py): - Errors: po-ingest-site-extractor - Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui - Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption), po-ingest-site-extractor, po-web-ui - DynamoDB throttle + system-error: purchase-orders, verified-sites, pending-site-review workorder-ingest (cdk/wo_stack.py): - Throttles: workorder-email-processor - Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption) - DynamoDB throttle + system-error: WorkOrders, WorkOrderComments DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation dimension set, so each table alarm is a Sum math expression across operations via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0). Refs INFRA-41 / audit H-8. * Drop NEEDS ADAM SIGN-OFF wording from alarm comments Duration alarm thresholds are owner-approved; remove the sign-off flag from po_stack.py and wo_stack.py comments. Threshold values, eval config, and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
# --- Shared alarm SNS topic (site-alerts) ---
# Imported once near the top so every alarm in this stack reuses the same
# Topic construct instance (avoids duplicate logical IDs). ALARM-only
# SnsAction; no OK action, per the CloudWatch-alarm preference. The
# topic's CMK (alias/seahaven-alarm-topics) lives on the topic itself.
alarm_topic = sns.Topic.from_topic_arn(
self,
"SiteAlertsTopic",
f"arn:aws:sns:{self.region}:{self.account}:site-alerts",
)
# --- S3 bucket for raw emails ---
email_bucket = s3.Bucket(
self,
"EmailBucket",
bucket_name=f"po-ingest-emails-{self.account}",
block_public_access=s3.BlockPublicAccess.BLOCK_ALL,
removal_policy=RemovalPolicy.RETAIN,
lifecycle_rules=[
s3.LifecycleRule(expiration=Duration.days(90)),
],
)
# --- Shared customer-managed CMK for sensitive DynamoDB tables ---
# Owned by the account-baseline app (alias/seahaven-dynamodb, INFRA-95 /
# M-3); ARN published to SSM. The purchase-orders table was migrated to
# SSE-KMS out-of-band, so declaring encryption_key here reconciles the
# drift and — via grant_read_write_data below — propagates the required
# kms:Decrypt/GenerateDataKey/DescribeKey to the consumer roles.
dynamodb_cmk = kms.Key.from_key_arn(
self,
"DynamoDbCmk",
ssm.StringParameter.value_for_string_parameter(
self, "/seahaven/dynamodb/cmk-arn"
),
)
# --- Purchase-orders DynamoDB table ---
# Owned by this stack. Streams enabled for the site-extractor pipeline.
# Other stacks (seahaven-slack-bot) reference this table via fromTableName().
po_table = dynamodb.Table(
self,
"PurchaseOrdersTable",
table_name="purchase-orders",
partition_key=dynamodb.Attribute(
name="po_number",
type=dynamodb.AttributeType.STRING,
),
billing_mode=dynamodb.BillingMode.PAY_PER_REQUEST,
removal_policy=RemovalPolicy.RETAIN,
stream=dynamodb.StreamViewType.NEW_IMAGE,
encryption=dynamodb.TableEncryption.CUSTOMER_MANAGED,
encryption_key=dynamodb_cmk,
)
# --- Secrets Manager for Anthropic API key ---
anthropic_secret = secretsmanager.Secret(
self,
"AnthropicApiKey",
secret_name="po-ingest/anthropic-api-key",
description="Anthropic API key for Coupa PO email parsing",
removal_policy=RemovalPolicy.RETAIN,
)
# --- DLQ for failed async invocations (INFRA-41 / audit H-8) ---
# SES → S3 → Lambda is async; without an OnFailure destination a failed
# parse (bad email, transient error) is silently dropped after Lambda's
# retries. CDK generates the queue name to avoid colliding with the
# interim CLI-created po-email-processor-dlq (removed post-deploy).
email_processor_dlq = sqs.Queue(
self,
"EmailProcessorDlq",
retention_period=Duration.days(14),
enforce_ssl=True,
)
# --- Lambda function ---
email_processor = lambda_.Function(
self,
"EmailProcessor",
function_name="po-email-processor",
runtime=lambda_.Runtime.PYTHON_3_12,
architecture=lambda_.Architecture.ARM_64,
handler="handler.handler",
code=lambda_.Code.from_asset(
Merge workorder-ingest into unified procurement repo (#22) * Merge workorder-ingest pipeline into unified repo Move PO lambdas under lambdas/po/, add WO pipeline under lambdas/wo/. Two independent CloudFormation stacks in one CDK app. Fix WO stack compliance: ARM64 architecture, 60-day log retention, aarch64 bundling, RETAIN on Anthropic secret. Remove stale CodePipeline buildspec. * Fix test_local.py import path and remove dead shared/models.py test_local.py referenced the old lambdas/email_processor path. Updated to lambdas/wo/email_processor. Removed shared/ directory entirely as nothing imports from it. * Escape HTML in both web UI dashboards to prevent XSS Both Function URLs are public (auth_type=NONE) and render email-derived content via f-strings. Attacker-crafted emails could inject scripts. Added html.escape() on all interpolated values in both PO and WO dashboards. * Add pagination to WO web UI scan get_work_orders() only fetched the first 1MB page from DynamoDB. Loop on LastEvaluatedKey to match the PO web UI pattern. * Fix esc(None) TypeError and javascript: scheme in PO web UI Coerce supplier name through `or ""` before escaping to handle nested None from DynamoDB. Add scheme allowlist on view_order_url to block javascript:/data: hrefs from LLM-extracted URLs. * Fix WO render_badge None guard, updated_at slice, and backfill path Add null guard to WO render_badge matching the PO version. Use `or ""` before slicing updated_at to handle explicit None values. Fix backfill_sites.py sys.path to use new lambdas/po/site_extractor. * Harden WO web UI and fix JS-context XSS in both dashboards - Use json.dumps for onclick URLs to prevent JS string breakout - Add .lower() to WO render_badge color lookup matching PO pattern - Add pagination to get_comments query - Cap get_work_orders to 500 results matching PO pattern * Apply ruff formatting to web UI handlers
2026-05-12 15:21:06 -04:00
"../lambdas/po/email_processor",
bundling=cdk.BundlingOptions(
image=lambda_.Runtime.PYTHON_3_12.bundling_image,
command=[
"bash",
"-c",
"pip install --platform manylinux2014_aarch64 --only-binary=:all: "
"-r requirements.txt -t /asset-output && "
"cp handler.py /asset-output/",
],
),
),
timeout=Duration.seconds(60),
memory_size=256,
log_retention=logs.RetentionDays.TWO_MONTHS,
dead_letter_queue=email_processor_dlq,
environment={
"PO_TABLE": "purchase-orders",
"ANTHROPIC_API_KEY_SECRET_ARN": anthropic_secret.secret_arn,
},
)
# Grant permissions
email_bucket.grant_read(email_processor)
po_table.grant_read_write_data(email_processor)
anthropic_secret.grant_read(email_processor)
# --- Errors alarm (INFRA-41 / audit H-8) ---
# ALARM-only (no OK action, per the CloudWatch-alarm preference) to the
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70) * Add CloudWatch alarm coverage for po-ingest and workorder-ingest Expands alarm coverage across both CDK stacks. All alarms are ALARM-only (no OK action) to the shared site-alerts SNS topic, with TreatMissingData NOT_BREACHING. The site-alerts topic is now imported once near the top of each stack so every alarm reuses one Topic instance. po-ingest (cdk/po_stack.py): - Errors: po-ingest-site-extractor - Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui - Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption), po-ingest-site-extractor, po-web-ui - DynamoDB throttle + system-error: purchase-orders, verified-sites, pending-site-review workorder-ingest (cdk/wo_stack.py): - Throttles: workorder-email-processor - Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption) - DynamoDB throttle + system-error: WorkOrders, WorkOrderComments DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation dimension set, so each table alarm is a Sum math expression across operations via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0). Refs INFRA-41 / audit H-8. * Drop NEEDS ADAM SIGN-OFF wording from alarm comments Duration alarm thresholds are owner-approved; remove the sign-off flag from po_stack.py and wo_stack.py comments. Threshold values, eval config, and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
# shared site-alerts topic. Any errored invocation in a 5-min window pages.
email_processor.metric_errors(
period=Duration.minutes(5),
statistic="Sum",
).create_alarm(
self,
"EmailProcessorErrorsAlarm",
alarm_name="po-email-processor-errors",
alarm_description="po-email-processor async invocation errors",
threshold=0,
evaluation_periods=1,
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70) * Add CloudWatch alarm coverage for po-ingest and workorder-ingest Expands alarm coverage across both CDK stacks. All alarms are ALARM-only (no OK action) to the shared site-alerts SNS topic, with TreatMissingData NOT_BREACHING. The site-alerts topic is now imported once near the top of each stack so every alarm reuses one Topic instance. po-ingest (cdk/po_stack.py): - Errors: po-ingest-site-extractor - Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui - Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption), po-ingest-site-extractor, po-web-ui - DynamoDB throttle + system-error: purchase-orders, verified-sites, pending-site-review workorder-ingest (cdk/wo_stack.py): - Throttles: workorder-email-processor - Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption) - DynamoDB throttle + system-error: WorkOrders, WorkOrderComments DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation dimension set, so each table alarm is a Sum math expression across operations via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0). Refs INFRA-41 / audit H-8. * Drop NEEDS ADAM SIGN-OFF wording from alarm comments Duration alarm thresholds are owner-approved; remove the sign-off flag from po_stack.py and wo_stack.py comments. Threshold values, eval config, and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
# --- Throttles alarm: po-email-processor ---
# Any throttled invocation (concurrency cap hit) in a 5-min window pages.
# ALARM-only to site-alerts; no OK action; NOT_BREACHING when no data.
email_processor.metric_throttles(
period=Duration.minutes(5),
statistic="Sum",
).create_alarm(
self,
"EmailProcessorThrottlesAlarm",
alarm_name="po-email-processor-throttles",
alarm_description="po-email-processor invocation throttles",
threshold=0,
evaluation_periods=1,
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
# --- DLQ messages-present alarm ---
# Pages when any message lands in the EmailProcessorDlq: a message here
# means a PO email was permanently dropped after Lambda exhausted its
# async retries. Maximum over a single 5-min window > 0 fires; missing
# data (no messages metric emitted) is not breaching. Reuses the shared
# site-alerts topic, ALARM-only, like the errors alarm above. The metric
# helper derives the QueueName dimension from the queue construct, so the
# alarm tracks the CDK-generated queue name without hardcoding it.
email_processor_dlq.metric_approximate_number_of_messages_visible(
period=Duration.minutes(5),
statistic="Maximum",
).create_alarm(
self,
"EmailProcessorDlqMessagesAlarm",
alarm_name="po-email-processor-dlq-messages",
alarm_description="po-email-processor DLQ has messages (dropped PO emails)",
threshold=0,
evaluation_periods=1,
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70) * Add CloudWatch alarm coverage for po-ingest and workorder-ingest Expands alarm coverage across both CDK stacks. All alarms are ALARM-only (no OK action) to the shared site-alerts SNS topic, with TreatMissingData NOT_BREACHING. The site-alerts topic is now imported once near the top of each stack so every alarm reuses one Topic instance. po-ingest (cdk/po_stack.py): - Errors: po-ingest-site-extractor - Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui - Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption), po-ingest-site-extractor, po-web-ui - DynamoDB throttle + system-error: purchase-orders, verified-sites, pending-site-review workorder-ingest (cdk/wo_stack.py): - Throttles: workorder-email-processor - Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption) - DynamoDB throttle + system-error: WorkOrders, WorkOrderComments DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation dimension set, so each table alarm is a Sum math expression across operations via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0). Refs INFRA-41 / audit H-8. * Drop NEEDS ADAM SIGN-OFF wording from alarm comments Duration alarm thresholds are owner-approved; remove the sign-off flag from po_stack.py and wo_stack.py comments. Threshold values, eval config, and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
# --- Duration alarm: po-email-processor (orphan adoption) ---
# Adopts the orphaned CLI alarm Lambda-Duration-po-email-processor under
# the repo's <fn>-duration naming (NEW logical name → no deploy collision;
# delete the orphan post-deploy). p99 / 45000 ms
# (75% of the 60s timeout) / eval 3 of 3 — tighter than the orphan's
# Maximum>=48000 / 1-of-1.
email_processor.metric_duration(
period=Duration.minutes(5),
statistic="p99",
).create_alarm(
self,
"EmailProcessorDurationAlarm",
alarm_name="po-email-processor-duration",
alarm_description="po-email-processor p99 duration approaching the 60s timeout",
threshold=45000,
evaluation_periods=3,
datapoints_to_alarm=2,
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
# S3 event notification → Lambda
email_bucket.add_event_notification(
s3.EventType.OBJECT_CREATED,
s3n.LambdaDestination(email_processor),
s3.NotificationKeyFilter(prefix="inbound/"),
)
# --- SES Receipt Rule ---
# Reuse the existing INBOUND_MAIL rule set (shared with workorder-ingest)
rule_set = ses.ReceiptRuleSet.from_receipt_rule_set_name(
self,
"ExistingRuleSet",
"INBOUND_MAIL",
)
rule_set.add_rule(
"PoEmailRule",
recipients=["amazon_po@int.seahaven.com"],
actions=[
ses_actions.S3(
bucket=email_bucket,
object_key_prefix="inbound/",
),
],
)
Land safe fixes from 2026-06-17 security sweep (#97) * Remove gratuitous KMS grant on shared DynamoDB CMK wo-email-processor held grant_encrypt_decrypt on the shared seahaven-dynamodb CMK, but the WorkOrders/WorkOrderComments tables are not encrypted with that CMK. The grant was dead weight that extended the WO processor's decrypt reach to the CMK protecting the purchase-orders table (cross-stack decrypt). Drop it to restore least privilege; re-add as part of the table CMK migration (INFRA-6). Refs: INFRA-6 * Require Secrets Manager key for Anthropic client Remove the silent fallback to a plaintext ANTHROPIC_API_KEY env var in both email processors; require ANTHROPIC_API_KEY_SECRET_ARN and raise if absent so a misconfigured deploy fails loudly instead of using an unmanaged key. Adapted from f175323 on security/sweep-2026-06-17. The From-header sender-domain allowlist from that commit is intentionally dropped: the From header is spoofable (INFRA-107, confirmed critical) and sender authentication is being reworked in a separate PR. Refs: INFRA-107 * Merge PO revisions and handle out-of-order events save_revision did a full put_item overwrite, so a revision omitting line_items/supplier permanently deleted them. save_new_po used a conditional put that silently dropped the PO when an out-of-order cancellation had already created a skeleton row. Switch both to field-level merge update_items: a revision now SETs only the fields it carries, and a new_po backfills data into a pre-existing Cancelled skeleton while preserving the Cancelled status. No email can now delete data established by an earlier one. * Gate web UIs behind auth and escape currency XSS The po-web-ui and workorder-web-ui handlers had no auth: any invocation path returned the full PO/WO DB. Add a fail-closed shared-secret gate (X-Auth-Token / Bearer, constant-time compared to WEB_UI_AUTH_TOKEN) so a future re-attached Function URL cannot re-expose the data (URLs removed under INFRA-74). Wire the token from the SSM String param /procurement-ingest/web-ui-auth-token. Also fix stored XSS in po-web-ui fmt_currency: the non-numeric fallback returned str(val) unescaped, so a prompt-injected email could make Claude emit total_amount as <script>. Escape it. Refs: INFRA-74 * Document sweep security fixes and merge semantics Update the README for the 2026-06-17 security sweep: required Secrets Manager key (no plaintext env fallback), web UI auth gate + SSM token setup step, output-escaping note, and the new PO revision/cancellation merge behavior. Adapted from d91f45e on security/sweep-2026-06-17; the sender allowlist documentation is dropped along with the allowlist itself (deferred to the INFRA-107 sender-authentication rework). Refs: INFRA-107 * fix: resolve web UI auth token from Secrets Manager at runtime Replace the plaintext SSM String parameter with a Secrets Manager secret referenced by ARN only. The token is fetched and cached at module level on first invocation, keeping shared secrets out of CloudFormation templates and Lambda environment variables. Refs: PR-97 * Add TTL to web UI auth token cache for rotation The web-ui handlers cached the Secrets Manager auth token at module level with no expiry, so a rotated secret was only picked up when the warm container recycled — an emergency rotation could take hours to take effect. Cache the fetched value for a 5-minute TTL instead, so a rotated token propagates within the TTL while still avoiding a Secrets Manager call on every request. Still fails closed when the secret is unset or unreadable. Refs: INFRA-74 * Log Secrets Manager failures in web UI auth token fetch The web UI auth gate correctly fails closed when the shared token cannot be read, but _get_auth_token() swallowed every exception silently. A Secrets Manager permission or config error then made every request 401 with no operational signal, leaving an outage indistinguishable from ordinary unauthenticated traffic. Add a module-level logger to both web_ui handlers and log the fetch failure with logger.exception() in the except block before returning None. Behavior is unchanged (still fails closed); the failure is now visible in CloudWatch. The secret value is never logged. The two handlers stay byte-consistent in the mirrored _get_auth_token() region. The companion finding on the CDK import of the shared procurement-ingest/web-ui-auth-token secret was evaluated and left as-is: the token is a single secret shared by both the PO and WO stacks, so from_secret_name_v2 (which scopes grant_read via the standard 6-char suffix wildcard) is correct; making it a CDK-managed Secret in both stacks would collide the two stacks on the same explicit secret name at deploy time. Refs: INFRA-74 * Make Cancelled PO status sticky via atomic write The PO merge path read status with a get_item (_is_cancelled) and then wrote with an unconditional update_item. Two defects followed from this: - Race (Issue A): a cancellation landing between the read and the write was silently un-cancelled by a revision carrying a non-cancelled po_status — a TOCTOU on a table with concurrent email processing. - Over-broad strip (Issue B): save_revision dropped po_status whenever the PO was Cancelled, so legitimate status updates on non-cancelled POs and status-less revisions were affected rather than only the true un-cancel transition. Enforce the invariant server-side instead. "Cancelled" is a sticky, authoritative status: once set, later new_po/revision emails may enrich other fields but must never move it to a non-cancelled status. When the payload carries a non-cancelled po_status, _merge_update issues the update_item guarded by ConditionExpression "attribute_not_exists(po_status) OR po_status <> :marker", evaluated atomically at write time, so a cancellation that lands first always wins. On ConditionalCheckFailedException the same fields are re-written without po_status/cancelled_at, enriching the record while Cancelled sticks. Payloads with no status change, or an already -Cancelled status, take a plain merge — the status is only ever suppressed on a real un-cancel. This removes the non-atomic get_item from the write path; _is_cancelled is deleted. Key schema and attribute names are unchanged, so the cross-stack purchase-orders contract (read-only by seahaven-slack-bot) holds. Add moto-backed tests covering un-cancel suppression with field enrichment, status-less merge onto a Cancelled PO, legitimate status updates on non-cancelled POs, new_po backfill of a Cancelled skeleton, fresh create/merge, and authoritative save_cancellation. Refs: #97
2026-07-15 20:17:46 -04:00
# --- Web UI auth token secret ---
# Shared secret for the web UI auth gate, stored in Secrets Manager and
# resolved at runtime so the token never appears in CloudFormation templates
# or Lambda environment variables. Create this secret before deploying
# either stack; both PO and WO stacks reference it by name.
web_ui_auth_secret = secretsmanager.Secret.from_secret_name_v2(
self,
"WebUiAuthToken",
"procurement-ingest/web-ui-auth-token",
)
# --- Web UI Lambda ---
web_ui = lambda_.Function(
self,
"WebUI",
function_name="po-web-ui",
runtime=lambda_.Runtime.PYTHON_3_12,
architecture=lambda_.Architecture.ARM_64,
handler="handler.handler",
Merge workorder-ingest into unified procurement repo (#22) * Merge workorder-ingest pipeline into unified repo Move PO lambdas under lambdas/po/, add WO pipeline under lambdas/wo/. Two independent CloudFormation stacks in one CDK app. Fix WO stack compliance: ARM64 architecture, 60-day log retention, aarch64 bundling, RETAIN on Anthropic secret. Remove stale CodePipeline buildspec. * Fix test_local.py import path and remove dead shared/models.py test_local.py referenced the old lambdas/email_processor path. Updated to lambdas/wo/email_processor. Removed shared/ directory entirely as nothing imports from it. * Escape HTML in both web UI dashboards to prevent XSS Both Function URLs are public (auth_type=NONE) and render email-derived content via f-strings. Attacker-crafted emails could inject scripts. Added html.escape() on all interpolated values in both PO and WO dashboards. * Add pagination to WO web UI scan get_work_orders() only fetched the first 1MB page from DynamoDB. Loop on LastEvaluatedKey to match the PO web UI pattern. * Fix esc(None) TypeError and javascript: scheme in PO web UI Coerce supplier name through `or ""` before escaping to handle nested None from DynamoDB. Add scheme allowlist on view_order_url to block javascript:/data: hrefs from LLM-extracted URLs. * Fix WO render_badge None guard, updated_at slice, and backfill path Add null guard to WO render_badge matching the PO version. Use `or ""` before slicing updated_at to handle explicit None values. Fix backfill_sites.py sys.path to use new lambdas/po/site_extractor. * Harden WO web UI and fix JS-context XSS in both dashboards - Use json.dumps for onclick URLs to prevent JS string breakout - Add .lower() to WO render_badge color lookup matching PO pattern - Add pagination to get_comments query - Cap get_work_orders to 500 results matching PO pattern * Apply ruff formatting to web UI handlers
2026-05-12 15:21:06 -04:00
code=lambda_.Code.from_asset("../lambdas/po/web_ui"),
timeout=Duration.seconds(60),
memory_size=256,
log_retention=logs.RetentionDays.TWO_MONTHS,
environment={
"PO_TABLE": "purchase-orders",
Land safe fixes from 2026-06-17 security sweep (#97) * Remove gratuitous KMS grant on shared DynamoDB CMK wo-email-processor held grant_encrypt_decrypt on the shared seahaven-dynamodb CMK, but the WorkOrders/WorkOrderComments tables are not encrypted with that CMK. The grant was dead weight that extended the WO processor's decrypt reach to the CMK protecting the purchase-orders table (cross-stack decrypt). Drop it to restore least privilege; re-add as part of the table CMK migration (INFRA-6). Refs: INFRA-6 * Require Secrets Manager key for Anthropic client Remove the silent fallback to a plaintext ANTHROPIC_API_KEY env var in both email processors; require ANTHROPIC_API_KEY_SECRET_ARN and raise if absent so a misconfigured deploy fails loudly instead of using an unmanaged key. Adapted from f175323 on security/sweep-2026-06-17. The From-header sender-domain allowlist from that commit is intentionally dropped: the From header is spoofable (INFRA-107, confirmed critical) and sender authentication is being reworked in a separate PR. Refs: INFRA-107 * Merge PO revisions and handle out-of-order events save_revision did a full put_item overwrite, so a revision omitting line_items/supplier permanently deleted them. save_new_po used a conditional put that silently dropped the PO when an out-of-order cancellation had already created a skeleton row. Switch both to field-level merge update_items: a revision now SETs only the fields it carries, and a new_po backfills data into a pre-existing Cancelled skeleton while preserving the Cancelled status. No email can now delete data established by an earlier one. * Gate web UIs behind auth and escape currency XSS The po-web-ui and workorder-web-ui handlers had no auth: any invocation path returned the full PO/WO DB. Add a fail-closed shared-secret gate (X-Auth-Token / Bearer, constant-time compared to WEB_UI_AUTH_TOKEN) so a future re-attached Function URL cannot re-expose the data (URLs removed under INFRA-74). Wire the token from the SSM String param /procurement-ingest/web-ui-auth-token. Also fix stored XSS in po-web-ui fmt_currency: the non-numeric fallback returned str(val) unescaped, so a prompt-injected email could make Claude emit total_amount as <script>. Escape it. Refs: INFRA-74 * Document sweep security fixes and merge semantics Update the README for the 2026-06-17 security sweep: required Secrets Manager key (no plaintext env fallback), web UI auth gate + SSM token setup step, output-escaping note, and the new PO revision/cancellation merge behavior. Adapted from d91f45e on security/sweep-2026-06-17; the sender allowlist documentation is dropped along with the allowlist itself (deferred to the INFRA-107 sender-authentication rework). Refs: INFRA-107 * fix: resolve web UI auth token from Secrets Manager at runtime Replace the plaintext SSM String parameter with a Secrets Manager secret referenced by ARN only. The token is fetched and cached at module level on first invocation, keeping shared secrets out of CloudFormation templates and Lambda environment variables. Refs: PR-97 * Add TTL to web UI auth token cache for rotation The web-ui handlers cached the Secrets Manager auth token at module level with no expiry, so a rotated secret was only picked up when the warm container recycled — an emergency rotation could take hours to take effect. Cache the fetched value for a 5-minute TTL instead, so a rotated token propagates within the TTL while still avoiding a Secrets Manager call on every request. Still fails closed when the secret is unset or unreadable. Refs: INFRA-74 * Log Secrets Manager failures in web UI auth token fetch The web UI auth gate correctly fails closed when the shared token cannot be read, but _get_auth_token() swallowed every exception silently. A Secrets Manager permission or config error then made every request 401 with no operational signal, leaving an outage indistinguishable from ordinary unauthenticated traffic. Add a module-level logger to both web_ui handlers and log the fetch failure with logger.exception() in the except block before returning None. Behavior is unchanged (still fails closed); the failure is now visible in CloudWatch. The secret value is never logged. The two handlers stay byte-consistent in the mirrored _get_auth_token() region. The companion finding on the CDK import of the shared procurement-ingest/web-ui-auth-token secret was evaluated and left as-is: the token is a single secret shared by both the PO and WO stacks, so from_secret_name_v2 (which scopes grant_read via the standard 6-char suffix wildcard) is correct; making it a CDK-managed Secret in both stacks would collide the two stacks on the same explicit secret name at deploy time. Refs: INFRA-74 * Make Cancelled PO status sticky via atomic write The PO merge path read status with a get_item (_is_cancelled) and then wrote with an unconditional update_item. Two defects followed from this: - Race (Issue A): a cancellation landing between the read and the write was silently un-cancelled by a revision carrying a non-cancelled po_status — a TOCTOU on a table with concurrent email processing. - Over-broad strip (Issue B): save_revision dropped po_status whenever the PO was Cancelled, so legitimate status updates on non-cancelled POs and status-less revisions were affected rather than only the true un-cancel transition. Enforce the invariant server-side instead. "Cancelled" is a sticky, authoritative status: once set, later new_po/revision emails may enrich other fields but must never move it to a non-cancelled status. When the payload carries a non-cancelled po_status, _merge_update issues the update_item guarded by ConditionExpression "attribute_not_exists(po_status) OR po_status <> :marker", evaluated atomically at write time, so a cancellation that lands first always wins. On ConditionalCheckFailedException the same fields are re-written without po_status/cancelled_at, enriching the record while Cancelled sticks. Payloads with no status change, or an already -Cancelled status, take a plain merge — the status is only ever suppressed on a real un-cancel. This removes the non-atomic get_item from the write path; _is_cancelled is deleted. Key schema and attribute names are unchanged, so the cross-stack purchase-orders contract (read-only by seahaven-slack-bot) holds. Add moto-backed tests covering un-cancel suppression with field enrichment, status-less merge onto a Cancelled PO, legitimate status updates on non-cancelled POs, new_po backfill of a Cancelled skeleton, fresh create/merge, and authoritative save_cancellation. Refs: #97
2026-07-15 20:17:46 -04:00
# Defense-in-depth shared secret for the web UI handler. The
# handler fails closed if this ARN is unset or the secret is
# missing, so any future invocation path cannot re-expose the
# PO DB unauthenticated. The secret value is fetched at runtime
# from Secrets Manager (not embedded in env vars or template).
"WEB_UI_AUTH_TOKEN_SECRET_ARN": web_ui_auth_secret.secret_arn,
},
)
po_table.grant_read_data(web_ui)
Land safe fixes from 2026-06-17 security sweep (#97) * Remove gratuitous KMS grant on shared DynamoDB CMK wo-email-processor held grant_encrypt_decrypt on the shared seahaven-dynamodb CMK, but the WorkOrders/WorkOrderComments tables are not encrypted with that CMK. The grant was dead weight that extended the WO processor's decrypt reach to the CMK protecting the purchase-orders table (cross-stack decrypt). Drop it to restore least privilege; re-add as part of the table CMK migration (INFRA-6). Refs: INFRA-6 * Require Secrets Manager key for Anthropic client Remove the silent fallback to a plaintext ANTHROPIC_API_KEY env var in both email processors; require ANTHROPIC_API_KEY_SECRET_ARN and raise if absent so a misconfigured deploy fails loudly instead of using an unmanaged key. Adapted from f175323 on security/sweep-2026-06-17. The From-header sender-domain allowlist from that commit is intentionally dropped: the From header is spoofable (INFRA-107, confirmed critical) and sender authentication is being reworked in a separate PR. Refs: INFRA-107 * Merge PO revisions and handle out-of-order events save_revision did a full put_item overwrite, so a revision omitting line_items/supplier permanently deleted them. save_new_po used a conditional put that silently dropped the PO when an out-of-order cancellation had already created a skeleton row. Switch both to field-level merge update_items: a revision now SETs only the fields it carries, and a new_po backfills data into a pre-existing Cancelled skeleton while preserving the Cancelled status. No email can now delete data established by an earlier one. * Gate web UIs behind auth and escape currency XSS The po-web-ui and workorder-web-ui handlers had no auth: any invocation path returned the full PO/WO DB. Add a fail-closed shared-secret gate (X-Auth-Token / Bearer, constant-time compared to WEB_UI_AUTH_TOKEN) so a future re-attached Function URL cannot re-expose the data (URLs removed under INFRA-74). Wire the token from the SSM String param /procurement-ingest/web-ui-auth-token. Also fix stored XSS in po-web-ui fmt_currency: the non-numeric fallback returned str(val) unescaped, so a prompt-injected email could make Claude emit total_amount as <script>. Escape it. Refs: INFRA-74 * Document sweep security fixes and merge semantics Update the README for the 2026-06-17 security sweep: required Secrets Manager key (no plaintext env fallback), web UI auth gate + SSM token setup step, output-escaping note, and the new PO revision/cancellation merge behavior. Adapted from d91f45e on security/sweep-2026-06-17; the sender allowlist documentation is dropped along with the allowlist itself (deferred to the INFRA-107 sender-authentication rework). Refs: INFRA-107 * fix: resolve web UI auth token from Secrets Manager at runtime Replace the plaintext SSM String parameter with a Secrets Manager secret referenced by ARN only. The token is fetched and cached at module level on first invocation, keeping shared secrets out of CloudFormation templates and Lambda environment variables. Refs: PR-97 * Add TTL to web UI auth token cache for rotation The web-ui handlers cached the Secrets Manager auth token at module level with no expiry, so a rotated secret was only picked up when the warm container recycled — an emergency rotation could take hours to take effect. Cache the fetched value for a 5-minute TTL instead, so a rotated token propagates within the TTL while still avoiding a Secrets Manager call on every request. Still fails closed when the secret is unset or unreadable. Refs: INFRA-74 * Log Secrets Manager failures in web UI auth token fetch The web UI auth gate correctly fails closed when the shared token cannot be read, but _get_auth_token() swallowed every exception silently. A Secrets Manager permission or config error then made every request 401 with no operational signal, leaving an outage indistinguishable from ordinary unauthenticated traffic. Add a module-level logger to both web_ui handlers and log the fetch failure with logger.exception() in the except block before returning None. Behavior is unchanged (still fails closed); the failure is now visible in CloudWatch. The secret value is never logged. The two handlers stay byte-consistent in the mirrored _get_auth_token() region. The companion finding on the CDK import of the shared procurement-ingest/web-ui-auth-token secret was evaluated and left as-is: the token is a single secret shared by both the PO and WO stacks, so from_secret_name_v2 (which scopes grant_read via the standard 6-char suffix wildcard) is correct; making it a CDK-managed Secret in both stacks would collide the two stacks on the same explicit secret name at deploy time. Refs: INFRA-74 * Make Cancelled PO status sticky via atomic write The PO merge path read status with a get_item (_is_cancelled) and then wrote with an unconditional update_item. Two defects followed from this: - Race (Issue A): a cancellation landing between the read and the write was silently un-cancelled by a revision carrying a non-cancelled po_status — a TOCTOU on a table with concurrent email processing. - Over-broad strip (Issue B): save_revision dropped po_status whenever the PO was Cancelled, so legitimate status updates on non-cancelled POs and status-less revisions were affected rather than only the true un-cancel transition. Enforce the invariant server-side instead. "Cancelled" is a sticky, authoritative status: once set, later new_po/revision emails may enrich other fields but must never move it to a non-cancelled status. When the payload carries a non-cancelled po_status, _merge_update issues the update_item guarded by ConditionExpression "attribute_not_exists(po_status) OR po_status <> :marker", evaluated atomically at write time, so a cancellation that lands first always wins. On ConditionalCheckFailedException the same fields are re-written without po_status/cancelled_at, enriching the record while Cancelled sticks. Payloads with no status change, or an already -Cancelled status, take a plain merge — the status is only ever suppressed on a real un-cancel. This removes the non-atomic get_item from the write path; _is_cancelled is deleted. Key schema and attribute names are unchanged, so the cross-stack purchase-orders contract (read-only by seahaven-slack-bot) holds. Add moto-backed tests covering un-cancel suppression with field enrichment, status-less merge onto a Cancelled PO, legitimate status updates on non-cancelled POs, new_po backfill of a Cancelled skeleton, fresh create/merge, and authoritative save_cancellation. Refs: #97
2026-07-15 20:17:46 -04:00
web_ui_auth_secret.grant_read(web_ui)
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70) * Add CloudWatch alarm coverage for po-ingest and workorder-ingest Expands alarm coverage across both CDK stacks. All alarms are ALARM-only (no OK action) to the shared site-alerts SNS topic, with TreatMissingData NOT_BREACHING. The site-alerts topic is now imported once near the top of each stack so every alarm reuses one Topic instance. po-ingest (cdk/po_stack.py): - Errors: po-ingest-site-extractor - Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui - Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption), po-ingest-site-extractor, po-web-ui - DynamoDB throttle + system-error: purchase-orders, verified-sites, pending-site-review workorder-ingest (cdk/wo_stack.py): - Throttles: workorder-email-processor - Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption) - DynamoDB throttle + system-error: WorkOrders, WorkOrderComments DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation dimension set, so each table alarm is a Sum math expression across operations via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0). Refs INFRA-41 / audit H-8. * Drop NEEDS ADAM SIGN-OFF wording from alarm comments Duration alarm thresholds are owner-approved; remove the sign-off flag from po_stack.py and wo_stack.py comments. Threshold values, eval config, and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
# --- Throttles alarm: po-web-ui ---
web_ui.metric_throttles(
period=Duration.minutes(5),
statistic="Sum",
).create_alarm(
self,
"WebUiThrottlesAlarm",
alarm_name="po-web-ui-throttles",
alarm_description="po-web-ui invocation throttles",
threshold=0,
evaluation_periods=1,
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
# --- Duration alarm: po-web-ui ---
# Net-new (no orphan exists for this function).
# p99 / 45000 ms (75% of the 60s timeout) / eval 3, datapoints 2.
web_ui.metric_duration(
period=Duration.minutes(5),
statistic="p99",
).create_alarm(
self,
"WebUiDurationAlarm",
alarm_name="po-web-ui-duration",
alarm_description="po-web-ui p99 duration approaching the 60s timeout",
threshold=45000,
evaluation_periods=3,
datapoints_to_alarm=2,
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
# Public Function URL removed 2026-06-08 (INFRA-74 / audit C-5): the
# unauthenticated FunctionUrlAuthType.NONE URL was deleted out-of-band
# via CLI. Removing the construct (and its auto-generated Principal:*
# invoke permission) reconciles IaC with the live state.
# --- Verified sites table (extracted from PO ship-to addresses) ---
verified_sites_table = dynamodb.Table(
self,
"VerifiedSitesTable",
table_name="verified-sites",
partition_key=dynamodb.Attribute(
name="siteCode",
type=dynamodb.AttributeType.STRING,
),
billing_mode=dynamodb.BillingMode.PAY_PER_REQUEST,
removal_policy=RemovalPolicy.RETAIN,
)
# by-state GSI removed 2026-06-03 (audit M-20): 0 reads in 30d against
# 518 WCU of write amplification. Re-add if a state-level query path ships.
# --- Site extractor Lambda (DynamoDB Streams → verified-sites) ---
site_extractor = lambda_.Function(
self,
"SiteExtractor",
function_name="po-ingest-site-extractor",
runtime=lambda_.Runtime.PYTHON_3_12,
architecture=lambda_.Architecture.ARM_64,
handler="handler.handler",
Merge workorder-ingest into unified procurement repo (#22) * Merge workorder-ingest pipeline into unified repo Move PO lambdas under lambdas/po/, add WO pipeline under lambdas/wo/. Two independent CloudFormation stacks in one CDK app. Fix WO stack compliance: ARM64 architecture, 60-day log retention, aarch64 bundling, RETAIN on Anthropic secret. Remove stale CodePipeline buildspec. * Fix test_local.py import path and remove dead shared/models.py test_local.py referenced the old lambdas/email_processor path. Updated to lambdas/wo/email_processor. Removed shared/ directory entirely as nothing imports from it. * Escape HTML in both web UI dashboards to prevent XSS Both Function URLs are public (auth_type=NONE) and render email-derived content via f-strings. Attacker-crafted emails could inject scripts. Added html.escape() on all interpolated values in both PO and WO dashboards. * Add pagination to WO web UI scan get_work_orders() only fetched the first 1MB page from DynamoDB. Loop on LastEvaluatedKey to match the PO web UI pattern. * Fix esc(None) TypeError and javascript: scheme in PO web UI Coerce supplier name through `or ""` before escaping to handle nested None from DynamoDB. Add scheme allowlist on view_order_url to block javascript:/data: hrefs from LLM-extracted URLs. * Fix WO render_badge None guard, updated_at slice, and backfill path Add null guard to WO render_badge matching the PO version. Use `or ""` before slicing updated_at to handle explicit None values. Fix backfill_sites.py sys.path to use new lambdas/po/site_extractor. * Harden WO web UI and fix JS-context XSS in both dashboards - Use json.dumps for onclick URLs to prevent JS string breakout - Add .lower() to WO render_badge color lookup matching PO pattern - Add pagination to get_comments query - Cap get_work_orders to 500 results matching PO pattern * Apply ruff formatting to web UI handlers
2026-05-12 15:21:06 -04:00
code=lambda_.Code.from_asset("../lambdas/po/site_extractor"),
timeout=Duration.seconds(60),
memory_size=256,
log_retention=logs.RetentionDays.TWO_MONTHS,
environment={
"VERIFIED_SITES_TABLE": verified_sites_table.table_name,
"PENDING_REVIEW_TABLE": "pending-site-review",
},
)
verified_sites_table.grant_read_write_data(site_extractor)
site_extractor.add_event_source(
lambda_event_sources.DynamoEventSource(
po_table,
starting_position=lambda_.StartingPosition.TRIM_HORIZON,
batch_size=10,
max_batching_window=Duration.seconds(30),
bisect_batch_on_error=True,
retry_attempts=3,
)
)
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70) * Add CloudWatch alarm coverage for po-ingest and workorder-ingest Expands alarm coverage across both CDK stacks. All alarms are ALARM-only (no OK action) to the shared site-alerts SNS topic, with TreatMissingData NOT_BREACHING. The site-alerts topic is now imported once near the top of each stack so every alarm reuses one Topic instance. po-ingest (cdk/po_stack.py): - Errors: po-ingest-site-extractor - Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui - Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption), po-ingest-site-extractor, po-web-ui - DynamoDB throttle + system-error: purchase-orders, verified-sites, pending-site-review workorder-ingest (cdk/wo_stack.py): - Throttles: workorder-email-processor - Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption) - DynamoDB throttle + system-error: WorkOrders, WorkOrderComments DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation dimension set, so each table alarm is a Sum math expression across operations via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0). Refs INFRA-41 / audit H-8. * Drop NEEDS ADAM SIGN-OFF wording from alarm comments Duration alarm thresholds are owner-approved; remove the sign-off flag from po_stack.py and wo_stack.py comments. Threshold values, eval config, and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
# --- Errors alarm: po-ingest-site-extractor ---
# Stream-consumer errors retry per the event-source config, but a
# persistent failure stalls the verified-sites pipeline. ALARM-only to
# site-alerts; no OK action; NOT_BREACHING when no data.
site_extractor.metric_errors(
period=Duration.minutes(5),
statistic="Sum",
).create_alarm(
self,
"SiteExtractorErrorsAlarm",
alarm_name="po-ingest-site-extractor-errors",
alarm_description="po-ingest-site-extractor invocation errors",
threshold=0,
evaluation_periods=1,
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
# --- Throttles alarm: po-ingest-site-extractor ---
site_extractor.metric_throttles(
period=Duration.minutes(5),
statistic="Sum",
).create_alarm(
self,
"SiteExtractorThrottlesAlarm",
alarm_name="po-ingest-site-extractor-throttles",
alarm_description="po-ingest-site-extractor invocation throttles",
threshold=0,
evaluation_periods=1,
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_THRESHOLD,
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
# --- Duration alarm: po-ingest-site-extractor ---
# Net-new (no orphan exists for this function).
# p99 / 45000 ms (75% of the 60s timeout) / eval 3, datapoints 2.
site_extractor.metric_duration(
period=Duration.minutes(5),
statistic="p99",
).create_alarm(
self,
"SiteExtractorDurationAlarm",
alarm_name="po-ingest-site-extractor-duration",
alarm_description="po-ingest-site-extractor p99 duration approaching the 60s timeout",
threshold=45000,
evaluation_periods=3,
datapoints_to_alarm=2,
comparison_operator=cloudwatch.ComparisonOperator.GREATER_THAN_OR_EQUAL_TO_THRESHOLD,
treat_missing_data=cloudwatch.TreatMissingData.NOT_BREACHING,
).add_alarm_action(cw_actions.SnsAction(alarm_topic))
cdk.CfnOutput(
self,
"VerifiedSitesTableName",
value=verified_sites_table.table_name,
description="Verified site addresses extracted from POs",
)
# --- Pending site review table (POs with no extractable site code) ---
pending_review_table = dynamodb.Table(
self,
"PendingSiteReviewTable",
table_name="pending-site-review",
partition_key=dynamodb.Attribute(
name="po_number",
type=dynamodb.AttributeType.STRING,
),
billing_mode=dynamodb.BillingMode.PAY_PER_REQUEST,
removal_policy=RemovalPolicy.RETAIN,
)
pending_review_table.grant_read_write_data(site_extractor)
verified_sites_table.grant_read_data(site_extractor)
Add CloudWatch alarm coverage for po-ingest and workorder-ingest (#70) * Add CloudWatch alarm coverage for po-ingest and workorder-ingest Expands alarm coverage across both CDK stacks. All alarms are ALARM-only (no OK action) to the shared site-alerts SNS topic, with TreatMissingData NOT_BREACHING. The site-alerts topic is now imported once near the top of each stack so every alarm reuses one Topic instance. po-ingest (cdk/po_stack.py): - Errors: po-ingest-site-extractor - Throttles: po-email-processor, po-ingest-site-extractor, po-web-ui - Duration (p99, >=45000ms, eval3/dp2): po-email-processor (orphan adoption), po-ingest-site-extractor, po-web-ui - DynamoDB throttle + system-error: purchase-orders, verified-sites, pending-site-review workorder-ingest (cdk/wo_stack.py): - Throttles: workorder-email-processor - Duration (p95, >=45000ms, eval3/dp2): workorder-email-processor (orphan adoption) - DynamoDB throttle + system-error: WorkOrders, WorkOrderComments DynamoDB ThrottledRequests/SystemErrors emit only at the TableName+Operation dimension set, so each table alarm is a Sum math expression across operations via the non-deprecated metric_*_for_operations helpers (metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0). Refs INFRA-41 / audit H-8. * Drop NEEDS ADAM SIGN-OFF wording from alarm comments Duration alarm thresholds are owner-approved; remove the sign-off flag from po_stack.py and wo_stack.py comments. Threshold values, eval config, and orphan-delete notes are unchanged.
2026-06-17 14:46:03 -04:00
# --- DynamoDB throttle + system-error alarms ---
# ThrottledRequests / SystemErrors emit at TableName + Operation only
# (verified against live CloudWatch: no TableName-only rollup exists, and
# metric_throttled_requests is deprecated/invalid in aws-cdk-lib 2.259.0).
# Each table currently has zero throttle/error datapoints, so the series
# only materialise on first occurrence — NOT_BREACHING keeps them OK until
# then.
_add_ddb_alarms(
self, "PurchaseOrdersTable", po_table, "purchase-orders", alarm_topic
)
_add_ddb_alarms(
self,
"VerifiedSitesTable",
verified_sites_table,
"verified-sites",
alarm_topic,
)
_add_ddb_alarms(
self,
"PendingSiteReviewTable",
pending_review_table,
"pending-site-review",
alarm_topic,
)